@bike4mind/cli 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/bin/bike4mind-cli.mjs +0 -5
- package/dist/AgentHistoryStore-B2NEOvSW.mjs +18014 -0
- package/dist/{ApiClient-BOWpVvTq.mjs → ApiClient-D0fQ2FT6.mjs} +2 -2
- package/dist/{BubblewrapRuntime-CkL9-gnG.mjs → BubblewrapRuntime-5bLPTwEC.mjs} +2 -2
- package/dist/{ConfigStore-DdHHCH2t.mjs → ConfigStore-qV7NrCgZ.mjs} +1154 -1365
- package/dist/{ProxyManager-C1-lgzEU.mjs → ProxyManager-B0-RuR2w.mjs} +21 -4
- package/dist/{SandboxOrchestrator-CIegJrCj.mjs → SandboxOrchestrator-BcUv9fQ3.mjs} +46 -12
- package/dist/{SandboxRuntimeAdapter-ChGlxSGQ.mjs → SandboxRuntimeAdapter-BgLUVTJL.mjs} +2 -2
- package/dist/{SandboxRuntimeAdapter-CKelGICD.mjs → SandboxRuntimeAdapter-BmHELLuM.mjs} +1 -1
- package/dist/{SeatbeltRuntime-Qqt19cAN.mjs → SeatbeltRuntime-C_Y8q8Mr.mjs} +9 -1
- package/dist/buildAgent-C-C-VGff.mjs +2001 -0
- package/dist/commands/acpCommand.mjs +8 -6
- package/dist/commands/apiCommand.mjs +1 -1
- package/dist/commands/doctorCommand.mjs +1 -1
- package/dist/commands/envCommand.mjs +1 -1
- package/dist/commands/headlessCommand.mjs +26 -16
- package/dist/commands/mcpCommand.mjs +3 -3
- package/dist/commands/pluginCommand.mjs +1 -1
- package/dist/commands/updateCommand.mjs +1 -1
- package/dist/index.mjs +444 -100
- package/dist/{package-Bqg2LSnH.mjs → package-CGZIoxcs.mjs} +1 -1
- package/dist/{serve-DO9Edl5S.mjs → serve-CPXqcEZr.mjs} +2 -2
- package/package.json +9 -10
- package/dist/AgentHistoryStore-kT9eNMRO.mjs +0 -39471
- package/dist/ProxyManager-B1jFWL7b.mjs +0 -3
- package/dist/SandboxOrchestrator-BbMDgjzr.mjs +0 -3
- package/dist/buildAgent-B_kArQ_Y.mjs +0 -833
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { t as DEFAULT_SANDBOX_CONFIG } from "./types-F61_hxmG.mjs";
|
|
3
|
-
import "crypto";
|
|
3
|
+
import { createHmac } from "crypto";
|
|
4
4
|
import { existsSync, promises } from "fs";
|
|
5
5
|
import os, { homedir } from "os";
|
|
6
6
|
import path from "path";
|
|
7
7
|
import { v4 } from "uuid";
|
|
8
8
|
import * as path$1 from "node:path";
|
|
9
|
-
import * as fs$
|
|
9
|
+
import * as fs$1 from "node:fs";
|
|
10
10
|
import * as z$2 from "zod";
|
|
11
11
|
import z, { ZodError, z as z$1 } from "zod";
|
|
12
12
|
import dayjs from "dayjs";
|
|
@@ -16,162 +16,9 @@ import relativeTime from "dayjs/plugin/relativeTime.js";
|
|
|
16
16
|
import localizedFormat from "dayjs/plugin/localizedFormat.js";
|
|
17
17
|
import { isAxiosError } from "axios";
|
|
18
18
|
import { homedir as homedir$1 } from "node:os";
|
|
19
|
-
import fs
|
|
19
|
+
import fs from "fs/promises";
|
|
20
20
|
process.env.APP_NAME;
|
|
21
21
|
process.env.WEBSITE_URL;
|
|
22
|
-
const SNIPPET_META_OPEN = "<!--snippet-meta";
|
|
23
|
-
const SNIPPET_META_CLOSE = "-->";
|
|
24
|
-
/** Index of the first non-whitespace character at or after `from`. */
|
|
25
|
-
const skipSpaceForward = (text, from) => {
|
|
26
|
-
let i = from;
|
|
27
|
-
while (i < text.length && /\s/.test(text[i])) i++;
|
|
28
|
-
return i;
|
|
29
|
-
};
|
|
30
|
-
/** Index just past the last non-whitespace character before `before`. */
|
|
31
|
-
const skipSpaceBackward = (text, before) => {
|
|
32
|
-
let i = before;
|
|
33
|
-
while (i > 0 && /\s/.test(text[i - 1])) i--;
|
|
34
|
-
return i;
|
|
35
|
-
};
|
|
36
|
-
/**
|
|
37
|
-
* Split a prompt into its `<!--snippet-meta {...}-->` sections and the text around them.
|
|
38
|
-
*
|
|
39
|
-
* Scanned with indexOf rather than a regex. The regex this replaces backtracked
|
|
40
|
-
* super-linearly on a marker whose JSON never closed, which invited a parse cap - but a
|
|
41
|
-
* cap could not be applied safely here: the section pattern was terminated by
|
|
42
|
-
* `(?=<!--snippet-meta|$)`, so capping moved `$` and changed section SHAPE, not just
|
|
43
|
-
* length. Past the cap a snippet re-emitted as a `text` section, and callers that skip
|
|
44
|
-
* snippets when collecting URLs to fetch (utils/src/llm/utils.ts) would start fetching
|
|
45
|
-
* them. A linear scan needs no cap, so the shape is the same at every input length.
|
|
46
|
-
*/
|
|
47
|
-
const extractSnippetMeta = (content) => {
|
|
48
|
-
const sections = [];
|
|
49
|
-
let cursor = 0;
|
|
50
|
-
let search = 0;
|
|
51
|
-
while (search < content.length) {
|
|
52
|
-
const open = content.indexOf(SNIPPET_META_OPEN, search);
|
|
53
|
-
if (open === -1) break;
|
|
54
|
-
const metaStart = skipSpaceForward(content, open + 16);
|
|
55
|
-
if (content[metaStart] !== "{") {
|
|
56
|
-
search = open + 16;
|
|
57
|
-
continue;
|
|
58
|
-
}
|
|
59
|
-
let close = content.indexOf(SNIPPET_META_CLOSE, metaStart);
|
|
60
|
-
while (close !== -1) {
|
|
61
|
-
const metaEnd = skipSpaceBackward(content, close);
|
|
62
|
-
if (metaEnd > metaStart && content[metaEnd - 1] === "}") break;
|
|
63
|
-
close = content.indexOf(SNIPPET_META_CLOSE, close + 3);
|
|
64
|
-
}
|
|
65
|
-
if (close === -1) break;
|
|
66
|
-
const bodyStart = close + 3;
|
|
67
|
-
const nextMarker = content.indexOf(SNIPPET_META_OPEN, bodyStart);
|
|
68
|
-
const bodyEnd = nextMarker === -1 ? content.length : nextMarker;
|
|
69
|
-
const textBefore = content.slice(cursor, open).trim();
|
|
70
|
-
if (textBefore) sections.push({
|
|
71
|
-
type: "text",
|
|
72
|
-
content: textBefore
|
|
73
|
-
});
|
|
74
|
-
try {
|
|
75
|
-
const meta = JSON.parse(content.slice(metaStart, skipSpaceBackward(content, close)));
|
|
76
|
-
const snippetContent = content.slice(bodyStart, bodyEnd).trim();
|
|
77
|
-
if (meta && snippetContent) sections.push({
|
|
78
|
-
type: "snippet",
|
|
79
|
-
meta,
|
|
80
|
-
content: snippetContent
|
|
81
|
-
});
|
|
82
|
-
} catch (e) {
|
|
83
|
-
console.error("Error parsing snippet meta:", e);
|
|
84
|
-
}
|
|
85
|
-
cursor = bodyEnd;
|
|
86
|
-
search = bodyEnd;
|
|
87
|
-
}
|
|
88
|
-
const remainingText = content.slice(cursor).trim();
|
|
89
|
-
if (remainingText) sections.push({
|
|
90
|
-
type: "text",
|
|
91
|
-
content: remainingText
|
|
92
|
-
});
|
|
93
|
-
return { sections };
|
|
94
|
-
};
|
|
95
|
-
function getHeader$1(headers, name) {
|
|
96
|
-
if (!headers || typeof headers !== "object") return null;
|
|
97
|
-
if (typeof headers.get === "function") {
|
|
98
|
-
const val = headers.get(name);
|
|
99
|
-
return typeof val === "string" ? val : null;
|
|
100
|
-
}
|
|
101
|
-
const value = headers[name] ?? headers[name.toLowerCase()];
|
|
102
|
-
return typeof value === "string" ? value : null;
|
|
103
|
-
}
|
|
104
|
-
function parseNumber(value) {
|
|
105
|
-
if (value === null || value === void 0) return null;
|
|
106
|
-
const num = Number(value);
|
|
107
|
-
return Number.isFinite(num) ? num : null;
|
|
108
|
-
}
|
|
109
|
-
/**
|
|
110
|
-
* Parse rate limit headers from an HTTP response.
|
|
111
|
-
*
|
|
112
|
-
* Reads standard headers:
|
|
113
|
-
* - `X-RateLimit-Limit` - max requests per window
|
|
114
|
-
* - `X-RateLimit-Remaining` - requests left
|
|
115
|
-
* - `X-RateLimit-Reset` - Unix epoch seconds when the window resets
|
|
116
|
-
* - `Retry-After` - seconds to wait (on 429 responses), or an HTTP-date
|
|
117
|
-
*/
|
|
118
|
-
function parseRateLimitHeaders(headers) {
|
|
119
|
-
const limitStr = getHeader$1(headers, "X-RateLimit-Limit") ?? getHeader$1(headers, "x-ratelimit-limit");
|
|
120
|
-
const remainingStr = getHeader$1(headers, "X-RateLimit-Remaining") ?? getHeader$1(headers, "x-ratelimit-remaining");
|
|
121
|
-
const resetStr = getHeader$1(headers, "X-RateLimit-Reset") ?? getHeader$1(headers, "x-ratelimit-reset");
|
|
122
|
-
const retryAfterStr = getHeader$1(headers, "Retry-After") ?? getHeader$1(headers, "retry-after");
|
|
123
|
-
const limit = parseNumber(limitStr);
|
|
124
|
-
const remaining = parseNumber(remainingStr);
|
|
125
|
-
let resetAt = null;
|
|
126
|
-
if (resetStr !== null) {
|
|
127
|
-
const resetNum = Number(resetStr);
|
|
128
|
-
if (Number.isFinite(resetNum)) resetAt = /* @__PURE__ */ new Date(resetNum * 1e3);
|
|
129
|
-
else {
|
|
130
|
-
const parsed = new Date(resetStr);
|
|
131
|
-
if (!isNaN(parsed.getTime())) resetAt = parsed;
|
|
132
|
-
}
|
|
133
|
-
}
|
|
134
|
-
let retryAfterMs = null;
|
|
135
|
-
if (retryAfterStr !== null) {
|
|
136
|
-
const retryNum = Number(retryAfterStr);
|
|
137
|
-
if (Number.isFinite(retryNum)) retryAfterMs = retryNum * 1e3;
|
|
138
|
-
else {
|
|
139
|
-
const parsed = new Date(retryAfterStr);
|
|
140
|
-
if (!isNaN(parsed.getTime())) retryAfterMs = Math.max(0, parsed.getTime() - Date.now());
|
|
141
|
-
}
|
|
142
|
-
}
|
|
143
|
-
let usagePercent = null;
|
|
144
|
-
if (limit !== null && limit > 0 && remaining !== null && remaining >= 0) usagePercent = Math.round((limit - remaining) / limit * 100);
|
|
145
|
-
return {
|
|
146
|
-
limit,
|
|
147
|
-
remaining,
|
|
148
|
-
resetAt,
|
|
149
|
-
retryAfterMs,
|
|
150
|
-
usagePercent
|
|
151
|
-
};
|
|
152
|
-
}
|
|
153
|
-
function isNearLimit(info, thresholdPercent = 80) {
|
|
154
|
-
if (info.usagePercent === null) return false;
|
|
155
|
-
return info.usagePercent >= thresholdPercent;
|
|
156
|
-
}
|
|
157
|
-
/**
|
|
158
|
-
* Build a structured log object for rate limit events.
|
|
159
|
-
* Used by all integration clients to emit consistent log lines.
|
|
160
|
-
*/
|
|
161
|
-
function buildRateLimitLogEntry(integration, endpoint, info, wasThrottled = false) {
|
|
162
|
-
return {
|
|
163
|
-
type: wasThrottled ? "RATE_LIMIT_ERROR" : "RATE_LIMIT",
|
|
164
|
-
integration,
|
|
165
|
-
endpoint,
|
|
166
|
-
limit: info.limit,
|
|
167
|
-
remaining: info.remaining,
|
|
168
|
-
resetAt: info.resetAt?.toISOString() ?? null,
|
|
169
|
-
retryAfterMs: info.retryAfterMs,
|
|
170
|
-
usagePercent: info.usagePercent,
|
|
171
|
-
wasThrottled,
|
|
172
|
-
timestamp: (/* @__PURE__ */ new Date()).toISOString()
|
|
173
|
-
};
|
|
174
|
-
}
|
|
175
22
|
//#endregion
|
|
176
23
|
//#region ../../b4m-core/hearth/dist/index.mjs
|
|
177
24
|
/**
|
|
@@ -319,23 +166,14 @@ z$1.object({
|
|
|
319
166
|
});
|
|
320
167
|
//#endregion
|
|
321
168
|
//#region ../../b4m-core/common/dist/index.mjs
|
|
322
|
-
let HttpStatus = /* @__PURE__ */ function(HttpStatus) {
|
|
323
|
-
HttpStatus[HttpStatus["Ok"] = 200] = "Ok";
|
|
324
|
-
HttpStatus[HttpStatus["Created"] = 201] = "Created";
|
|
325
|
-
HttpStatus[HttpStatus["BadRequest"] = 400] = "BadRequest";
|
|
326
|
-
HttpStatus[HttpStatus["Unauthorized"] = 401] = "Unauthorized";
|
|
327
|
-
HttpStatus[HttpStatus["Forbidden"] = 403] = "Forbidden";
|
|
328
|
-
HttpStatus[HttpStatus["NotFound"] = 404] = "NotFound";
|
|
329
|
-
HttpStatus[HttpStatus["Conflict"] = 409] = "Conflict";
|
|
330
|
-
HttpStatus[HttpStatus["UnprocessableEntity"] = 422] = "UnprocessableEntity";
|
|
331
|
-
HttpStatus[HttpStatus["TooManyRequests"] = 429] = "TooManyRequests";
|
|
332
|
-
HttpStatus[HttpStatus["InternalServerError"] = 500] = "InternalServerError";
|
|
333
|
-
HttpStatus[HttpStatus["BadGateway"] = 502] = "BadGateway";
|
|
334
|
-
return HttpStatus;
|
|
335
|
-
}({});
|
|
336
169
|
var HTTPError = class extends Error {
|
|
337
170
|
statusCode;
|
|
338
171
|
additionalInfo;
|
|
172
|
+
/**
|
|
173
|
+
* Set on a 5xx that reports a third party's failure the server handled correctly (a source site
|
|
174
|
+
* timing out), so `errorHandler` logs it at warn rather than paging as a server fault.
|
|
175
|
+
*/
|
|
176
|
+
expected;
|
|
339
177
|
constructor(statusCode, message, additionalInfo) {
|
|
340
178
|
super(message);
|
|
341
179
|
this.statusCode = statusCode;
|
|
@@ -367,47 +205,6 @@ var UnprocessableEntityError = class extends HTTPError {
|
|
|
367
205
|
this.name = "UnprocessableEntityError";
|
|
368
206
|
}
|
|
369
207
|
};
|
|
370
|
-
var BadRequestError = class extends HTTPError {
|
|
371
|
-
additionalInfo;
|
|
372
|
-
constructor(message, additionalInfo) {
|
|
373
|
-
super(400, message, additionalInfo);
|
|
374
|
-
this.additionalInfo = additionalInfo;
|
|
375
|
-
this.name = "BadRequestError";
|
|
376
|
-
}
|
|
377
|
-
};
|
|
378
|
-
var UnauthorizedError = class extends HTTPError {
|
|
379
|
-
additionalInfo;
|
|
380
|
-
constructor(message, additionalInfo) {
|
|
381
|
-
super(401, message, additionalInfo);
|
|
382
|
-
this.additionalInfo = additionalInfo;
|
|
383
|
-
this.name = "UnauthorizedError";
|
|
384
|
-
}
|
|
385
|
-
};
|
|
386
|
-
var ForbiddenError = class extends HTTPError {
|
|
387
|
-
additionalInfo;
|
|
388
|
-
constructor(message, additionalInfo) {
|
|
389
|
-
super(403, message, additionalInfo);
|
|
390
|
-
this.additionalInfo = additionalInfo;
|
|
391
|
-
this.name = "ForbiddenError";
|
|
392
|
-
}
|
|
393
|
-
};
|
|
394
|
-
var TooManyRequestsError = class extends HTTPError {
|
|
395
|
-
additionalInfo;
|
|
396
|
-
constructor(message, additionalInfo) {
|
|
397
|
-
super(429, message, additionalInfo);
|
|
398
|
-
this.additionalInfo = additionalInfo;
|
|
399
|
-
this.name = "TooManyRequestsError";
|
|
400
|
-
}
|
|
401
|
-
};
|
|
402
|
-
var CorruptedFileError = class extends HTTPError {
|
|
403
|
-
additionalInfo;
|
|
404
|
-
constructor(fileName, fileType, corruptionDetails, additionalInfo) {
|
|
405
|
-
const message = `File '${fileName}' (${fileType}) appears to be corrupted${corruptionDetails ? `: ${corruptionDetails}` : ""}. Please try uploading the file again.`;
|
|
406
|
-
super(422, message, additionalInfo);
|
|
407
|
-
this.additionalInfo = additionalInfo;
|
|
408
|
-
this.name = "CorruptedFileError";
|
|
409
|
-
}
|
|
410
|
-
};
|
|
411
208
|
function isZodError(err) {
|
|
412
209
|
return Boolean(err && (err instanceof ZodError || err.name === "ZodError"));
|
|
413
210
|
}
|
|
@@ -800,19 +597,7 @@ z$1.object({
|
|
|
800
597
|
content: z$1.string(),
|
|
801
598
|
metadata: ArtifactMetadataSchema.optional()
|
|
802
599
|
});
|
|
803
|
-
|
|
804
|
-
* Regex sub-pattern (as a string) that matches the attribute portion of an
|
|
805
|
-
* `<artifact ...>` opening tag. It handles:
|
|
806
|
-
* - newlines inside the attribute list (AI sometimes wraps long tags),
|
|
807
|
-
* - `>` characters inside double- or single-quoted attribute values.
|
|
808
|
-
*
|
|
809
|
-
* Exported as a string (not a compiled RegExp) so each consumer can
|
|
810
|
-
* compose it into their own regex with the flags they need, avoiding
|
|
811
|
-
* shared mutable `lastIndex` state.
|
|
812
|
-
*
|
|
813
|
-
* Usage: `new RegExp('<artifact\\s+(' + ARTIFACT_ATTRS_PATTERN + ')>...')`
|
|
814
|
-
*/
|
|
815
|
-
const ARTIFACT_ATTRS_PATTERN = String.raw`(?:[^>"']|"[^"]*"|'[^']*')*`;
|
|
600
|
+
String.raw`(?:[^>"']|"[^"]*"|'[^']*')*`;
|
|
816
601
|
const ClaudeArtifactMimeTypes = {
|
|
817
602
|
REACT: "application/vnd.ant.react",
|
|
818
603
|
HTML: "text/html",
|
|
@@ -826,45 +611,6 @@ const ClaudeArtifactMimeTypes = {
|
|
|
826
611
|
PYTHON: "application/vnd.ant.python",
|
|
827
612
|
BLOG_DRAFT: "application/vnd.b4m.blog-draft"
|
|
828
613
|
};
|
|
829
|
-
/**
|
|
830
|
-
* Map a MIME type (or AI-provider artifact-type string) to an internal {@link ArtifactType}.
|
|
831
|
-
*
|
|
832
|
-
* Single source of truth - consumed by the artifact parsers (b4m-core/utils + client) and the
|
|
833
|
-
* tool_result dedup in ChatCompletionProcess. Exact blessed-type matches first (case-insensitive,
|
|
834
|
-
* since MIME types are), then language/format inference; returns null if unrecognized.
|
|
835
|
-
*
|
|
836
|
-
* This previously lived as three hand-maintained copies that drifted - e.g. the lattice
|
|
837
|
-
* tool emits `application/vnd.b4m.lattice` but a copy matched `application/vnd.ant.lattice`,
|
|
838
|
-
* letting lattice tool_result artifacts dodge the dedup set.
|
|
839
|
-
*/
|
|
840
|
-
function mapMimeTypeToArtifactType(mimeType) {
|
|
841
|
-
if (!mimeType) return null;
|
|
842
|
-
const normalized = mimeType.toLowerCase().trim();
|
|
843
|
-
switch (normalized) {
|
|
844
|
-
case ClaudeArtifactMimeTypes.REACT.toLowerCase(): return "react";
|
|
845
|
-
case ClaudeArtifactMimeTypes.HTML.toLowerCase(): return "html";
|
|
846
|
-
case ClaudeArtifactMimeTypes.SVG.toLowerCase(): return "svg";
|
|
847
|
-
case ClaudeArtifactMimeTypes.MERMAID.toLowerCase(): return "mermaid";
|
|
848
|
-
case ClaudeArtifactMimeTypes.RECHARTS.toLowerCase(): return "recharts";
|
|
849
|
-
case ClaudeArtifactMimeTypes.CHESS.toLowerCase(): return "chess";
|
|
850
|
-
case ClaudeArtifactMimeTypes.CODE.toLowerCase(): return "code";
|
|
851
|
-
case ClaudeArtifactMimeTypes.MARKDOWN.toLowerCase(): return "code";
|
|
852
|
-
case ClaudeArtifactMimeTypes.LATTICE.toLowerCase(): return "lattice";
|
|
853
|
-
case ClaudeArtifactMimeTypes.PYTHON.toLowerCase(): return "python";
|
|
854
|
-
case ClaudeArtifactMimeTypes.BLOG_DRAFT.toLowerCase(): return "blog-draft";
|
|
855
|
-
}
|
|
856
|
-
if (normalized.includes("jsx") || normalized.includes("react")) return "react";
|
|
857
|
-
if (normalized.includes("javascript") || normalized.includes("typescript")) return "code";
|
|
858
|
-
if (normalized.includes("python") || normalized === "text/x-python") return "python";
|
|
859
|
-
if (normalized.includes("java") || normalized.includes("c++") || normalized.includes("rust") || normalized.includes("go") || normalized.includes("ruby") || normalized.includes("php") || normalized.includes("swift") || normalized.includes("kotlin") || normalized.includes("csharp") || normalized.includes("c#")) return "code";
|
|
860
|
-
if (normalized.includes("html") || normalized.includes("xhtml")) return "html";
|
|
861
|
-
if (normalized.includes("svg")) return "svg";
|
|
862
|
-
if (normalized.includes("markdown") || normalized.includes("md")) return "code";
|
|
863
|
-
if (normalized.includes("mermaid")) return "mermaid";
|
|
864
|
-
if (normalized.includes("recharts") || normalized.includes("chart")) return "recharts";
|
|
865
|
-
if (normalized.includes("chess")) return "chess";
|
|
866
|
-
return null;
|
|
867
|
-
}
|
|
868
614
|
let KnowledgeType = /* @__PURE__ */ function(KnowledgeType) {
|
|
869
615
|
/**
|
|
870
616
|
* A knowledge that is from a URL.
|
|
@@ -1062,9 +808,8 @@ const IMAGE_SIZE_CONSTRAINTS = {
|
|
|
1062
808
|
},
|
|
1063
809
|
/**
|
|
1064
810
|
* dall-e-3 is no longer in ImageModels, but the generate path still accepts its sizes
|
|
1065
|
-
* for callers holding a persisted one.
|
|
1066
|
-
*
|
|
1067
|
-
* (OPENAI_LEGACY_IMAGE_SIZES), so the split is documentation rather than enforcement.
|
|
811
|
+
* for callers holding a persisted one. Reached by LEGACY_DALL_E_3_MODEL_ID rather than an
|
|
812
|
+
* enum member; isSupportedImageSize measures each dall-e tier against its own list.
|
|
1068
813
|
*/
|
|
1069
814
|
DALL_E_3: {
|
|
1070
815
|
sizes: [
|
|
@@ -1154,6 +899,7 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
|
|
|
1154
899
|
ChatModels["CLAUDE_4_8_OPUS"] = "claude-opus-4-8";
|
|
1155
900
|
ChatModels["CLAUDE_FABLE_5"] = "claude-fable-5";
|
|
1156
901
|
ChatModels["CLAUDE_5_OPUS"] = "claude-opus-5";
|
|
902
|
+
ChatModels["CLAUDE_5_5_OPUS"] = "claude-opus-5-5";
|
|
1157
903
|
ChatModels["JURASSIC2_ULTRA"] = "ai21.j2-ultra-v1";
|
|
1158
904
|
ChatModels["JURASSIC2_MID"] = "ai21.j2-mid-v1";
|
|
1159
905
|
ChatModels["GEMINI_3_5_FLASH"] = "gemini-3.5-flash";
|
|
@@ -1195,12 +941,6 @@ let ChatModels = /* @__PURE__ */ function(ChatModels) {
|
|
|
1195
941
|
const CHAT_MODELS = Object.values(ChatModels);
|
|
1196
942
|
const supportedChatModels = z$1.enum(ChatModels);
|
|
1197
943
|
/**
|
|
1198
|
-
* Every `ChatModels` Gemini entry is named `gemini...` (see the GEMINI block above) - a prefix
|
|
1199
|
-
* check tracks that naming convention automatically as new Gemini models are added, unlike an
|
|
1200
|
-
* explicit id list that would need updating in lockstep and could silently miss one.
|
|
1201
|
-
*/
|
|
1202
|
-
const isGeminiModelId = (model) => model.startsWith("gemini");
|
|
1203
|
-
/**
|
|
1204
944
|
* Models that support the reasoning_effort parameter.
|
|
1205
945
|
* o1-preview and o1-mini do NOT support reasoning_effort.
|
|
1206
946
|
*/
|
|
@@ -1221,137 +961,7 @@ const REASONING_SUPPORTED_MODELS = /* @__PURE__ */ new Set([
|
|
|
1221
961
|
"gpt-5.6-luna",
|
|
1222
962
|
"gpt-5.6-terra"
|
|
1223
963
|
]);
|
|
1224
|
-
|
|
1225
|
-
* GPT-5-family reasoning models whose tool calling breaks on
|
|
1226
|
-
* `/v1/chat/completions` when `reasoning_effort` is also sent. OpenAI requires
|
|
1227
|
-
* this combination to go through `/v1/responses` instead. The failure mode
|
|
1228
|
-
* differs by model:
|
|
1229
|
-
* - GPT-5.4 (and -mini/-nano) hard-reject with a 400:
|
|
1230
|
-
* "Function tools with reasoning_effort are not supported for <model> in
|
|
1231
|
-
* /v1/chat/completions. Please use /v1/responses instead."
|
|
1232
|
-
* - GPT-5 / -mini / -nano / 5.1 / 5.2 return 200 but silently degrade: the
|
|
1233
|
-
* model *narrates* the tool call in its text ("Calling the tool now...")
|
|
1234
|
-
* instead of emitting a real `tool_calls` entry, so no tool ever executes.
|
|
1235
|
-
* This surfaced as the /opti optimizer's "Draft with AI" doing nothing on
|
|
1236
|
-
* GPT-5: the same request on a model with `reasoning_effort`
|
|
1237
|
-
* dropped (or on Claude) fires the tool correctly.
|
|
1238
|
-
*
|
|
1239
|
-
* We drop `reasoning_effort` when tools are sent for these models so tool
|
|
1240
|
-
* calling continues to work on `/v1/chat/completions`. Dropping it only forgoes
|
|
1241
|
-
* explicit effort control - the model still reasons at its default.
|
|
1242
|
-
*
|
|
1243
|
-
* NOTE: for the base GPT-5 narrator family (`RESPONSES_API_TOOL_MODELS`), the
|
|
1244
|
-
* adapter now routes tool turns to `/v1/responses` instead - where reasoning +
|
|
1245
|
-
* tools work together, so `reasoning_effort` is kept. This drop remains as
|
|
1246
|
-
* defense-in-depth for the (now-unreached) chat path and covers the GPT-5.4
|
|
1247
|
-
* family, which is NOT routed to Responses (its drop-path already works).
|
|
1248
|
-
*
|
|
1249
|
-
* O-series reasoning models (o1/o3/o4) are intentionally excluded: they call
|
|
1250
|
-
* tools correctly with `reasoning_effort` on `/v1/chat/completions`.
|
|
1251
|
-
*
|
|
1252
|
-
* Invariant: every member here MUST also be in `REASONING_SUPPORTED_MODELS`.
|
|
1253
|
-
* The gate in `openaiBackend.ts` short-circuits when a model doesn't support
|
|
1254
|
-
* reasoning at all, so adding a non-reasoning model here would make the gate
|
|
1255
|
-
* a no-op and silently leak `reasoning_effort` to the request.
|
|
1256
|
-
*/
|
|
1257
|
-
const REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS = /* @__PURE__ */ new Set([
|
|
1258
|
-
"gpt-5",
|
|
1259
|
-
"gpt-5-mini",
|
|
1260
|
-
"gpt-5-nano",
|
|
1261
|
-
"gpt-5.1",
|
|
1262
|
-
"gpt-5.2",
|
|
1263
|
-
"gpt-5.4",
|
|
1264
|
-
"gpt-5.4-mini",
|
|
1265
|
-
"gpt-5.4-nano",
|
|
1266
|
-
"gpt-5.6-sol",
|
|
1267
|
-
"gpt-5.6-luna",
|
|
1268
|
-
"gpt-5.6-terra"
|
|
1269
|
-
]);
|
|
1270
|
-
/**
|
|
1271
|
-
* GPT-5 reasoning models that silently *narrate* tool calls on
|
|
1272
|
-
* `/v1/chat/completions` (return 200 with the call written as text instead of a
|
|
1273
|
-
* real `tool_calls` entry, so nothing executes). The adapter
|
|
1274
|
-
* routes these to OpenAI's `/v1/responses` API when function tools are present,
|
|
1275
|
-
* where reasoning + tools work together and `reasoning_effort` can be kept.
|
|
1276
|
-
*
|
|
1277
|
-
* Deliberately excludes the GPT-5.4 family: it *hard-errors* (400) on that
|
|
1278
|
-
* combination and is already handled by dropping `reasoning_effort` on the chat
|
|
1279
|
-
* path (see `REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS`), so it stays on
|
|
1280
|
-
* the working chat path to keep this routing's blast radius small. Also excludes
|
|
1281
|
-
* `*-chat-latest` (non-reasoning) and O-series (tools work there already).
|
|
1282
|
-
*
|
|
1283
|
-
* Invariant: every member MUST also be in `REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS`
|
|
1284
|
-
* so the chat path still drops `reasoning_effort` as a fallback if routing is bypassed.
|
|
1285
|
-
*/
|
|
1286
|
-
const RESPONSES_API_TOOL_MODELS = /* @__PURE__ */ new Set([
|
|
1287
|
-
"gpt-5",
|
|
1288
|
-
"gpt-5-mini",
|
|
1289
|
-
"gpt-5-nano",
|
|
1290
|
-
"gpt-5.1",
|
|
1291
|
-
"gpt-5.2",
|
|
1292
|
-
"gpt-5.6-sol",
|
|
1293
|
-
"gpt-5.6-luna",
|
|
1294
|
-
"gpt-5.6-terra"
|
|
1295
|
-
]);
|
|
1296
|
-
/**
|
|
1297
|
-
* Models that only support temperature=1 (no custom temperature).
|
|
1298
|
-
* Includes:
|
|
1299
|
-
* - All reasoning models (OpenAI requires temp=1 when reasoning is active)
|
|
1300
|
-
* - chat-latest variants that enforce this constraint
|
|
1301
|
-
* - GPT-5.5, which rejects custom temperature even though it does not expose
|
|
1302
|
-
* reasoning controls
|
|
1303
|
-
*/
|
|
1304
|
-
const FIXED_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
|
|
1305
|
-
...Array.from(REASONING_SUPPORTED_MODELS),
|
|
1306
|
-
"gpt-5.1-chat-latest",
|
|
1307
|
-
"gpt-5.2-chat-latest",
|
|
1308
|
-
"gpt-5.5"
|
|
1309
|
-
]);
|
|
1310
|
-
/**
|
|
1311
|
-
* Models that do not accept the temperature parameter at all.
|
|
1312
|
-
* The API will reject requests that include temperature for these models.
|
|
1313
|
-
*/
|
|
1314
|
-
const NO_TEMPERATURE_MODELS = /* @__PURE__ */ new Set([
|
|
1315
|
-
"claude-opus-4-7",
|
|
1316
|
-
"global.anthropic.claude-opus-4-7",
|
|
1317
|
-
"claude-opus-4-8",
|
|
1318
|
-
"global.anthropic.claude-opus-4-8",
|
|
1319
|
-
"claude-sonnet-5",
|
|
1320
|
-
"global.anthropic.claude-sonnet-5",
|
|
1321
|
-
"claude-fable-5",
|
|
1322
|
-
"claude-opus-5",
|
|
1323
|
-
"kimi-k3",
|
|
1324
|
-
"kimi-k2.7-code",
|
|
1325
|
-
"kimi-k2.7-code-highspeed",
|
|
1326
|
-
"kimi-k2.6",
|
|
1327
|
-
"kimi-k2.5",
|
|
1328
|
-
"deepseek-flash",
|
|
1329
|
-
"deepseek-v4-pro"
|
|
1330
|
-
]);
|
|
1331
|
-
/**
|
|
1332
|
-
* Models whose safety classifiers can decline a request with `stop_reason: 'refusal'`
|
|
1333
|
-
* (HTTP 200, empty or partial content) - Claude Fable 5's GA classifiers target research
|
|
1334
|
-
* biology and most cybersecurity content and occasionally false-positive on benign adjacent
|
|
1335
|
-
* work. Per Anthropic's GA guidance a refusal from these is opt-in recoverable: rather than
|
|
1336
|
-
* surfacing a hard refusal, the backend throws so the completion loop's existing fallback
|
|
1337
|
-
* machinery continues the request on Opus 5 (whose classifiers intervene far less often).
|
|
1338
|
-
* A refusal from any *other* model is a genuine decline and surfaces unchanged. Keep in
|
|
1339
|
-
* sync with the `claude-fable-5` fallback preference chain in `adminSettings/fallback.ts`.
|
|
1340
|
-
*/
|
|
1341
|
-
const REFUSAL_FALLBACK_MODELS = /* @__PURE__ */ new Set(["claude-fable-5"]);
|
|
1342
|
-
/**
|
|
1343
|
-
* Bedrock-hosted Claude models that do NOT support prompt caching (`cache_control`).
|
|
1344
|
-
* Sending `cache_control` to these models triggers a Bedrock deserialization error:
|
|
1345
|
-
* `tools.N.cache_control: Extra inputs are not permitted`
|
|
1346
|
-
*
|
|
1347
|
-
* AWS Bedrock added prompt caching for Claude 3.5 Haiku and Claude 3.7 Sonnet (and later);
|
|
1348
|
-
* the OG Claude 3 Haiku and the v1 Claude 3.5 Sonnet were not retrofitted.
|
|
1349
|
-
*
|
|
1350
|
-
* Keep this set narrow - default behavior is to apply caching when `cacheStrategy.enableCaching`
|
|
1351
|
-
* is true. Add a model here only when we have concrete evidence (a Bedrock validation error)
|
|
1352
|
-
* that it rejects `cache_control`.
|
|
1353
|
-
*/
|
|
1354
|
-
const BEDROCK_NO_PROMPT_CACHING_MODELS = /* @__PURE__ */ new Set(["anthropic.claude-3-haiku-20240307-v1:0", "anthropic.claude-3-5-sonnet-20240620-v1:0"]);
|
|
964
|
+
[...Array.from(REASONING_SUPPORTED_MODELS)];
|
|
1355
965
|
/**
|
|
1356
966
|
* Speech to Text Models
|
|
1357
967
|
*
|
|
@@ -1397,13 +1007,6 @@ z$1.enum({
|
|
|
1397
1007
|
...SpeechToTextModels,
|
|
1398
1008
|
...VideoModels
|
|
1399
1009
|
});
|
|
1400
|
-
/** Returns true if the model is deprecated on or before the provided date (default: now). */
|
|
1401
|
-
const isModelDeprecated = (model, now = /* @__PURE__ */ new Date()) => {
|
|
1402
|
-
if (!model.deprecationDate) return false;
|
|
1403
|
-
const todayYMD = new Date(now.toISOString().slice(0, 10));
|
|
1404
|
-
const cutoff = /* @__PURE__ */ new Date(model.deprecationDate + "T00:00:00Z");
|
|
1405
|
-
return todayYMD.getTime() >= cutoff.getTime();
|
|
1406
|
-
};
|
|
1407
1010
|
/**
|
|
1408
1011
|
* Valid status values for sub-quests.
|
|
1409
1012
|
* Canonical vocabulary - the mongoose schema, zod schemas, and client all
|
|
@@ -1599,7 +1202,14 @@ const RealtimeVoiceUsageTransaction = BaseCreditTransaction.extend({
|
|
|
1599
1202
|
});
|
|
1600
1203
|
const ToolUsageTransaction = BaseCreditTransaction.extend({
|
|
1601
1204
|
type: z$1.literal("tool_usage"),
|
|
1602
|
-
|
|
1205
|
+
/**
|
|
1206
|
+
* The model that actually incurred the tool cost (e.g. 'gpt-image-2'), NOT the chat
|
|
1207
|
+
* model of the quest that ran the tool. The row is one aggregate over every charging
|
|
1208
|
+
* tool call in the quest, so this is set only when exactly one model charged; a quest
|
|
1209
|
+
* that charged on two or more models leaves it unset rather than naming one of them.
|
|
1210
|
+
* Per-call attribution always lives on the `feature: 'tool'` UsageEventModel rows.
|
|
1211
|
+
*/
|
|
1212
|
+
model: z$1.string().optional(),
|
|
1603
1213
|
questId: z$1.string(),
|
|
1604
1214
|
sessionId: z$1.string()
|
|
1605
1215
|
});
|
|
@@ -1685,8 +1295,8 @@ z$1.object({
|
|
|
1685
1295
|
ownerType: z$1.enum(CreditHolderType),
|
|
1686
1296
|
sessionId: z$1.string().optional(),
|
|
1687
1297
|
/**
|
|
1688
|
-
* Data lake this call is 1:1 attributable to (ingestion embeds
|
|
1689
|
-
* embedding can span multiple lakes and is never attributed here). Unset for
|
|
1298
|
+
* Data lake this call is 1:1 attributable to (ingestion embeds and research-run judge calls -
|
|
1299
|
+
* a query embedding can span multiple lakes and is never attributed here). Unset for
|
|
1690
1300
|
* every other feature/call.
|
|
1691
1301
|
*/
|
|
1692
1302
|
dataLakeId: z$1.string().optional(),
|
|
@@ -1841,7 +1451,6 @@ const FIELD_GROUPS = [
|
|
|
1841
1451
|
"dispatch",
|
|
1842
1452
|
"availability"
|
|
1843
1453
|
];
|
|
1844
|
-
const isFieldGroup = (value) => FIELD_GROUPS.includes(value);
|
|
1845
1454
|
/** YYYY-MM-DD, the format every date-ish ModelInfo field already uses. */
|
|
1846
1455
|
const CALENDAR_DATE = z$1.string().regex(/^\d{4}-\d{2}-\d{2}$/, "expected a YYYY-MM-DD calendar date");
|
|
1847
1456
|
const ReasoningWrite = z$1.strictObject({
|
|
@@ -2656,6 +2265,7 @@ const DATA_LAKE_STABLE_STATUSES = [
|
|
|
2656
2265
|
"deleted"
|
|
2657
2266
|
];
|
|
2658
2267
|
DATA_LAKE_STATUSES.filter((s) => !DATA_LAKE_STABLE_STATUSES.includes(s));
|
|
2268
|
+
const DATA_LAKE_ORIGINS = ["curated", "connector-fed"];
|
|
2659
2269
|
z$1.object({
|
|
2660
2270
|
/**
|
|
2661
2271
|
* The granting lake's Mongo `_id`. ALWAYS a persisted DB lake: a hardcoded/fallback lake has no
|
|
@@ -2703,6 +2313,14 @@ z$1.object({
|
|
|
2703
2313
|
removedAt: z$1.date(),
|
|
2704
2314
|
expiresAt: z$1.date()
|
|
2705
2315
|
});
|
|
2316
|
+
/** Mirrors the read model's vocabulary deliberately (aliased, not re-declared, so the two can
|
|
2317
|
+
* never drift) - an audit reader learns one principal shape for both halves of the trail. */
|
|
2318
|
+
const LAKE_CONFIG_CHANGE_PRINCIPAL_KINDS = [
|
|
2319
|
+
"user",
|
|
2320
|
+
"agent",
|
|
2321
|
+
"apiKey",
|
|
2322
|
+
"system"
|
|
2323
|
+
];
|
|
2706
2324
|
/**
|
|
2707
2325
|
* Every `IDataLake` field, classified as audited or not. A TOTAL map keyed by `keyof IDataLake`,
|
|
2708
2326
|
* exactly like `LAKE_FIELD_VISIBILITY` in redactLakeForActor.ts and for the same reason: a list of
|
|
@@ -2731,6 +2349,7 @@ const LAKE_CONFIG_FIELD_AUDIT = {
|
|
|
2731
2349
|
auditQueryTextEnabled: "audited",
|
|
2732
2350
|
lakeMemoryEnabled: "audited",
|
|
2733
2351
|
status: "audited",
|
|
2352
|
+
origin: "audited",
|
|
2734
2353
|
createdByUserId: "audited",
|
|
2735
2354
|
lastUpdatedByUserId: "excluded",
|
|
2736
2355
|
fileCount: "excluded",
|
|
@@ -2739,15 +2358,58 @@ const LAKE_CONFIG_FIELD_AUDIT = {
|
|
|
2739
2358
|
embeddingSpendMicroUsd: "excluded",
|
|
2740
2359
|
lastSyncAt: "excluded",
|
|
2741
2360
|
lastHealthCheckedAt: "excluded",
|
|
2361
|
+
lastInconsistencyScanAt: "excluded",
|
|
2742
2362
|
filesDeletedAt: "excluded",
|
|
2743
2363
|
filesArchivedAt: "excluded",
|
|
2744
2364
|
lakeMemoryExtractionAt: "excluded",
|
|
2745
2365
|
lakeMemoryCursor: "excluded",
|
|
2746
2366
|
lakeMemoryPurgedAt: "excluded",
|
|
2747
2367
|
inconsistencyReport: "excluded",
|
|
2748
|
-
inconsistencyComputedAt: "excluded"
|
|
2368
|
+
inconsistencyComputedAt: "excluded",
|
|
2369
|
+
modelInconsistencyRunAt: "excluded"
|
|
2749
2370
|
};
|
|
2750
2371
|
[...Object.keys(LAKE_CONFIG_FIELD_AUDIT).filter((field) => LAKE_CONFIG_FIELD_AUDIT[field] === "audited")];
|
|
2372
|
+
z$1.object({
|
|
2373
|
+
dataLakeId: z$1.string(),
|
|
2374
|
+
/** Denormalized from the lake at offer time: a recipient route can filter by it without a join. */
|
|
2375
|
+
organizationId: z$1.string().nullish(),
|
|
2376
|
+
/** The actor who made the offer - always the eventual `grantedByUserId` of an accepted transfer. */
|
|
2377
|
+
offeredByUserId: z$1.string(),
|
|
2378
|
+
recipientUserId: z$1.string(),
|
|
2379
|
+
status: z$1.enum([
|
|
2380
|
+
"pending",
|
|
2381
|
+
"accepted",
|
|
2382
|
+
"declined",
|
|
2383
|
+
"cancelled",
|
|
2384
|
+
"expired"
|
|
2385
|
+
]),
|
|
2386
|
+
expiresAt: z$1.date(),
|
|
2387
|
+
/** Set by the atomic `resolve` when the offer leaves `pending`; null while it is still open. */
|
|
2388
|
+
resolvedAt: z$1.date().nullish(),
|
|
2389
|
+
/**
|
|
2390
|
+
* The lake's effective owner ids AT OFFER TIME. Accept re-resolves them and refuses if they moved,
|
|
2391
|
+
* so an offer can never apply a transfer over another transfer (or a departure succession) that
|
|
2392
|
+
* happened in between. Snapshotted rather than recomputed because "stale" is exactly the case
|
|
2393
|
+
* where today's answer differs from the one the offer was made under.
|
|
2394
|
+
*/
|
|
2395
|
+
priorOwnerUserIds: z$1.array(z$1.string()),
|
|
2396
|
+
offeredVia: z$1.enum([
|
|
2397
|
+
"grant-owner",
|
|
2398
|
+
"creator",
|
|
2399
|
+
"platform-admin",
|
|
2400
|
+
"org-admin"
|
|
2401
|
+
]),
|
|
2402
|
+
/**
|
|
2403
|
+
* The principal to attribute the applied transfer to, resolved by the route at OFFER time (only a
|
|
2404
|
+
* route can tell an API key from a session). Carried so the eventual audit row keeps naming the
|
|
2405
|
+
* same principal the offer was made under, rather than whichever principal happens to accept.
|
|
2406
|
+
*/
|
|
2407
|
+
auditPrincipal: z$1.object({
|
|
2408
|
+
principalKind: z$1.enum(LAKE_CONFIG_CHANGE_PRINCIPAL_KINDS),
|
|
2409
|
+
principalId: z$1.string(),
|
|
2410
|
+
onBehalfOfUserId: z$1.string().optional()
|
|
2411
|
+
}).optional()
|
|
2412
|
+
});
|
|
2751
2413
|
/**
|
|
2752
2414
|
* SRE Agent Trio - Shared Types
|
|
2753
2415
|
*
|
|
@@ -3099,52 +2761,7 @@ let SupportedFabFileMimeTypes = /* @__PURE__ */ function(SupportedFabFileMimeTyp
|
|
|
3099
2761
|
SupportedFabFileMimeTypes["CONF"] = "text/plain";
|
|
3100
2762
|
return SupportedFabFileMimeTypes;
|
|
3101
2763
|
}({});
|
|
3102
|
-
|
|
3103
|
-
* The canonical set of MIME types the ingest pipeline can actually chunk +
|
|
3104
|
-
* vectorize. Kept in lockstep with the `SmartChunker` switch in
|
|
3105
|
-
* `@bike4mind/fab-pipeline`.
|
|
3106
|
-
*/
|
|
3107
|
-
const SUPPORTED_FAB_FILE_MIME_TYPES = new Set(Object.values(SupportedFabFileMimeTypes));
|
|
3108
|
-
/**
|
|
3109
|
-
* Type guard: is a claimed MIME type one we actually support ingesting?
|
|
3110
|
-
*
|
|
3111
|
-
* Used to gate uploads so unsupported/binary files (e.g. `.exe`) are rejected
|
|
3112
|
-
* with a clear error instead of being stored and silently "vectorized" into 0
|
|
3113
|
-
* chunks. Node-free so it can run on both the client (upload UI) and server
|
|
3114
|
-
* (ingest endpoints).
|
|
3115
|
-
*/
|
|
3116
|
-
function isSupportedFabFileMimeType(mimeType) {
|
|
3117
|
-
return !!mimeType && SUPPORTED_FAB_FILE_MIME_TYPES.has(mimeType);
|
|
3118
|
-
}
|
|
3119
|
-
/**
|
|
3120
|
-
* Known MIME types emitted by the generated-audio endpoints (TTS + sound
|
|
3121
|
-
* effects). Deliberately SEPARATE from SupportedFabFileMimeTypes: that set is
|
|
3122
|
-
* the ingest pipeline's chunk+vectorize allowlist, and audio is intentionally
|
|
3123
|
-
* NOT vectorizable and NOT attachable to a chat completion (no model accepts
|
|
3124
|
-
* audio input). Kept only for extension mapping / documentation; the runtime
|
|
3125
|
-
* guard below matches any `audio/*` so unusual provider subtypes are still
|
|
3126
|
-
* treated as audio (i.e. still excluded from every LLM path).
|
|
3127
|
-
*/
|
|
3128
|
-
let AudioMimeType = /* @__PURE__ */ function(AudioMimeType) {
|
|
3129
|
-
AudioMimeType["MP3"] = "audio/mpeg";
|
|
3130
|
-
AudioMimeType["WAV"] = "audio/wav";
|
|
3131
|
-
AudioMimeType["OPUS"] = "audio/opus";
|
|
3132
|
-
AudioMimeType["AAC"] = "audio/aac";
|
|
3133
|
-
AudioMimeType["FLAC"] = "audio/flac";
|
|
3134
|
-
AudioMimeType["PCM"] = "audio/pcm";
|
|
3135
|
-
AudioMimeType["OGG"] = "audio/ogg";
|
|
3136
|
-
AudioMimeType["WEBM"] = "audio/webm";
|
|
3137
|
-
return AudioMimeType;
|
|
3138
|
-
}({});
|
|
3139
|
-
/**
|
|
3140
|
-
* Is this an audio MIME type? Matches any `audio/*` (not just the known set)
|
|
3141
|
-
* so the LLM-exclusion and vectorization-skip guards fail safe for any audio
|
|
3142
|
-
* subtype a provider might emit.
|
|
3143
|
-
*/
|
|
3144
|
-
function isAudioMimeType(mimeType) {
|
|
3145
|
-
if (!mimeType) return false;
|
|
3146
|
-
return mimeType.split(";")[0].trim().toLowerCase().startsWith("audio/");
|
|
3147
|
-
}
|
|
2764
|
+
new Set(Object.values(SupportedFabFileMimeTypes));
|
|
3148
2765
|
/** Reads back the classifier set by the tagged-error helpers above; `undefined` for untagged errors. */
|
|
3149
2766
|
function getQuestErrorCode(error) {
|
|
3150
2767
|
const code = error?.additionalInfo?.errorCode;
|
|
@@ -3223,77 +2840,12 @@ const AGENT_QUEST_MANIFEST = {
|
|
|
3223
2840
|
}
|
|
3224
2841
|
};
|
|
3225
2842
|
/**
|
|
3226
|
-
* The catalog row in force for a model and unit at a given time: newest
|
|
3227
|
-
* effectiveFrom <= at AMONG ROWS OF THAT UNIT. Units are independent price
|
|
3228
|
-
* streams - a newer per_minute row must never shadow the in-force per_token
|
|
3229
|
-
* row (that would silently revert token billing to the adapter literal).
|
|
3230
|
-
* Rows are append-only, so this is the whole time-travel story (see
|
|
3231
|
-
* ModelPriceTypes).
|
|
3232
|
-
*/
|
|
3233
|
-
function resolveModelPriceRow(rows, modelId, unit, at) {
|
|
3234
|
-
let inForce;
|
|
3235
|
-
for (const row of rows) {
|
|
3236
|
-
if (row.modelId !== modelId || row.unit !== unit) continue;
|
|
3237
|
-
if (row.effectiveFrom.getTime() > at.getTime()) continue;
|
|
3238
|
-
if (!inForce || row.effectiveFrom.getTime() > inForce.effectiveFrom.getTime()) inForce = row;
|
|
3239
|
-
}
|
|
3240
|
-
return inForce;
|
|
3241
|
-
}
|
|
3242
|
-
/**
|
|
3243
|
-
* Overlay catalog prices onto assembled ModelInfo. Only per_token rows apply
|
|
3244
|
-
* here (they feed getTextModelCost via ModelInfo.pricing); per_minute and
|
|
3245
|
-
* per_image rows are consumed by their own settlement paths. A model with no
|
|
3246
|
-
* per_token row in force keeps its adapter literal - the fallback that keeps
|
|
3247
|
-
* zero-config self-host deployments working.
|
|
3248
|
-
*/
|
|
3249
|
-
function applyModelPriceCatalog(models, rows, at = /* @__PURE__ */ new Date()) {
|
|
3250
|
-
if (rows.length === 0) return models;
|
|
3251
|
-
return models.map((model) => {
|
|
3252
|
-
if (model.type !== "text") return model;
|
|
3253
|
-
const row = resolveModelPriceRow(rows, model.id, "per_token", at);
|
|
3254
|
-
if (!row) return model;
|
|
3255
|
-
const pricing = {};
|
|
3256
|
-
for (const [threshold, tier] of Object.entries(row.pricing)) pricing[Number(threshold)] = tier;
|
|
3257
|
-
return {
|
|
3258
|
-
...model,
|
|
3259
|
-
pricing
|
|
3260
|
-
};
|
|
3261
|
-
});
|
|
3262
|
-
}
|
|
3263
|
-
/**
|
|
3264
2843
|
* Input headroom the chat path holds back on top of a request's reserved output.
|
|
3265
2844
|
* safeInputWindow (ChatCompletionProcess) subtracts it, so a text entry whose output
|
|
3266
2845
|
* reserve leaves less than this has no room for a prompt at all.
|
|
3267
2846
|
*/
|
|
3268
2847
|
const CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS = 1e3;
|
|
3269
2848
|
/**
|
|
3270
|
-
* Fallback context window for a media row whose catalog value is not usable (a discovery feed's
|
|
3271
|
-
* literal 0, meaning "not applicable" rather than a real budget - see isMediaModelType). Shared by
|
|
3272
|
-
* effectiveContextWindow (@bike4mind/utils, server) and useTokenLimits (apps/client, browser) so
|
|
3273
|
-
* the two do not drift onto different placeholder numbers for the same "unknown" case.
|
|
3274
|
-
*/
|
|
3275
|
-
const DEFAULT_UNKNOWN_CONTEXT_WINDOW = 2e5;
|
|
3276
|
-
/**
|
|
3277
|
-
* The model types this build's ModelInfo consumers narrow on. ModelRecord.type is
|
|
3278
|
-
* wider (embedding / tts / realtime-voice), so the read path drops and counts any
|
|
3279
|
-
* record outside this set: an old build must degrade to "I do not see the new video
|
|
3280
|
-
* models", never to a runtime narrowing failure.
|
|
3281
|
-
*/
|
|
3282
|
-
const MODEL_INFO_TYPES = [
|
|
3283
|
-
"text",
|
|
3284
|
-
"image",
|
|
3285
|
-
"speech-to-text",
|
|
3286
|
-
"video"
|
|
3287
|
-
];
|
|
3288
|
-
const isRenderableModelType = (type) => MODEL_INFO_TYPES.includes(type);
|
|
3289
|
-
/**
|
|
3290
|
-
* Whether a ModelInfo type returns media (image/video) rather than tokens. Shared by
|
|
3291
|
-
* safeInputWindow/effectiveContextWindow (@bike4mind/utils, server) and useTokenLimits
|
|
3292
|
-
* (apps/client, browser) so "what counts as media" cannot drift between the two halves of the
|
|
3293
|
-
* same guard - both need it, and common is the one package already safe to import from either.
|
|
3294
|
-
*/
|
|
3295
|
-
const isMediaModelType = (type) => type === "image" || type === "video";
|
|
3296
|
-
/**
|
|
3297
2849
|
* The modalities promptMeta.model.type is allowed to record. Deliberately NARROWER than
|
|
3298
2850
|
* MODEL_INFO_TYPES: 'speech-to-text' is served by its own transcription route, never by a
|
|
3299
2851
|
* completion, so recording it would put a value in the field that no reader narrows on.
|
|
@@ -3305,162 +2857,6 @@ const PROMPT_META_MODEL_TYPES = [
|
|
|
3305
2857
|
"image",
|
|
3306
2858
|
"video"
|
|
3307
2859
|
];
|
|
3308
|
-
/**
|
|
3309
|
-
* The record -> ModelInfo adapter: one place where every ModelInfo field a
|
|
3310
|
-
* catalog record does not carry gets its default. Each default degrades to the
|
|
3311
|
-
* visible, recoverable behavior rather than the silent or expensive one.
|
|
3312
|
-
*
|
|
3313
|
-
* Pricing is never sourced here: applyModelPriceCatalog overlays the ModelPrice
|
|
3314
|
-
* rows afterwards, and an empty map trips the [UNPRICED_MODEL] alarm on first
|
|
3315
|
-
* billed use, which is the intended fail-loud path.
|
|
3316
|
-
*/
|
|
3317
|
-
function toModelInfo(record) {
|
|
3318
|
-
const retired = record.lifecycle?.status === "retired";
|
|
3319
|
-
const disabled = record.disabled === true || record.autoDisabled === true || retired;
|
|
3320
|
-
return {
|
|
3321
|
-
id: record.id,
|
|
3322
|
-
type: record.type,
|
|
3323
|
-
name: record.name,
|
|
3324
|
-
backend: record.backend,
|
|
3325
|
-
contextWindow: record.contextWindow,
|
|
3326
|
-
max_tokens: record.maxOutputTokens ?? Math.min(record.contextWindow, 4096),
|
|
3327
|
-
...record.maxOutputTokens === void 0 ? { maxOutputTokensDerived: true } : {},
|
|
3328
|
-
pricing: {},
|
|
3329
|
-
can_stream: record.canStream,
|
|
3330
|
-
can_think: record.reasoning?.supported ?? false,
|
|
3331
|
-
thinkingStyle: toThinkingStyle(record),
|
|
3332
|
-
adapterFamily: record.adapterFamily,
|
|
3333
|
-
dispatchProfile: record.dispatchProfile,
|
|
3334
|
-
supportsVision: record.supportsVision,
|
|
3335
|
-
supportsTools: record.supportsTools,
|
|
3336
|
-
supportsImageVariation: record.supportsImageVariation ?? false,
|
|
3337
|
-
supportsSafetyTolerance: record.supportsSafetyTolerance,
|
|
3338
|
-
freeToRun: record.freeToRun,
|
|
3339
|
-
private: record.private ?? false,
|
|
3340
|
-
disabled,
|
|
3341
|
-
disabledReason: record.disabledReason ?? record.autoDisabledReason ?? (retired ? "retired by the provider" : void 0),
|
|
3342
|
-
deprecationDate: record.lifecycle?.deprecationDate,
|
|
3343
|
-
replacedBy: record.lifecycle?.replacedBy,
|
|
3344
|
-
trainingCutoff: record.trainingCutoff,
|
|
3345
|
-
releaseDate: record.releaseDate,
|
|
3346
|
-
logoFile: record.logoFile,
|
|
3347
|
-
rank: record.rank,
|
|
3348
|
-
description: record.description,
|
|
3349
|
-
isSlowModel: record.isSlowModel
|
|
3350
|
-
};
|
|
3351
|
-
}
|
|
3352
|
-
/**
|
|
3353
|
-
* ModelInfo.thinkingStyle only describes the two Anthropic request shapes. Other
|
|
3354
|
-
* reasoning styles map to undefined rather than to a wrong shape; the backends
|
|
3355
|
-
* treat unset as their own default.
|
|
3356
|
-
*/
|
|
3357
|
-
function toThinkingStyle(record) {
|
|
3358
|
-
switch (record.reasoning?.style) {
|
|
3359
|
-
case "anthropic-adaptive": return "adaptive";
|
|
3360
|
-
case "anthropic-legacy": return "legacy";
|
|
3361
|
-
default: return;
|
|
3362
|
-
}
|
|
3363
|
-
}
|
|
3364
|
-
function fromThinkingStyle(style) {
|
|
3365
|
-
if (style === "adaptive") return "anthropic-adaptive";
|
|
3366
|
-
if (style === "legacy") return "anthropic-legacy";
|
|
3367
|
-
}
|
|
3368
|
-
/** Who makes the model, when the id namespace says so: [region.]<vendor>.<model>. */
|
|
3369
|
-
const BEDROCK_REGION_PREFIX = /^(us|eu|apac|global)\./;
|
|
3370
|
-
const VENDOR_BY_BACKEND = {
|
|
3371
|
-
["openai"]: "openai",
|
|
3372
|
-
["anthropic"]: "anthropic",
|
|
3373
|
-
["gemini"]: "google",
|
|
3374
|
-
["xai"]: "xai",
|
|
3375
|
-
["kimi"]: "moonshotai",
|
|
3376
|
-
["deepseek"]: "deepseek",
|
|
3377
|
-
["bfl"]: "black-forest-labs",
|
|
3378
|
-
["aws"]: "amazon",
|
|
3379
|
-
["voyageai"]: "voyageai",
|
|
3380
|
-
["ollama"]: "ollama",
|
|
3381
|
-
["local-image"]: "local",
|
|
3382
|
-
["bedrock"]: "amazon"
|
|
3383
|
-
};
|
|
3384
|
-
/**
|
|
3385
|
-
* ModelInfo carries no vendor (that is one of the four disagreeing taxonomies
|
|
3386
|
-
* this catalog replaces), so the inverse adapter derives it: the backend answers
|
|
3387
|
-
* it for direct providers, and for Bedrock the id namespace does.
|
|
3388
|
-
*/
|
|
3389
|
-
/**
|
|
3390
|
-
* Bedrock id prefixes that name the same maker as a different string. AWS spells
|
|
3391
|
-
* Kimi K2.5 `moonshotai.` and K2 Thinking `moonshot.`, so the raw prefix would
|
|
3392
|
-
* file one vendor's two models under two vendors and split them in the admin
|
|
3393
|
-
* dashboard. Canonicalized to the spelling the direct backend and the models.dev
|
|
3394
|
-
* provider both use.
|
|
3395
|
-
*/
|
|
3396
|
-
const BEDROCK_VENDOR_ALIASES = { moonshot: "moonshotai" };
|
|
3397
|
-
function inferVendor(info) {
|
|
3398
|
-
if (info.backend === "bedrock") {
|
|
3399
|
-
const withoutRegion = String(info.id).replace(BEDROCK_REGION_PREFIX, "");
|
|
3400
|
-
const dot = withoutRegion.indexOf(".");
|
|
3401
|
-
if (dot > 0) {
|
|
3402
|
-
const prefix = withoutRegion.slice(0, dot);
|
|
3403
|
-
return BEDROCK_VENDOR_ALIASES[prefix] ?? prefix;
|
|
3404
|
-
}
|
|
3405
|
-
}
|
|
3406
|
-
return VENDOR_BY_BACKEND[info.backend] ?? String(info.backend);
|
|
3407
|
-
}
|
|
3408
|
-
/**
|
|
3409
|
-
* ModelInfo -> ModelRecord, the inverse of toModelInfo. Two callers: the
|
|
3410
|
-
* fallback seed generator (adapter literals become seed rows) and the merge's
|
|
3411
|
-
* base tier (a seeded model becomes a record that a catalog row can then claim
|
|
3412
|
-
* groups of).
|
|
3413
|
-
*
|
|
3414
|
-
* `pricing` has no ModelInfo spelling here and is deliberately absent rather
|
|
3415
|
-
* than guessed: catalog rows never carry it. The dispatch group round-trips
|
|
3416
|
-
* (ModelInfo carries it since dispatch consumes it), but no feed may author it -
|
|
3417
|
-
* a wrong value there mis-routes a request, so it stays seed- or
|
|
3418
|
-
* operator-sourced, or comes from the seed-side DispatchResolver.
|
|
3419
|
-
*
|
|
3420
|
-
* Round-tripping normalizes the optional booleans toModelInfo defaults
|
|
3421
|
-
* (can_think, private, disabled, supportsImageVariation): undefined becomes an
|
|
3422
|
-
* explicit false. That is only observable for a model whose merged record a
|
|
3423
|
-
* catalog row actually owns a group of.
|
|
3424
|
-
*/
|
|
3425
|
-
function toModelRecord(info) {
|
|
3426
|
-
return {
|
|
3427
|
-
id: info.id,
|
|
3428
|
-
vendor: inferVendor(info),
|
|
3429
|
-
backend: info.backend,
|
|
3430
|
-
type: info.type,
|
|
3431
|
-
name: info.name,
|
|
3432
|
-
contextWindow: info.contextWindow,
|
|
3433
|
-
maxOutputTokens: info.maxOutputTokensDerived === true ? void 0 : info.max_tokens,
|
|
3434
|
-
canStream: info.can_stream,
|
|
3435
|
-
reasoning: info.can_think === void 0 && info.thinkingStyle === void 0 ? void 0 : {
|
|
3436
|
-
supported: info.can_think === true,
|
|
3437
|
-
style: fromThinkingStyle(info.thinkingStyle)
|
|
3438
|
-
},
|
|
3439
|
-
adapterFamily: info.adapterFamily,
|
|
3440
|
-
dispatchProfile: info.dispatchProfile,
|
|
3441
|
-
supportsVision: info.supportsVision,
|
|
3442
|
-
supportsTools: info.supportsTools,
|
|
3443
|
-
supportsImageVariation: info.supportsImageVariation,
|
|
3444
|
-
supportsSafetyTolerance: info.supportsSafetyTolerance,
|
|
3445
|
-
lifecycle: {
|
|
3446
|
-
...info.deprecationDate ? {
|
|
3447
|
-
status: "deprecated",
|
|
3448
|
-
deprecationDate: info.deprecationDate
|
|
3449
|
-
} : { status: "active" },
|
|
3450
|
-
...info.replacedBy ? { replacedBy: info.replacedBy } : {}
|
|
3451
|
-
},
|
|
3452
|
-
description: info.description,
|
|
3453
|
-
logoFile: info.logoFile,
|
|
3454
|
-
rank: info.rank,
|
|
3455
|
-
isSlowModel: info.isSlowModel,
|
|
3456
|
-
trainingCutoff: info.trainingCutoff,
|
|
3457
|
-
releaseDate: info.releaseDate,
|
|
3458
|
-
private: info.private,
|
|
3459
|
-
freeToRun: info.freeToRun,
|
|
3460
|
-
disabled: info.disabled,
|
|
3461
|
-
disabledReason: info.disabledReason
|
|
3462
|
-
};
|
|
3463
|
-
}
|
|
3464
2860
|
const DEFAULT_PRICE_MARGIN = 1.2;
|
|
3465
2861
|
const DEFAULT_USD_TO_CREDITS_RATE = 6e-4;
|
|
3466
2862
|
/**
|
|
@@ -3508,140 +2904,6 @@ const CREDITS_PER_USD_COST = (() => {
|
|
|
3508
2904
|
console.warn(`[pricing] Env-configured margin/rate derive ${derived} credits per USD cost; using defaults`);
|
|
3509
2905
|
return Math.round(DEFAULT_PRICE_MARGIN / DEFAULT_USD_TO_CREDITS_RATE);
|
|
3510
2906
|
})();
|
|
3511
|
-
/**
|
|
3512
|
-
* Converts a USD cost to credits, including markup, rounding UP (minimum 1).
|
|
3513
|
-
*
|
|
3514
|
-
* Use for reservations, eligibility checks, estimates, and display - anywhere
|
|
3515
|
-
* a deterministic, conservative number is needed. Final settlement of variable
|
|
3516
|
-
* usage should use usdToCreditsStochastic so users pay the exact fraction in
|
|
3517
|
-
* expectation instead of the round-up.
|
|
3518
|
-
*
|
|
3519
|
-
* Examples (defaults):
|
|
3520
|
-
* $1 USD = 2000 credits (1.2x markup at $0.0006/credit)
|
|
3521
|
-
* $0.001 USD = 2 credits
|
|
3522
|
-
* $0.0001 USD = 1 credit (minimum)
|
|
3523
|
-
*
|
|
3524
|
-
* @param rate - Credits per $1, defaults to the platform-wide CREDITS_PER_USD_COST.
|
|
3525
|
-
* Pass an override for a provider-specific admin-tunable rate (e.g. a
|
|
3526
|
-
* billing-sensitive external compute path priced independently of the
|
|
3527
|
-
* platform default).
|
|
3528
|
-
*/
|
|
3529
|
-
const usdToCredits = (usd, rate = CREDITS_PER_USD_COST) => {
|
|
3530
|
-
return Math.max(1, Math.ceil(usd * rate));
|
|
3531
|
-
};
|
|
3532
|
-
/**
|
|
3533
|
-
* Uniform draw in [0, 1) from the platform CSPRNG. Billing draws MUST be
|
|
3534
|
-
* unpredictable: a caller who can foresee the stream could time requests to
|
|
3535
|
-
* land on the no-charge side of every draw. Node >= 19 and all browsers
|
|
3536
|
-
* expose globalThis.crypto; the Math.random fallback exists only so tests
|
|
3537
|
-
* and exotic runtimes do not crash, and it warns once.
|
|
3538
|
-
*/
|
|
3539
|
-
let warnedNonCryptoRng = false;
|
|
3540
|
-
const cryptoUniform = () => {
|
|
3541
|
-
const cryptoObj = globalThis.crypto;
|
|
3542
|
-
if (cryptoObj?.getRandomValues) {
|
|
3543
|
-
const buf = /* @__PURE__ */ new Uint32Array(1);
|
|
3544
|
-
cryptoObj.getRandomValues(buf);
|
|
3545
|
-
return buf[0] / 4294967296;
|
|
3546
|
-
}
|
|
3547
|
-
if (!warnedNonCryptoRng) {
|
|
3548
|
-
warnedNonCryptoRng = true;
|
|
3549
|
-
console.warn("[pricing] globalThis.crypto unavailable; billing rounding falls back to Math.random");
|
|
3550
|
-
}
|
|
3551
|
-
return Math.random();
|
|
3552
|
-
};
|
|
3553
|
-
/**
|
|
3554
|
-
* Converts a USD cost to credits with markup using UNBIASED stochastic
|
|
3555
|
-
* rounding: the integer part is always charged, and the fractional part
|
|
3556
|
-
* charges one extra credit with probability equal to the fraction.
|
|
3557
|
-
* E[charge] equals the exact fractional cost at every call size - no
|
|
3558
|
-
* round-up overcharge, no 1-credit minimum, and no free-below-threshold
|
|
3559
|
-
* leak. Because draws are independent and priced at exact cost, no calling
|
|
3560
|
-
* pattern (splitting, retrying, aborting) changes expected cost.
|
|
3561
|
-
*
|
|
3562
|
-
* Server-side settlement only. Do NOT use for display, estimates, or
|
|
3563
|
-
* reservations (it is non-deterministic; use usdToCredits). Zero or
|
|
3564
|
-
* non-finite cost charges 0.
|
|
3565
|
-
*
|
|
3566
|
-
* @param usd - The raw provider cost in USD to convert
|
|
3567
|
-
* @param rng - Uniform [0,1) source, injectable for tests; defaults to CSPRNG
|
|
3568
|
-
* @param rate - Credits per $1, defaults to the platform-wide CREDITS_PER_USD_COST.
|
|
3569
|
-
* Callers settling a provider-specific admin-tunable rate must pass the SAME
|
|
3570
|
-
* rate value that was used at reservation time (snapshot it, don't re-read
|
|
3571
|
-
* a live admin setting), or the settlement math mixes scales.
|
|
3572
|
-
*/
|
|
3573
|
-
const usdToCreditsStochastic = (usd, rng = cryptoUniform, rate = CREDITS_PER_USD_COST) => {
|
|
3574
|
-
const raw = usd * rate;
|
|
3575
|
-
if (!Number.isFinite(raw) || raw <= 0) return 0;
|
|
3576
|
-
const base = Math.floor(raw);
|
|
3577
|
-
const fraction = raw - base;
|
|
3578
|
-
return base + (rng() < fraction ? 1 : 0);
|
|
3579
|
-
};
|
|
3580
|
-
/**
|
|
3581
|
-
* Output tokens the pre-flight credit hold prices when a request's max_tokens
|
|
3582
|
-
* ceiling exceeds it.
|
|
3583
|
-
*
|
|
3584
|
-
* max_tokens is a *ceiling*, not a prediction: adaptive reasoning models stop at
|
|
3585
|
-
* end_turn well short of it, so pricing the hold at the full window (128K on the
|
|
3586
|
-
* flagship models) reserved several thousand credits per turn regardless of answer
|
|
3587
|
-
* length. The excess was always refunded at settlement, but the hold IS the
|
|
3588
|
-
* insufficient-funds gate, so users whose balance sat between their real cost and
|
|
3589
|
-
* the worst case were falsely blocked.
|
|
3590
|
-
*
|
|
3591
|
-
* 16K covers the realistic long answer with headroom - the largest replies seen in
|
|
3592
|
-
* practice are HTML-artifact turns at roughly 10-11K output tokens (see
|
|
3593
|
-
* buildThinkingParams in llm-adapters/thinkingParams.ts, whose max_tokens floor was
|
|
3594
|
-
* sized off the same measurement).
|
|
3595
|
-
*
|
|
3596
|
-
* UNDER-RESERVATION IS ACCEPTED, NOT PREVENTED. A turn that emits more than this
|
|
3597
|
-
* settles as a shortfall debit at reconciliation, which is already a supported path
|
|
3598
|
-
* (provider-basis settlement could always exceed a hold priced on the local
|
|
3599
|
-
* estimate). The two sites clamp the shortfall differently: chat's
|
|
3600
|
-
* computeSettlementDelta (services/llm/ChatCompletionProcess.ts) floors the debit at
|
|
3601
|
-
* the balance snapshot taken at its OWN admission and reports the remainder as
|
|
3602
|
-
* writtenOffCredits; cliCompletions.ts does an unclamped $inc on success and only
|
|
3603
|
-
* logs an ALERT if it lands negative. Neither is a non-negativity guarantee: the
|
|
3604
|
-
* chat snapshot predates any sibling turn's spend (see that function's own "best
|
|
3605
|
-
* effort" note), and when the shortfall still fits the stale snapshot the debit
|
|
3606
|
-
* applies in full - writtenOffCredits 0, and no BILLING_SHORTFALL_CLAMP log - so a
|
|
3607
|
-
* concurrent holder can land negative there too. Either way the resulting balance
|
|
3608
|
-
* fails the *next* turn's own admission gate - but that bound is per turn, not per
|
|
3609
|
-
* holder: turns admitted concurrently are each checked against the balance at their
|
|
3610
|
-
* own admission and settle against a snapshot that predates their siblings' spend,
|
|
3611
|
-
* so a holder running turns in parallel can be shorted once per in-flight turn, not
|
|
3612
|
-
* once total.
|
|
3613
|
-
*/
|
|
3614
|
-
const PREFLIGHT_RESERVATION_OUTPUT_TOKENS = 16384;
|
|
3615
|
-
/**
|
|
3616
|
-
* Reservation ceiling for models that spend reasoning tokens inside their output
|
|
3617
|
-
* budget (see reasonsWithinOutputBudget in llm-adapters/thinkingParams.ts). Those
|
|
3618
|
-
* tokens bill as output on top of the visible answer, so the 16K figure above -
|
|
3619
|
-
* which was measured on visible artifact size alone - under-reserves them badly.
|
|
3620
|
-
*
|
|
3621
|
-
* Deliberately below ADAPTIVE_THINKING_MAX_TOKENS_FLOOR (64K), which sizes the
|
|
3622
|
-
* request's real ceiling and therefore has to cover the worst case: exceeding it
|
|
3623
|
-
* truncates a reply mid-tag, an unrecoverable failure. A hold has no such duty -
|
|
3624
|
-
* exceeding it settles as a shortfall debit - so it is sized for the long turn
|
|
3625
|
-
* (roughly 3x the largest observed visible answer, leaving the rest for the trace)
|
|
3626
|
-
* rather than the worst one, which is what keeps the gate off affordable requests.
|
|
3627
|
-
*/
|
|
3628
|
-
const PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS = 32768;
|
|
3629
|
-
/**
|
|
3630
|
-
* Output-token figure to price a pre-flight credit hold at, given the max_tokens
|
|
3631
|
-
* this request will actually send. Never raises the caller's ceiling: a request
|
|
3632
|
-
* that asks for less than the cap holds only what it can possibly spend.
|
|
3633
|
-
*
|
|
3634
|
-
* Reserving only, never gating: the per-member org credit cap is still priced on
|
|
3635
|
-
* the unshrunk ceiling at both call sites, since that check has no settlement
|
|
3636
|
-
* counterpart to correct an under-estimate. That is a strictly larger figure than
|
|
3637
|
-
* the hold, not a true upper bound on the turn - it prices one model round trip,
|
|
3638
|
-
* at the uncached input rate, on the primary model - so it still under-counts a
|
|
3639
|
-
* multi-round tool loop, a cache-write turn, or a fallback hop onto pricier pricing.
|
|
3640
|
-
*
|
|
3641
|
-
* @param reasonsWithinOutputBudget - reasonsWithinOutputBudget(modelInfo); passed as
|
|
3642
|
-
* a boolean because common cannot import llm-adapters.
|
|
3643
|
-
*/
|
|
3644
|
-
const reservationOutputTokens = (requestedMaxTokens, reasonsWithinOutputBudget = false) => Math.min(requestedMaxTokens, reasonsWithinOutputBudget ? PREFLIGHT_RESERVATION_REASONING_OUTPUT_TOKENS : PREFLIGHT_RESERVATION_OUTPUT_TOKENS);
|
|
3645
2907
|
z.enum([
|
|
3646
2908
|
"openai",
|
|
3647
2909
|
"test",
|
|
@@ -3701,7 +2963,22 @@ const PromptBatchQuerySchema = z$1.object({
|
|
|
3701
2963
|
personal: z$1.boolean().optional()
|
|
3702
2964
|
});
|
|
3703
2965
|
z$1.object({ queries: z$1.array(PromptBatchQuerySchema).min(1).max(32).refine((qs) => new Set(qs.map((q) => q.key)).size === qs.length, { message: "Batch query keys must be unique" }) });
|
|
3704
|
-
|
|
2966
|
+
/**
|
|
2967
|
+
* Request schema for POST /api/chat - the simplified external chat surface.
|
|
2968
|
+
*
|
|
2969
|
+
* Shared between the Next.js API handler (apps/client/pages/api/chat.ts, which
|
|
2970
|
+
* validates req.body with this exact object) and the OpenAPI registry
|
|
2971
|
+
* (b4m-core/common/src/openapi), so the published contract cannot drift from
|
|
2972
|
+
* what the handler actually accepts. The model is optional here and resolved
|
|
2973
|
+
* server-side from admin settings when omitted.
|
|
2974
|
+
*
|
|
2975
|
+
* Public-API rule: no `.catch()` / top-level `.transform()`. Both silently mutate
|
|
2976
|
+
* caller input (fail-quiet) and are opaque to zod-to-openapi. `historyCount` uses
|
|
2977
|
+
* `.default()` (fail loud on a bad value); unknown tool ids are filtered in the
|
|
2978
|
+
* handler (see filterKnownTools) instead of by a schema transform. This keeps the
|
|
2979
|
+
* schema fully OpenAPI-representable with no doc projection needed.
|
|
2980
|
+
*/
|
|
2981
|
+
const SimplifiedChatRequestSchema = z$1.object({
|
|
3705
2982
|
sessionId: z$1.string().nullish(),
|
|
3706
2983
|
message: z$1.string(),
|
|
3707
2984
|
organizationId: z$1.string().optional(),
|
|
@@ -3730,7 +3007,15 @@ z$1.object({
|
|
|
3730
3007
|
includeSystemPrompt: z$1.boolean().optional(),
|
|
3731
3008
|
systemPrompt: z$1.string().max(PROMPT_TEXT_MAX).optional().describe("System-prompt text for this request only, never persisted. Rendered as a defended block appended after every other system-prompt source, with prose instructing the model to defer to organization, session and data-lake guidance. Over the cap is a 422, never truncated.")
|
|
3732
3009
|
});
|
|
3733
|
-
|
|
3010
|
+
/**
|
|
3011
|
+
* Async ACK returned on the default (wait:false) path of POST /api/chat. The
|
|
3012
|
+
* `type`/`errorCode` pair below is the same classifier the `wait: true` body and
|
|
3013
|
+
* the polled quest (`GET /api/quests/{id}`) carry, so it is modelled once here -
|
|
3014
|
+
* the rest of those two bodies is NOT described by this schema. The
|
|
3015
|
+
* handler assembles the ack body inline (apps/client/pages/api/chat.ts), so
|
|
3016
|
+
* this schema MUST stay in sync with that `res.json({...})` shape.
|
|
3017
|
+
*/
|
|
3018
|
+
const ChatAckSchema = z$1.object({
|
|
3734
3019
|
id: z$1.string(),
|
|
3735
3020
|
status: z$1.string(),
|
|
3736
3021
|
message_received: z$1.boolean(),
|
|
@@ -3751,7 +3036,33 @@ z$1.object({
|
|
|
3751
3036
|
poll_url: z$1.string().optional()
|
|
3752
3037
|
})
|
|
3753
3038
|
});
|
|
3754
|
-
|
|
3039
|
+
/**
|
|
3040
|
+
* The quest a `wait: false` caller polls at `GET /api/quests/{id}` - the outcome
|
|
3041
|
+
* of the turn the ACK above only acknowledged.
|
|
3042
|
+
*
|
|
3043
|
+
* Deliberately the OUTCOME SUBSET, not the whole quest: that endpoint is a plain
|
|
3044
|
+
* handler rather than a contract, so this models only what decides whether the
|
|
3045
|
+
* turn succeeded, and a poll body carries further fields (`images`, `files`,
|
|
3046
|
+
* `toolPayloads`, `promptMeta`, ...). Must stay in sync with that handler's
|
|
3047
|
+
* `res.json` shape (apps/client/pages/api/quests/[id]/index.ts) - unlike a
|
|
3048
|
+
* contract-registered request/response schema, nothing validates this at
|
|
3049
|
+
* runtime. The "parses against the published ChatQuestPollResultSchema"
|
|
3050
|
+
* integration test (index.integration.test.ts) only proves the handler's
|
|
3051
|
+
* CURRENT response satisfies this schema - a non-strict `z.object` strips
|
|
3052
|
+
* unknown keys rather than rejecting them, and only `id` is required, so a
|
|
3053
|
+
* field the handler starts returning without a matching addition here keeps
|
|
3054
|
+
* that test green. The real per-field coverage lives in the sibling
|
|
3055
|
+
* assertions in that same test file; a shape addition still needs a schema
|
|
3056
|
+
* update by hand.
|
|
3057
|
+
*
|
|
3058
|
+
* A failed turn is still `status: 'done'` with the failure text in `reply`, so
|
|
3059
|
+
* `reply` alone cannot tell an answer from a failure - `type` and `errorCode` are
|
|
3060
|
+
* what separate a CLASSIFIED failure. A run recovered from a timeout with partial
|
|
3061
|
+
* content is not one of those: `terminalRecoveryFor` (questTimeoutRecovery.ts)
|
|
3062
|
+
* flips only `status` to preserve the surviving content, so it polls back as
|
|
3063
|
+
* `type: 'message'` even though it never finished.
|
|
3064
|
+
*/
|
|
3065
|
+
const ChatQuestPollResultSchema = z$1.object({
|
|
3755
3066
|
id: z$1.string(),
|
|
3756
3067
|
status: z$1.enum([
|
|
3757
3068
|
"stopped",
|
|
@@ -3782,7 +3093,18 @@ const ApiErrorSchema = z$1.object({
|
|
|
3782
3093
|
*/
|
|
3783
3094
|
name: z$1.string().optional()
|
|
3784
3095
|
});
|
|
3785
|
-
|
|
3096
|
+
/**
|
|
3097
|
+
* Error envelope for the 422 a credit-metered endpoint returns for two unrelated
|
|
3098
|
+
* reasons: "your body is invalid" and "you cannot afford this". `errorCode` is
|
|
3099
|
+
* what separates them - `insufficientCreditsError` (see insufficientCredits.ts)
|
|
3100
|
+
* tags the credit case, so its absence means an ordinary validation failure.
|
|
3101
|
+
*
|
|
3102
|
+
* Derived from `ApiErrorSchema` rather than re-declaring `error`/`request_id`:
|
|
3103
|
+
* both of those 422s are *thrown*, so errorHandler serves the body and adds
|
|
3104
|
+
* `name`. Extending is what keeps that documented here (and what drops it again
|
|
3105
|
+
* on the sunset date) instead of leaving a bespoke copy behind to drift.
|
|
3106
|
+
*/
|
|
3107
|
+
const InsufficientCreditsErrorSchema = ApiErrorSchema.extend({ errorCode: z$1.literal("insufficient_credits").optional() });
|
|
3786
3108
|
const supportedVoiceGenerationVendor = z.enum(["openai", "elevenlabs"]);
|
|
3787
3109
|
const voiceOutputFormatSchema = z.enum([
|
|
3788
3110
|
"mp3",
|
|
@@ -3793,13 +3115,12 @@ const voiceOutputFormatSchema = z.enum([
|
|
|
3793
3115
|
"pcm"
|
|
3794
3116
|
]);
|
|
3795
3117
|
const voiceResponseEncodingSchema = z.enum(["binary", "base64"]);
|
|
3796
|
-
const
|
|
3118
|
+
const TTS_ABSOLUTE_MAX_INPUT_CHARS = Math.max(...Object.values({
|
|
3797
3119
|
openai: 4096,
|
|
3798
3120
|
elevenlabs: 1e4
|
|
3799
|
-
};
|
|
3800
|
-
const TTS_ABSOLUTE_MAX_INPUT_CHARS = Math.max(...Object.values(TTS_MAX_INPUT_CHARS));
|
|
3121
|
+
}));
|
|
3801
3122
|
const ttsLanguageCodeSchema = z.string().regex(/^[a-z]{2}$/, "languageCode must be a lowercase ISO 639-1 code, e.g. \"en\" or \"ja\"");
|
|
3802
|
-
z.object({
|
|
3123
|
+
const ttsRequestSchema = z.object({
|
|
3803
3124
|
text: z.string().min(1).max(TTS_ABSOLUTE_MAX_INPUT_CHARS),
|
|
3804
3125
|
provider: supportedVoiceGenerationVendor.optional(),
|
|
3805
3126
|
model: z.string().optional(),
|
|
@@ -3825,7 +3146,18 @@ const audioSaveSkippedReasonSchema = z.enum([
|
|
|
3825
3146
|
"file_too_large",
|
|
3826
3147
|
"error"
|
|
3827
3148
|
]);
|
|
3828
|
-
|
|
3149
|
+
/**
|
|
3150
|
+
* JSON body of `POST /api/ai/tts` when the caller asks for `encoding: 'base64'`.
|
|
3151
|
+
* The default `binary` encoding returns raw audio bytes instead and has no JSON
|
|
3152
|
+
* shape.
|
|
3153
|
+
*
|
|
3154
|
+
* The save + provider fields are all optional because the handler spreads them in
|
|
3155
|
+
* only when they apply: the save fields are absent when no copy was attempted
|
|
3156
|
+
* (`preview: true`, or the saveGeneratedAudio preference is off), and
|
|
3157
|
+
* `provider`/`fallbackFrom` appear only when the requested provider was
|
|
3158
|
+
* unavailable and another one stood in.
|
|
3159
|
+
*/
|
|
3160
|
+
const ttsBase64ResponseSchema = z.object({
|
|
3829
3161
|
/** Base64-encoded audio payload. */
|
|
3830
3162
|
audio: z.string(),
|
|
3831
3163
|
format: voiceOutputFormatSchema,
|
|
@@ -3839,7 +3171,32 @@ z.object({
|
|
|
3839
3171
|
/** The originally requested provider that could not serve the request. */
|
|
3840
3172
|
fallbackFrom: supportedVoiceGenerationVendor.optional()
|
|
3841
3173
|
});
|
|
3842
|
-
|
|
3174
|
+
/**
|
|
3175
|
+
* Error body for `POST /api/ai/tts`, shared by the 401, 422, 429 and 502; the 413
|
|
3176
|
+
* has a shape of its own (`ttsResponseTooLargeSchema`). `errorCode` is present only
|
|
3177
|
+
* on the conditions that carry a classifier; an ordinary validation 422 has none.
|
|
3178
|
+
*
|
|
3179
|
+
* Extends `ApiErrorSchema` because what decides whether a body carries the fields
|
|
3180
|
+
* errorHandler adds - `request_id`, and `name` until its 2026-12-01 sunset - is
|
|
3181
|
+
* whether the body was THROWN, not which status it wears, and three of these four
|
|
3182
|
+
* statuses are reachable both ways:
|
|
3183
|
+
*
|
|
3184
|
+
* - 401: thrown by apiKeyAuth on a rejected key; written by `auth` when no
|
|
3185
|
+
* credential was presented at all, and by the handler for
|
|
3186
|
+
* `provider_not_configured` and for an upstream credential rejection
|
|
3187
|
+
* (`provider_rejected`).
|
|
3188
|
+
* - 422: thrown by request validation and by the char-limit / format guards;
|
|
3189
|
+
* written by the handler for `insufficient_credits` and for an upstream 422.
|
|
3190
|
+
* - 429: thrown by apiKeyRateLimit; written by the handler on an upstream 429.
|
|
3191
|
+
* - 502: only ever written.
|
|
3192
|
+
*
|
|
3193
|
+
* So those two are genuinely optional here, and splitting this per status would be
|
|
3194
|
+
* wrong for the first three. The 502 does advertise both without ever sending them;
|
|
3195
|
+
* that is not worth a fourth error schema on one route, and `request_id` missing from
|
|
3196
|
+
* this handler's written bodies is a gap in the handler rather than something to
|
|
3197
|
+
* enshrine in a schema.
|
|
3198
|
+
*/
|
|
3199
|
+
const ttsErrorResponseSchema = ApiErrorSchema.extend({
|
|
3843
3200
|
provider: supportedVoiceGenerationVendor.optional(),
|
|
3844
3201
|
errorCode: z.enum([
|
|
3845
3202
|
"insufficient_credits",
|
|
@@ -3847,7 +3204,21 @@ ApiErrorSchema.extend({
|
|
|
3847
3204
|
"provider_rejected"
|
|
3848
3205
|
]).optional()
|
|
3849
3206
|
});
|
|
3850
|
-
|
|
3207
|
+
/**
|
|
3208
|
+
* 413 body: the audio was generated and billed but exceeds the serverless
|
|
3209
|
+
* response-size cap. When a browsable copy was saved, `fileUrl` is how the caller
|
|
3210
|
+
* retrieves the audio it paid for.
|
|
3211
|
+
*
|
|
3212
|
+
* Not derived from `ApiErrorSchema`: every 413 on this route is written, never
|
|
3213
|
+
* thrown, so errorHandler never serves one and `name` is genuinely absent rather than
|
|
3214
|
+
* optional - a stronger claim than `ttsErrorResponseSchema` can make for its own
|
|
3215
|
+
* statuses, see the note there. Two writers, and only the first matches the paragraph
|
|
3216
|
+
* above: the exceedsTtsResponseLimit guard, and the upstream-4xx passthrough relaying
|
|
3217
|
+
* a provider 413, where nothing was generated or billed and there is no `fileUrl`. An
|
|
3218
|
+
* oversized *request* body is a third 413 that never reaches this schema at all -
|
|
3219
|
+
* Next's own body parser answers it in plain text before the router runs.
|
|
3220
|
+
*/
|
|
3221
|
+
const ttsResponseTooLargeSchema = z.object({
|
|
3851
3222
|
error: z.string(),
|
|
3852
3223
|
provider: supportedVoiceGenerationVendor,
|
|
3853
3224
|
saved: z.literal(true).optional(),
|
|
@@ -3860,7 +3231,15 @@ z.enum(["openai"]);
|
|
|
3860
3231
|
* New vendors are added here and in the `aiSoundService` factory.
|
|
3861
3232
|
*/
|
|
3862
3233
|
const supportedSoundGenerationVendor = z.enum(["elevenlabs"]);
|
|
3863
|
-
|
|
3234
|
+
/**
|
|
3235
|
+
* Inbound request body for `POST /api/ai/sound-effects`.
|
|
3236
|
+
*
|
|
3237
|
+
* `durationSeconds` and `promptInfluence` bounds mirror the ElevenLabs
|
|
3238
|
+
* sound-generation limits (0.5-30s for the default eleven_text_to_sound_v2
|
|
3239
|
+
* model, prompt influence 0-1). `format` is the provider-specific output
|
|
3240
|
+
* encoding token (e.g. `mp3_44100_128`).
|
|
3241
|
+
*/
|
|
3242
|
+
const soundEffectsRequestSchema = z.object({
|
|
3864
3243
|
provider: supportedSoundGenerationVendor.default("elevenlabs"),
|
|
3865
3244
|
text: z.string().min(1).max(1e3),
|
|
3866
3245
|
durationSeconds: z.number().min(.5).max(30).optional(),
|
|
@@ -3878,13 +3257,23 @@ const supportedMusicGenerationVendor = z.enum(["elevenlabs"]);
|
|
|
3878
3257
|
* change the request surface needs to accept it.
|
|
3879
3258
|
*/
|
|
3880
3259
|
const supportedMusicModel = z.enum(["music_v1"]);
|
|
3881
|
-
|
|
3882
|
-
|
|
3260
|
+
/**
|
|
3261
|
+
* Inbound request body for `POST /api/ai/music`.
|
|
3262
|
+
*
|
|
3263
|
+
* `lengthMs` upper bound is capped below the ElevenLabs Music API ceiling to fit
|
|
3264
|
+
* the serving function's time budget (see MAX_MUSIC_LENGTH_MS). It carries a
|
|
3265
|
+
* default rather than being optional so the billed
|
|
3266
|
+
* length is always known up front (the reserve/settle path needs a deterministic
|
|
3267
|
+
* cost before generation) and the route can force that exact length on the
|
|
3268
|
+
* provider. `format` is the provider-specific output encoding token (e.g.
|
|
3269
|
+
* `mp3_44100_128`).
|
|
3270
|
+
*/
|
|
3271
|
+
const musicRequestSchema = z.object({
|
|
3883
3272
|
provider: supportedMusicGenerationVendor.default("elevenlabs"),
|
|
3884
3273
|
prompt: z.string().min(1).max(2e3),
|
|
3885
3274
|
lengthMs: z.number().int().min(3e3).max(12e4).default(1e4),
|
|
3886
3275
|
forceInstrumental: z.boolean().optional(),
|
|
3887
|
-
modelId: supportedMusicModel.default(
|
|
3276
|
+
modelId: supportedMusicModel.default("music_v1"),
|
|
3888
3277
|
format: z.string().optional()
|
|
3889
3278
|
});
|
|
3890
3279
|
VIDEO_SIZE_CONSTRAINTS.SORA.durations;
|
|
@@ -3980,32 +3369,40 @@ const AGENT_EXECUTION_STATUSES = [
|
|
|
3980
3369
|
"aborted"
|
|
3981
3370
|
];
|
|
3982
3371
|
/**
|
|
3983
|
-
* Why a session summarization happened, stamped on `ISession.summaryTrigger`. Single source for
|
|
3984
|
-
*
|
|
3985
|
-
*
|
|
3986
|
-
*
|
|
3987
|
-
*
|
|
3988
|
-
*
|
|
3989
|
-
*
|
|
3372
|
+
* Why a session summarization happened, stamped on `ISession.summaryTrigger`. Single source for the
|
|
3373
|
+
* places that each used to spell this list out: the Session zod schema (schemas/actions.ts), the
|
|
3374
|
+
* entity type (types/entities/SessionTypes.ts), the Mongoose path (packages/database SessionModel)
|
|
3375
|
+
* and the session.summarize event payload (apps/client server/utils/eventBus.ts). They drifted -
|
|
3376
|
+
* the Mongoose enum said 'milestone'/'growth' for two values nothing produces - and a drift there
|
|
3377
|
+
* is invisible on the update path, because BaseModel's findOneAndUpdate writes without
|
|
3378
|
+
* runValidators.
|
|
3990
3379
|
*
|
|
3991
|
-
*
|
|
3992
|
-
*
|
|
3993
|
-
*
|
|
3994
|
-
*
|
|
3380
|
+
* Those surfaces now name PERSISTED_SESSION_SUMMARY_TRIGGERS below, not the full union: a stored
|
|
3381
|
+
* field may only carry a reason a run HAPPENED. 'throttling' is the exception that forced the
|
|
3382
|
+
* split - shouldSummarizeSession (b4m-core/services ChatCompletionFeatures) returns it as the
|
|
3383
|
+
* reason it DECLINED to summarize, so it describes no run and belongs to a decision, not a
|
|
3384
|
+
* document. It is typed by SummarizationDecision there, not by the session field.
|
|
3995
3385
|
*
|
|
3996
3386
|
* 'manual' means someone asked for one notebook's summary. The admin sweep (apps/client
|
|
3997
3387
|
* server/events/spider.ts) summarizes every un-summarized notebook of the admin who ran it in one
|
|
3998
3388
|
* billed pass, so it stamps 'spider' instead: without that, one deliberate click and a whole sweep
|
|
3999
3389
|
* are indistinguishable when someone investigates unexpected summarization spend.
|
|
4000
3390
|
*/
|
|
4001
|
-
|
|
3391
|
+
/**
|
|
3392
|
+
* The triggers a document may actually carry - every reason a summarization HAPPENED. The event
|
|
3393
|
+
* payload and createSessionParametersSchema both name this list rather than the full union below,
|
|
3394
|
+
* so 'throttling' cannot be published, cannot be stored, and therefore cannot reach a copy path.
|
|
3395
|
+
* Add a new reason-it-happened here, not to SESSION_SUMMARY_TRIGGERS, and every one of those
|
|
3396
|
+
* boundaries picks it up.
|
|
3397
|
+
*/
|
|
3398
|
+
const PERSISTED_SESSION_SUMMARY_TRIGGERS = [
|
|
4002
3399
|
"manual",
|
|
4003
3400
|
"project",
|
|
4004
3401
|
"earlyMilestone",
|
|
4005
3402
|
"contentGrowth",
|
|
4006
|
-
"throttling",
|
|
4007
3403
|
"spider"
|
|
4008
3404
|
];
|
|
3405
|
+
[...PERSISTED_SESSION_SUMMARY_TRIGGERS];
|
|
4009
3406
|
/**
|
|
4010
3407
|
* Operator allow-list for the client-authored Mongo filter carried on a `subscribe_query` frame.
|
|
4011
3408
|
*
|
|
@@ -5173,7 +4570,7 @@ const SessionCreatedAction = shareableDocumentSchema.extend({
|
|
|
5173
4570
|
claudeConversationId: z$1.string().optional(),
|
|
5174
4571
|
summary: z$1.string().optional(),
|
|
5175
4572
|
summaryAt: z$1.date().optional(),
|
|
5176
|
-
summaryTrigger: z$1.enum(
|
|
4573
|
+
summaryTrigger: z$1.enum(PERSISTED_SESSION_SUMMARY_TRIGGERS).optional(),
|
|
5177
4574
|
deletedAt: z$1.date().optional(),
|
|
5178
4575
|
tags: z$1.array(z$1.object({
|
|
5179
4576
|
name: z$1.string(),
|
|
@@ -5370,7 +4767,14 @@ const PermissionRequestAction = z$1.object({
|
|
|
5370
4767
|
executionId: z$1.string(),
|
|
5371
4768
|
toolName: z$1.string(),
|
|
5372
4769
|
toolInput: z$1.unknown(),
|
|
5373
|
-
iteration: z$1.number()
|
|
4770
|
+
iteration: z$1.number(),
|
|
4771
|
+
/**
|
|
4772
|
+
* Provider tool_use id of the specific gated call this card is asking about.
|
|
4773
|
+
* The client echoes it back on `permission_response` so the server can bind
|
|
4774
|
+
* the answer to THIS pause rather than the latest one that happens to share
|
|
4775
|
+
* a tool name - see `handlePermissionResponse`'s toolCallId check.
|
|
4776
|
+
*/
|
|
4777
|
+
toolCallId: z$1.string().optional()
|
|
5374
4778
|
});
|
|
5375
4779
|
const ChildExecutionSnapshotSchema = z$1.lazy(() => z$1.object({
|
|
5376
4780
|
executionId: z$1.string(),
|
|
@@ -5392,7 +4796,8 @@ const ReconnectResultAction = z$1.object({
|
|
|
5392
4796
|
pendingPermission: z$1.object({
|
|
5393
4797
|
toolName: z$1.string(),
|
|
5394
4798
|
toolInput: z$1.unknown(),
|
|
5395
|
-
requestedAt: z$1.union([z$1.string(), z$1.date()])
|
|
4799
|
+
requestedAt: z$1.union([z$1.string(), z$1.date()]),
|
|
4800
|
+
toolCallId: z$1.string().optional()
|
|
5396
4801
|
}).optional(),
|
|
5397
4802
|
totalCreditsUsed: z$1.number().optional(),
|
|
5398
4803
|
iterationCount: z$1.number().optional(),
|
|
@@ -5484,7 +4889,29 @@ z$1.discriminatedUnion("action", [
|
|
|
5484
4889
|
PermissionRequestAction,
|
|
5485
4890
|
ReconnectResultAction
|
|
5486
4891
|
]);
|
|
5487
|
-
|
|
4892
|
+
/**
|
|
4893
|
+
* Public wire schemas for the agent-executor (ReAct) endpoints:
|
|
4894
|
+
* `POST /api/v1/agent-executions` and `GET /api/v1/agent-executions/{id}`.
|
|
4895
|
+
*
|
|
4896
|
+
* These are the REST twin of the WebSocket `agent_execute` command surface
|
|
4897
|
+
* (apps/client/server/websocket/agentExecute.ts). Both transports funnel into the
|
|
4898
|
+
* same `startAgentExecution` service, but the wire shapes are deliberately separate:
|
|
4899
|
+
* the WS payload carries UI-only fields (routing provenance, an optimistic-bubble
|
|
4900
|
+
* back-reference) that must never become published API surface, and public fields are
|
|
4901
|
+
* snake_case per CONVENTIONS.md section 2 while the WS command is camelCase.
|
|
4902
|
+
*
|
|
4903
|
+
* Public-API rules apply here: no `.catch()`, no top-level `.transform()`.
|
|
4904
|
+
*/
|
|
4905
|
+
/**
|
|
4906
|
+
* Request body for `POST /api/v1/agent-executions`.
|
|
4907
|
+
*
|
|
4908
|
+
* `session_id` is required rather than defaulted (unlike `POST /api/chat`, which falls
|
|
4909
|
+
* back to the caller's last notebook): the session is what determines which agent
|
|
4910
|
+
* profile the executor builds, so guessing it would silently change the run's
|
|
4911
|
+
* behaviour. Everything else is optional and falls back to admin defaults or the
|
|
4912
|
+
* agent's own orchestration profile.
|
|
4913
|
+
*/
|
|
4914
|
+
const AgentExecutionStartRequestSchema = z$1.object({
|
|
5488
4915
|
session_id: z$1.string().min(1),
|
|
5489
4916
|
message: z$1.string().min(1),
|
|
5490
4917
|
/** Falls back to the deployment's default chat model when omitted. */
|
|
@@ -5541,7 +4968,11 @@ z$1.object({
|
|
|
5541
4968
|
*/
|
|
5542
4969
|
enable_artifacts: z$1.boolean().optional()
|
|
5543
4970
|
});
|
|
5544
|
-
|
|
4971
|
+
/**
|
|
4972
|
+
* 202 ACK for `POST /api/v1/agent-executions`. The run is fire-and-forget: nothing is
|
|
4973
|
+
* streamed back over REST, so the caller polls `poll_url` until `status` is terminal.
|
|
4974
|
+
*/
|
|
4975
|
+
const AgentExecutionAckSchema = z$1.object({
|
|
5545
4976
|
id: z$1.string(),
|
|
5546
4977
|
status: z$1.literal("pending"),
|
|
5547
4978
|
session_id: z$1.string(),
|
|
@@ -5571,7 +5002,15 @@ const AgentExecutionStepSchema = z$1.object({
|
|
|
5571
5002
|
/** Set on `action` steps: the tool the agent invoked. */
|
|
5572
5003
|
tool_name: z$1.string().optional()
|
|
5573
5004
|
});
|
|
5574
|
-
|
|
5005
|
+
/**
|
|
5006
|
+
* Poll response for `GET /api/v1/agent-executions/{id}`.
|
|
5007
|
+
*
|
|
5008
|
+
* `steps` is the live trace: it grows while the run is in flight (read from the
|
|
5009
|
+
* checkpoint) and freezes at the final one. `answer` is null until the run reaches a
|
|
5010
|
+
* terminal status, and stays null on `failed` / `aborted` - where `error` carries the
|
|
5011
|
+
* reason instead.
|
|
5012
|
+
*/
|
|
5013
|
+
const AgentExecutionStatusResponseSchema = z$1.object({
|
|
5575
5014
|
id: z$1.string(),
|
|
5576
5015
|
status: z$1.enum([
|
|
5577
5016
|
"pending",
|
|
@@ -5603,7 +5042,8 @@ z$1.object({
|
|
|
5603
5042
|
created_at: z$1.string(),
|
|
5604
5043
|
updated_at: z$1.string()
|
|
5605
5044
|
});
|
|
5606
|
-
|
|
5045
|
+
/** Path parameter for `GET /api/v1/agent-executions/{id}`. */
|
|
5046
|
+
const AgentExecutionIdParamSchema = z$1.object({ id: z$1.string().min(1) });
|
|
5607
5047
|
/**
|
|
5608
5048
|
* Tool schema matching ICompletionOptionTools.toolSchema. The Zod surface only
|
|
5609
5049
|
* covers wire-format fields (toolFn is server-side). Replaces the historical
|
|
@@ -5653,7 +5093,17 @@ const CompletionMessageSchema = z$1.object({
|
|
|
5653
5093
|
content: z$1.union([z$1.string(), z$1.array(z$1.any())]),
|
|
5654
5094
|
cache: z$1.boolean().optional()
|
|
5655
5095
|
});
|
|
5656
|
-
|
|
5096
|
+
/**
|
|
5097
|
+
* Schema for CLI LLM completion requests
|
|
5098
|
+
* Shared between Next.js API route (dev) and Lambda function (production)
|
|
5099
|
+
*
|
|
5100
|
+
* `response_format`, `stream`, `tools`, `temperature`, and `max_tokens` are all
|
|
5101
|
+
* accepted at the top level (OpenAI-compatible, matching how every major LLM
|
|
5102
|
+
* SDK shapes a completion request) AND nested under `options` (legacy shape).
|
|
5103
|
+
* Use `normalizeCompletionRequest()` to collapse both surfaces into the
|
|
5104
|
+
* canonical `options.<field>` location before downstream consumption.
|
|
5105
|
+
*/
|
|
5106
|
+
const CompletionRequestSchema = z$1.object({
|
|
5657
5107
|
model: z$1.string(),
|
|
5658
5108
|
messages: z$1.array(CompletionMessageSchema),
|
|
5659
5109
|
response_format: ResponseFormatSchema.optional(),
|
|
@@ -5722,12 +5172,20 @@ const CompletionSseErrorEventSchema = z$1.object({
|
|
|
5722
5172
|
requestId: z$1.string().optional(),
|
|
5723
5173
|
code: z$1.enum(QUEST_ERROR_CODES).optional()
|
|
5724
5174
|
});
|
|
5725
|
-
|
|
5175
|
+
/** One `data:` event in the `text/event-stream` completions response. */
|
|
5176
|
+
const CompletionStreamEventSchema = z$1.union([
|
|
5726
5177
|
CompletionMetaEventSchema,
|
|
5727
5178
|
CompletionContentEventSchema,
|
|
5728
5179
|
CompletionSseErrorEventSchema
|
|
5729
5180
|
]);
|
|
5730
|
-
|
|
5181
|
+
/**
|
|
5182
|
+
* Server-side tool execution schemas for POST /api/ai/v1/tools.
|
|
5183
|
+
*
|
|
5184
|
+
* Plain Zod (no `.openapi()`) so any runtime can import them; the OpenAPI layer
|
|
5185
|
+
* annotates them via the contract. The tool-name enum MUST stay in sync with
|
|
5186
|
+
* SUPPORTED_TOOLS in apps/client/server/cli/toolsHandler.shared.ts.
|
|
5187
|
+
*/
|
|
5188
|
+
const ToolExecutionRequestSchema = z$1.object({
|
|
5731
5189
|
toolName: z$1.enum([
|
|
5732
5190
|
"weather_info",
|
|
5733
5191
|
"web_search",
|
|
@@ -6446,13 +5904,6 @@ const ImageOutputFormatSchema = z$1.enum([
|
|
|
6446
5904
|
"webp"
|
|
6447
5905
|
]);
|
|
6448
5906
|
/**
|
|
6449
|
-
* Degrade a shared output-format setting to what BFL and Gemini accept, so selecting
|
|
6450
|
-
* webp for gpt-image cannot fail an unrelated render after a model switch.
|
|
6451
|
-
*/
|
|
6452
|
-
function toNonWebpOutputFormat(format) {
|
|
6453
|
-
return format === "webp" ? "png" : format;
|
|
6454
|
-
}
|
|
6455
|
-
/**
|
|
6456
5907
|
* Maps legacy/removed image model IDs to their current replacements.
|
|
6457
5908
|
* Prevents Zod validation failures when clients send stale persisted model names.
|
|
6458
5909
|
*
|
|
@@ -6611,7 +6062,13 @@ const SessionTagSchema = z$1.object({
|
|
|
6611
6062
|
name: z$1.string(),
|
|
6612
6063
|
strength: z$1.number()
|
|
6613
6064
|
});
|
|
6614
|
-
|
|
6065
|
+
/**
|
|
6066
|
+
* Request schema for PUT /api/sessions/{id}. This is the exact field allowlist
|
|
6067
|
+
* sessionService.updateSession enforces (b4m-core/services/src/sessionService/update.ts,
|
|
6068
|
+
* which extends this schema with `id`) - shared so the public contract can never
|
|
6069
|
+
* document a field the service silently drops, or vice versa.
|
|
6070
|
+
*/
|
|
6071
|
+
const SessionUpdateRequestSchema = z$1.object({
|
|
6615
6072
|
name: z$1.string().min(1).optional(),
|
|
6616
6073
|
knowledgeIds: z$1.array(z$1.string()).optional(),
|
|
6617
6074
|
artifactIds: z$1.array(z$1.string()).optional(),
|
|
@@ -6621,8 +6078,14 @@ z$1.object({
|
|
|
6621
6078
|
lakeScope: z$1.array(z$1.string()).nullable().optional().describe("The data lakes this session grounds on, as lake tags (the `datalakeTag` of each lake from GET /api/data-lakes). Send a list to ground only on those lakes, `[]` to ground on no lake at all, or `null` to clear the choice so retrieval falls back to every lake you can reach. Omit to leave the current choice unchanged. Tags naming a lake you cannot reach are ignored at retrieval time rather than rejected here. Narrowing the scope does not by itself turn retrieval on: pair it with `forceKnowledgeRetrieval: true` for a session that is not already grounded. Conversely `[]` leaves a grounded session nothing to retrieve from, so its forced retrieval is skipped rather than run against every lake."),
|
|
6622
6079
|
propagateToProjects: z$1.boolean().optional().describe("Defaults to true when omitted. When knowledgeIds grows, the newly-added file ids are also appended to every project that contains this session, granting every member of that project access to those files. This propagation is append-only and cannot be undone through the UI - pass false if newly-attached files should not be shared with the project.")
|
|
6623
6080
|
});
|
|
6624
|
-
|
|
6625
|
-
z$1.object({
|
|
6081
|
+
/** Path parameter for session-scoped endpoints, e.g. GET/PUT /api/sessions/{id}. */
|
|
6082
|
+
const SessionIdParamSchema = z$1.object({ id: z$1.string().min(1) });
|
|
6083
|
+
/**
|
|
6084
|
+
* Practical response subset for PUT /api/sessions/{id} - the fields a caller needs to
|
|
6085
|
+
* confirm an update took effect. ISession (types/entities/SessionTypes.ts) carries many
|
|
6086
|
+
* more server-internal fields not documented as public API surface here.
|
|
6087
|
+
*/
|
|
6088
|
+
const SessionResponseSchema = z$1.object({
|
|
6626
6089
|
id: z$1.string(),
|
|
6627
6090
|
name: z$1.string(),
|
|
6628
6091
|
userId: z$1.string(),
|
|
@@ -6751,15 +6214,6 @@ const CHUNK_STALL_REASONS = [
|
|
|
6751
6214
|
"unchunkedPaused"
|
|
6752
6215
|
];
|
|
6753
6216
|
/**
|
|
6754
|
-
* Whether a file is stalled by the convergence kill switch, by any arm. THE predicate every
|
|
6755
|
-
* reader uses, so adding a stall reason reaches health, convergence and retrieval without separate
|
|
6756
|
-
* comparisons drifting apart. Also the in-memory mirror of a Mongo
|
|
6757
|
-
* `chunkStallReason: { $in: [...CHUNK_STALL_REASONS] }`.
|
|
6758
|
-
*/
|
|
6759
|
-
function isChunkStalled(reason) {
|
|
6760
|
-
return CHUNK_STALL_REASONS.includes(reason);
|
|
6761
|
-
}
|
|
6762
|
-
/**
|
|
6763
6217
|
* Which reasons leave the file with NO passages, as opposed to passages with no vectors. A `Record`
|
|
6764
6218
|
* over every reason rather than a hand-written subset array: a new stall reason then cannot compile
|
|
6765
6219
|
* until it is classified, where a member missing from a literal array would just make a health count
|
|
@@ -6786,76 +6240,7 @@ const CHUNK_STALL_NOTICES = {
|
|
|
6786
6240
|
};
|
|
6787
6241
|
CHUNK_STALL_NOTICES.vectorizePaused;
|
|
6788
6242
|
CHUNK_STALL_NOTICES.rechunkPaused;
|
|
6789
|
-
|
|
6790
|
-
* TRANSITIONAL, and the ONE stall predicate every RETRIEVAL path must use until #2016's migration
|
|
6791
|
-
* has run in every environment. Reads the new field, then falls back to the legacy prose that the
|
|
6792
|
-
* pre-migration rows still carry in `notes`.
|
|
6793
|
-
*
|
|
6794
|
-
* It exists for the FORWARD window only: `migratorInvocation` is a `dependsOn` of the web stack
|
|
6795
|
-
* only (infra/web.ts); the queue stack has none, so the executor can serve forced retrieval and
|
|
6796
|
-
* `knowledge_base_search` while rows still carry the marker in `notes` and no `chunkStallReason`. A
|
|
6797
|
-
* row stalled by the chunk arm then reads as a plain unindexed file: `isRetrievalExcluded` drops it
|
|
6798
|
-
* upstream of the withhold on a vectorizedOnly lake, and `partitionByIndexAvailability` calls it
|
|
6799
|
-
* servable everywhere else. The turn answers around a passage-less file and reports FULL coverage -
|
|
6800
|
-
* the silent degradation this whole path exists to prevent.
|
|
6801
|
-
*
|
|
6802
|
-
* A code ROLLBACK is the mirror image and this arm CANNOT cover it: the rows are already migrated
|
|
6803
|
-
* (`chunkStallReason` set, `notes` unset) and the code restored is pre-#2016, which does not contain
|
|
6804
|
-
* this function. Nothing reverts the data on its own either - `migratorInvocation` only ever runs
|
|
6805
|
-
* `up` and `migrate down` is a manual CLI step - so `migrate down` is a REQUIRED step of any
|
|
6806
|
-
* rollback past #2016, not an optional tidy-up. What this arm does buy is that `down()` is safe to
|
|
6807
|
-
* run FIRST: whichever stack is still new keeps honoring the prose it restores, so a staggered
|
|
6808
|
-
* rollback has no window where a restored marker is invisible. `down()` is a PARTIAL restore
|
|
6809
|
-
* though - it skips a row whose owner typed a note after `up()`, and that row grades as unstalled
|
|
6810
|
-
* on both stacks once the field is dropped. See its own comment.
|
|
6811
|
-
*
|
|
6812
|
-
* Deliberately NOT used by the grading/health/UI readers: they are gated behind the web stack, and
|
|
6813
|
-
* a legacy row there renders the notice line AND the identical text as the owner's note.
|
|
6814
|
-
*
|
|
6815
|
-
* Mirrored in Mongo by `buildFabFileSearchQuery`'s `vectorizedOnly` exemption. Delete the legacy arm
|
|
6816
|
-
* from both together, one release after the migration has landed everywhere.
|
|
6817
|
-
*
|
|
6818
|
-
* Pinned to the two reasons the migration backfilled rather than every notice: `unchunkedPaused`
|
|
6819
|
-
* postdates it, so no row carries its prose, and including it would read an owner who happens to type
|
|
6820
|
-
* that sentence into `notes` as stalled.
|
|
6821
|
-
*/
|
|
6822
|
-
const LEGACY_CHUNK_STALL_NOTES = [CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused];
|
|
6823
|
-
function isChunkStalledFile(file) {
|
|
6824
|
-
return isChunkStalled(file.chunkStallReason) || LEGACY_CHUNK_STALL_NOTES.includes(file.notes ?? "");
|
|
6825
|
-
}
|
|
6826
|
-
/**
|
|
6827
|
-
* `FabFile.chunkRebuildRequestedAt`: stamped by `resetChunkStateByIds` in the SAME write that
|
|
6828
|
-
* clears a file's chunk rollups, so "this file's passages are being rebuilt" can never be lost the
|
|
6829
|
-
* way the pair of steps that creates the state can be. The reset and the queue send are two
|
|
6830
|
-
* operations - kill the producer between them, or lose the consumer's marker write, and the file
|
|
6831
|
-
* sits at `chunkCount: 0` with `error: null` and no stall reason, a shape indistinguishable from an
|
|
6832
|
-
* image or a still-uploading row. It then drops out of lake health's denominator, out of the
|
|
6833
|
-
* convergence plan and out of the retrieval withhold at the same moment: every rollup says its
|
|
6834
|
-
* passages are gone, and nothing reports it.
|
|
6835
|
-
*
|
|
6836
|
-
* Deliberately NOT the `rechunkPaused` stall reason pre-written by the producer, which is the obvious
|
|
6837
|
-
* fix and the wrong one: that marker means "halted, needs an administrator", so a file awaiting an
|
|
6838
|
-
* ORDINARY rebuild would read to every reader as permanently paused for the whole rebuild - search
|
|
6839
|
-
* would tell readers it does not return on its own, health would hard-fail P3, and "Rebuild
|
|
6840
|
-
* passages" would offer to repair a file that is already repairing. A flag that cries wolf on the
|
|
6841
|
-
* normal path is worse than the rare window it closes.
|
|
6842
|
-
*
|
|
6843
|
-
* So the two facts are distinct states, and the consumer UPGRADES one to the other: pending means
|
|
6844
|
-
* "in flight, returns on its own", the paused note means "halted, needs intervention". A LOST
|
|
6845
|
-
* upgrade therefore degrades to mislabelled-but-visible rather than invisible, which is the trade
|
|
6846
|
-
* this field exists to make - invisibility is the real harm, labelling is secondary.
|
|
6847
|
-
*
|
|
6848
|
-
* A dedicated field on purpose, and the precedent #2016 followed for the other two machine-written
|
|
6849
|
-
* facts: while they all shared `notes` every writer of that field clobbered the others, including
|
|
6850
|
-
* the user's own note.
|
|
6851
|
-
*
|
|
6852
|
-
* Cleared by `commitFabFileChunks` (the rebuild landed) and by the chunk handler's pause write (the
|
|
6853
|
-
* rebuild was halted instead). A file carrying `error` is settled regardless - see
|
|
6854
|
-
* `isMemberIndexingInFlight`, which is where the precedence between these three lives.
|
|
6855
|
-
*/
|
|
6856
|
-
function isChunkRebuildPending(requestedAt) {
|
|
6857
|
-
return requestedAt !== null && requestedAt !== void 0 && requestedAt !== "";
|
|
6858
|
-
}
|
|
6243
|
+
CHUNK_STALL_NOTICES.vectorizePaused, CHUNK_STALL_NOTICES.rechunkPaused;
|
|
6859
6244
|
/** Ceiling so "adjustable" cannot mean "unbounded" in either direction. */
|
|
6860
6245
|
const LAKE_ACCESS_AUDIT_RETENTION_MAX_DAYS = 2555;
|
|
6861
6246
|
/**
|
|
@@ -6949,47 +6334,8 @@ function defaultEmbeddingModelForEnv() {
|
|
|
6949
6334
|
const selfHost = process.env.B4M_SELF_HOST === "true";
|
|
6950
6335
|
const hasOllama = !!process.env.OLLAMA_BASE_URL?.trim();
|
|
6951
6336
|
const hasCloudEmbeddingKey = !isPlaceholderApiKey(process.env.OPENAI_API_KEY) || !isPlaceholderApiKey(process.env.VOYAGE_API_KEY);
|
|
6952
|
-
if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
|
|
6953
|
-
return "text-embedding-3-small";
|
|
6954
|
-
}
|
|
6955
|
-
/**
|
|
6956
|
-
* True when this deployment can embed with no provider API key at all: a cloud stage reaches
|
|
6957
|
-
* Bedrock through its task/execution role's AWS credentials.
|
|
6958
|
-
*
|
|
6959
|
-
* Requires POSITIVE evidence of an execution role rather than merely "not self-host". A plain
|
|
6960
|
-
* `next dev` session and a CI job both leave B4M_SELF_HOST unset while holding no AWS credentials
|
|
6961
|
-
* at all, so an absence test would send them to the Bedrock SDK for an opaque `CredentialsProvider
|
|
6962
|
-
* Error` in place of the actionable OPENAI_KEY_MISSING_MESSAGE naming the key to set - the same
|
|
6963
|
-
* actionable-to-opaque trade this fallback exists to avoid, just in a different keyless place.
|
|
6964
|
-
*
|
|
6965
|
-
* BOTH runtimes must be covered, and they carry different markers. The Lambdas (vectorize
|
|
6966
|
-
* subscriber, the crons, the Next API routes) get AWS_LAMBDA_FUNCTION_NAME; ChatCompletion is a
|
|
6967
|
-
* Fargate service (infra/chatCompletion.ts) and gets the ECS task-role URI instead. Since
|
|
6968
|
-
* knowledgeBaseSearch runs inside that container, keying on the Lambda marker alone would leave
|
|
6969
|
-
* chat knowledge-base search failing on exactly the keyless stages this fallback is for.
|
|
6970
|
-
*
|
|
6971
|
-
* SST_RESOURCE_App is the third arm and the one this repo can prove: SST sets it on anything it
|
|
6972
|
-
* links, Lambda and Service alike (infra/chatCompletion.ts:129 and infra/agentExecutor.ts:54 both
|
|
6973
|
-
* note that linking alone exposes SST_RESOURCE_*). It covers the Fargate task whether or not the
|
|
6974
|
-
* ECS credential URI is present, and it is absent from a plain `next dev` and from CI, which is
|
|
6975
|
-
* the case that matters. `sst dev` does set it - correctly, since that session runs against real
|
|
6976
|
-
* AWS credentials.
|
|
6977
|
-
*
|
|
6978
|
-
* Self-host is excluded outright because it has no such role - its keyless path is the local
|
|
6979
|
-
* Ollama embedder (`isLocalEmbedderAvailable` in toolAvailability.ts), not Bedrock.
|
|
6980
|
-
*
|
|
6981
|
-
* Answers "is Bedrock reachable here", NOT "should we use it" - a keyed stage is keyless-capable
|
|
6982
|
-
* too, so this must only ever be asked ALONGSIDE a resolved credential table that came back empty.
|
|
6983
|
-
* `resolveEmbeddingWithKeylessFallback` is where the two questions are paired for callers free to
|
|
6984
|
-
* choose the model, and is what such a caller should use instead of asking this directly. The
|
|
6985
|
-
* direct callers are the ones that additionally need the answer BEFORE resolving, to decide
|
|
6986
|
-
* policy: toolAvailability reports whether embedding-backed tools are usable at all, and
|
|
6987
|
-
* data-lakes/semantic-search decides whether a substitution is permitted for this request before
|
|
6988
|
-
* it knows whether one is needed. Both still pair it with the table.
|
|
6989
|
-
*/
|
|
6990
|
-
function hasKeylessCloudEmbedder() {
|
|
6991
|
-
if (process.env.B4M_SELF_HOST === "true") return false;
|
|
6992
|
-
return !!(process.env.AWS_LAMBDA_FUNCTION_NAME || process.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI || process.env.AWS_CONTAINER_CREDENTIALS_FULL_URI || process.env.SST_RESOURCE_App);
|
|
6337
|
+
if (selfHost && hasOllama && !hasCloudEmbeddingKey) return "qwen3-embedding:0.6b";
|
|
6338
|
+
return "text-embedding-3-small";
|
|
6993
6339
|
}
|
|
6994
6340
|
z$1.union([
|
|
6995
6341
|
z$1.enum(OpenAIEmbeddingModel),
|
|
@@ -7182,20 +6528,6 @@ Call \`search_knowledge_base\` BEFORE answering when that library would settle t
|
|
|
7182
6528
|
Do not search when the answer is already in front of you or out of scope: general knowledge (definitions, mathematics, established theory, public facts); anything answerable from this conversation, from an attached document, or from content already retrieved for you this turn; or a request to transform, summarize or reformat text the user has just supplied. If the library has already been searched on this turn, do not search it again for the same question - a repeat spends a round trip to return the same passages.
|
|
7183
6529
|
|
|
7184
6530
|
When a search does not turn up what was asked for, say so plainly rather than filling the gap from training data, and never imply an answer came from the user's documents when it did not.`;
|
|
7185
|
-
/**
|
|
7186
|
-
* Default text for the formatting system message. Runtime fallback used by
|
|
7187
|
-
* `includeHardcodedSystemMessage` (b4m-core/utils/src/llm/utils.ts) when the `FormatPromptTemplate`
|
|
7188
|
-
* admin setting is blank; that setting's own default is intentionally '' - keep this the sole home.
|
|
7189
|
-
*
|
|
7190
|
-
* Deliberately scoped to formatting ONLY. The previous wording ("Adhere to specific formatting
|
|
7191
|
-
* requests...") read as a general compliance instruction and bled into WHETHER to answer: as the
|
|
7192
|
-
* only system content it roughly halved refusal quality. The opening clause is the fix - it fences
|
|
7193
|
-
* this message off from the answer/abstain decision. Injected only when `UseFormatPrompt` is on.
|
|
7194
|
-
*
|
|
7195
|
-
* NOTE: a stored settings row pins its own wording, so changing this default does not reach an
|
|
7196
|
-
* existing deployment that has already saved a value - the row must be edited in admin settings too.
|
|
7197
|
-
*/
|
|
7198
|
-
const FORMAT_PROMPT_TEMPLATE = `Formatting only - nothing here decides whether or how fully to answer. Format replies to maintain the integrity of the requested style; default to markdown for text. Preserve proper structure for poems, songs, or haikus. When the user specifies an output format (e.g. TypeScript), use that format for the parts you do answer.`;
|
|
7199
6531
|
z$1.enum([
|
|
7200
6532
|
"openaiDemoKey",
|
|
7201
6533
|
"anthropicDemoKey",
|
|
@@ -7239,13 +6571,16 @@ z$1.enum([
|
|
|
7239
6571
|
"EnableDataLakeSlackAdd",
|
|
7240
6572
|
"EnableDataLakeGroundingMode",
|
|
7241
6573
|
"EnableLakeMemory",
|
|
6574
|
+
"EnableLakeModelInconsistencyDetection",
|
|
7242
6575
|
"EnableDataLakeVectorSearch",
|
|
7243
6576
|
"EnableRetrievalSupersessionCollapse",
|
|
7244
6577
|
"PauseLakeConvergence",
|
|
7245
6578
|
"LakeConvergenceBulkChangeSharePct",
|
|
7246
6579
|
"EnforceLakeReadGrants",
|
|
7247
6580
|
"EnableDataLakeDrivePoll",
|
|
6581
|
+
"EnableDataLakeGitHub",
|
|
7248
6582
|
"EnforceLakeAdmission",
|
|
6583
|
+
"EnforceLakeOriginOnIngest",
|
|
7249
6584
|
"EnableBriefcase",
|
|
7250
6585
|
"EnableBriefcaseDefault",
|
|
7251
6586
|
"EnableImageTemplates",
|
|
@@ -7451,7 +6786,13 @@ const IntentClassifierConfigSchema = z$1.object({
|
|
|
7451
6786
|
fallbackModels: z$1.array(z$1.string()).default(["gemini-2.5-flash-lite", "gpt-5.4-nano"])
|
|
7452
6787
|
});
|
|
7453
6788
|
const OrchestrationDefaultsSchema = z$1.object({
|
|
7454
|
-
/**
|
|
6789
|
+
/**
|
|
6790
|
+
* Tool names the synthetic profile is allowed to invoke. A DEFAULT toolbelt, not a gate:
|
|
6791
|
+
* an agentless chat dispatch ships the user's ambient Smart Tools and the executor UNIONS
|
|
6792
|
+
* them onto this list (`pickEffectiveEnabledTools`), so narrowing this narrows what the
|
|
6793
|
+
* agent brings of its own rather than capping what the user may select. `deniedTools` below
|
|
6794
|
+
* is the gate.
|
|
6795
|
+
*/
|
|
7455
6796
|
allowedTools: z$1.array(z$1.string()).default([
|
|
7456
6797
|
"web_search",
|
|
7457
6798
|
"retrieve_knowledge_content",
|
|
@@ -9212,6 +8553,16 @@ const settingsMap = {
|
|
|
9212
8553
|
order: 91,
|
|
9213
8554
|
dependsOn: "EnableDataLakes"
|
|
9214
8555
|
}),
|
|
8556
|
+
EnableLakeModelInconsistencyDetection: makeBooleanSetting({
|
|
8557
|
+
key: "EnableLakeModelInconsistencyDetection",
|
|
8558
|
+
name: "Data Lakes: Model-driven contradiction pass",
|
|
8559
|
+
defaultValue: false,
|
|
8560
|
+
description: "Gate for the model-driven reading pass (#3057) that finds cross-document contradictions the lexical pattern rules cannot - two documents stating incompatible things in ordinary prose. Off by default: unlike the free lexical pass, this reads corpus content through an LLM, so it costs real money per run. Findings land in the same durable findings collection (detector: 'model') as the lexical pass, triggered the same way (POST /api/data-lakes/:id/inconsistencies?detector=model), gated separately here and rate-limited far lower per caller. Detect only - see the guardrail on corpusInconsistency.ts.",
|
|
8561
|
+
category: "Experimental",
|
|
8562
|
+
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8563
|
+
order: 98,
|
|
8564
|
+
dependsOn: "EnableDataLakes"
|
|
8565
|
+
}),
|
|
9215
8566
|
EnableDataLakeVectorSearch: makeBooleanSetting({
|
|
9216
8567
|
key: "EnableDataLakeVectorSearch",
|
|
9217
8568
|
name: "Data Lakes: Use Atlas $vectorSearch",
|
|
@@ -9283,6 +8634,16 @@ const settingsMap = {
|
|
|
9283
8634
|
order: 95,
|
|
9284
8635
|
dependsOn: "EnableDataLakes"
|
|
9285
8636
|
}),
|
|
8637
|
+
EnableDataLakeGitHub: makeBooleanSetting({
|
|
8638
|
+
key: "EnableDataLakeGitHub",
|
|
8639
|
+
name: "Data Lakes: GitHub repository source",
|
|
8640
|
+
defaultValue: false,
|
|
8641
|
+
description: "Server-side gate for connecting a GitHub repository to a data lake through the read-only GitHub App (contents:read + metadata:read on the one selected repository). Off by default while the connect, ingest and purge pieces land dark; every GitHub lake route answers 403 until it is on.",
|
|
8642
|
+
category: "Experimental",
|
|
8643
|
+
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8644
|
+
order: 96,
|
|
8645
|
+
dependsOn: "EnableDataLakes"
|
|
8646
|
+
}),
|
|
9286
8647
|
EnforceLakeAdmission: makeBooleanSetting({
|
|
9287
8648
|
key: "EnforceLakeAdmission",
|
|
9288
8649
|
name: "Data Lakes: Enforce the admission contract",
|
|
@@ -9298,6 +8659,21 @@ const settingsMap = {
|
|
|
9298
8659
|
"lake"
|
|
9299
8660
|
] }
|
|
9300
8661
|
}),
|
|
8662
|
+
EnforceLakeOriginOnIngest: makeBooleanSetting({
|
|
8663
|
+
key: "EnforceLakeOriginOnIngest",
|
|
8664
|
+
name: "Data Lakes: Enforce curated-lake origin on ingest",
|
|
8665
|
+
defaultValue: true,
|
|
8666
|
+
description: "ON by default: unattended ingest (the Drive folder sync) refuses to add content to a lake whose owner declared it curated. OFF makes the refusal advisory and lets the write through. Unlike the admission contract this ships ON, because it refuses on an explicit owner declaration rather than a heuristic, and because the origin backfill marks every lake that currently has a connector as connector-fed - so at rollout this refuses nothing that exists. The lake rung is the one that matters; the org and owner rungs disable it across every lake in that scope at once. A flip is not instantaneous: the settings cache is per-instance, so it applies immediately on the instance that served the change and within ~5 min elsewhere.",
|
|
8667
|
+
category: "Experimental",
|
|
8668
|
+
group: API_SERVICE_GROUPS.EXPERIMENTAL.id,
|
|
8669
|
+
order: 97,
|
|
8670
|
+
dependsOn: "EnableDataLakes",
|
|
8671
|
+
scope: { settableAt: [
|
|
8672
|
+
"organization",
|
|
8673
|
+
"owner",
|
|
8674
|
+
"lake"
|
|
8675
|
+
] }
|
|
8676
|
+
}),
|
|
9301
8677
|
EnableBriefcase: makeBooleanSetting({
|
|
9302
8678
|
key: "EnableBriefcase",
|
|
9303
8679
|
name: "Enable Briefcase",
|
|
@@ -12074,7 +11450,28 @@ const CitableSourceSchema = z$1.object({
|
|
|
12074
11450
|
practiceAreas: z$1.array(z$1.string()).optional(),
|
|
12075
11451
|
chunkId: z$1.string().optional(),
|
|
12076
11452
|
relevanceScore: z$1.number().optional(),
|
|
12077
|
-
fullContext: z$1.string().optional()
|
|
11453
|
+
fullContext: z$1.string().optional(),
|
|
11454
|
+
/** web_search's own thumbnail/image cluster for this source, gated on `withImages`. */
|
|
11455
|
+
thumbnail: z$1.string().optional(),
|
|
11456
|
+
images: z$1.array(z$1.string()).optional(),
|
|
11457
|
+
/**
|
|
11458
|
+
* Ids of the other cited sources this one provably disagrees with (#3041). Declared rather
|
|
11459
|
+
* than left to the loose object, for the same reason chunkId/fullContext are: a writer that
|
|
11460
|
+
* stamps the wrong shape should fail here, not render a badge that silently names nobody.
|
|
11461
|
+
*/
|
|
11462
|
+
conflictsWith: z$1.array(z$1.string()).optional(),
|
|
11463
|
+
/** web_search's provider-located place (WebSearchPlace), the only source of map coordinates. */
|
|
11464
|
+
place: z$1.object({
|
|
11465
|
+
id: z$1.string(),
|
|
11466
|
+
name: z$1.string(),
|
|
11467
|
+
lat: z$1.number(),
|
|
11468
|
+
lng: z$1.number(),
|
|
11469
|
+
rating: z$1.number().optional(),
|
|
11470
|
+
reviews: z$1.number().optional(),
|
|
11471
|
+
category: z$1.string().optional(),
|
|
11472
|
+
address: z$1.string().optional(),
|
|
11473
|
+
thumbnail: z$1.string().optional()
|
|
11474
|
+
}).optional()
|
|
12078
11475
|
}).optional()
|
|
12079
11476
|
});
|
|
12080
11477
|
/**
|
|
@@ -12461,7 +11858,44 @@ const RetrievalSummarySchema = z$1.object({
|
|
|
12461
11858
|
* absence is weaker evidence than presence. Date-bound any rollup: turns predating this field
|
|
12462
11859
|
* carry nothing, and no backfill is possible - a past turn's grant rows have moved on.
|
|
12463
11860
|
*/
|
|
12464
|
-
grantedLakeIdsUsed: z$1.array(z$1.string()).optional()
|
|
11861
|
+
grantedLakeIdsUsed: z$1.array(z$1.string()).optional(),
|
|
11862
|
+
/**
|
|
11863
|
+
* How many lakes were excluded from this turn's scope because the caller lacks the access to
|
|
11864
|
+
* search them, and why (#3055). Resolved at the seed alongside `lakeScope`, from a dedicated
|
|
11865
|
+
* count-only query (see excludedByAccessCount on getDynamicDataLakeAccess - NOT derived from
|
|
11866
|
+
* the candidate set `lakeScope` comes from, which already has the gate enforced datastore-side
|
|
11867
|
+
* and so cannot see this population).
|
|
11868
|
+
*
|
|
11869
|
+
* ABSENT MEANS NOT RECORDED, never "nothing was excluded" - a turn with nothing excluded records
|
|
11870
|
+
* `count: 0` explicitly. Three distinct causes collapse into this one absent state and are not
|
|
11871
|
+
* distinguishable from it: a turn predating this field, a turn whose `retrieval` was written
|
|
11872
|
+
* only by a tool arm rather than by the seed, and the count-only query itself failing or not
|
|
11873
|
+
* being wired on this host (mirrors `lakeViewComplete`'s contract on the access resolver: a
|
|
11874
|
+
* failure must report unknown, never a false zero).
|
|
11875
|
+
*
|
|
11876
|
+
* COUNT AND REASON ONLY, DELIBERATELY. Never a lake id, name, or tag: the caller may not be
|
|
11877
|
+
* permitted to know a given excluded lake exists at all, and this field must stay safe to show
|
|
11878
|
+
* them regardless of which specific lake(s) it is counting. `reason` is a closed enum, not free
|
|
11879
|
+
* text - prose could leak a lake's identity through phrasing - so a future exclusion cause (e.g.
|
|
11880
|
+
* an archived or quota-limited lake) adds an enum value here rather than a description.
|
|
11881
|
+
*
|
|
11882
|
+
* 'access' is the only reason today: the caller's org membership or the lake's public listing
|
|
11883
|
+
* surfaced it as a candidate (they could see it exists) but they hold neither its own
|
|
11884
|
+
* gate/entitlement nor an ownership or grant exception for it.
|
|
11885
|
+
*
|
|
11886
|
+
* A session-preauthorized lake (unionPreauthorizedLakeAccess) that is ALSO gate-dropped from
|
|
11887
|
+
* this account-wide count is corrected, not merely narrow: the seed's targeted measurement
|
|
11888
|
+
* (measureIdentityNamedExclusion, ChatCompletionProcess's promptMeta seed) excludes exactly the
|
|
11889
|
+
* tags this turn successfully admitted via preauthorization before running the gate query, so an
|
|
11890
|
+
* admitted-and-searched lake never reports here as excluded. This account-wide number itself
|
|
11891
|
+
* (excludedByAccessCount on getDynamicDataLakeAccess) is still computed before that union and is
|
|
11892
|
+
* NOT corrected the same way - only the per-turn targeted measurement is, which is what a
|
|
11893
|
+
* preauthorized session's own narrowing always uses (see sessionNamesALake's call site).
|
|
11894
|
+
*/
|
|
11895
|
+
excludedLakes: z$1.object({
|
|
11896
|
+
count: z$1.number().int().nonnegative(),
|
|
11897
|
+
reason: z$1.enum(["access"])
|
|
11898
|
+
}).optional()
|
|
12465
11899
|
});
|
|
12466
11900
|
/**
|
|
12467
11901
|
* Why a grounded turn's library scan stopped short of the whole library.
|
|
@@ -12669,7 +12103,7 @@ const MeSubscriptionSchema = z$1.object({
|
|
|
12669
12103
|
/** ISO 8601. When the current billing period ends - not a cancellation date. */
|
|
12670
12104
|
current_period_ends_at: z$1.string()
|
|
12671
12105
|
});
|
|
12672
|
-
z$1.object({
|
|
12106
|
+
const MeResponseSchema = z$1.object({
|
|
12673
12107
|
/** Stable B4M user id. Safe to key an integrator's own records on. */
|
|
12674
12108
|
id: z$1.string(),
|
|
12675
12109
|
/** Display name. Never the email address. */
|
|
@@ -13319,13 +12753,6 @@ const DATA_LAKES = [{
|
|
|
13319
12753
|
}
|
|
13320
12754
|
})()];
|
|
13321
12755
|
new Set(DATA_LAKES.map((l) => l.id));
|
|
13322
|
-
/**
|
|
13323
|
-
* Canonical normalization for entitlement keys + `requiredEntitlement` values - the ONE
|
|
13324
|
-
* rule, applied at write time (create/update/stamp) and at match time. Mirrors the
|
|
13325
|
-
* entitlement registry's `normalizeTag` (trim + lowercase) so a value authored in any
|
|
13326
|
-
* casing matches the lowercase keys the resolver produces.
|
|
13327
|
-
*/
|
|
13328
|
-
const normalizeEntitlementKey = (key) => key.trim().toLowerCase();
|
|
13329
12756
|
const sha256Regex = /^[a-f0-9]{64}$/;
|
|
13330
12757
|
const requiredUserTagValue = z.string().trim().min(1).max(100).refine((s) => !/[,;]/.test(s), "User tag must be a single tag with no commas or semicolons (e.g. \"vip\" or \"Sales Team\")");
|
|
13331
12758
|
z.object({
|
|
@@ -13335,7 +12762,8 @@ z.object({
|
|
|
13335
12762
|
fileTagPrefix: z.string().trim().min(2).max(30).refine((s) => s.endsWith(":"), "Tag prefix must end with \":\" (e.g. \"acme:\")").refine((s) => !hasBlankTagPrefixSegment(s), "Tag prefix segments must be non-empty (e.g. \"acme:\" or \"acme:legal:\")").refine((s) => !isReservedTagPrefix(s), `Tag prefix cannot use the reserved "${DATALAKE_TAG_PREFIX}" namespace`),
|
|
13336
12763
|
requiredUserTag: requiredUserTagValue.optional(),
|
|
13337
12764
|
requiredEntitlement: z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")").optional(),
|
|
13338
|
-
organizationId: z.string().optional()
|
|
12765
|
+
organizationId: z.string().optional(),
|
|
12766
|
+
origin: z.enum(DATA_LAKE_ORIGINS).optional()
|
|
13339
12767
|
});
|
|
13340
12768
|
z.object({
|
|
13341
12769
|
name: z.string().min(1).max(200).optional(),
|
|
@@ -13347,7 +12775,8 @@ z.object({
|
|
|
13347
12775
|
requiredEntitlement: z.union([z.literal(""), z.string().min(3).max(100).refine((s) => s.includes(":") && s.split(":").every((part) => part.length > 0), "Entitlement key must be namespaced with non-empty parts (e.g. \"product:pro\")")]).optional(),
|
|
13348
12776
|
auditQueryTextEnabled: z.boolean().optional(),
|
|
13349
12777
|
lakeMemoryEnabled: z.boolean().optional(),
|
|
13350
|
-
requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional()
|
|
12778
|
+
requiredPassageTokenTarget: z.number().int().min(64).max(OVERSIZED_PASSAGE_TOKEN_THRESHOLD).nullable().optional(),
|
|
12779
|
+
origin: z.enum(DATA_LAKE_ORIGINS).optional()
|
|
13351
12780
|
});
|
|
13352
12781
|
z.object({
|
|
13353
12782
|
groundingMode: z.enum(DATA_LAKE_GROUNDING_MODES).optional(),
|
|
@@ -13489,7 +12918,380 @@ z$1.object({
|
|
|
13489
12918
|
model: ImageTemplateModelSchema,
|
|
13490
12919
|
settings: ImageTemplateSettingsSchema
|
|
13491
12920
|
}).omit({ model: true }).partial();
|
|
13492
|
-
|
|
12921
|
+
/**
|
|
12922
|
+
* Identity factory that pins an endpoint contract's type. The `const` type
|
|
12923
|
+
* parameter preserves the concrete request-schema type so downstream adapters
|
|
12924
|
+
* can infer the validated body type (`z.infer<contract['request']>`) rather than
|
|
12925
|
+
* collapsing to the `z.ZodTypeAny` constraint.
|
|
12926
|
+
*
|
|
12927
|
+
* Otherwise deliberately a no-op at runtime (no registry side effect, no
|
|
12928
|
+
* `.openapi()`) so a contract stays a plain, transport-agnostic value that any
|
|
12929
|
+
* runtime can import - the one exception is the `pathParams`/`queryParams`
|
|
12930
|
+
* overlap check below, a plain assertion with no side effect of its own.
|
|
12931
|
+
*/
|
|
12932
|
+
function defineEndpoint(contract) {
|
|
12933
|
+
if (contract.pathParams && contract.queryParams) {
|
|
12934
|
+
const queryKeys = new Set(Object.keys(contract.queryParams.shape));
|
|
12935
|
+
const overlap = Object.keys(contract.pathParams.shape).filter((key) => queryKeys.has(key));
|
|
12936
|
+
if (overlap.length > 0) throw new Error(`defineEndpoint(${contract.operationId}): pathParams and queryParams both declare ${JSON.stringify(overlap)}. Next merges both into req.query by name, so the path segment would silently win over the query value. Rename one side.`);
|
|
12937
|
+
}
|
|
12938
|
+
return contract;
|
|
12939
|
+
}
|
|
12940
|
+
defineEndpoint({
|
|
12941
|
+
method: "post",
|
|
12942
|
+
path: "/api/chat",
|
|
12943
|
+
operationId: "sendChatMessage",
|
|
12944
|
+
summary: "Send a chat message",
|
|
12945
|
+
description: "Sends a message to the AI and creates a quest to process it. By default (async) the call returns immediately with a quest id; poll `GET /api/quests/{id}` for the reply. Send `wait: true` to block until the reply is ready and receive it inline. A tool that produced machine-readable state reports it under `toolPayloads` - an array of `{ type, payload }` entries in emission order, alongside (never instead of) the prose reply - on the `wait: true` body and on the polled quest. A turn can FAIL after the ACK - reported on the polled quest as `type: \"error\"`, since the ACK was already sent. A terminal `status: \"stopped\"` (a missing session, a user-cancelled turn) is ALSO a failure even without `type: \"error\"` - it carries an explanatory string in `reply`/`replies` rather than an answer. `type` is the failure signal for the classified failure classes (an abort, a provider timeout, credit exhaustion); `errorCode` is an optional refinement present only when the failure is a classified billing reason. A recovered stuck quest is NOT in that list: one that still has renderable content resolves as a success by design. `QUEST_ERROR_CODES` has two members, but only `insufficient_credits` is raised as a quest errorCode by any current throw site on this endpoint - `spend_cap_exceeded` is thrown only by the embed chat route's pre-flight, which fires outside the process try/catch that would classify it onto a quest. A caller must treat `type: \"error\"` OR a terminal `status: \"stopped\"` as failure even when `errorCode` is absent, and must not read `reply` as an answer without checking those first. Authenticate with an API key (`b4m_live_`) or a JWT.",
|
|
12946
|
+
tags: ["AI"],
|
|
12947
|
+
auth: "apiKeyOrJwt",
|
|
12948
|
+
scopes: ["ai:chat", "ai:generate"],
|
|
12949
|
+
request: SimplifiedChatRequestSchema,
|
|
12950
|
+
requestExample: {
|
|
12951
|
+
message: "How do I reset my password?",
|
|
12952
|
+
toolMode: "smart"
|
|
12953
|
+
},
|
|
12954
|
+
emitsRateLimitHeaders: true,
|
|
12955
|
+
responses: {
|
|
12956
|
+
200: {
|
|
12957
|
+
description: "Message accepted - NOT a completed turn. The default (async) path returns this queued ACK; the outcome arrives on `GET /api/quests/{id}` (see the `sendChatMessage200PollResult` schema). With `wait: true` the body additionally carries the completed reply (`response`/`responses`), `toolPayloads`, `createdAt`, and `performance` timings - fields not modelled here yet; the synchronous response shape is a follow-up. A turn that FAILS still resolves with `200`, never a 4xx, on both that `wait: true` body and the polled quest (`GET /api/quests/{id}`) - the prose explaining why lands in `reply`/`response` like any other answer, so the reply text alone cannot tell a failure from an answer. `type` is the field that can: both surfaces carry it unconditionally, so match on `type: \"error\"` first - it covers credit exhaustion, a provider timeout or overload, and an in-process aborted turn; a real answer carries the turn's actual completion type instead (`\"message\"` for an ordinary reply). Two related states do NOT set `type: \"error\"`: a user-cancelled turn resolves as `status: \"stopped\"` with `type` left at `\"message\"`, and a recovered stuck quest that still has renderable content resolves as a success (`status: \"done\"`, no error) by design, to avoid destroying content to report a failure. `errorCode` then names the failure reason, but only for the billing failures that have one - `\"insufficient_credits\"` today; it is absent on every other `type: \"error\"` turn, so never use its absence to infer success. On a real answer `errorCode` is absent from the `wait: true` body. Contrast the tts/music/soundEffects contracts, which reject synchronously with a 422 carrying the same `errorCode` vocabulary.",
|
|
12958
|
+
schema: ChatAckSchema,
|
|
12959
|
+
pollResult: {
|
|
12960
|
+
schema: ChatQuestPollResultSchema,
|
|
12961
|
+
description: "Outcome fields of the quest polled at `GET /api/quests/{id}` after this ACK. A finished turn that failed is `status: \"done\"` with `type: \"error\"` and the failure text in `reply`, so a caller reading `reply` alone cannot tell a failure from an answer - check `type` first, and also treat a terminal `status: \"stopped\"` (a missing session, a user-cancelled turn) as failure even though it never sets `type`. `errorCode` is an optional refinement of `type: \"error\"`, present only for a classified billing failure; credit exhaustion arrives here as `insufficient_credits`, the same vocabulary the synchronous 422s on `/api/ai/music`, `/api/ai/sound-effects` and `/api/ai/tts` use. `QUEST_ERROR_CODES` publishes a second member, `spend_cap_exceeded`, but no current throw site on this endpoint raises it as a quest errorCode: its only one, the embed chat route's pre-flight 422, fires outside the process try/catch that would classify it onto the quest. Most `type: \"error\"` turns - an abort, a provider timeout or overload - have NO `errorCode`; its absence does not mean success, only that the failure is unclassified. A recovered stuck quest that still has renderable content is not a failure at all: it keeps `type: \"message\"` even though it did not finish, so a caller gets the content rather than an error. The poll body carries further fields not modelled here, including `images`, `files`, `toolPayloads`, `promptMeta`, and the attachment report (`attachmentNotices`/`attachmentDelivery`) - only the outcome subset is modelled here.",
|
|
12962
|
+
example: {
|
|
12963
|
+
id: "664f1c2b9a1e4d0012ab34cd",
|
|
12964
|
+
status: "done",
|
|
12965
|
+
type: "error",
|
|
12966
|
+
errorCode: "insufficient_credits",
|
|
12967
|
+
reply: "You're out of credits. This request needs about 12 credits, but only 3 are available."
|
|
12968
|
+
}
|
|
12969
|
+
}
|
|
12970
|
+
},
|
|
12971
|
+
400: {
|
|
12972
|
+
description: "No usable default chat model is configured and none was supplied.",
|
|
12973
|
+
schema: ApiErrorSchema
|
|
12974
|
+
},
|
|
12975
|
+
404: {
|
|
12976
|
+
description: "No notebook/session exists to attach the message to.",
|
|
12977
|
+
schema: ApiErrorSchema
|
|
12978
|
+
},
|
|
12979
|
+
422: {
|
|
12980
|
+
description: "Request body failed schema validation.",
|
|
12981
|
+
schema: ApiErrorSchema
|
|
12982
|
+
},
|
|
12983
|
+
429: {
|
|
12984
|
+
description: "Per-user rate limit exceeded.",
|
|
12985
|
+
schema: ApiErrorSchema
|
|
12986
|
+
}
|
|
12987
|
+
},
|
|
12988
|
+
codeSample: {
|
|
12989
|
+
authToken: "b4m_live_<key>",
|
|
12990
|
+
streaming: false,
|
|
12991
|
+
body: {
|
|
12992
|
+
message: "How do I reset my password?",
|
|
12993
|
+
toolMode: "smart"
|
|
12994
|
+
}
|
|
12995
|
+
}
|
|
12996
|
+
});
|
|
12997
|
+
defineEndpoint({
|
|
12998
|
+
method: "post",
|
|
12999
|
+
path: "/api/v1/agent-executions",
|
|
13000
|
+
operationId: "startAgentExecution",
|
|
13001
|
+
summary: "Start an agent execution",
|
|
13002
|
+
description: "Runs the tool-using agent (ReAct) loop against a session - the same pipeline the product UI's Agent Mode toggle dispatches to, and the REST equivalent of the `agent_execute` WebSocket command. The run is asynchronous: this returns `202` with an execution id, and the caller polls `GET /api/v1/agent-executions/{id}` until `status` is terminal (`completed`, `failed`, or `aborted`). Nothing is streamed back over REST - for live iteration events, use the WebSocket route instead. Omit `agent_id` to get the profile the session's own surface resolves to, which is what reproduces the in-app toggle. The final reply is also written to the session as a normal chat message, so it appears in history. Naming a tool in `tools` also PRE-APPROVES it for the run: there is no interactive client to answer a permission prompt, so a run that calls an approval-gated tool you did not name fails with that tool named in `error` rather than hanging. Authenticate with an API key (`b4m_live_`) or a JWT.",
|
|
13003
|
+
tags: ["AI"],
|
|
13004
|
+
auth: "apiKeyOrJwt",
|
|
13005
|
+
scopes: ["ai:chat", "ai:generate"],
|
|
13006
|
+
request: AgentExecutionStartRequestSchema,
|
|
13007
|
+
requestExample: {
|
|
13008
|
+
session_id: "<sessionId>",
|
|
13009
|
+
message: "Audit this data set and summarize what stands out."
|
|
13010
|
+
},
|
|
13011
|
+
emitsRateLimitHeaders: true,
|
|
13012
|
+
responses: {
|
|
13013
|
+
202: {
|
|
13014
|
+
description: "Run accepted and dispatched. Poll `tracking_info.poll_url` for status and the answer.",
|
|
13015
|
+
schema: AgentExecutionAckSchema
|
|
13016
|
+
},
|
|
13017
|
+
400: {
|
|
13018
|
+
description: "No usable default chat model is configured and none was supplied.",
|
|
13019
|
+
schema: ApiErrorSchema
|
|
13020
|
+
},
|
|
13021
|
+
404: {
|
|
13022
|
+
description: "No session with the given `session_id` is owned by the caller, or `organization_id` names an organization the caller does not belong to. \"Exists but is not yours\" is reported as 404 too, so neither session ids nor org membership can be probed through this endpoint.",
|
|
13023
|
+
schema: ApiErrorSchema
|
|
13024
|
+
},
|
|
13025
|
+
409: {
|
|
13026
|
+
description: "The caller already has the maximum number of agent runs in flight - a per-user cap shared with runs started from the product UI. Wait for one to reach a terminal status before starting another; aborting a run is currently only possible over the WebSocket route.",
|
|
13027
|
+
schema: ApiErrorSchema
|
|
13028
|
+
},
|
|
13029
|
+
429: {
|
|
13030
|
+
description: "Per-user rate limit exceeded.",
|
|
13031
|
+
schema: ApiErrorSchema
|
|
13032
|
+
},
|
|
13033
|
+
502: {
|
|
13034
|
+
description: "The executor could not be dispatched. The run did not start, so a retry is safe.",
|
|
13035
|
+
schema: ApiErrorSchema
|
|
13036
|
+
}
|
|
13037
|
+
},
|
|
13038
|
+
codeSample: {
|
|
13039
|
+
authToken: "b4m_live_<key>",
|
|
13040
|
+
streaming: false,
|
|
13041
|
+
body: {
|
|
13042
|
+
session_id: "<sessionId>",
|
|
13043
|
+
message: "Audit this data set and summarize what stands out."
|
|
13044
|
+
}
|
|
13045
|
+
}
|
|
13046
|
+
});
|
|
13047
|
+
defineEndpoint({
|
|
13048
|
+
method: "get",
|
|
13049
|
+
path: "/api/v1/agent-executions/{id}",
|
|
13050
|
+
operationId: "getAgentExecution",
|
|
13051
|
+
summary: "Get an agent execution",
|
|
13052
|
+
description: "Returns the status, reasoning trace, and (once terminal) the final answer of a run started by `POST /api/v1/agent-executions`. `steps` grows while the run is in flight, so polling this endpoint is also how a REST caller follows the loop. Safe (GET) requests on this route are exempt from the per-day API-key quota so polling a single run costs one daily slot, not one per poll; the per-minute burst limit still applies.",
|
|
13053
|
+
tags: ["AI"],
|
|
13054
|
+
auth: "apiKeyOrJwt",
|
|
13055
|
+
scopes: ["ai:chat", "ai:generate"],
|
|
13056
|
+
pathParams: AgentExecutionIdParamSchema,
|
|
13057
|
+
emitsRateLimitHeaders: true,
|
|
13058
|
+
responses: {
|
|
13059
|
+
200: {
|
|
13060
|
+
description: "The execution, its trace, and its answer if it has finished.",
|
|
13061
|
+
schema: AgentExecutionStatusResponseSchema
|
|
13062
|
+
},
|
|
13063
|
+
404: {
|
|
13064
|
+
description: "No execution with that id is visible to the caller.",
|
|
13065
|
+
schema: ApiErrorSchema
|
|
13066
|
+
},
|
|
13067
|
+
429: {
|
|
13068
|
+
description: "Per-user rate limit exceeded.",
|
|
13069
|
+
schema: ApiErrorSchema
|
|
13070
|
+
}
|
|
13071
|
+
},
|
|
13072
|
+
codeSample: {
|
|
13073
|
+
authToken: "b4m_live_<key>",
|
|
13074
|
+
streaming: false,
|
|
13075
|
+
body: {}
|
|
13076
|
+
}
|
|
13077
|
+
});
|
|
13078
|
+
defineEndpoint({
|
|
13079
|
+
method: "put",
|
|
13080
|
+
path: "/api/sessions/{id}",
|
|
13081
|
+
operationId: "updateSession",
|
|
13082
|
+
summary: "Update a session",
|
|
13083
|
+
description: "Updates a session (called a \"notebook\" in the product UI): its name, attached knowledge files, tags, or retrieval settings. Set `knowledgeIds` and `forceKnowledgeRetrieval: true` together to enable grounded retrieval for `POST /api/chat` against this session - retrieval is gated by these session fields, not by the chat request. `lakeScope` narrows that retrieval to a chosen set of data lakes; omit it to leave the current choice unchanged, or send `null` to clear it so retrieval reaches every lake you can access. Authenticate with an API key (`b4m_live_`) or a JWT. Warning: adding to `knowledgeIds` shares those files with every member of every project containing this session by default (see `propagateToProjects`), and that sharing cannot be undone through the UI.",
|
|
13084
|
+
tags: ["Sessions"],
|
|
13085
|
+
auth: "apiKeyOrJwt",
|
|
13086
|
+
scopes: ["notebooks:write"],
|
|
13087
|
+
pathParams: SessionIdParamSchema,
|
|
13088
|
+
request: SessionUpdateRequestSchema,
|
|
13089
|
+
emitsRateLimitHeaders: true,
|
|
13090
|
+
requestExample: {
|
|
13091
|
+
knowledgeIds: ["<fabFileId>"],
|
|
13092
|
+
forceKnowledgeRetrieval: true
|
|
13093
|
+
},
|
|
13094
|
+
responses: {
|
|
13095
|
+
200: {
|
|
13096
|
+
description: "The updated session.",
|
|
13097
|
+
schema: SessionResponseSchema
|
|
13098
|
+
},
|
|
13099
|
+
404: {
|
|
13100
|
+
description: "No session exists with the given id.",
|
|
13101
|
+
schema: ApiErrorSchema
|
|
13102
|
+
}
|
|
13103
|
+
},
|
|
13104
|
+
codeSample: {
|
|
13105
|
+
authToken: "b4m_live_<key>",
|
|
13106
|
+
streaming: false,
|
|
13107
|
+
body: {
|
|
13108
|
+
knowledgeIds: ["<fabFileId>"],
|
|
13109
|
+
forceKnowledgeRetrieval: true
|
|
13110
|
+
}
|
|
13111
|
+
}
|
|
13112
|
+
});
|
|
13113
|
+
defineEndpoint({
|
|
13114
|
+
method: "post",
|
|
13115
|
+
path: "/api/ai/v1/tools",
|
|
13116
|
+
operationId: "executeTool",
|
|
13117
|
+
summary: "Execute a server-side tool",
|
|
13118
|
+
description: "Runs one of the built-in server-side tools (`weather_info`, `web_search`, `web_fetch`) and returns its result as JSON. Authenticate with a JWT access token only - API keys are NOT accepted on this endpoint. Rate-limited to 100 requests/hour. `request_id` echoes the X-Request-ID response header.",
|
|
13119
|
+
tags: ["AI"],
|
|
13120
|
+
auth: "jwtOnly",
|
|
13121
|
+
request: ToolExecutionRequestSchema,
|
|
13122
|
+
requestExample: {
|
|
13123
|
+
toolName: "web_search",
|
|
13124
|
+
input: { query: "how to reset a password" }
|
|
13125
|
+
},
|
|
13126
|
+
responses: {
|
|
13127
|
+
200: {
|
|
13128
|
+
description: "Tool executed successfully (`success` is always true here).",
|
|
13129
|
+
schema: ToolExecutionResponseSchema,
|
|
13130
|
+
example: {
|
|
13131
|
+
success: true,
|
|
13132
|
+
result: { summary: "Top results for the query." },
|
|
13133
|
+
executionTimeMs: 842,
|
|
13134
|
+
request_id: "abc-123"
|
|
13135
|
+
}
|
|
13136
|
+
},
|
|
13137
|
+
400: {
|
|
13138
|
+
description: "Malformed JSON body.",
|
|
13139
|
+
schema: ApiErrorSchema
|
|
13140
|
+
},
|
|
13141
|
+
401: {
|
|
13142
|
+
description: "Missing or invalid JWT (an API key is rejected here).",
|
|
13143
|
+
schema: ApiErrorSchema
|
|
13144
|
+
},
|
|
13145
|
+
429: {
|
|
13146
|
+
description: "Rate limit exceeded (100 requests/hour).",
|
|
13147
|
+
schema: ApiErrorSchema
|
|
13148
|
+
},
|
|
13149
|
+
500: {
|
|
13150
|
+
description: "Tool execution failed (`success: false` with `error`) or an unexpected server error.",
|
|
13151
|
+
schema: z$1.union([ToolExecutionResponseSchema, ApiErrorSchema]),
|
|
13152
|
+
bespokeErrorShape: "A tool that ran but failed returns the full ToolExecutionResponse (success: false), not an error envelope."
|
|
13153
|
+
}
|
|
13154
|
+
},
|
|
13155
|
+
codeSample: {
|
|
13156
|
+
authToken: "<access_token>",
|
|
13157
|
+
streaming: false,
|
|
13158
|
+
body: {
|
|
13159
|
+
toolName: "web_search",
|
|
13160
|
+
input: { query: "how to reset a password" }
|
|
13161
|
+
}
|
|
13162
|
+
}
|
|
13163
|
+
});
|
|
13164
|
+
defineEndpoint({
|
|
13165
|
+
method: "post",
|
|
13166
|
+
path: "/api/ai/v1/completions",
|
|
13167
|
+
operationId: "createCompletion",
|
|
13168
|
+
summary: "Create a chat completion",
|
|
13169
|
+
description: "OpenAI-compatible completion. The response is ALWAYS an SSE stream (`text/event-stream`), regardless of the `stream` flag: a `meta` event, then `content`/`tool_use` events carrying `usage`/`credits`, terminated by `data: [DONE]`. Once the stream has opened the HTTP status stays 200 and failures arrive as an in-band `error` event. The terminal event carries `stopReason`; treat `max_tokens` as a TRUNCATED reply rather than a complete one, and note that omitting `max_tokens` on the request lets the server size the output ceiling for the model (recommended for reasoning models, which spend thinking tokens inside that ceiling). A message `content` may be a string or an array of parts; image parts are accepted in OpenAI Chat (`image_url`), OpenAI Responses (`input_image`) or Anthropic (`image` with a `source`) form and are translated to whatever the target model speaks. Authenticate with an API key (`b4m_live_`) or a JWT.\n\nBILLING FAILURES. Headers are flushed before authentication or pricing, so unlike the JSON surfaces this endpoint has no pre-stream `422` + `errorCode: \"insufficient_credits\"` to pair with: credit exhaustion ALWAYS arrives as the in-band `error` event, whether it is caught by the reservation before the first token or by settlement mid-generation. Branch on that event's `code` (`insufficient_credits` - buy credits; `spend_cap_exceeded` - the owner is solvent but this key hit its admin-set ceiling, so raise the cap), never on `message`, which is prose and may change. `code` is absent on unclassified failures.",
|
|
13170
|
+
tags: ["AI"],
|
|
13171
|
+
auth: "apiKeyOrJwt",
|
|
13172
|
+
scopes: ["ai:chat", "ai:generate"],
|
|
13173
|
+
request: CompletionRequestSchema,
|
|
13174
|
+
requestExample: {
|
|
13175
|
+
model: "claude-opus-4-8",
|
|
13176
|
+
messages: [{
|
|
13177
|
+
role: "user",
|
|
13178
|
+
content: "How do I reset my password?"
|
|
13179
|
+
}],
|
|
13180
|
+
max_tokens: 500
|
|
13181
|
+
},
|
|
13182
|
+
streaming: true,
|
|
13183
|
+
responses: {
|
|
13184
|
+
200: {
|
|
13185
|
+
description: "SSE stream of completion events (see the CompletionStreamEvent shape).",
|
|
13186
|
+
contentType: "text/event-stream",
|
|
13187
|
+
schema: CompletionStreamEventSchema,
|
|
13188
|
+
example: {
|
|
13189
|
+
type: "content",
|
|
13190
|
+
text: "To reset your password, click 'Forgot password' on the login screen.",
|
|
13191
|
+
usage: {
|
|
13192
|
+
inputTokens: 42,
|
|
13193
|
+
outputTokens: 12
|
|
13194
|
+
},
|
|
13195
|
+
credits: {
|
|
13196
|
+
used: 1,
|
|
13197
|
+
usdCost: 37e-5
|
|
13198
|
+
},
|
|
13199
|
+
stopReason: "end_turn"
|
|
13200
|
+
}
|
|
13201
|
+
},
|
|
13202
|
+
400: {
|
|
13203
|
+
description: "Malformed JSON body (rejected before the stream opens).",
|
|
13204
|
+
schema: ApiErrorSchema
|
|
13205
|
+
}
|
|
13206
|
+
},
|
|
13207
|
+
codeSample: {
|
|
13208
|
+
authToken: "b4m_live_<key>",
|
|
13209
|
+
streaming: true,
|
|
13210
|
+
body: {
|
|
13211
|
+
model: "claude-opus-4-8",
|
|
13212
|
+
messages: [{
|
|
13213
|
+
role: "user",
|
|
13214
|
+
content: "How do I reset my password?"
|
|
13215
|
+
}],
|
|
13216
|
+
max_tokens: 500
|
|
13217
|
+
}
|
|
13218
|
+
}
|
|
13219
|
+
});
|
|
13220
|
+
defineEndpoint({
|
|
13221
|
+
method: "post",
|
|
13222
|
+
path: "/api/ai/tts",
|
|
13223
|
+
operationId: "synthesizeSpeech",
|
|
13224
|
+
summary: "Synthesize speech from text",
|
|
13225
|
+
description: "Generates speech from text using OpenAI or ElevenLabs. The default `encoding: \"binary\"` streams raw audio bytes with an `audio/*` Content-Type; `encoding: \"base64\"` returns JSON instead. When the requested provider has no usable key (or the provider rejects it), another configured provider stands in and the substitution is reported via `provider`/`fallbackFrom` and the `X-B4M-Tts-Provider*` headers. Input length is capped per provider (OpenAI 4096 characters, ElevenLabs 10000), and an output `format` the chosen provider cannot produce is rejected with a 422 before any provider cost is incurred. Generated audio is saved to the file browser by default (opt out per-user via the saveGeneratedAudio preference, or per-call with `preview`); the outcome is reported via `saved`/`fabFileId` and the `X-B4M-Audio-*` headers. Authenticate with an API key (`b4m_live_`) or a JWT.",
|
|
13226
|
+
tags: ["Audio"],
|
|
13227
|
+
auth: "apiKeyOrJwt",
|
|
13228
|
+
scopes: ["ai:generate"],
|
|
13229
|
+
request: ttsRequestSchema,
|
|
13230
|
+
requestExample: {
|
|
13231
|
+
text: "Your password has been reset.",
|
|
13232
|
+
provider: "openai",
|
|
13233
|
+
voice: "alloy",
|
|
13234
|
+
format: "mp3"
|
|
13235
|
+
},
|
|
13236
|
+
responses: {
|
|
13237
|
+
200: {
|
|
13238
|
+
description: "Speech synthesized. The default `encoding: \"binary\"` returns raw audio bytes whose Content-Type follows the requested `format` (`audio/mpeg` for mp3, else `audio/wav`, `audio/opus`, `audio/aac`, `audio/flac`, `audio/pcm`); `encoding: \"base64\"` returns the JSON body.",
|
|
13239
|
+
schema: ttsBase64ResponseSchema,
|
|
13240
|
+
example: {
|
|
13241
|
+
audio: "SUQzBAAAAAAA...",
|
|
13242
|
+
format: "mp3",
|
|
13243
|
+
contentType: "audio/mpeg",
|
|
13244
|
+
saved: true,
|
|
13245
|
+
fabFileId: "664f1c2b9a1e4d0012ab34cd"
|
|
13246
|
+
},
|
|
13247
|
+
alsoReturns: [
|
|
13248
|
+
{ contentType: "audio/mpeg" },
|
|
13249
|
+
{ contentType: "audio/wav" },
|
|
13250
|
+
{ contentType: "audio/opus" },
|
|
13251
|
+
{ contentType: "audio/aac" },
|
|
13252
|
+
{ contentType: "audio/flac" },
|
|
13253
|
+
{ contentType: "audio/pcm" }
|
|
13254
|
+
],
|
|
13255
|
+
headers: {
|
|
13256
|
+
"X-B4M-Tts-Provider": "The provider that produced the audio. Present only when a fallback happened.",
|
|
13257
|
+
"X-B4M-Tts-Provider-Fallback-From": "The originally requested provider that could not serve the request. Present only on a fallback.",
|
|
13258
|
+
"X-B4M-Audio-Saved": "Whether a browsable copy was saved to the file browser (\"true\"/\"false\").",
|
|
13259
|
+
"X-B4M-Audio-Fab-File-Id": "Id of the saved file. Present only when the copy was saved."
|
|
13260
|
+
}
|
|
13261
|
+
},
|
|
13262
|
+
401: {
|
|
13263
|
+
description: "Missing/invalid credentials, or no provider has a usable key (`provider_not_configured`).",
|
|
13264
|
+
schema: ttsErrorResponseSchema
|
|
13265
|
+
},
|
|
13266
|
+
413: {
|
|
13267
|
+
description: "The audio was generated (and billed) but is too large to return over this endpoint. Retrieve it from `fileUrl` when a browsable copy was saved.",
|
|
13268
|
+
schema: ttsResponseTooLargeSchema
|
|
13269
|
+
},
|
|
13270
|
+
422: {
|
|
13271
|
+
description: "Request body failed validation, the text exceeds the provider character limit, the provider cannot produce the requested `format`, or the caller cannot afford the synthesis - the last of those is the only one tagged `errorCode: \"insufficient_credits\"`, so match on the classifier rather than the status to tell a billing failure from a bad request.",
|
|
13272
|
+
schema: ttsErrorResponseSchema
|
|
13273
|
+
},
|
|
13274
|
+
429: {
|
|
13275
|
+
description: "The provider rate-limited the request.",
|
|
13276
|
+
schema: ttsErrorResponseSchema
|
|
13277
|
+
},
|
|
13278
|
+
502: {
|
|
13279
|
+
description: "The provider failed to generate speech.",
|
|
13280
|
+
schema: ttsErrorResponseSchema
|
|
13281
|
+
}
|
|
13282
|
+
},
|
|
13283
|
+
emitsRateLimitHeaders: true,
|
|
13284
|
+
codeSample: {
|
|
13285
|
+
authToken: "b4m_live_<key>",
|
|
13286
|
+
streaming: false,
|
|
13287
|
+
body: {
|
|
13288
|
+
text: "Your password has been reset.",
|
|
13289
|
+
provider: "openai",
|
|
13290
|
+
voice: "alloy",
|
|
13291
|
+
encoding: "base64"
|
|
13292
|
+
}
|
|
13293
|
+
}
|
|
13294
|
+
});
|
|
13493
13295
|
/**
|
|
13494
13296
|
* Response details shared by the endpoints that return generated audio as raw
|
|
13495
13297
|
* bytes (music, sound effects). Not a contract - just the pieces both of their
|
|
@@ -13526,8 +13328,179 @@ const generatedAudioBody = () => ({
|
|
|
13526
13328
|
alsoReturns: GENERATED_AUDIO_CONTENT_TYPES.slice(1).map((contentType) => ({ contentType })),
|
|
13527
13329
|
headers: GENERATED_AUDIO_SAVE_HEADERS
|
|
13528
13330
|
});
|
|
13529
|
-
({
|
|
13530
|
-
|
|
13331
|
+
defineEndpoint({
|
|
13332
|
+
method: "post",
|
|
13333
|
+
path: "/api/ai/music",
|
|
13334
|
+
operationId: "generateMusic",
|
|
13335
|
+
summary: "Generate background music",
|
|
13336
|
+
description: "Generates an instrumental or vocal background-music track from a text prompt and returns the raw audio bytes. `lengthMs` (3000-120000, default 10000) is forced on the provider, so the generated track always matches the billed length; credits are reserved before generation and refunded if it fails. Generated audio is saved to the file browser by default (opt out via the saveGeneratedAudio preference); the outcome is reported via the `X-B4M-Audio-Saved` / `X-B4M-Audio-Fab-File-Id` / `X-B4M-Audio-File-Url` response headers - use `X-B4M-Audio-File-Url` to fetch the saved copy, since `GET /api/files/{id}` fails closed until moderation completes. Authenticate with an API key (`b4m_live_`) or a JWT.",
|
|
13337
|
+
tags: ["Audio"],
|
|
13338
|
+
auth: "apiKeyOrJwt",
|
|
13339
|
+
scopes: ["ai:generate"],
|
|
13340
|
+
request: musicRequestSchema,
|
|
13341
|
+
requestExample: {
|
|
13342
|
+
prompt: "calm lo-fi study beat with soft piano",
|
|
13343
|
+
lengthMs: 3e4,
|
|
13344
|
+
forceInstrumental: true
|
|
13345
|
+
},
|
|
13346
|
+
responses: {
|
|
13347
|
+
200: {
|
|
13348
|
+
description: "Raw audio bytes; the Content-Type follows the requested `format` (mp3 by default).",
|
|
13349
|
+
...generatedAudioBody()
|
|
13350
|
+
},
|
|
13351
|
+
400: {
|
|
13352
|
+
description: "The billing user or organization could not be resolved.",
|
|
13353
|
+
schema: ApiErrorSchema
|
|
13354
|
+
},
|
|
13355
|
+
422: {
|
|
13356
|
+
description: "Request body failed validation, or the caller cannot afford the track - the latter is tagged `errorCode: \"insufficient_credits\"` (the balance is short, or the org member credit cap is exhausted).",
|
|
13357
|
+
schema: InsufficientCreditsErrorSchema
|
|
13358
|
+
},
|
|
13359
|
+
502: {
|
|
13360
|
+
description: "The provider failed to generate the track; reserved credits are refunded.",
|
|
13361
|
+
schema: ApiErrorSchema
|
|
13362
|
+
},
|
|
13363
|
+
503: {
|
|
13364
|
+
description: "No provider API key is configured for this deployment.",
|
|
13365
|
+
schema: ApiErrorSchema
|
|
13366
|
+
}
|
|
13367
|
+
},
|
|
13368
|
+
emitsRateLimitHeaders: true,
|
|
13369
|
+
codeSample: {
|
|
13370
|
+
authToken: "b4m_live_<key>",
|
|
13371
|
+
streaming: false,
|
|
13372
|
+
body: {
|
|
13373
|
+
prompt: "calm lo-fi study beat with soft piano",
|
|
13374
|
+
lengthMs: 3e4,
|
|
13375
|
+
forceInstrumental: true
|
|
13376
|
+
}
|
|
13377
|
+
}
|
|
13378
|
+
});
|
|
13379
|
+
defineEndpoint({
|
|
13380
|
+
method: "post",
|
|
13381
|
+
path: "/api/ai/sound-effects",
|
|
13382
|
+
operationId: "generateSoundEffect",
|
|
13383
|
+
summary: "Generate a sound effect",
|
|
13384
|
+
description: "Generates a short sound effect from a text description and returns the raw audio bytes. Omitting `durationSeconds` lets the provider pick the length (and bills at its default); `promptInfluence` trades prompt fidelity (1) against variation (0). Credits are reserved before generation and refunded if it fails. Generated audio is saved to the file browser by default (opt out via the saveGeneratedAudio preference); the outcome is reported via the `X-B4M-Audio-Saved` / `X-B4M-Audio-Fab-File-Id` / `X-B4M-Audio-File-Url` response headers - use `X-B4M-Audio-File-Url` to fetch the saved copy, since `GET /api/files/{id}` fails closed until moderation completes. Authenticate with an API key (`b4m_live_`) or a JWT.",
|
|
13385
|
+
tags: ["Audio"],
|
|
13386
|
+
auth: "apiKeyOrJwt",
|
|
13387
|
+
scopes: ["ai:generate"],
|
|
13388
|
+
request: soundEffectsRequestSchema,
|
|
13389
|
+
requestExample: {
|
|
13390
|
+
text: "heavy wooden door creaking open",
|
|
13391
|
+
durationSeconds: 3,
|
|
13392
|
+
promptInfluence: .5
|
|
13393
|
+
},
|
|
13394
|
+
responses: {
|
|
13395
|
+
200: {
|
|
13396
|
+
description: "Raw audio bytes; the Content-Type follows the requested `format` (mp3 by default).",
|
|
13397
|
+
...generatedAudioBody()
|
|
13398
|
+
},
|
|
13399
|
+
400: {
|
|
13400
|
+
description: "The billing user or organization could not be resolved.",
|
|
13401
|
+
schema: ApiErrorSchema
|
|
13402
|
+
},
|
|
13403
|
+
422: {
|
|
13404
|
+
description: "Request body failed validation, or the caller cannot afford the effect - the latter is tagged `errorCode: \"insufficient_credits\"` (the balance is short, or the org member credit cap is exhausted).",
|
|
13405
|
+
schema: InsufficientCreditsErrorSchema
|
|
13406
|
+
},
|
|
13407
|
+
502: {
|
|
13408
|
+
description: "The provider failed to generate the effect; reserved credits are refunded.",
|
|
13409
|
+
schema: ApiErrorSchema
|
|
13410
|
+
},
|
|
13411
|
+
503: {
|
|
13412
|
+
description: "No provider API key is configured for this deployment.",
|
|
13413
|
+
schema: ApiErrorSchema
|
|
13414
|
+
}
|
|
13415
|
+
},
|
|
13416
|
+
emitsRateLimitHeaders: true,
|
|
13417
|
+
codeSample: {
|
|
13418
|
+
authToken: "b4m_live_<key>",
|
|
13419
|
+
streaming: false,
|
|
13420
|
+
body: {
|
|
13421
|
+
text: "heavy wooden door creaking open",
|
|
13422
|
+
durationSeconds: 3,
|
|
13423
|
+
promptInfluence: .5
|
|
13424
|
+
}
|
|
13425
|
+
}
|
|
13426
|
+
});
|
|
13427
|
+
defineEndpoint({
|
|
13428
|
+
method: "get",
|
|
13429
|
+
path: "/api/v1/me",
|
|
13430
|
+
operationId: "getMe",
|
|
13431
|
+
summary: "Get the authenticated caller",
|
|
13432
|
+
description: "Returns the authenticated caller: stable id, display name, plan tier, personal credit balance, and entitlement keys. The subject is always the credential holder - this endpoint accepts no user id, owner id, or impersonation parameter of any kind, so a key can only ever read its own owner. `credits.balance` is the caller's personal ledger; a call billed to an organization draws on a pool this number does not describe. Gate on `tier != \"free\"` for \"is this caller paying\" and on `subscription.price_id` for which product - the `basic`/`pro` rungs come from an internal plan ladder and do not track a plan's marketing name. Responses are never cacheable. Authenticate with an API key (`b4m_live_`) carrying `me:read`, or a JWT.",
|
|
13433
|
+
tags: ["Account"],
|
|
13434
|
+
auth: "apiKeyOrJwt",
|
|
13435
|
+
scopes: ["me:read"],
|
|
13436
|
+
emitsRateLimitHeaders: true,
|
|
13437
|
+
responses: {
|
|
13438
|
+
200: {
|
|
13439
|
+
description: "The caller, their tier, their personal credit balance, and their entitlement keys.",
|
|
13440
|
+
schema: MeResponseSchema,
|
|
13441
|
+
example: {
|
|
13442
|
+
id: "507f1f77bcf86cd799439011",
|
|
13443
|
+
name: "Ada Lovelace",
|
|
13444
|
+
tier: "basic",
|
|
13445
|
+
subscription: {
|
|
13446
|
+
plan_name: "Professional",
|
|
13447
|
+
price_id: "price_123",
|
|
13448
|
+
interval: "monthly",
|
|
13449
|
+
current_period_ends_at: "2026-10-18T00:00:00.000Z"
|
|
13450
|
+
},
|
|
13451
|
+
credits: { balance: 31667 },
|
|
13452
|
+
entitlements: ["base"]
|
|
13453
|
+
}
|
|
13454
|
+
},
|
|
13455
|
+
429: {
|
|
13456
|
+
description: "Per-user rate limit exceeded.",
|
|
13457
|
+
schema: ApiErrorSchema
|
|
13458
|
+
}
|
|
13459
|
+
},
|
|
13460
|
+
codeSample: {
|
|
13461
|
+
authToken: "b4m_live_<key>",
|
|
13462
|
+
streaming: false,
|
|
13463
|
+
body: {}
|
|
13464
|
+
}
|
|
13465
|
+
});
|
|
13466
|
+
/**
|
|
13467
|
+
* The fence language the model writes to place an inline map of search-result places in a reply.
|
|
13468
|
+
*
|
|
13469
|
+
* Same contract as SEARCH_RESULT_CARDS_LANGUAGE (searchResultCards.ts): `\w`-only and lowercase,
|
|
13470
|
+
* taught by WEB_SEARCH_MAP_PROMPT (@bike4mind/services), rendered by the reply renderer
|
|
13471
|
+
* (apps/client .../Session/PromptReplies.tsx), and rewritten to a plain list for every other
|
|
13472
|
+
* surface by `stripSearchResultCardFences`.
|
|
13473
|
+
*/
|
|
13474
|
+
const LOCATION_MAP_LANGUAGE = "b4m_map";
|
|
13475
|
+
function googleMapsSearchUrl(name, placeId) {
|
|
13476
|
+
const params = new URLSearchParams({
|
|
13477
|
+
api: "1",
|
|
13478
|
+
query: name
|
|
13479
|
+
});
|
|
13480
|
+
if (placeId?.startsWith("ChIJ")) params.set("query_place_id", placeId);
|
|
13481
|
+
return `https://www.google.com/maps/search/?${params.toString()}`;
|
|
13482
|
+
}
|
|
13483
|
+
/**
|
|
13484
|
+
* The fence language the model writes to place image cards inline in a reply.
|
|
13485
|
+
*
|
|
13486
|
+
* Two families of consumer must agree on this constant:
|
|
13487
|
+
* - WEB_SEARCH_CARDS_PROMPT (@bike4mind/services) teaches the model to emit it, and the reply
|
|
13488
|
+
* renderer (apps/client .../Session/PromptReplies.tsx) intercepts it to render cards instead
|
|
13489
|
+
* of raw text.
|
|
13490
|
+
* - Every other surface that reads reply markdown for an export, download, copy, publish, or
|
|
13491
|
+
* Slack delivery must strip the fenced block entirely with `stripSearchResultCardFences`
|
|
13492
|
+
* below, rather than passing the model-authored card JSON through verbatim. See that
|
|
13493
|
+
* function's own doc comment for the current list of call sites.
|
|
13494
|
+
*
|
|
13495
|
+
* MUST contain only `\w` characters: both the renderer (`/language-(\w+)/`) and the notebook
|
|
13496
|
+
* curation extractor (```` /```(\w+)?/ ````) capture the language with `\w+`, so a hyphen would
|
|
13497
|
+
* silently truncate this and the cards would never render or be skipped.
|
|
13498
|
+
*
|
|
13499
|
+
* MUST already be lowercase: some consumers lowercase the language they capture before
|
|
13500
|
+
* comparing, so an uppercase-containing value would pass the `\w`-only rule above while silently
|
|
13501
|
+
* breaking that comparison.
|
|
13502
|
+
*/
|
|
13503
|
+
const SEARCH_RESULT_CARDS_LANGUAGE = "b4m_cards";
|
|
13531
13504
|
z$1.enum(["user", "convergence"]).optional().catch(void 0), z$1.string().optional();
|
|
13532
13505
|
/**
|
|
13533
13506
|
* Blessed, self-hosted artifact library script paths (root-relative).
|
|
@@ -13589,50 +13562,6 @@ const OPTIONAL_DEP_BLESSED_SCRIPT_PATHS = Object.values({
|
|
|
13589
13562
|
}
|
|
13590
13563
|
}).map((d) => d.path);
|
|
13591
13564
|
[...REACT_BLESSED_SCRIPT_PATHS, ...OPTIONAL_DEP_BLESSED_SCRIPT_PATHS];
|
|
13592
|
-
function isGPTImageModel(model) {
|
|
13593
|
-
if (!model) return false;
|
|
13594
|
-
return OPENAI_IMAGE_MODELS.includes(model) || model.startsWith("gpt-image-");
|
|
13595
|
-
}
|
|
13596
|
-
/**
|
|
13597
|
-
* Flux Ultra is the only BFL generation model driven by `aspect_ratio`; the Pro family takes
|
|
13598
|
-
* discrete `width`/`height` and ignores aspect_ratio outright. Must stay in sync with the branch
|
|
13599
|
-
* in `BFLImageService.generate`, which is what actually builds the request body.
|
|
13600
|
-
*/
|
|
13601
|
-
function isBflUltraImageModel(model) {
|
|
13602
|
-
return model === "flux-pro-1.1-ultra";
|
|
13603
|
-
}
|
|
13604
|
-
/** Returns true specifically for gpt-image-2 (including versioned snapshots like gpt-image-2-2026-04-21). */
|
|
13605
|
-
function isGPTImage2Model(model) {
|
|
13606
|
-
if (!model) return false;
|
|
13607
|
-
return model === "gpt-image-2" || model.startsWith("gpt-image-2");
|
|
13608
|
-
}
|
|
13609
|
-
/**
|
|
13610
|
-
* Whether a user can access a model. Access is any-of (mirrors the Q3b data-lake
|
|
13611
|
-
* rule, `getAccessibleDataLakes`): a non-admin reaches the model via
|
|
13612
|
-
* `allowedUserTags ∩ userTags` OR `allowedEntitlements ∩ entitlementKeys`.
|
|
13613
|
-
*
|
|
13614
|
-
* `entitlementKeys` is optional - when omitted/empty the entitlement branch is
|
|
13615
|
-
* inert, so a model with no `allowedEntitlements` behaves exactly as before
|
|
13616
|
-
* (tag-only). This lets a tag-less subscriber reach an entitlement-gated model
|
|
13617
|
-
* while leaving every existing tag-gated model unchanged (zero regression).
|
|
13618
|
-
*
|
|
13619
|
-
* Pure + zero-dependency (only types + the shared key normalizer), so it lives
|
|
13620
|
-
* in `@bike4mind/common` as the SINGLE source of truth - imported by the core
|
|
13621
|
-
* `@bike4mind/utils` re-export (server/services) AND the client
|
|
13622
|
-
* `useAccessibleModels` hook, which previously kept a hand-rolled twin "to avoid
|
|
13623
|
-
* AWS SDK imports". Common is browser-safe, so there is no longer any reason to
|
|
13624
|
-
* duplicate the logic.
|
|
13625
|
-
*/
|
|
13626
|
-
function isModelAccessible(model, userTags, isAdmin = false, entitlementKeys = []) {
|
|
13627
|
-
if (!model.enabled) return false;
|
|
13628
|
-
if (isAdmin) return true;
|
|
13629
|
-
const normalizedUserTags = userTags.map((tag) => tag.toLowerCase());
|
|
13630
|
-
const normalizedAllowedTags = (model.allowedUserTags ?? []).map((tag) => tag.toLowerCase());
|
|
13631
|
-
if (normalizedUserTags.some((tag) => normalizedAllowedTags.includes(tag))) return true;
|
|
13632
|
-
const normalizedKeys = entitlementKeys.map(normalizeEntitlementKey);
|
|
13633
|
-
const normalizedAllowedEntitlements = (model.allowedEntitlements ?? []).map(normalizeEntitlementKey);
|
|
13634
|
-
return normalizedKeys.some((key) => normalizedAllowedEntitlements.includes(key));
|
|
13635
|
-
}
|
|
13636
13565
|
[...OPENAI_IMAGE_MODELS, ...GEMINI_IMAGE_MODELS];
|
|
13637
13566
|
const DashboardParamsSchema = z$1.object({
|
|
13638
13567
|
dashboardDataSources: z$1.array(z$1.object({
|
|
@@ -13759,13 +13688,6 @@ OpenAIImageGenerationInput.extend({
|
|
|
13759
13688
|
image: z$1.string(),
|
|
13760
13689
|
output_format: ImageOutputFormatSchema.nullable().optional()
|
|
13761
13690
|
});
|
|
13762
|
-
const isUnlimitedHistory = (historyCount) => historyCount === -1;
|
|
13763
|
-
/**
|
|
13764
|
-
* A usable number for callers that need one (page size, overflow math, telemetry). Unlimited
|
|
13765
|
-
* history has no count, so it resolves to the default page size, which is what an unwindowed
|
|
13766
|
-
* request has always actually fetched.
|
|
13767
|
-
*/
|
|
13768
|
-
const resolveHistoryFetchLimit = (historyCount) => isUnlimitedHistory(historyCount) || historyCount == null ? 14 : historyCount;
|
|
13769
13691
|
z$1.object({
|
|
13770
13692
|
/** Notebook session ID */
|
|
13771
13693
|
sessionId: z$1.string(),
|
|
@@ -14287,6 +14209,18 @@ const ArtifactVersionMetaSchema = z$1.object({
|
|
|
14287
14209
|
}),
|
|
14288
14210
|
sha256Index: z$1.string()
|
|
14289
14211
|
});
|
|
14212
|
+
/** One no-sign-in share link. `token` is the capability itself and is stripped from every
|
|
14213
|
+
* serialized response, so it is optional here: an owner-facing read carries the metadata
|
|
14214
|
+
* (id to revoke by, timestamps, per-link view count) with no token at all. `id` is the
|
|
14215
|
+
* subdocument `_id`, rendered as a string. */
|
|
14216
|
+
const ShareTokenEntrySchema = z$1.object({
|
|
14217
|
+
id: z$1.string().optional(),
|
|
14218
|
+
token: z$1.string().optional(),
|
|
14219
|
+
createdAt: z$1.date().optional(),
|
|
14220
|
+
revokedAt: z$1.date().nullish(),
|
|
14221
|
+
viewCount: z$1.int().nonnegative().prefault(0),
|
|
14222
|
+
lastViewedAt: z$1.date().nullish()
|
|
14223
|
+
});
|
|
14290
14224
|
/** Reserved slugs - must include every tier URL token so a slug can't shadow routing. */
|
|
14291
14225
|
const RESERVED_SLUGS = [
|
|
14292
14226
|
"api",
|
|
@@ -14410,6 +14344,10 @@ z$1.object({
|
|
|
14410
14344
|
shareToken: z$1.string().optional(),
|
|
14411
14345
|
/** When `shareToken` was last minted/rotated; drives the owner-facing "link created" surface. */
|
|
14412
14346
|
shareTokenUpdatedAt: z$1.date().nullish(),
|
|
14347
|
+
/** Every share link ever minted, revoked ones included. Mirrors the two fields above during
|
|
14348
|
+
* the rollout and becomes the source of truth once the backfill has run everywhere; `token`
|
|
14349
|
+
* is stripped from serialized responses, so an owner-facing read sees only the metadata. */
|
|
14350
|
+
shareTokens: z$1.array(ShareTokenEntrySchema).prefault([]),
|
|
14413
14351
|
/** Collaboration gate: who (among viewers) may annotate. Orthogonal to
|
|
14414
14352
|
* `visibility`. Defaults to `none` (read-only) until the owner opts in. */
|
|
14415
14353
|
commentPolicy: CommentPolicySchema.prefault("none"),
|
|
@@ -14440,6 +14378,9 @@ z$1.object({
|
|
|
14440
14378
|
declaredApiEndpoints: z$1.array(z$1.string()).prefault([]),
|
|
14441
14379
|
/** Rendered body snapshot for reply/fabfile viewer pages (markdown or text). */
|
|
14442
14380
|
renderedBody: z$1.string().optional(),
|
|
14381
|
+
/** Snapshot of the source reply's citables (reply source only), so a `b4m_map` fence in
|
|
14382
|
+
* `renderedBody` can still resolve its place ids after the source Quest is edited or deleted. */
|
|
14383
|
+
citables: z$1.array(CitableSourceSchema).optional(),
|
|
14443
14384
|
publishedAt: z$1.date(),
|
|
14444
14385
|
previousVersionMeta: ArtifactVersionMetaSchema.optional(),
|
|
14445
14386
|
/** Full version history (oldest to newest); each entry's bytes are archived at
|
|
@@ -14693,170 +14634,70 @@ z$1.object({
|
|
|
14693
14634
|
createdAt: z$1.union([z$1.string(), z$1.date()]).optional(),
|
|
14694
14635
|
updatedAt: z$1.union([z$1.string(), z$1.date()]).optional()
|
|
14695
14636
|
});
|
|
14637
|
+
ClaudeArtifactMimeTypes.RECHARTS, ClaudeArtifactMimeTypes.MERMAID, ClaudeArtifactMimeTypes.LATTICE, ClaudeArtifactMimeTypes.BLOG_DRAFT, ClaudeArtifactMimeTypes.CHESS;
|
|
14638
|
+
[...IMAGE_SIZE_CONSTRAINTS.DALL_E_2.sizes, ...IMAGE_SIZE_CONSTRAINTS.DALL_E_3.sizes];
|
|
14696
14639
|
/**
|
|
14697
|
-
*
|
|
14698
|
-
*
|
|
14699
|
-
* (packages/scripts/datalake/backfill-chunk-char-length.ts) compute the same number server-side
|
|
14700
|
-
* without reading chunk text out of the database. Deliberately NOT `text.length` (UTF-16 code
|
|
14701
|
-
* units): the two differ on astral characters (surrogate pairs), and the write path and the
|
|
14702
|
-
* backfill must agree exactly.
|
|
14703
|
-
*/
|
|
14704
|
-
const countCodePoints = (text) => {
|
|
14705
|
-
let count = 0;
|
|
14706
|
-
for (const _ch of text) count++;
|
|
14707
|
-
return count;
|
|
14708
|
-
};
|
|
14709
|
-
/**
|
|
14710
|
-
* Default character budget for a super-linear parse pass. A starting point only -
|
|
14711
|
-
* see the sizing rule below; any call site in front of a measured parser should pass
|
|
14712
|
-
* an explicit `max` derived from that parser's own curve.
|
|
14640
|
+
* Zero-width space spliced into a defanged marker. It is invisible wherever the text is
|
|
14641
|
+
* rendered, so escaping never changes what the reader sees - only what the parser matches.
|
|
14713
14642
|
*/
|
|
14714
|
-
const
|
|
14643
|
+
const ZERO_WIDTH_SPACE = "";
|
|
14715
14644
|
/**
|
|
14716
|
-
*
|
|
14645
|
+
* Defangs think-marker-shaped substrings inside provider-authored reasoning text so they can
|
|
14646
|
+
* never be mistaken for the real control markers adapters wrap around that same text.
|
|
14717
14647
|
*
|
|
14718
|
-
* A
|
|
14719
|
-
*
|
|
14720
|
-
*
|
|
14721
|
-
*
|
|
14722
|
-
*
|
|
14723
|
-
*
|
|
14724
|
-
|
|
14725
|
-
|
|
14726
|
-
|
|
14727
|
-
|
|
14728
|
-
* cleaners in services/lib/turndown.ts and the snippet-meta extractor in common/utils.ts
|
|
14729
|
-
* both started here and both ended up as single-pass scans instead, because a cap that
|
|
14730
|
-
* was small enough to bound their cost was also small enough to stop them working on
|
|
14731
|
-
* real content. When the parser is linear, the cap can go.
|
|
14732
|
-
*
|
|
14733
|
-
* SIZING RULE - if you do cap, pick `max` from the parser's measured cost AT the cap,
|
|
14734
|
-
* not from the headroom above real input. For an O(n^2)/O(n^3) inner loop, "far above
|
|
14735
|
-
* anything legitimate" is not a bound: the email cleaners were cubic, so a 512k cap
|
|
14736
|
-
* still ran for minutes. Measure the worst-case shape at the cap you intend to ship and
|
|
14737
|
-
* make sure the number of milliseconds is one you can defend.
|
|
14738
|
-
*
|
|
14739
|
-
* Four further traps this helper cannot solve for you:
|
|
14740
|
-
* - A per-item cap is not a total bound when the item count is also attacker-chosen
|
|
14741
|
-
* (a .pptx picks its own slide count). Carry a running budget across the loop.
|
|
14742
|
-
* - Cap what is SCANNED, not what is returned. Where the result is rendered or
|
|
14743
|
-
* stored, re-append `input.slice(max)` unchanged: truncating the value silently
|
|
14744
|
-
* drops content, and a cut inside an element can strand its closing tag. Splicing
|
|
14745
|
-
* the original halves back together also avoids leaving a lone surrogate, since
|
|
14746
|
-
* `slice` cuts by UTF-16 code unit.
|
|
14747
|
-
* - Splicing the tail back is not enough when the cap changes the SHAPE of the parse
|
|
14748
|
-
* rather than only its length - a pattern terminated by `$`, say, whose sections then
|
|
14749
|
-
* end early and get classified differently. Check what an over-cap input parses INTO,
|
|
14750
|
-
* not just that its characters survive.
|
|
14751
|
-
* - Cap in front of a best-effort cleaner or detector, where degrading quietly is
|
|
14752
|
-
* acceptable. Do not cap in front of a parser whose output drives execution (a
|
|
14753
|
-
* tool-call parser): silently parsing a prefix there means running a subset of what
|
|
14754
|
-
* was asked. If such a parser caps internally, its callers must pass it text already
|
|
14755
|
-
* scoped to the construct being parsed.
|
|
14756
|
-
*
|
|
14757
|
-
* @returns the input unchanged when within budget, otherwise its first `max` chars.
|
|
14758
|
-
*/
|
|
14759
|
-
function capForParse(input, max = DEFAULT_PARSE_CAP) {
|
|
14760
|
-
if (!Number.isFinite(max) || max < 0) throw new RangeError(`capForParse: max must be a non-negative finite number, got ${max}`);
|
|
14761
|
-
return input.length <= max ? input : input.slice(0, max);
|
|
14762
|
-
}
|
|
14763
|
-
/**
|
|
14764
|
-
* Exactly 'WIDTHxHEIGHT' - nothing else, no surrounding whitespace and no third segment.
|
|
14765
|
-
* Anchored on purpose: this module is the single source of the size rule, so a value that
|
|
14766
|
-
* only looks like a resolution ('1024x1024x1024') must not be measured as one.
|
|
14767
|
-
*/
|
|
14768
|
-
const SIZE_PATTERN = /^(\d+)x(\d+)$/;
|
|
14769
|
-
/**
|
|
14770
|
-
* Splits a 'WIDTHxHEIGHT' string into numeric edges, or null when it is not a pair
|
|
14771
|
-
* of positive numbers. Lets callers tell a resolution that breaks a rule apart from
|
|
14772
|
-
* a value that expresses no resolution at all ('auto', 'wide', undefined).
|
|
14773
|
-
*/
|
|
14774
|
-
function parseSizeEdges(size) {
|
|
14775
|
-
if (typeof size !== "string") return null;
|
|
14776
|
-
const match = SIZE_PATTERN.exec(size);
|
|
14777
|
-
if (!match) return null;
|
|
14778
|
-
const width = Number(match[1]);
|
|
14779
|
-
const height = Number(match[2]);
|
|
14780
|
-
if (!width || !height) return null;
|
|
14781
|
-
return {
|
|
14782
|
-
width,
|
|
14783
|
-
height
|
|
14784
|
-
};
|
|
14785
|
-
}
|
|
14786
|
-
/**
|
|
14787
|
-
* True when a custom gpt-image-2 resolution meets OpenAI's documented limits.
|
|
14788
|
-
* gpt-image-2 accepts any resolution satisfying these, not only the presets in
|
|
14789
|
-
* IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.sizes, so a flat preset check would reject
|
|
14790
|
-
* valid custom sizes.
|
|
14791
|
-
*/
|
|
14792
|
-
function satisfiesGptImage2Constraints({ width, height }) {
|
|
14793
|
-
const { maxEdge, minTotalPixels, maxTotalPixels, edgeMultiple, maxAspectRatio } = IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.constraints;
|
|
14794
|
-
const longEdge = Math.max(width, height);
|
|
14795
|
-
const shortEdge = Math.min(width, height);
|
|
14796
|
-
const totalPixels = width * height;
|
|
14797
|
-
return longEdge <= maxEdge && width % edgeMultiple === 0 && height % edgeMultiple === 0 && longEdge / shortEdge <= maxAspectRatio && totalPixels >= minTotalPixels && totalPixels <= maxTotalPixels;
|
|
14648
|
+
* A reasoning delta is provider output, not our control plane - a model can say `<think>` or
|
|
14649
|
+
* `</think>` as literal content (reasoning about the protocol itself, or a leading/trailing
|
|
14650
|
+
* `</think>` from a provider that already delimits its own monologue). Adapters wrap the whole
|
|
14651
|
+
* delta in real markers via plain string concatenation, so an unescaped literal is
|
|
14652
|
+
* indistinguishable from a genuine open/close once it lands in the same string. Call this on
|
|
14653
|
+
* every raw reasoning delta before it is concatenated with THINK_OPEN_TAG/THINK_CLOSE_TAG.
|
|
14654
|
+
*/
|
|
14655
|
+
function escapeThinkMarkers(text) {
|
|
14656
|
+
if (!text) return text;
|
|
14657
|
+
return text.replace(/<(\/?)think>/g, `<${ZERO_WIDTH_SPACE}$1think>`);
|
|
14798
14658
|
}
|
|
14799
|
-
/**
|
|
14800
|
-
|
|
14801
|
-
|
|
14802
|
-
|
|
14803
|
-
|
|
14804
|
-
|
|
14805
|
-
|
|
14806
|
-
|
|
14807
|
-
* OpenAI image size rule - generate, edit, the variation endpoint, the cost estimate and
|
|
14808
|
-
* the settings UI all resolve through it, so a tier changing its accepted sizes is a
|
|
14809
|
-
* one-line edit to IMAGE_SIZE_CONSTRAINTS rather than a hunt through retyped lists.
|
|
14810
|
-
*
|
|
14811
|
-
* gpt-image-2 takes 'auto', its presets, or any custom WIDTHxHEIGHT meeting
|
|
14812
|
-
* satisfiesGptImage2Constraints. The gpt-image-1 family is limited to its three fixed
|
|
14813
|
-
* sizes. Anything else is treated as legacy dall-e.
|
|
14814
|
-
*
|
|
14815
|
-
* OpenAI image models only: BFL, Gemini and xAI sizes are validated by their own adapters,
|
|
14816
|
-
* and passing one of those models here would measure it against the wrong list.
|
|
14817
|
-
*/
|
|
14818
|
-
function isSupportedImageSize(model, size) {
|
|
14819
|
-
if (typeof size !== "string") return false;
|
|
14820
|
-
if (isGPTImage2Model(model)) {
|
|
14821
|
-
if (size === IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.autoSize || IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.sizes.includes(size)) return true;
|
|
14822
|
-
const edges = parseSizeEdges(size);
|
|
14823
|
-
return edges !== null && satisfiesGptImage2Constraints(edges);
|
|
14659
|
+
/** One less than the longer marker's length: the most characters a real marker prefix can span. */
|
|
14660
|
+
const MAX_PARTIAL_MARKER_LENGTH = 7;
|
|
14661
|
+
/** Length of the longest suffix of `text` that is a proper prefix of either marker token. */
|
|
14662
|
+
function partialMarkerSuffixLength(text) {
|
|
14663
|
+
const max = Math.min(MAX_PARTIAL_MARKER_LENGTH, text.length);
|
|
14664
|
+
for (let len = max; len > 0; len--) {
|
|
14665
|
+
const suffix = text.slice(-len);
|
|
14666
|
+
if ("<think>".startsWith(suffix) || "</think>".startsWith(suffix)) return len;
|
|
14824
14667
|
}
|
|
14825
|
-
|
|
14826
|
-
return OPENAI_LEGACY_IMAGE_SIZES.includes(size);
|
|
14668
|
+
return 0;
|
|
14827
14669
|
}
|
|
14828
14670
|
/**
|
|
14829
|
-
*
|
|
14830
|
-
* defaults to 1024x1024 today; the indirection keeps that a per-tier decision.
|
|
14831
|
-
*/
|
|
14832
|
-
function fallbackImageSize(model) {
|
|
14833
|
-
if (isGPTImage2Model(model)) return IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.defaultSize;
|
|
14834
|
-
if (isGPTImageModel(model)) return IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_1.defaultSize;
|
|
14835
|
-
return IMAGE_SIZE_CONSTRAINTS.DALL_E_2.defaultSize;
|
|
14836
|
-
}
|
|
14837
|
-
/**
|
|
14838
|
-
* The size the generate endpoint should send for a GPT-Image `model`, given what the
|
|
14839
|
-
* caller asked for. Returns the input unchanged when nothing needs correcting, so a
|
|
14840
|
-
* caller can compare the two to decide whether to warn.
|
|
14841
|
-
*
|
|
14842
|
-
* Absent: gpt-image-2 gets 'auto' - it is the only tier that can pick its own - and the
|
|
14843
|
-
* gpt-image-1 family gets its fixed default.
|
|
14671
|
+
* Stateful counterpart to escapeThinkMarkers for text that arrives in streamed pieces.
|
|
14844
14672
|
*
|
|
14845
|
-
*
|
|
14846
|
-
*
|
|
14847
|
-
*
|
|
14848
|
-
*
|
|
14849
|
-
*
|
|
14673
|
+
* escapeThinkMarkers alone is only safe on a complete string: adapters call it once per
|
|
14674
|
+
* delta, but a provider is free to split a marker-shaped substring across two adjacent
|
|
14675
|
+
* deltas (e.g. 'wrote <th' then 'ink> tag'). Escaping each half independently leaves both
|
|
14676
|
+
* halves unescaped, and concatenating them reassembles a literal `<think>`/`</think>` that
|
|
14677
|
+
* is then indistinguishable from the real control marker wrapped around the same text.
|
|
14850
14678
|
*
|
|
14851
|
-
*
|
|
14852
|
-
*
|
|
14853
|
-
|
|
14854
|
-
|
|
14855
|
-
|
|
14856
|
-
|
|
14857
|
-
|
|
14858
|
-
|
|
14859
|
-
|
|
14679
|
+
* This holds back any trailing substring of the buffered text that could still extend into
|
|
14680
|
+
* a marker (up to `<think>`/`</think>`'s length minus one) until the next push resolves it
|
|
14681
|
+
* one way or the other, or flush() is called at the end of the reasoning span.
|
|
14682
|
+
*/
|
|
14683
|
+
function createThinkMarkerEscaper() {
|
|
14684
|
+
let pending = "";
|
|
14685
|
+
return {
|
|
14686
|
+
push(chunk) {
|
|
14687
|
+
if (!chunk) return "";
|
|
14688
|
+
const combined = pending + chunk;
|
|
14689
|
+
const holdLength = partialMarkerSuffixLength(combined);
|
|
14690
|
+
const safeLength = combined.length - holdLength;
|
|
14691
|
+
const safe = combined.slice(0, safeLength);
|
|
14692
|
+
pending = combined.slice(safeLength);
|
|
14693
|
+
return escapeThinkMarkers(safe);
|
|
14694
|
+
},
|
|
14695
|
+
flush() {
|
|
14696
|
+
const remaining = pending;
|
|
14697
|
+
pending = "";
|
|
14698
|
+
return escapeThinkMarkers(remaining);
|
|
14699
|
+
}
|
|
14700
|
+
};
|
|
14860
14701
|
}
|
|
14861
14702
|
/** Generation was cut off against the output-token ceiling. */
|
|
14862
14703
|
const TRUNCATED_FINISH_REASON = "max_tokens";
|
|
@@ -14884,6 +14725,71 @@ function isEarlyStop(stopReason) {
|
|
|
14884
14725
|
return !!stopReason && EARLY_STOP_FINISH_REASONS.has(stopReason);
|
|
14885
14726
|
}
|
|
14886
14727
|
/**
|
|
14728
|
+
* Signs the image URLs `web_search` writes into its tool output, so `/api/search-image` (the
|
|
14729
|
+
* same-origin proxy that fetches them server-side) can verify a URL actually came from our own
|
|
14730
|
+
* search-provider results, not from a hostile page's snippet text steering the model into writing
|
|
14731
|
+
* an attacker-controlled URL with exfiltrated data in the query string.
|
|
14732
|
+
*
|
|
14733
|
+
* The signature is a trailing `&b4mSig=<hex>` (or `?b4mSig=<hex>` with no prior query string)
|
|
14734
|
+
* appended by plain string concatenation - never by reparsing the URL through `URLSearchParams`,
|
|
14735
|
+
* which re-serializes every existing param (e.g. turning a literal space into `+`) and would send
|
|
14736
|
+
* a byte-different URL to a host whose own signature (an imgix/S3-presigned link) covers the exact
|
|
14737
|
+
* query string the provider returned. Anchoring the signature to the END of the string, and
|
|
14738
|
+
* requiring it to be the only occurrence, is what makes verification unambiguous: anything a
|
|
14739
|
+
* tamperer appends after signing - including a second `b4mSig` - changes what's left of the anchor
|
|
14740
|
+
* match, which changes the canonical text, which invalidates the signature. There is deliberately
|
|
14741
|
+
* no "read the first/last `b4mSig` param and ignore the rest" step, since that step is exactly
|
|
14742
|
+
* where an earlier version of this file's bypass lived (verify read the URL's first `b4mSig` via
|
|
14743
|
+
* `searchParams.get` while signing canonicalized by deleting *every* `b4mSig`, so a validly-signed
|
|
14744
|
+
* URL with a second, attacker-authored `b4mSig` appended still verified, and was then forwarded to
|
|
14745
|
+
* `safeFetch` with that attacker data still attached).
|
|
14746
|
+
*
|
|
14747
|
+
* The model is already told to copy an `Images:` line's URL verbatim, so no new instruction is
|
|
14748
|
+
* needed for the signature to survive.
|
|
14749
|
+
*
|
|
14750
|
+
* The signature has no expiry: a stored reply, citable, published page, or curation transcript
|
|
14751
|
+
* persists indefinitely, and nothing re-signs a URL on read, so a TTL would make every card
|
|
14752
|
+
* permanently break once it lapsed rather than bound anything meaningful. What actually bounds
|
|
14753
|
+
* a leaked signed URL's usefulness is the same thing that bounds the route otherwise: it's only
|
|
14754
|
+
* reachable at all behind `jwtOnly` auth (apps/client/pages/api/search-image.ts) and is capped by
|
|
14755
|
+
* a per-user rate limit there.
|
|
14756
|
+
*/
|
|
14757
|
+
const SIGNATURE_PARAM = "b4mSig";
|
|
14758
|
+
const TRAILING_SIGNATURE_RE = new RegExp(`[?&]${SIGNATURE_PARAM}=([0-9a-f]{32})$`);
|
|
14759
|
+
const KNOWN_PLACEHOLDER_SECRETS = /* @__PURE__ */ new Set([
|
|
14760
|
+
"",
|
|
14761
|
+
"my-secret-placeholder-value",
|
|
14762
|
+
"not-configured"
|
|
14763
|
+
]);
|
|
14764
|
+
/**
|
|
14765
|
+
* True for an empty or known-placeholder signing secret. Shared so any caller that would
|
|
14766
|
+
* otherwise sign-and-fail (e.g. web_search deciding whether to pay for image results at all)
|
|
14767
|
+
* uses the same definition `verifyImageUrlSignature` fails closed on, rather than a second one
|
|
14768
|
+
* that could drift out of sync.
|
|
14769
|
+
*/
|
|
14770
|
+
function isPlaceholderImageSigningSecret(secret) {
|
|
14771
|
+
return KNOWN_PLACEHOLDER_SECRETS.has(secret ?? "");
|
|
14772
|
+
}
|
|
14773
|
+
function computeSignature(canonical, secret) {
|
|
14774
|
+
return createHmac("sha256", secret).update(canonical).digest("hex").slice(0, 32);
|
|
14775
|
+
}
|
|
14776
|
+
/**
|
|
14777
|
+
* Appends expiry + signature query params by string concatenation (never reparses or
|
|
14778
|
+
* re-serializes the existing query string - see the file-level comment). Returns the URL
|
|
14779
|
+
* unchanged if it fails to parse as a URL, or if it already carries a trailing signature.
|
|
14780
|
+
*/
|
|
14781
|
+
function signImageUrl(rawUrl, secret) {
|
|
14782
|
+
try {
|
|
14783
|
+
new URL(rawUrl);
|
|
14784
|
+
} catch {
|
|
14785
|
+
return rawUrl;
|
|
14786
|
+
}
|
|
14787
|
+
if (TRAILING_SIGNATURE_RE.test(rawUrl)) return rawUrl;
|
|
14788
|
+
const separator = rawUrl.includes("?") ? "&" : "?";
|
|
14789
|
+
const signature = computeSignature(rawUrl, secret);
|
|
14790
|
+
return `${rawUrl}${separator}${SIGNATURE_PARAM}=${signature}`;
|
|
14791
|
+
}
|
|
14792
|
+
/**
|
|
14887
14793
|
* promptMeta.functionCalls fields that must never reach a viewer who only holds a
|
|
14888
14794
|
* "read this conversation" grant - a session share, a live subscription, a bug-report
|
|
14889
14795
|
* egress to a third party (Slack/email), or a session clone made by a share holder
|
|
@@ -14917,6 +14823,13 @@ const OWNER_ONLY_FUNCTION_CALL_FIELDS = ["returnValue", "error"];
|
|
|
14917
14823
|
* and a share/subscribe/clone grant authorizes reading the conversation, not re-reading the
|
|
14918
14824
|
* owner's corpus through it. The sibling `chunkId` is deliberately kept - an opaque id is not
|
|
14919
14825
|
* content, and `citables[].id` (the file id) is already unredacted beside it.
|
|
14826
|
+
*
|
|
14827
|
+
* `conflictsWith` (#3041) is kept for the same reason, recorded here so the next reader does not
|
|
14828
|
+
* re-derive it: it holds `fabFileId`s of other cited sources, so it adds a RELATIONSHIP between
|
|
14829
|
+
* chips the viewer can already see rather than a slice of the owner's corpus. That holds only while
|
|
14830
|
+
* the field stays ids - a future version carrying the conflicting SENTENCES (the detector has them:
|
|
14831
|
+
* InconsistencyEvidence.excerpt) would be `fullContext`'s class exactly and would have to join the
|
|
14832
|
+
* list below rather than ride along inside this one.
|
|
14920
14833
|
*/
|
|
14921
14834
|
const OWNER_ONLY_CITABLE_METADATA_FIELDS = ["fullContext"];
|
|
14922
14835
|
[...OWNER_ONLY_FUNCTION_CALL_FIELDS.map((field) => `promptMeta.functionCalls.${field}`), ...OWNER_ONLY_CITABLE_METADATA_FIELDS.map((field) => `promptMeta.citables.metadata.${field}`)];
|
|
@@ -14953,47 +14866,6 @@ z$1.array(triggerWordSchema).max(20, "Up to 20 trigger words allowed.").transfor
|
|
|
14953
14866
|
}
|
|
14954
14867
|
return out;
|
|
14955
14868
|
});
|
|
14956
|
-
/**
|
|
14957
|
-
* Serve gate for uploaded FabFiles: hold-until-scanned, fail-closed on ALL mime
|
|
14958
|
-
* types, not just images. A file is serveable only once moderation has run to
|
|
14959
|
-
* completion on it, REGARDLESS of its declared `mimeType`.
|
|
14960
|
-
*
|
|
14961
|
-
* Why gate non-images too: `mimeType` is client-declared at upload time and is only
|
|
14962
|
-
* corrected by the S3 scan ~1-2s later (see `moderateUploadedFile`'s byte-sniffing). If this
|
|
14963
|
-
* gate special-cased "non-images always serveable" based on that same untrusted declared
|
|
14964
|
-
* mimeType, a file uploaded as `application/pdf` but actually a PNG (or vice versa) would be
|
|
14965
|
-
* served during that window before the sniff/scan ever runs. Gating on `moderationStatus`
|
|
14966
|
-
* alone closes that window for every file, image or not.
|
|
14967
|
-
*
|
|
14968
|
-
* `moderationStatus` semantics:
|
|
14969
|
-
* - 'clean' -> serveable
|
|
14970
|
-
* - 'pending' | 'scanning' -> NOT serveable (not yet through the scan)
|
|
14971
|
-
* - 'blocked' -> NOT serveable (confirmed block / unscannable format)
|
|
14972
|
-
* - null | undefined -> NOT serveable (fail-closed; legacy rows are
|
|
14973
|
-
* backfilled to 'clean', see backfill-fabfile-moderation-status.ts)
|
|
14974
|
-
*
|
|
14975
|
-
* Non-image files (PDFs, docs, text, ...) are NOT scanned by Rekognition, but they still
|
|
14976
|
-
* pass through `moderateUploadedFile`/`objectCreated`, which resolves them to 'clean'
|
|
14977
|
-
* immediately (no image bytes to hold on), so the hold is brief (one S3 event round trip),
|
|
14978
|
-
* not an indefinite block.
|
|
14979
|
-
*/
|
|
14980
|
-
function isImageServeable(f) {
|
|
14981
|
-
return f.moderationStatus === "clean";
|
|
14982
|
-
}
|
|
14983
|
-
/**
|
|
14984
|
-
* Is this mime type an image? `image/svg+xml` counts, since vision models receive it
|
|
14985
|
-
* as an image.
|
|
14986
|
-
*
|
|
14987
|
-
* The repo has ~50 inline `startsWith('image/')` checks and they disagree on the edges
|
|
14988
|
-
* (case, null handling). Only the ones on the attachment pipeline - composer upload,
|
|
14989
|
-
* chat context assembly, attachment capability warnings - have been converted here.
|
|
14990
|
-
* Icon pickers, avatar validation and resize eligibility still carry their own copies:
|
|
14991
|
-
* same question, unrelated subsystems, and folding them in would have made this a
|
|
14992
|
-
* repo-wide diff.
|
|
14993
|
-
*/
|
|
14994
|
-
function isImageAttachment(mimeType) {
|
|
14995
|
-
return typeof mimeType === "string" && mimeType.toLowerCase().startsWith("image/");
|
|
14996
|
-
}
|
|
14997
14869
|
IMAGE_SIZE_CONSTRAINTS.BFL.minWidth, IMAGE_SIZE_CONSTRAINTS.BFL.maxWidth;
|
|
14998
14870
|
Array.from(new Set([
|
|
14999
14871
|
{
|
|
@@ -15901,71 +15773,10 @@ Array.from(new Set([
|
|
|
15901
15773
|
const [, top] = v.target.split("/");
|
|
15902
15774
|
return `/${top}`;
|
|
15903
15775
|
})));
|
|
15904
|
-
function getHeader(headers, name) {
|
|
15905
|
-
if (!headers || typeof headers !== "object") return null;
|
|
15906
|
-
if (typeof headers.get === "function") {
|
|
15907
|
-
const value = headers.get(name);
|
|
15908
|
-
return typeof value === "string" ? value : null;
|
|
15909
|
-
}
|
|
15910
|
-
const value = headers[name] ?? headers[name.toLowerCase()];
|
|
15911
|
-
return typeof value === "string" ? value : null;
|
|
15912
|
-
}
|
|
15913
|
-
function parseCount(value) {
|
|
15914
|
-
if (value === null) return null;
|
|
15915
|
-
const trimmed = value.trim();
|
|
15916
|
-
if (!trimmed) return null;
|
|
15917
|
-
const parsed = Number(trimmed);
|
|
15918
|
-
return Number.isFinite(parsed) ? parsed : null;
|
|
15919
|
-
}
|
|
15920
|
-
const UNIT_MS = {
|
|
15921
|
-
ms: 1,
|
|
15922
|
-
s: 1e3,
|
|
15923
|
-
m: 6e4,
|
|
15924
|
-
h: 36e5
|
|
15925
|
-
};
|
|
15926
|
-
const DURATION_PART = /(\d+(?:\.\d+)?)(ms|h|m|s)/g;
|
|
15927
|
-
/**
|
|
15928
|
-
* Parse a Go-style duration ("6ms", "0s", "1m30s", "1h2m3s") to milliseconds.
|
|
15929
|
-
*
|
|
15930
|
-
* Exported for its own tests: it is the part of this module that can be wrong in a way the
|
|
15931
|
-
* numbers still look plausible.
|
|
15932
|
-
*/
|
|
15933
|
-
function parseDurationMs(value) {
|
|
15934
|
-
if (typeof value !== "string") return null;
|
|
15935
|
-
const trimmed = value.trim();
|
|
15936
|
-
if (!trimmed) return null;
|
|
15937
|
-
DURATION_PART.lastIndex = 0;
|
|
15938
|
-
let total = 0;
|
|
15939
|
-
let matched = 0;
|
|
15940
|
-
let consumed = 0;
|
|
15941
|
-
for (const part of trimmed.matchAll(DURATION_PART)) {
|
|
15942
|
-
total += Number(part[1]) * UNIT_MS[part[2]];
|
|
15943
|
-
consumed += part[0].length;
|
|
15944
|
-
matched += 1;
|
|
15945
|
-
}
|
|
15946
|
-
if (matched === 0 || consumed !== trimmed.length) return null;
|
|
15947
|
-
return total;
|
|
15948
|
-
}
|
|
15949
|
-
/** Read both rate-limit dimensions off a provider response. */
|
|
15950
|
-
function parseEmbeddingRateLimitHeaders(headers) {
|
|
15951
|
-
return {
|
|
15952
|
-
limitTokens: parseCount(getHeader(headers, "x-ratelimit-limit-tokens")),
|
|
15953
|
-
limitRequests: parseCount(getHeader(headers, "x-ratelimit-limit-requests")),
|
|
15954
|
-
remainingTokens: parseCount(getHeader(headers, "x-ratelimit-remaining-tokens")),
|
|
15955
|
-
remainingRequests: parseCount(getHeader(headers, "x-ratelimit-remaining-requests")),
|
|
15956
|
-
resetTokensMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-tokens")),
|
|
15957
|
-
resetRequestsMs: parseDurationMs(getHeader(headers, "x-ratelimit-reset-requests"))
|
|
15958
|
-
};
|
|
15959
|
-
}
|
|
15960
|
-
/** True when the provider reported at least one usable ceiling. */
|
|
15961
|
-
function hasUsableLimits(snapshot) {
|
|
15962
|
-
return snapshot.limitTokens !== null || snapshot.limitRequests !== null;
|
|
15963
|
-
}
|
|
15964
15776
|
dayjs.extend(utc);
|
|
15965
15777
|
dayjs.extend(timezone);
|
|
15966
15778
|
dayjs.extend(relativeTime);
|
|
15967
15779
|
dayjs.extend(localizedFormat);
|
|
15968
|
-
var dayjsConfig_default = dayjs;
|
|
15969
15780
|
/**
|
|
15970
15781
|
* Default retryable errors for LLM API calls
|
|
15971
15782
|
*/
|
|
@@ -16133,42 +15944,6 @@ async function withRetry(fn, options = {}) {
|
|
|
16133
15944
|
}
|
|
16134
15945
|
}
|
|
16135
15946
|
}
|
|
16136
|
-
function readZipEntryBounded(entry, maxBytes) {
|
|
16137
|
-
return new Promise((resolve, reject) => {
|
|
16138
|
-
const stream = entry.internalStream("nodebuffer");
|
|
16139
|
-
let parts = [];
|
|
16140
|
-
let byteLength = 0;
|
|
16141
|
-
let settled = false;
|
|
16142
|
-
stream.on("data", (chunk) => {
|
|
16143
|
-
if (settled) return;
|
|
16144
|
-
byteLength += chunk.length;
|
|
16145
|
-
if (byteLength > maxBytes) {
|
|
16146
|
-
settled = true;
|
|
16147
|
-
stream.pause();
|
|
16148
|
-
parts = [];
|
|
16149
|
-
resolve({
|
|
16150
|
-
ok: false,
|
|
16151
|
-
reason: "too-large"
|
|
16152
|
-
});
|
|
16153
|
-
return;
|
|
16154
|
-
}
|
|
16155
|
-
parts.push(chunk);
|
|
16156
|
-
}).on("error", (error) => {
|
|
16157
|
-
if (settled) return;
|
|
16158
|
-
settled = true;
|
|
16159
|
-
parts = [];
|
|
16160
|
-
reject(error);
|
|
16161
|
-
}).on("end", () => {
|
|
16162
|
-
if (settled) return;
|
|
16163
|
-
settled = true;
|
|
16164
|
-
resolve({
|
|
16165
|
-
ok: true,
|
|
16166
|
-
text: Buffer.concat(parts).toString("utf8"),
|
|
16167
|
-
byteLength
|
|
16168
|
-
});
|
|
16169
|
-
}).resume();
|
|
16170
|
-
});
|
|
16171
|
-
}
|
|
16172
15947
|
//#endregion
|
|
16173
15948
|
//#region src/utils/apiUrl.ts
|
|
16174
15949
|
/**
|
|
@@ -16290,6 +16065,19 @@ function getEnvironmentName(configApiConfig) {
|
|
|
16290
16065
|
return "Self-Hosted";
|
|
16291
16066
|
}
|
|
16292
16067
|
//#endregion
|
|
16068
|
+
//#region src/utils/validateSessionId.ts
|
|
16069
|
+
/**
|
|
16070
|
+
* Session and resume ids arrive from the environment (`B4M_SESSION_ID`,
|
|
16071
|
+
* `B4M_RESUME_ID`) and are used as filesystem path components by the session
|
|
16072
|
+
* store and the debug logger. Restrict them to a strict charset (which still
|
|
16073
|
+
* covers UUIDs) so a hostile launcher cannot traverse out of the base dir via
|
|
16074
|
+
* e.g. `B4M_SESSION_ID=../config`.
|
|
16075
|
+
*/
|
|
16076
|
+
const SESSION_ID_PATTERN = /^[A-Za-z0-9_-]+$/;
|
|
16077
|
+
function isValidSessionId(value) {
|
|
16078
|
+
return SESSION_ID_PATTERN.test(value);
|
|
16079
|
+
}
|
|
16080
|
+
//#endregion
|
|
16293
16081
|
//#region src/config/toolSafety.ts
|
|
16294
16082
|
/**
|
|
16295
16083
|
* Tool safety categories determine when permission is required
|
|
@@ -16405,13 +16193,13 @@ function formatFileSize(bytes) {
|
|
|
16405
16193
|
function tryReadContextFile(dir, filename, source) {
|
|
16406
16194
|
const filePath = path$1.join(dir, filename);
|
|
16407
16195
|
try {
|
|
16408
|
-
const stats = fs$
|
|
16196
|
+
const stats = fs$1.lstatSync(filePath);
|
|
16409
16197
|
if (stats.isDirectory()) return null;
|
|
16410
16198
|
if (stats.isSymbolicLink()) return { error: `${source === "global" ? "Global" : "Project"} ${filename} is a symlink (not allowed for security)` };
|
|
16411
16199
|
if (stats.size > 102400) return { error: `${source === "global" ? "Global" : "Project"} ${filename} exceeds 100KB limit (${formatFileSize(stats.size)})` };
|
|
16412
16200
|
return {
|
|
16413
16201
|
filename,
|
|
16414
|
-
content: fs$
|
|
16202
|
+
content: fs$1.readFileSync(filePath, "utf-8"),
|
|
16415
16203
|
source,
|
|
16416
16204
|
path: filePath
|
|
16417
16205
|
};
|
|
@@ -16513,9 +16301,10 @@ const logger = class Logger {
|
|
|
16513
16301
|
* Initialize the logger with a session ID
|
|
16514
16302
|
*/
|
|
16515
16303
|
async initialize(sessionId) {
|
|
16304
|
+
if (!isValidSessionId(sessionId)) throw new Error(`Invalid session id "${sessionId}": must match ${SESSION_ID_PATTERN.source}`);
|
|
16516
16305
|
this.sessionId = sessionId;
|
|
16517
16306
|
const debugDir = path.join(os.homedir(), ".bike4mind", "debug");
|
|
16518
|
-
await fs
|
|
16307
|
+
await fs.mkdir(debugDir, { recursive: true });
|
|
16519
16308
|
this.logFilePath = path.join(debugDir, `${sessionId}.txt`);
|
|
16520
16309
|
await this.writeToFile("INFO", "=== CLI SESSION START ===");
|
|
16521
16310
|
}
|
|
@@ -16567,7 +16356,7 @@ const logger = class Logger {
|
|
|
16567
16356
|
if (!this.fileLoggingEnabled || !this.logFilePath) return;
|
|
16568
16357
|
try {
|
|
16569
16358
|
const logEntry = `[${(/* @__PURE__ */ new Date()).toISOString().replace("T", " ").substring(0, 19)}] [${level}] ${message}\n`;
|
|
16570
|
-
await fs
|
|
16359
|
+
await fs.appendFile(this.logFilePath, logEntry, "utf-8");
|
|
16571
16360
|
} catch (error) {
|
|
16572
16361
|
console.error("File logging failed:", error);
|
|
16573
16362
|
}
|
|
@@ -16684,11 +16473,11 @@ const logger = class Logger {
|
|
|
16684
16473
|
if (!this.fileLoggingEnabled) return;
|
|
16685
16474
|
try {
|
|
16686
16475
|
const debugDir = path.join(os.homedir(), ".bike4mind", "debug");
|
|
16687
|
-
const files = await fs
|
|
16476
|
+
const files = await fs.readdir(debugDir);
|
|
16688
16477
|
const thirtyDaysAgo = Date.now() - 2592e6;
|
|
16689
16478
|
for (const file of files) {
|
|
16690
16479
|
const filePath = path.join(debugDir, file);
|
|
16691
|
-
if ((await fs
|
|
16480
|
+
if ((await fs.stat(filePath)).mtime.getTime() < thirtyDaysAgo) await fs.unlink(filePath);
|
|
16692
16481
|
}
|
|
16693
16482
|
} catch (error) {
|
|
16694
16483
|
console.error("Failed to cleanup old logs:", error);
|
|
@@ -17640,7 +17429,7 @@ var ConfigStore = class {
|
|
|
17640
17429
|
*/
|
|
17641
17430
|
async saveSandboxConfig(sandbox) {
|
|
17642
17431
|
await this.load();
|
|
17643
|
-
this.globalConfig.sandbox = sandbox;
|
|
17432
|
+
this.globalConfig.sandbox = structuredClone(sandbox);
|
|
17644
17433
|
await this.save();
|
|
17645
17434
|
}
|
|
17646
17435
|
/**
|
|
@@ -17931,4 +17720,4 @@ var ConfigStore = class {
|
|
|
17931
17720
|
}
|
|
17932
17721
|
};
|
|
17933
17722
|
//#endregion
|
|
17934
|
-
export {
|
|
17723
|
+
export { withRetry as $, NotFoundError as A, WORK_ITEM_STATUSES as B, CREDIT_DEDUCT_TRANSACTION_TYPES as C, MODEL_INFO_FIELD_GROUP_OF as D, LOCATION_MAP_LANGUAGE as E, REVIEW_GATE_STATUS_VALUES as F, googleMapsSearchUrl as G, escapeThinkMarkers as H, SEARCH_RESULT_CARDS_LANGUAGE as I, isRetryableError as J, isEarlyStop as K, SUBQUEST_STATUS_VALUES as L, OpenAIEmbeddingModel as M, PROMPT_TEXT_MAX as N, McpServerName as O, PermissionDeniedError as P, signImageUrl as Q, SupportedFabFileMimeTypes as R, CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS as S, DEGENERATE_FINISH_REASON as T, getMcpProviderMetadata as U, createThinkMarkerEscaper as V, getQuestErrorCode as W, obfuscateApiKey as X, isUserInitiatedAbort as Y, secureParameters as Z, AGENT_QUEST_ID as _, canTrustTool as a, ApiKeyType as b, SESSION_ID_PATTERN as c, LOCAL_DEV_URL as d, ACTOR_COLOR_SLOTS as et, getCreditsUrl as f, resolveApiEndpoint as g, requireApiUrl as h, loadContextFiles as i, selfClaimedActorKindSchema as it, OllamaEmbeddingModel as j, ModelBackend as k, isValidSessionId as l, parseApiUrl as m, logger as n, actorKindMarker as nt, getToolCategory as o, getEnvironmentName as p, isPlaceholderImageSigningSecret as q, extractCompactInstructions as r, actorKindSchema as rt, isReadOnlyTool as s, ConfigStore as t, actorColorIndex as tt, ApiEndpointUnconfiguredError as u, AGENT_QUEST_MANIFEST as v, ChatModels as w, BedrockEmbeddingModel as x, AGENT_QUEST_MCP_URI as y, VoyageAIEmbeddingModel as z };
|