@bike4mind/cli 0.20.2 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { $ as VoyageAIEmbeddingModel, A as HTTPError, At as resolveHistoryFetchLimit, B as PermissionDeniedError, Bt as isNearLimit, Ct as isRetryableError, D as FIXED_TEMPERATURE_MODELS, Dt as isZodError, E as FIELD_GROUP_OF, Et as isUserInitiatedAbort, F as ModelBackend, Ft as usdToCredits, G as SpeechToTextModels, H as REASONING_SUPPORTED_MODELS, I as NO_TEMPERATURE_MODELS, It as usdToCreditsStochastic, J as TooManyRequestsError, K as SupportedFabFileMimeTypes, L as NotFoundError, Lt as withRetry, M as ImageModels, Mt as settingsMap, N as InternalServerError, Nt as toModelInfo, O as FORMAT_PROMPT_TEMPLATE, Ot as mapMimeTypeToArtifactType, P as MODEL_INFO_FIELD_GROUP_OF, Pt as toModelRecord, Q as VideoModels, R as OllamaEmbeddingModel, Rt as buildRateLimitLogEntry, S as CorruptedFileError, St as isRenderableModelType, Tt as isUnlimitedHistory, U as REFUSAL_FALLBACK_MODELS, V as REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS, Vt as parseRateLimitHeaders, W as RESPONSES_API_TOOL_MODELS, X as UnprocessableEntityError, Y as UnauthorizedError, Z as VIDEO_SIZE_CONSTRAINTS, _ as BadRequestError, _t as isImageServeable, at as getMcpProviderMetadata, bt as isModelDeprecated, ct as isAudioMimeType, et as WORK_ITEM_STATUSES, ft as isFieldGroup, g as BFL_SAFETY_TOLERANCE, gt as isImageAttachment, h as BEDROCK_NO_PROMPT_CACHING_MODELS, ht as isGeminiModelId, it as defaultEmbeddingModelForEnv, j as HttpStatus, jt as secureParameters, k as ForbiddenError, kt as obfuscateApiKey, lt as isChunkRebuildPending, m as ApiKeyType, mt as isGPTImageModel, n as logger, nt as calculateRetryDelay, ot as getQuestErrorCode, p as ARTIFACT_ATTRS_PATTERN, pt as isGPTImage2Model, q as TTS_MAX_INPUT_CHARS, rt as dayjsConfig_default, st as getRetryAfterMs, tt as applyModelPriceCatalog, ut as isConvergencePausedNote, v as BedrockEmbeddingModel, vt as isMediaModelType, w as DEFAULT_UNKNOWN_CONTEXT_WINDOW, wt as isSupportedFabFileMimeType, x as ChatModels, xt as isPlaceholderApiKey, y as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, yt as isModelAccessible, z as OpenAIEmbeddingModel, zt as extractSnippetMeta } from "./ConfigStore-cIyF7hDg.mjs";
2
+ import { $ as SupportedFabFileMimeTypes, A as HTTPError, At as isPlaceholderApiKey, B as OPENAI_GPT_IMAGE_1_IMAGE_SIZES, Bt as reservationOutputTokens, Ct as isGPTImageModel, D as FIXED_TEMPERATURE_MODELS, Dt as isMediaModelType, E as FIELD_GROUP_OF, Et as isImageServeable, F as MODEL_INFO_FIELD_GROUP_OF, Ft as isUserInitiatedAbort, G as PermissionDeniedError, Gt as toModelRecord, H as OllamaEmbeddingModel, Ht as secureParameters, I as McpServerName, It as isZodError, J as REFUSAL_FALLBACK_MODELS, Jt as withRetry, K as REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS, Kt as usdToCredits, L as ModelBackend, Lt as mapMimeTypeToArtifactType, M as IMAGE_SIZE_CONSTRAINTS, Mt as isRetryableError, N as ImageModels, Nt as isSupportedFabFileMimeType, O as FORMAT_PROMPT_TEMPLATE, Ot as isModelAccessible, P as InternalServerError, Pt as isUnlimitedHistory, Q as SpeechToTextModels, R as NO_TEMPERATURE_MODELS, Rt as obfuscateApiKey, S as CorruptedFileError, St as isGPTImage2Model, Tt as isImageAttachment, U as OpenAIEmbeddingModel, Ut as settingsMap, V as OPENAI_GPT_IMAGE_2_IMAGE_SIZES, Vt as resolveHistoryFetchLimit, Wt as toModelInfo, Y as RESPONSES_API_TOOL_MODELS, _ as BadRequestError, _t as isAudioMimeType, at as VideoModels, ct as applyModelPriceCatalog, dt as defaultEmbeddingModelForEnv, en as buildRateLimitLogEntry, et as TTS_MAX_INPUT_CHARS, ft as getMcpProviderMetadata, g as BFL_SAFETY_TOLERANCE, gt as hasUsableLimits, h as BEDROCK_NO_PROMPT_CACHING_MODELS, ht as hasKeylessCloudEmbedder, it as VIDEO_SIZE_CONSTRAINTS, j as HttpStatus, jt as isRenderableModelType, k as ForbiddenError, kt as isModelDeprecated, lt as calculateRetryDelay, m as ApiKeyType, mt as getRetryAfterMs, n as logger, nn as isNearLimit, nt as UnauthorizedError, ot as VoyageAIEmbeddingModel, p as ARTIFACT_ATTRS_PATTERN, pt as getQuestErrorCode, q as REASONING_SUPPORTED_MODELS, qt as usdToCreditsStochastic, rn as parseRateLimitHeaders, rt as UnprocessableEntityError, st as WORK_ITEM_STATUSES, tn as extractSnippetMeta, tt as TooManyRequestsError, ut as dayjsConfig_default, v as BedrockEmbeddingModel, vt as isChunkRebuildPending, w as DEFAULT_UNKNOWN_CONTEXT_WINDOW, wt as isGeminiModelId, x as ChatModels, xt as isFieldGroup, y as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, yt as isChunkStalledFile, z as NotFoundError, zt as parseEmbeddingRateLimitHeaders } from "./ConfigStore-8_0WsN5r.mjs";
3
3
  import { n as isPathAllowed, t as assertPathAllowed } from "./pathValidation-D8tjkQXE-1HwvsuYT.mjs";
4
4
  import { n as isTerminalShellStatus, t as getShellSessionManager } from "./ShellSessionManager-6o8KZzl1-vrbPAUTq.mjs";
5
5
  import { execFile, execFileSync, spawn } from "child_process";
@@ -21,6 +21,7 @@ import * as turndownPluginGfm from "@joplin/turndown-plugin-gfm";
21
21
  import * as cheerio from "cheerio";
22
22
  import FirecrawlDefault, { FirecrawlError } from "@mendable/firecrawl-js";
23
23
  import { lookup } from "node:dns/promises";
24
+ import mongoose, { isObjectIdOrHexString } from "mongoose";
24
25
  import random from "lodash/random.js";
25
26
  import sum from "lodash/sum.js";
26
27
  import times from "lodash/times.js";
@@ -48,13 +49,12 @@ import { NodeHttpHandler } from "@smithy/node-http-handler";
48
49
  import "@opensearch-project/opensearch";
49
50
  import "@aws-sdk/credential-provider-node";
50
51
  import "@opensearch-project/opensearch/aws-v3";
51
- import mongoose from "mongoose";
52
52
  import { parse } from "shell-quote";
53
53
  import { homedir as homedir$1 } from "node:os";
54
54
  import { EventEmitter } from "events";
55
+ import { CloudWatchClient, PutMetricDataCommand, StandardUnit } from "@aws-sdk/client-cloudwatch";
55
56
  import { fileURLToPath } from "url";
56
57
  import { Anthropic, RateLimitError } from "@anthropic-ai/sdk";
57
- import { CloudWatchClient, PutMetricDataCommand, StandardUnit } from "@aws-sdk/client-cloudwatch";
58
58
  import { GoogleGenAI } from "@google/genai";
59
59
  import pick from "lodash/pick.js";
60
60
  import { Stream } from "openai/streaming";
@@ -66,6 +66,7 @@ import { StreamableHTTPClientTransport } from "@modelcontextprotocol/sdk/client/
66
66
  import { Client } from "@modelcontextprotocol/sdk/client/index.js";
67
67
  import { getDomain } from "tldts";
68
68
  import * as dotenv from "dotenv";
69
+ import { createHash as createHash$1 } from "node:crypto";
69
70
  import invert from "lodash/invert.js";
70
71
  import * as util from "node:util";
71
72
  import * as zlib from "node:zlib";
@@ -920,6 +921,7 @@ const DEMO_KEY_MAP = {
920
921
  [ApiKeyType.gemini]: "geminiDemoKey",
921
922
  [ApiKeyType.xai]: "xaiApiKey",
922
923
  [ApiKeyType.kimi]: "moonshotApiKey",
924
+ [ApiKeyType.deepseek]: "deepseekApiKey",
923
925
  [ApiKeyType.bfl]: "bflApiKey",
924
926
  [ApiKeyType.voyageai]: "voyageApiKey",
925
927
  [ApiKeyType.elevenlabs]: "elevenLabsServerApiKey"
@@ -977,6 +979,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
977
979
  ApiKeyType.bfl,
978
980
  ApiKeyType.xai,
979
981
  ApiKeyType.kimi,
982
+ ApiKeyType.deepseek,
980
983
  ApiKeyType.voyageai
981
984
  ], adapters) : Promise.resolve([]), adapters.getSettingsByNames([
982
985
  "openaiDemoKey",
@@ -985,6 +988,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
985
988
  "bflApiKey",
986
989
  "xaiApiKey",
987
990
  "moonshotApiKey",
991
+ "deepseekApiKey",
988
992
  "voyageApiKey",
989
993
  "ollamaBackend",
990
994
  "EnableOllama"
@@ -997,6 +1001,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
997
1001
  const bflUserKey = userKeyMap.get(ApiKeyType.bfl) || null;
998
1002
  const xaiUserKey = userKeyMap.get(ApiKeyType.xai) || null;
999
1003
  const kimiUserKey = userKeyMap.get(ApiKeyType.kimi) || null;
1004
+ const deepseekUserKey = userKeyMap.get(ApiKeyType.deepseek) || null;
1000
1005
  const voyageaiUserKey = userKeyMap.get(ApiKeyType.voyageai) || null;
1001
1006
  const openaiDemoKey = adminSettings["openaiDemoKey"];
1002
1007
  const anthropicDemoKey = adminSettings["anthropicDemoKey"];
@@ -1004,6 +1009,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
1004
1009
  const bflDemoKey = adminSettings["bflApiKey"];
1005
1010
  const xaiDemoKey = adminSettings["xaiApiKey"];
1006
1011
  const kimiDemoKey = adminSettings["moonshotApiKey"];
1012
+ const deepseekDemoKey = adminSettings["deepseekApiKey"];
1007
1013
  const voyageaiDemoKey = adminSettings["voyageApiKey"];
1008
1014
  const ollamaBackend = adminSettings["ollamaBackend"];
1009
1015
  const enableOllama = adminSettings["EnableOllama"];
@@ -1022,6 +1028,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
1022
1028
  bfl: keyOrExpired(bflUserKey) || bflDemoKey || null,
1023
1029
  xai: keyOrExpired(xaiUserKey) || xaiDemoKey || envKey("XAI_API_KEY"),
1024
1030
  kimi: keyOrExpired(kimiUserKey) || kimiDemoKey || envKey("MOONSHOT_API_KEY"),
1031
+ deepseek: keyOrExpired(deepseekUserKey) || deepseekDemoKey || envKey("DEEPSEEK_API_KEY"),
1025
1032
  voyageai: keyOrExpired(voyageaiUserKey) || voyageaiDemoKey || null,
1026
1033
  ollama: (ollamaEnabled ? ollamaBackend || null : null) || envKey("OLLAMA_BASE_URL"),
1027
1034
  imageGen: envKey("IMAGE_GEN_BASE_URL")
@@ -1836,7 +1843,27 @@ const webFetchTool = {
1836
1843
  })
1837
1844
  };
1838
1845
  //#endregion
1839
- //#region ../../b4m-core/services/dist/websearch-DpUKKZyj.mjs
1846
+ //#region ../../b4m-core/services/dist/websearch-BLmQCbHG.mjs
1847
+ /**
1848
+ * The coarse recency bucket both providers speak, as the smallest one containing `recencyDays`.
1849
+ * Null when there is no constraint, or when the window is wider than the widest bucket - a
1850
+ * "within 10 years" filter is not a filter, and sending one would exclude undated pages for nothing.
1851
+ */
1852
+ function recencyBucket(recencyDays) {
1853
+ if (typeof recencyDays !== "number" || !Number.isFinite(recencyDays) || recencyDays <= 0) return null;
1854
+ if (recencyDays <= 1) return "day";
1855
+ if (recencyDays <= 7) return "week";
1856
+ if (recencyDays <= 31) return "month";
1857
+ if (recencyDays <= 366) return "year";
1858
+ return null;
1859
+ }
1860
+ /** SerpAPI spells the buckets `qdr:d|w|m|y` on the `tbs` parameter. */
1861
+ const SERPAPI_QDR = {
1862
+ day: "qdr:d",
1863
+ week: "qdr:w",
1864
+ month: "qdr:m",
1865
+ year: "qdr:y"
1866
+ };
1840
1867
  const DEFAULT_NUM_RESULTS = 3;
1841
1868
  const SEARCH_TIMEOUT_MS = 6e4;
1842
1869
  /**
@@ -1845,14 +1872,14 @@ const SEARCH_TIMEOUT_MS = 6e4;
1845
1872
  * on a non-OK response so the tool surfaces the failure. Exported (re-exported from index) for the
1846
1873
  * REST endpoint and existing tests.
1847
1874
  */
1848
- async function serpApiSearch(adapters, query, num_results) {
1875
+ async function serpApiSearch(adapters, query, num_results, options) {
1849
1876
  const apiKey = await (0, apiKeyService_exports.getSerperKey)(adapters);
1850
1877
  const url = new URL("https://serpapi.com/search");
1851
1878
  if (!apiKey) {
1852
1879
  Logger.globalInstance.error("❌ WebSearch Tool: No API key configured. Skipping search.");
1853
1880
  return { organic_results: [] };
1854
1881
  }
1855
- url.search = new URLSearchParams({
1882
+ const searchParams = new URLSearchParams({
1856
1883
  engine: "google",
1857
1884
  api_key: apiKey,
1858
1885
  q: query,
@@ -1861,7 +1888,10 @@ async function serpApiSearch(adapters, query, num_results) {
1861
1888
  gl: "us",
1862
1889
  hl: "en",
1863
1890
  num: (num_results || DEFAULT_NUM_RESULTS).toString()
1864
- }).toString();
1891
+ });
1892
+ const bucket = recencyBucket(options?.recencyDays);
1893
+ if (bucket) searchParams.set("tbs", SERPAPI_QDR[bucket]);
1894
+ url.search = searchParams.toString();
1865
1895
  const controller = new AbortController();
1866
1896
  const timeoutId = setTimeout(() => controller.abort(), SEARCH_TIMEOUT_MS);
1867
1897
  let response;
@@ -1889,8 +1919,8 @@ async function serpApiSearch(adapters, query, num_results) {
1889
1919
  function createSerpApiProvider(adapters) {
1890
1920
  return {
1891
1921
  name: "serpapi",
1892
- async search(query, numResults) {
1893
- const data = await serpApiSearch(adapters, query, numResults);
1922
+ async search(query, numResults, options) {
1923
+ const data = await serpApiSearch(adapters, query, numResults, options);
1894
1924
  return (Array.isArray(data.organic_results) ? data.organic_results : []).filter((r) => !!r && typeof r.link === "string").map((r) => ({
1895
1925
  title: r.title ?? r.link,
1896
1926
  url: r.link,
@@ -1929,16 +1959,19 @@ function parseSearxngResults(data, numResults) {
1929
1959
  function createSearxngProvider(baseUrl) {
1930
1960
  return {
1931
1961
  name: "searxng",
1932
- async search(query, numResults) {
1962
+ async search(query, numResults, options) {
1933
1963
  const limit = numResults && numResults > 0 ? numResults : DEFAULT_NUM_RESULTS;
1934
1964
  const trimmed = baseUrl.replace(/\/+$/, "");
1935
1965
  const url = new URL(`${trimmed}/search`);
1936
- url.search = new URLSearchParams({
1966
+ const params = new URLSearchParams({
1937
1967
  q: query,
1938
1968
  format: "json",
1939
1969
  language: "en",
1940
1970
  safesearch: "1"
1941
- }).toString();
1971
+ });
1972
+ const bucket = recencyBucket(options?.recencyDays);
1973
+ if (bucket) params.set("time_range", bucket);
1974
+ url.search = params.toString();
1942
1975
  const controller = new AbortController();
1943
1976
  const timeoutId = setTimeout(() => controller.abort(), SEARCH_TIMEOUT_MS);
1944
1977
  try {
@@ -2061,7 +2094,7 @@ const webSearchTool = {
2061
2094
  })
2062
2095
  };
2063
2096
  //#endregion
2064
- //#region ../../b4m-core/services/dist/toolGenerators-D3QFkvc-.mjs
2097
+ //#region ../../b4m-core/services/dist/toolGenerators-DGjRmthM.mjs
2065
2098
  const diceRoll = async (parameters) => {
2066
2099
  if (!parameters?.sides || !parameters?.times) throw new Error("Tool dice roll: Missing required parameters");
2067
2100
  return sum(times(parameters.times, () => random(1, parameters.sides))).toString();
@@ -2629,6 +2662,27 @@ const promptEnhancementTool = {
2629
2662
  }
2630
2663
  })
2631
2664
  };
2665
+ /**
2666
+ * Is this value shaped like something Mongoose can cast to an `_id`?
2667
+ *
2668
+ * Tool arguments are composed by the model out of conversation text and reach us as unvalidated
2669
+ * JSON, so an id parameter routinely holds something that is not an id at all - a filename token,
2670
+ * an arXiv number, a bare integer. Mongoose casts `_id` and throws a CastError on those, which a
2671
+ * generic catch upstream then reports as an outage rather than the bad argument it is (#2530).
2672
+ * Call this before handing a model-supplied id to `findById` and answer a false the same way the
2673
+ * surface answers a genuinely missing row.
2674
+ *
2675
+ * `isObjectIdOrHexString`, not `isValidObjectId`: the latter also accepts a number and casts it to
2676
+ * a fabricated id, and a model emitting `{"file_id": 12}` gives us exactly that despite the
2677
+ * `string` type. Same choice, same reason, as `usableObjectIds` in @bike4mind/db-core, which is
2678
+ * the array-shaped version of this check.
2679
+ *
2680
+ * NOT usable for artifact ids (`artifact_<...>`), which are matched on a string `id` field rather
2681
+ * than `_id` - see `createArtifactId` in @bike4mind/common.
2682
+ */
2683
+ function isObjectIdShaped(id) {
2684
+ return isObjectIdOrHexString(id);
2685
+ }
2632
2686
  let _showUserQuestion = null;
2633
2687
  /**
2634
2688
  * Inject the CLI callback that displays the question UI.
@@ -2748,7 +2802,7 @@ const askUserQuestionTool = {
2748
2802
  * re-export them without pulling the full tool graph. `index.ts` re-exports them
2749
2803
  * so the server barrel's public API is unchanged.
2750
2804
  */
2751
- const generateTools = (userId, user, logger, { db, retrievalFilter, kbScope, inlinedAttachmentIds, fullyInlinedAttachmentIds, suppressLakeArms, sessionRetrievalTags, questId, getAbortSignal }, storage, imageGenerateStorage, statusUpdate, onStart, onFinish, llm, config, model, imageProcessorLambdaName, tools, allowedDirectories, entitlementKeys = [], sessionId, codeMinifier, availableModels, onToolLlmUsage) => {
2805
+ const generateTools = (userId, user, logger, { db, retrievalFilter, kbScope, inlinedAttachmentIds, fullyInlinedAttachmentIds, suppressLakeArms, sessionRetrievalTags, sessionPreauthorizedLakeIds, questId, getAbortSignal }, storage, imageGenerateStorage, statusUpdate, onStart, onFinish, llm, config, model, imageProcessorLambdaName, tools, allowedDirectories, entitlementKeys = [], sessionId, codeMinifier, availableModels, onToolLlmUsage) => {
2752
2806
  const context = {
2753
2807
  userId,
2754
2808
  user,
@@ -2772,6 +2826,7 @@ const generateTools = (userId, user, logger, { db, retrievalFilter, kbScope, inl
2772
2826
  fullyInlinedAttachmentIds,
2773
2827
  suppressLakeArms,
2774
2828
  sessionRetrievalTags,
2829
+ sessionPreauthorizedLakeIds,
2775
2830
  codeMinifier,
2776
2831
  availableModels,
2777
2832
  onToolLlmUsage,
@@ -4240,9 +4295,9 @@ const latticeAddEntityTool = {
4240
4295
  createdAt: /* @__PURE__ */ new Date(),
4241
4296
  updatedAt: /* @__PURE__ */ new Date()
4242
4297
  };
4243
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4298
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4244
4299
  const model = await context.db.latticeModels.findById(modelId);
4245
- if (model) {
4300
+ if (model && model.userId === context.userId) {
4246
4301
  const existingIndex = model.data.entities.findIndex((e) => e.id === entityId);
4247
4302
  if (existingIndex >= 0) model.data.entities[existingIndex] = entityData;
4248
4303
  else model.data.entities.push(entityData);
@@ -4252,7 +4307,23 @@ const latticeAddEntityTool = {
4252
4307
  updatedAt: /* @__PURE__ */ new Date()
4253
4308
  });
4254
4309
  context.logger.info(`[Lattice] Added entity ${entityId} to model ${modelId}`);
4255
- } else context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4310
+ } else if (model) {
4311
+ context.logger.warn(`[Lattice] Access denied: caller does not own model ${modelId}`);
4312
+ return JSON.stringify({
4313
+ success: false,
4314
+ action: "ADD_ENTITY",
4315
+ modelId,
4316
+ error: `Access denied: you do not have permission to modify model ${modelId}`
4317
+ });
4318
+ } else {
4319
+ context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4320
+ return JSON.stringify({
4321
+ success: false,
4322
+ action: "ADD_ENTITY",
4323
+ modelId,
4324
+ error: `Model ${modelId} not found`
4325
+ });
4326
+ }
4256
4327
  } catch (error) {
4257
4328
  context.logger.error(`[Lattice] Failed to persist entity to database:`, error);
4258
4329
  }
@@ -4368,9 +4439,9 @@ const latticeSetValueTool = {
4368
4439
  else if (rawValue.toLowerCase() === "true") value = true;
4369
4440
  else if (rawValue.toLowerCase() === "false") value = false;
4370
4441
  const entityId = entityName.toLowerCase().replace(/\s+/g, "_");
4371
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4442
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4372
4443
  const model = await context.db.latticeModels.findById(modelId);
4373
- if (model) {
4444
+ if (model && model.userId === context.userId) {
4374
4445
  const entity = model.data.entities.find((e) => e.id === entityId || e.name === entityName);
4375
4446
  if (entity) {
4376
4447
  const attrIndex = entity.attributes.findIndex((a) => a.key === attributeKey);
@@ -4390,7 +4461,23 @@ const latticeSetValueTool = {
4390
4461
  });
4391
4462
  context.logger.info(`[Lattice] Set ${entityId}.${attributeKey} = ${value} in model ${modelId}`);
4392
4463
  } else context.logger.warn(`[Lattice] Entity ${entityName} not found in model ${modelId}`);
4393
- } else context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4464
+ } else if (model) {
4465
+ context.logger.warn(`[Lattice] Access denied: caller does not own model ${modelId}`);
4466
+ return JSON.stringify({
4467
+ success: false,
4468
+ action: "SET_VALUE",
4469
+ modelId,
4470
+ error: `Access denied: you do not have permission to modify model ${modelId}`
4471
+ });
4472
+ } else {
4473
+ context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4474
+ return JSON.stringify({
4475
+ success: false,
4476
+ action: "SET_VALUE",
4477
+ modelId,
4478
+ error: `Model ${modelId} not found`
4479
+ });
4480
+ }
4394
4481
  } catch (error) {
4395
4482
  context.logger.error(`[Lattice] Failed to persist value to database:`, error);
4396
4483
  }
@@ -4487,9 +4574,9 @@ const latticeCreateRuleTool = {
4487
4574
  };
4488
4575
  const outputEntityId = parsedRule.outputEntity.toLowerCase().replace(/\s+/g, "_");
4489
4576
  let entityCreatedMessage = "";
4490
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4577
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4491
4578
  const model = await context.db.latticeModels.findById(modelId);
4492
- if (model) {
4579
+ if (model && model.userId === context.userId) {
4493
4580
  if (!model.data.entities.some((e) => e.id === outputEntityId || e.name.toLowerCase() === parsedRule.outputEntity.toLowerCase()) && parsedRule.outputEntity !== "unknown") {
4494
4581
  const now = /* @__PURE__ */ new Date();
4495
4582
  const newEntity = {
@@ -4526,7 +4613,23 @@ const latticeCreateRuleTool = {
4526
4613
  updatedAt: /* @__PURE__ */ new Date()
4527
4614
  });
4528
4615
  context.logger.info(`[Lattice] Created rule ${ruleId} in model ${modelId}`);
4529
- } else context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4616
+ } else if (model) {
4617
+ context.logger.warn(`[Lattice] Access denied: caller does not own model ${modelId}`);
4618
+ return JSON.stringify({
4619
+ success: false,
4620
+ action: "CREATE_RULE",
4621
+ modelId,
4622
+ error: `Access denied: you do not have permission to modify model ${modelId}`
4623
+ });
4624
+ } else {
4625
+ context.logger.warn(`[Lattice] Model ${modelId} not found in database`);
4626
+ return JSON.stringify({
4627
+ success: false,
4628
+ action: "CREATE_RULE",
4629
+ modelId,
4630
+ error: `Model ${modelId} not found`
4631
+ });
4632
+ }
4530
4633
  } catch (error) {
4531
4634
  context.logger.error(`[Lattice] Failed to persist rule to database:`, error);
4532
4635
  }
@@ -6413,6 +6516,130 @@ var EmbeddingAuthError = class extends Error {
6413
6516
  this.provider = provider;
6414
6517
  }
6415
6518
  };
6519
+ /**
6520
+ * Passive reporting of embedding-provider rate-limit ceilings.
6521
+ *
6522
+ * A rate limit belongs to the provider organization behind the key, and every embedding response
6523
+ * already carries the ceiling in its headers, so reading them costs no extra request and no extra
6524
+ * tokens. Providers that do not report them (Bedrock, Ollama) simply never produce an observation.
6525
+ *
6526
+ * This module owns the "what did the provider say" half only. It has no opinion about data lakes
6527
+ * or about the throughput levers configured against these numbers; interpreting a ceiling against
6528
+ * a lever belongs to the layer that knows what the levers are.
6529
+ */
6530
+ /**
6531
+ * Remaining/limit ratio at or below which the provider counts as under pressure. A bulk re-index
6532
+ * draws its window down steadily, so this sits low enough that ordinary throughput does not trip
6533
+ * it and only genuine starvation does.
6534
+ */
6535
+ const PRESSURE_RATIO = .1;
6536
+ /**
6537
+ * Pressure lasts as long as the window does, and every call in that window reports it. Throttle to
6538
+ * one line per interval so a starved ingest leaves a readable trace instead of flooding the log.
6539
+ */
6540
+ const PRESSURE_LOG_INTERVAL_MS = 6e4;
6541
+ /**
6542
+ * Process-local, and deliberately so: a cold start re-reports what it measures rather than leaving
6543
+ * a gap shared storage would have to close. Keyed by provider+model+account, so the map is bounded
6544
+ * by the model list times the number of distinct credentials the process serves.
6545
+ */
6546
+ const stateByKey = /* @__PURE__ */ new Map();
6547
+ /**
6548
+ * A broken reporter is indistinguishable from a steady ceiling - both are silence - so the first
6549
+ * fault has to be loud. Subsequent ones drop to debug: whatever breaks here breaks on every
6550
+ * embedding call, and a bulk re-index would drown the log in it.
6551
+ */
6552
+ let hasReportedFailure = false;
6553
+ const keyFor = (provider, model, account) => `${provider}:${model}:${account}`;
6554
+ const ceilingChanged = (previous, next) => previous.limitTokens !== next.limitTokens || previous.limitRequests !== next.limitRequests;
6555
+ const describeCeiling = (snapshot) => `${snapshot.limitTokens ?? "unreported"} tokens/min, ${snapshot.limitRequests ?? "unreported"} requests/min`;
6556
+ /** Ratio of the window still available, or null when the provider did not report that dimension. */
6557
+ const remainingRatio = (remaining, limit) => {
6558
+ if (remaining === null || limit === null || limit <= 0) return null;
6559
+ return remaining / limit;
6560
+ };
6561
+ const pressuredDimensions = (snapshot) => {
6562
+ const tokens = remainingRatio(snapshot.remainingTokens, snapshot.limitTokens);
6563
+ const requests = remainingRatio(snapshot.remainingRequests, snapshot.limitRequests);
6564
+ const dimensions = [];
6565
+ if (tokens !== null && tokens <= PRESSURE_RATIO) dimensions.push("tokens");
6566
+ if (requests !== null && requests <= PRESSURE_RATIO) dimensions.push("requests");
6567
+ return dimensions;
6568
+ };
6569
+ /**
6570
+ * Read the rate-limit headers off an embedding response and report the ceiling when it is worth
6571
+ * reporting: the first sighting in this process, a change since the last sighting, or the window
6572
+ * running down. Returns the observation when the provider reported a usable ceiling, else null.
6573
+ *
6574
+ * `account` identifies the provider account the reading belongs to and is part of the memo key,
6575
+ * not just the log line. The credential is resolved per user - a stored personal key beats the
6576
+ * platform key in `getEffectiveLLMApiKeys` - so one process can see several accounts on the same
6577
+ * provider+model. Without the discriminator their readings would collapse into one entry that
6578
+ * flaps between unrelated ceilings and attributes each figure to whoever reads the log next. The
6579
+ * caller supplies it; it must never be key material.
6580
+ *
6581
+ * Never throws. This hangs off the hot path of every embedding call, and a reporting fault must
6582
+ * not be able to fail an embedding that otherwise succeeded.
6583
+ */
6584
+ function recordEmbeddingRateLimitHeaders(provider, model, account, headers, now = Date.now()) {
6585
+ try {
6586
+ const snapshot = parseEmbeddingRateLimitHeaders(headers);
6587
+ if (!hasUsableLimits(snapshot)) return null;
6588
+ const key = keyFor(provider, model, account);
6589
+ const previous = stateByKey.get(key);
6590
+ const observation = {
6591
+ provider,
6592
+ model,
6593
+ account,
6594
+ snapshot,
6595
+ observedAt: now
6596
+ };
6597
+ const subject = `${provider} ${model} (account ${account})`;
6598
+ if (!previous) Logger.globalInstance.info(`[embedding-limits] ${subject} ceiling measured: ${describeCeiling(snapshot)}`, {
6599
+ provider,
6600
+ model,
6601
+ account,
6602
+ limitTokens: snapshot.limitTokens,
6603
+ limitRequests: snapshot.limitRequests
6604
+ });
6605
+ else if (ceilingChanged(previous.last.snapshot, snapshot)) Logger.globalInstance.warn(`[embedding-limits] ${subject} ceiling CHANGED: was ${describeCeiling(previous.last.snapshot)}, now ${describeCeiling(snapshot)}. Reconcile any throughput lever governed by this account against the new figure.`, {
6606
+ provider,
6607
+ model,
6608
+ account,
6609
+ previousLimitTokens: previous.last.snapshot.limitTokens,
6610
+ previousLimitRequests: previous.last.snapshot.limitRequests,
6611
+ limitTokens: snapshot.limitTokens,
6612
+ limitRequests: snapshot.limitRequests
6613
+ });
6614
+ const pressured = pressuredDimensions(snapshot);
6615
+ const dueForPressureLog = previous?.lastPressureLogAt == null || now - previous.lastPressureLogAt >= PRESSURE_LOG_INTERVAL_MS;
6616
+ const logPressure = pressured.length > 0 && dueForPressureLog;
6617
+ if (logPressure) Logger.globalInstance.warn(`[embedding-limits] ${subject} is at or below ${PRESSURE_RATIO * 100}% of its ${pressured.join(" and ")} window`, {
6618
+ provider,
6619
+ model,
6620
+ account,
6621
+ remainingTokens: snapshot.remainingTokens,
6622
+ remainingRequests: snapshot.remainingRequests,
6623
+ limitTokens: snapshot.limitTokens,
6624
+ limitRequests: snapshot.limitRequests,
6625
+ resetTokensMs: snapshot.resetTokensMs,
6626
+ resetRequestsMs: snapshot.resetRequestsMs
6627
+ });
6628
+ stateByKey.set(key, {
6629
+ last: observation,
6630
+ lastPressureLogAt: logPressure ? now : previous?.lastPressureLogAt ?? null
6631
+ });
6632
+ return observation;
6633
+ } catch (error) {
6634
+ const message = `[embedding-limits] failed to record rate-limit headers: ${error}`;
6635
+ if (hasReportedFailure) Logger.globalInstance.debug(message);
6636
+ else {
6637
+ hasReportedFailure = true;
6638
+ Logger.globalInstance.warn(message);
6639
+ }
6640
+ return null;
6641
+ }
6642
+ }
6416
6643
  const OPENAI_EMBEDDING_MODEL_MAP = {
6417
6644
  [OpenAIEmbeddingModel.TEXT_EMBEDDING_3_SMALL]: {
6418
6645
  provider: "OpenAI",
@@ -6433,34 +6660,72 @@ const OPENAI_EMBEDDING_MODEL_MAP = {
6433
6660
  dimensions: [1536]
6434
6661
  }
6435
6662
  };
6436
- var OpenAIEmbeddingService = class OpenAIEmbeddingService {
6663
+ /**
6664
+ * Non-reversible stand-in for a credential, for use where two accounts have to be told apart in a
6665
+ * log. Same construction as the API-key logging hash in the request middleware. Never emit the key.
6666
+ */
6667
+ const fingerprintCredential = (apiKey) => `key:${createHash("sha256").update(apiKey).digest("hex").slice(0, 16)}`;
6668
+ /**
6669
+ * Total by construction. The only caller runs inside processSingleBatch's classifying try, where a
6670
+ * throw would be misread as a provider error and re-issue the batch.
6671
+ */
6672
+ const headerOrNull = (httpResponse, name) => {
6673
+ try {
6674
+ return httpResponse.headers?.get(name) ?? null;
6675
+ } catch {
6676
+ return null;
6677
+ }
6678
+ };
6679
+ /**
6680
+ * The ceilings `generateEmbeddingBatch` splits on, at module scope and exported because a cost
6681
+ * PREFLIGHT has to model the same split before it spends (packages/scripts/retrieval/capturePlan.ts).
6682
+ * A second copy of these numbers in a script cannot track a provider change.
6683
+ */
6684
+ const OPENAI_MAX_INPUTS_PER_REQUEST = 2048;
6685
+ const OPENAI_MAX_TOKENS_PER_INPUT = 8192;
6686
+ /**
6687
+ * Effective token limit with a 10% safety buffer.
6688
+ * The tiktoken fallback (text.length/3) deliberately overestimates to be safe,
6689
+ * but DB token counts may have been produced by a different tokenizer (Bedrock, Voyage)
6690
+ * that underestimates. The buffer keeps us clear of the hard limit under tokenizer variance.
6691
+ */
6692
+ const OPENAI_EFFECTIVE_TOKEN_LIMIT = Math.floor(27e4);
6693
+ var OpenAIEmbeddingService = class {
6437
6694
  client;
6438
6695
  model;
6439
- /** Hard limit imposed by OpenAI's embeddings API. */
6440
- static MAX_TOKENS_PER_REQUEST = 3e5;
6441
- /**
6442
- * Effective token limit with a 10% safety buffer.
6443
- * The tiktoken fallback (text.length/3) deliberately overestimates to be safe,
6444
- * but DB token counts may have been produced by a different tokenizer (Bedrock, Voyage)
6445
- * that underestimates. The buffer keeps us clear of the hard limit under tokenizer variance.
6446
- */
6447
- static EFFECTIVE_TOKEN_LIMIT = Math.floor(OpenAIEmbeddingService.MAX_TOKENS_PER_REQUEST * .9);
6696
+ credentialFingerprint;
6448
6697
  constructor(apiKey, model = OpenAIEmbeddingModel.TEXT_EMBEDDING_ADA_002) {
6449
6698
  this.client = new OpenAI({ apiKey });
6450
6699
  this.validateModel(model);
6451
6700
  this.model = model;
6701
+ this.credentialFingerprint = fingerprintCredential(apiKey);
6702
+ }
6703
+ /**
6704
+ * Report the provider ceiling carried on a response we already received. Covers ingest and
6705
+ * query alike: both reach OpenAI through this class, so neither needs its own sampling point.
6706
+ *
6707
+ * The ceiling belongs to the organization behind the key, and the key is resolved per user
6708
+ * (getEffectiveLLMApiKeys prefers a stored personal key over the platform one), so the reading
6709
+ * has to say whose it is. `openai-organization` is the provider's own answer to that; the
6710
+ * credential fingerprint covers the case where the response omits it, and still keeps two
6711
+ * distinct keys as two readings rather than one that flaps between them.
6712
+ */
6713
+ recordRateLimit(httpResponse) {
6714
+ const account = headerOrNull(httpResponse, "openai-organization") || this.credentialFingerprint;
6715
+ recordEmbeddingRateLimitHeaders("OpenAI", this.model, account, httpResponse.headers);
6452
6716
  }
6453
6717
  validateModel(model) {
6454
6718
  if (!OPENAI_EMBEDDING_MODEL_MAP[model]) throw new Error(`Invalid OpenAI embedding model: ${model}`);
6455
6719
  }
6456
6720
  async generateEmbedding(text) {
6457
- const response = await this.client.embeddings.create({
6721
+ const { data: body, response: httpResponse } = await this.client.embeddings.create({
6458
6722
  model: this.model,
6459
6723
  input: text
6460
- }).catch((error) => {
6724
+ }).withResponse().catch((error) => {
6461
6725
  throw this.toActionableAuthError(error);
6462
6726
  });
6463
- if (response.data && response.data.length > 0) return response.data[0].embedding;
6727
+ this.recordRateLimit(httpResponse);
6728
+ if (body.data && body.data.length > 0) return body.data[0].embedding;
6464
6729
  throw new Error("No embedding data received from OpenAI");
6465
6730
  }
6466
6731
  /**
@@ -6491,8 +6756,6 @@ var OpenAIEmbeddingService = class OpenAIEmbeddingService {
6491
6756
  */
6492
6757
  async generateEmbeddingBatch(texts, tokenCounts) {
6493
6758
  if (texts.length === 0) return [];
6494
- const MAX_INPUTS_PER_REQUEST = 2048;
6495
- const MAX_TOKENS_PER_INPUT = 8192;
6496
6759
  let tokens;
6497
6760
  let needsRecalculation = false;
6498
6761
  if (!tokenCounts || tokenCounts.length !== texts.length) needsRecalculation = true;
@@ -6507,12 +6770,12 @@ var OpenAIEmbeddingService = class OpenAIEmbeddingService {
6507
6770
  let totalTokens = 0;
6508
6771
  for (let i = 0; i < texts.length; i++) {
6509
6772
  const tokenCount = tokens[i];
6510
- if (tokenCount > MAX_TOKENS_PER_INPUT) throw new Error(`Input at index ${i} exceeds ${MAX_TOKENS_PER_INPUT} token limit (${tokenCount} tokens)`);
6773
+ if (tokenCount > 8192) throw new Error(`Input at index ${i} exceeds ${OPENAI_MAX_TOKENS_PER_INPUT} token limit (${tokenCount} tokens)`);
6511
6774
  totalTokens += tokenCount;
6512
6775
  }
6513
6776
  Logger.globalInstance.debug(`[OpenAI] Batch embedding: ${texts.length} inputs, ${totalTokens} total tokens`);
6514
- const batches = this.createBatches(texts, tokens, MAX_INPUTS_PER_REQUEST, OpenAIEmbeddingService.EFFECTIVE_TOKEN_LIMIT);
6515
- Logger.globalInstance.debug(`[OpenAI] Split into ${batches.length} batch(es) (effective limit: ${OpenAIEmbeddingService.EFFECTIVE_TOKEN_LIMIT} tokens)`);
6777
+ const batches = this.createBatches(texts, tokens, OPENAI_MAX_INPUTS_PER_REQUEST, OPENAI_EFFECTIVE_TOKEN_LIMIT);
6778
+ Logger.globalInstance.debug(`[OpenAI] Split into ${batches.length} batch(es) (effective limit: ${OPENAI_EFFECTIVE_TOKEN_LIMIT} tokens)`);
6516
6779
  if (batches.length === 1) return await this.processSingleBatch(batches[0].texts);
6517
6780
  const allEmbeddings = new Array(texts.length);
6518
6781
  for (const batch of batches) (await this.processSingleBatch(batch.texts)).forEach((embedding, batchIndex) => {
@@ -6578,8 +6841,8 @@ var OpenAIEmbeddingService = class OpenAIEmbeddingService {
6578
6841
  async processSingleBatch(texts, preCalculatedTokens) {
6579
6842
  const tokenCounts = preCalculatedTokens || await this.calculateTokenCounts(texts);
6580
6843
  const batchTokens = tokenCounts.reduce((sum, count) => sum + count, 0);
6581
- if (batchTokens > OpenAIEmbeddingService.EFFECTIVE_TOKEN_LIMIT) {
6582
- Logger.globalInstance.warn(`[OpenAI] Batch exceeds effective token limit (${batchTokens}/${OpenAIEmbeddingService.EFFECTIVE_TOKEN_LIMIT} tokens), splitting recursively`);
6844
+ if (batchTokens > 27e4) {
6845
+ Logger.globalInstance.warn(`[OpenAI] Batch exceeds effective token limit (${batchTokens}/${OPENAI_EFFECTIVE_TOKEN_LIMIT} tokens), splitting recursively`);
6583
6846
  const mid = Math.ceil(texts.length / 2);
6584
6847
  const firstHalf = texts.slice(0, mid);
6585
6848
  const secondHalf = texts.slice(mid);
@@ -6588,13 +6851,14 @@ var OpenAIEmbeddingService = class OpenAIEmbeddingService {
6588
6851
  const [firstEmbeddings, secondEmbeddings] = await Promise.all([this.processSingleBatch(firstHalf, firstTokens), this.processSingleBatch(secondHalf, secondTokens)]);
6589
6852
  return [...firstEmbeddings, ...secondEmbeddings];
6590
6853
  }
6591
- for (let i = 0; i < tokenCounts.length; i++) if (tokenCounts[i] > 8192) throw new Error(`Text at index ${i} exceeds OpenAI's 8192 token limit per input (${tokenCounts[i]} tokens). This indicates a data integrity issue - chunk should have been smaller. This chunk cannot be processed and the entire batch must fail.`);
6854
+ for (let i = 0; i < tokenCounts.length; i++) if (tokenCounts[i] > 8192) throw new Error(`Text at index ${i} exceeds OpenAI's ${OPENAI_MAX_TOKENS_PER_INPUT} token limit per input (${tokenCounts[i]} tokens). This indicates a data integrity issue - chunk should have been smaller. This chunk cannot be processed and the entire batch must fail.`);
6592
6855
  try {
6593
- const response = await this.client.embeddings.create({
6856
+ const { data: body, response: httpResponse } = await this.client.embeddings.create({
6594
6857
  model: this.model,
6595
6858
  input: texts
6596
- });
6597
- if (response.data && response.data.length > 0) return response.data.sort((a, b) => a.index - b.index).map((item) => item.embedding);
6859
+ }).withResponse();
6860
+ this.recordRateLimit(httpResponse);
6861
+ if (body.data && body.data.length > 0) return body.data.sort((a, b) => a.index - b.index).map((item) => item.embedding);
6598
6862
  else throw new Error("No embedding data received from OpenAI");
6599
6863
  } catch (error) {
6600
6864
  if (error instanceof OpenAI.AuthenticationError) throw this.toActionableAuthError(error);
@@ -6993,7 +7257,25 @@ const getProviderFromModel = (modelName) => {
6993
7257
  * ("OpenAI rejected the embedding request") instead of the actionable missing-credential path.
6994
7258
  */
6995
7259
  const EXPIRED_KEY_SENTINEL = "expired";
6996
- const usableKey = (value) => value && value !== EXPIRED_KEY_SENTINEL ? value : null;
7260
+ /**
7261
+ * A placeholder is rejected for the same reason the sentinel is, and the two sibling answers to
7262
+ * "is this key usable" both already do it (modelDiscoveryService/credentials.ts,
7263
+ * toolAvailability.ts, and defaultEmbeddingModelForEnv's own key test). Keeping a placeholder here
7264
+ * would report `missing: null`, so the keyless fallback would never fire and EmbeddingFactory would
7265
+ * then throw on the placeholder itself - the PR's headline case failing silently rather than
7266
+ * substituting. `.trim()` because a whitespace-only value is no key either.
7267
+ */
7268
+ const usableKey = (value) => {
7269
+ const trimmed = value?.trim();
7270
+ if (!trimmed || trimmed === EXPIRED_KEY_SENTINEL || isPlaceholderApiKey(trimmed)) return null;
7271
+ return trimmed;
7272
+ };
7273
+ /**
7274
+ * The slot is missing because THIS CALLER's key expired, not because the deployment holds none.
7275
+ * Bedrock has no credential and Ollama's base URL carries no expiry, so only the two keyed cloud
7276
+ * providers can be in this state. See the keyless-fallback doc comment for why it matters.
7277
+ */
7278
+ const isExpiredCallerKey = (missing, keyTable) => missing === "openai" && keyTable?.openai === EXPIRED_KEY_SENTINEL || missing === "voyageai" && keyTable?.voyageai === EXPIRED_KEY_SENTINEL;
6997
7279
  /**
6998
7280
  * Map an embedding provider plus the caller's resolved key table to the config
6999
7281
  * `EmbeddingFactory` expects, and report which credential is missing if any.
@@ -7013,6 +7295,13 @@ const usableKey = (value) => value && value !== EXPIRED_KEY_SENTINEL ? value : n
7013
7295
  *
7014
7296
  * Adding a provider means editing this function and its table test, not auditing
7015
7297
  * every call site.
7298
+ *
7299
+ * `keyTable` is always an ANSWER about the caller's credentials, never a failure channel. `null` /
7300
+ * `undefined` mean "resolved: this caller holds none", and both this function and the keyless
7301
+ * fallback below act on that - substituting the keyless embedder is a real decision with a real
7302
+ * vector space attached. A caller whose own key lookup THREW must therefore not pass the failure in
7303
+ * here; it has to report unknown instead, or an unavailable Mongo becomes a confident Titan on a
7304
+ * fully keyed production stage.
7016
7305
  */
7017
7306
  function resolveEmbeddingConfig(provider, keyTable) {
7018
7307
  switch (provider) {
@@ -7036,19 +7325,78 @@ function resolveEmbeddingConfig(provider, keyTable) {
7036
7325
  missing: "voyageai"
7037
7326
  };
7038
7327
  }
7039
- case ModelBackend.Ollama: return keyTable?.ollama ? {
7040
- config: { ollamaBaseUrl: keyTable.ollama },
7041
- missing: null
7042
- } : {
7043
- config: {},
7044
- missing: "ollama"
7045
- };
7328
+ case ModelBackend.Ollama: {
7329
+ const baseUrl = keyTable?.ollama?.trim();
7330
+ return baseUrl ? {
7331
+ config: { ollamaBaseUrl: baseUrl },
7332
+ missing: null
7333
+ } : {
7334
+ config: {},
7335
+ missing: "ollama"
7336
+ };
7337
+ }
7046
7338
  case ModelBackend.Bedrock: return {
7047
7339
  config: {},
7048
7340
  missing: null
7049
7341
  };
7050
7342
  }
7051
7343
  }
7344
+ /**
7345
+ * Resolve a config for `model`, falling back to keyless Bedrock when this deployment holds no
7346
+ * credential for the provider `model` needs but can reach Bedrock with its own AWS role.
7347
+ *
7348
+ * WHY THIS EXISTS HERE and not in `defaultEmbeddingModelForEnv`: "does this deployment have a
7349
+ * cloud embedding key" is unanswerable from process.env on a hosted stage - an SST secret arrives
7350
+ * as a linked Resource, so OPENAI_API_KEY is absent on production exactly as it is on a preview.
7351
+ * The key table passed in here is the first point that actually knows, which is why the decision
7352
+ * belongs at this seam.
7353
+ *
7354
+ * Related to but NOT the same as EmbeddingFactory.getDefaultEmbeddingModel, which ranks providers
7355
+ * from scratch (OpenAI > VoyageAI > Ollama > Bedrock). This keeps the model the admin asked for
7356
+ * whenever it is reachable and only substitutes the keyless one otherwise - so a deployment
7357
+ * holding only a Voyage key still falls back to Bedrock here, where the factory would pick
7358
+ * voyage-3. Deliberate: this is a reachability backstop, not a second opinion on the setting.
7359
+ *
7360
+ * ONLY FOR CALLERS FREE TO CHOOSE THE MODEL - i.e. the model came from the `defaultEmbeddingModel`
7361
+ * admin setting. A caller that must hit one specific vector space MUST keep using
7362
+ * `resolveEmbeddingConfig` and fail, because a fallback there would silently compare or write
7363
+ * across incompatible spaces:
7364
+ * - V2 mementos are pinned to MEMENTO_EMBEDDING_MODEL at 512 truncated dims (see embedding.ts);
7365
+ * - V1 mementos (mementoEmbedding.ts, getRelevantMementos.ts) read the admin default and so LOOK
7366
+ * free to choose, but neither live write path stamps `Memento.embeddingModel` - only the
7367
+ * reembedMementos backfill does. Their vectors are ranked by in-process cosine with no width
7368
+ * guard and no Atlas index, so a substitution here would drop 1024-dim vectors into a field
7369
+ * holding 1536-dim ones with nothing recording which is which, and nothing able to tell them
7370
+ * apart afterwards. Stamping V1 is the prerequisite for including it, not this helper.
7371
+ * - alternateModelAnn embeds one query per model bucket to match each chunk's recorded stamp.
7372
+ *
7373
+ * Returns the model actually used, so callers stamp what they embedded with rather than what they
7374
+ * asked for - that is what keeps `fabFileChunk`'s recorded `embeddingModel` honest.
7375
+ *
7376
+ * TWO credential states are deliberately NOT treated as "this deployment is keyless":
7377
+ * - `missing: 'ollama'` - a self-host that set no OLLAMA_BASE_URL has no AWS role either, and
7378
+ * OPENAI_KEY_MISSING_MESSAGE naming OPENAI_API_KEY / OLLAMA_BASE_URL is the actionable error
7379
+ * there. `hasKeylessCloudEmbedder()` already excludes self-host; this is belt-and-braces.
7380
+ * - an EXPIRED caller key. `getEffectiveLLMApiKeys` returns the `'expired'` sentinel instead of
7381
+ * falling through to the platform demo key, deliberately, so the user is told their key
7382
+ * expired rather than silently moved onto the platform's (see the reasoning in the
7383
+ * reactivate-collateral-deactivated-api-keys migration). `usableKey` normalizes that to null
7384
+ * for the CREDENTIAL check, which is right - but read as "this deployment holds no key" it
7385
+ * would substitute Titan for that one caller on keyed production, querying a vector space the
7386
+ * corpus was never written in. The deployment's own key state is unchanged by one expiry, so
7387
+ * the requested model is returned and the actionable expired-key error stands.
7388
+ */
7389
+ function resolveEmbeddingWithKeylessFallback(model, keyTable) {
7390
+ const resolved = resolveEmbeddingConfig(getProviderFromModel(model), keyTable);
7391
+ if (!resolved.missing || resolved.missing === "ollama" || isExpiredCallerKey(resolved.missing, keyTable) || !hasKeylessCloudEmbedder()) return {
7392
+ ...resolved,
7393
+ model
7394
+ };
7395
+ return {
7396
+ ...resolveEmbeddingConfig(ModelBackend.Bedrock, null),
7397
+ model: BedrockEmbeddingModel.TITAN_TEXT_EMBEDDINGS_V2
7398
+ };
7399
+ }
7052
7400
  const ChunkSchema = z$1.object({
7053
7401
  text: z$1.string(),
7054
7402
  tokenCount: z$1.number()
@@ -8116,6 +8464,50 @@ async function fetchWithoutRedirects(url, timeoutMs) {
8116
8464
  validateStatus: (status) => status >= 200 && status < 300 || status >= 300 && status < 400
8117
8465
  });
8118
8466
  }
8467
+ const BLOCK_LEVEL_SELECTOR = `*:not(${"a, span, em, strong, b, i, u, code, kbd, samp, var, sub, sup, small, abbr, cite, q, time, mark, s, del, ins, bdi, bdo, wbr, ruby, rt, rp".split(", ").join("):not(")}):not(td):not(th)`;
8468
+ /**
8469
+ * Extract readable text from the WHOLE document, not just `<p>` elements. The single collector
8470
+ * this replaced was `<p>`-only and fell back to the raw HTML when it found none: on a page whose
8471
+ * content isn't inside `<p>` (an RFC page using `<pre>`) that meant the fallback fired and stored
8472
+ * markup verbatim; on a page with real substance in headings, list items, table cells or code
8473
+ * blocks alongside its `<p>`s, that content was silently dropped.
8474
+ *
8475
+ * `head` (title/meta/script/style all live there, and the caller already reads `<title>`
8476
+ * separately) plus any stray `script`/`style`/`noscript` outside it are removed before extraction,
8477
+ * so none of that reaches what gets embedded. `<pre>` content is pulled out and stashed BEFORE the
8478
+ * rest of the document is collapsed, and spliced back in verbatim afterward - it needs to skip the
8479
+ * whitespace-collapse below (a code block's leading-space indentation is meaningful, unlike prose
8480
+ * whitespace) but still needs to land in the right place relative to everything else. Table cells
8481
+ * get a trailing space (still the same row, but no longer jammed into the next cell's word); every
8482
+ * other block-level element gets a trailing newline; runs of whitespace and blank lines are then
8483
+ * collapsed. Returns `''` when nothing extractable was found, so the caller stores nothing rather
8484
+ * than falling back to raw HTML.
8485
+ */
8486
+ function extractReadableText($) {
8487
+ $("head, script, style, noscript").remove();
8488
+ $("br").replaceWith("\n");
8489
+ const nonce = Math.random().toString(36).slice(2) + Date.now().toString(36);
8490
+ const markerFor = (index) => `\uE000PRE${nonce}_${index}\uE000`;
8491
+ const markerPattern = new RegExp(`\\uE000PRE${nonce}_(\\d+)\\uE000`, "g");
8492
+ const preBlocks = [];
8493
+ $("pre").each((_index, element) => {
8494
+ const text = $(element).text();
8495
+ if (text) {
8496
+ preBlocks.push(text);
8497
+ $(element).replaceWith(`${markerFor(preBlocks.length - 1)}\n`);
8498
+ } else $(element).remove();
8499
+ });
8500
+ $("td, th").each((_index, cell) => {
8501
+ $(cell).after(" ");
8502
+ });
8503
+ $(BLOCK_LEVEL_SELECTOR).each((_index, element) => {
8504
+ $(element).after("\n");
8505
+ });
8506
+ return $.root().text().split("\n").map((line) => line.replace(/[ \t]+/g, " ").trim()).filter(Boolean).join("\n").replace(markerPattern, (match, indexStr) => {
8507
+ const index = Number(indexStr);
8508
+ return index >= 0 && index < preBlocks.length ? preBlocks[index] : match;
8509
+ });
8510
+ }
8119
8511
  async function fetchAndParseURL(url, { logger }) {
8120
8512
  logger.updateMetadata({ failedUrl: null });
8121
8513
  try {
@@ -8149,16 +8541,13 @@ async function fetchAndParseURL(url, { logger }) {
8149
8541
  const htmlContent = body.toString("utf8");
8150
8542
  const $ = cheerio.load(htmlContent);
8151
8543
  title = $("title").text() || lastPathSegment(currentUrl);
8152
- let textContent = "";
8153
- $("body").find("p").each((index, element) => {
8154
- textContent += $(element).text() + "\n";
8155
- });
8156
- urlContent = textContent || htmlContent;
8544
+ urlContent = extractReadableText($);
8157
8545
  }
8158
8546
  const original = redactUrlCredentials(url);
8159
8547
  const final = redactUrlCredentials(currentUrl);
8160
8548
  const fetched = original === final ? original : `${original} -> ${final}`;
8161
- logger.log(`Fetched ${title} with mimetype ${urlMimeType} and parsed ${fetched}`);
8549
+ if (urlContent === "") logger.log(`Fetched ${title} with mimetype ${urlMimeType} and parsed ${fetched}, but no extractable text was found`);
8550
+ else logger.log(`Fetched ${title} with mimetype ${urlMimeType} and parsed ${fetched}`);
8162
8551
  return {
8163
8552
  title,
8164
8553
  textContent: urlContent,
@@ -9821,16 +10210,16 @@ function parseSettingsHooks(settingsJson) {
9821
10210
  return null;
9822
10211
  }
9823
10212
  }
9824
- let cached;
10213
+ let cached$1;
9825
10214
  /**
9826
10215
  * Lazily-built process-hook singleton from `B4M_SETTINGS_JSON`. Returns null when
9827
10216
  * no hooks are configured, so call sites can `void getProcessHooks()?.fireStop()`.
9828
10217
  */
9829
10218
  function getProcessHooks() {
9830
- if (cached !== void 0) return cached;
10219
+ if (cached$1 !== void 0) return cached$1;
9831
10220
  const hooks = parseSettingsHooks(process.env.B4M_SETTINGS_JSON);
9832
- cached = hooks ? new ProcessHooks(hooks) : null;
9833
- return cached;
10221
+ cached$1 = hooks ? new ProcessHooks(hooks) : null;
10222
+ return cached$1;
9834
10223
  }
9835
10224
  //#endregion
9836
10225
  //#region src/agents/interactionModeClamp.ts
@@ -12901,10 +13290,117 @@ const vm = require('node:vm');
12901
13290
  const STDOUT_HEAD_BYTES = ${5e3};
12902
13291
  const STDOUT_TAIL_BYTES = ${2e3};
12903
13292
  const HARD_PER_LINE_BYTES = ${5e4};
13293
+ const MIRROR_TAIL_FLUSH_MS = ${100};
12904
13294
 
12905
13295
  let stdoutChunks = [];
12906
- let stdoutBytes = 0;
12907
13296
  let truncated = false;
13297
+ // --- Mirror state --------------------------------------------------------
13298
+ // Mirrors this run's stdout to the main thread as it is produced, in the same
13299
+ // head + marker + tail shape collectStdout() produces, so a retired run and a
13300
+ // completed one report the same thing by the same rule.
13301
+ //
13302
+ // The two halves are cost-bounded differently. The head is mirrored line by
13303
+ // line, so a chatty loop stops paying per line once the head is full. Past
13304
+ // that the tail is kept locally in a rolling window and posted on a timer, so
13305
+ // the message rate stops tracking the line rate entirely.
13306
+ let currentRunId = null;
13307
+ let mirroredHeadBytes = 0;
13308
+ let headMirrorFull = false;
13309
+ let tailChunks = [];
13310
+ let tailBytes = 0;
13311
+ let elidedBytes = 0;
13312
+ let tailFlushTimer = null;
13313
+
13314
+ /**
13315
+ * What the rolling tail may hold: whatever the head did not use of the same
13316
+ * HEAD + TAIL total collectStdout() reports within. A fixed TAIL budget made
13317
+ * the two disagree whenever the head came up short - a single 6KB first line
13318
+ * does not fit the head, so the mirror would have kept 2KB of a run that
13319
+ * collectStdout() reports whole, and called it truncated. mirroredHeadBytes
13320
+ * is frozen once the head is full, so this is stable for the rest of the run.
13321
+ */
13322
+ function tailBudget() {
13323
+ return STDOUT_HEAD_BYTES + STDOUT_TAIL_BYTES - mirroredHeadBytes;
13324
+ }
13325
+
13326
+ function postToMain(msg) {
13327
+ try { parentPort.postMessage(msg); } catch { /* worker being torn down; nothing to preserve */ }
13328
+ }
13329
+ function cancelTailFlush() {
13330
+ if (tailFlushTimer === null) return;
13331
+ clearTimeout(tailFlushTimer);
13332
+ tailFlushTimer = null;
13333
+ }
13334
+ function flushTail() {
13335
+ if (currentRunId === null || !headMirrorFull) return;
13336
+ const joined = tailChunks.join('\n');
13337
+ // The rolling window is trimmed line by line, so it can only exceed the
13338
+ // budget by holding ONE line longer than the whole budget. Slice to the
13339
+ // same last-N-chars rule collectStdout() uses, which both matches that
13340
+ // path and keeps the flush payload bounded - a guest printing 50KB lines
13341
+ // would otherwise re-send 50KB on every tick.
13342
+ const overflow = Math.max(0, joined.length - tailBudget());
13343
+ postToMain({
13344
+ type: 'stdoutTail',
13345
+ id: currentRunId,
13346
+ tail: overflow > 0 ? joined.slice(overflow) : joined,
13347
+ // Counted from what was actually DROPPED - lines the rolling window
13348
+ // evicted, plus whatever this payload's own slice cuts - rather than
13349
+ // derived from the byte totals. The derived form read zero on the first
13350
+ // flush by construction (every line was still in the head, so the
13351
+ // subtraction cancelled), which made "truncated" unreportable on exactly
13352
+ // the run the mirror exists for.
13353
+ elidedBytes: elidedBytes + overflow,
13354
+ });
13355
+ }
13356
+ function scheduleTailFlush() {
13357
+ if (tailFlushTimer !== null) return;
13358
+ tailFlushTimer = setTimeout(() => {
13359
+ tailFlushTimer = null;
13360
+ flushTail();
13361
+ }, MIRROR_TAIL_FLUSH_MS);
13362
+ }
13363
+ function mirrorLine(capped) {
13364
+ if (currentRunId === null) return;
13365
+ if (!headMirrorFull) {
13366
+ // Does THIS line fit, rather than "is the running total already over".
13367
+ // Both of the orderings tried before this were wrong in one direction
13368
+ // each: gating on the running total let one line of up to
13369
+ // HARD_PER_LINE_BYTES past a 5KB budget (mirrored head ~55KB, disagreeing
13370
+ // with collectStdout's head and with the "~7K chars" codeExecuteTool
13371
+ // advertises to the model), while adding first and checking after moved
13372
+ // the boundary but kept the crossing line in the head - so the tail was
13373
+ // still empty at the immediate flush below and a run killed right there
13374
+ // dropped the last line before the hang and reported itself complete.
13375
+ //
13376
+ // A fit check does both: the head stops at STDOUT_HEAD_BYTES exactly, and
13377
+ // the line that did not fit STARTS the tail, so the flush that fires on
13378
+ // this same call carries it.
13379
+ if (mirroredHeadBytes + capped.length + 1 <= STDOUT_HEAD_BYTES) {
13380
+ mirroredHeadBytes += capped.length + 1;
13381
+ postToMain({ type: 'stdout', id: currentRunId, chunk: capped });
13382
+ return;
13383
+ }
13384
+ headMirrorFull = true;
13385
+ tailChunks.push(capped);
13386
+ tailBytes += capped.length + 1;
13387
+ // Post once immediately: a run killed before the first timed flush would
13388
+ // otherwise report a short mirror as if it were complete.
13389
+ flushTail();
13390
+ return;
13391
+ }
13392
+ tailChunks.push(capped);
13393
+ tailBytes += capped.length + 1;
13394
+ // Never evict the only line held: a line larger than the whole budget is
13395
+ // still the last thing the run printed, which is what the mirror is for.
13396
+ const budget = tailBudget();
13397
+ while (tailBytes > budget && tailChunks.length > 1) {
13398
+ const dropped = tailChunks.shift();
13399
+ tailBytes -= dropped.length + 1;
13400
+ elidedBytes += dropped.length + 1;
13401
+ }
13402
+ scheduleTailFlush();
13403
+ }
12908
13404
  function captureLine(args) {
12909
13405
  const line = args.map(a => {
12910
13406
  if (typeof a === 'string') return a;
@@ -12916,7 +13412,7 @@ function captureLine(args) {
12916
13412
  ? line.slice(0, HARD_PER_LINE_BYTES) + ' [...line truncated]'
12917
13413
  : line;
12918
13414
  stdoutChunks.push(capped);
12919
- stdoutBytes += capped.length + 1;
13415
+ mirrorLine(capped);
12920
13416
  }
12921
13417
  function jsonReplacer(_k, v) {
12922
13418
  if (v instanceof Error) return { name: v.name, message: v.message };
@@ -13017,7 +13513,11 @@ parentPort.on('message', async (msg) => {
13017
13513
  }
13018
13514
  if (msg.type === 'runCode') {
13019
13515
  const t0 = Date.now();
13020
- stdoutChunks = []; stdoutBytes = 0; truncated = false;
13516
+ stdoutChunks = []; truncated = false;
13517
+ cancelTailFlush();
13518
+ currentRunId = msg.id;
13519
+ mirroredHeadBytes = 0; headMirrorFull = false;
13520
+ tailChunks = []; tailBytes = 0; elidedBytes = 0;
13021
13521
  let error = null;
13022
13522
  const wrapped = '(async () => {\n' + msg.code + '\n})()';
13023
13523
  try {
@@ -13029,6 +13529,11 @@ parentPort.on('message', async (msg) => {
13029
13529
  } catch (e) {
13030
13530
  error = serializeError(e);
13031
13531
  }
13532
+ // Stop mirroring before the authoritative result goes out, so a late
13533
+ // console.log from an abandoned continuation cannot attach to this run,
13534
+ // and a pending tail flush cannot land after it.
13535
+ currentRunId = null;
13536
+ cancelTailFlush();
13032
13537
  parentPort.postMessage({
13033
13538
  type: 'runResult',
13034
13539
  id: msg.id,
@@ -13042,6 +13547,17 @@ parentPort.on('message', async (msg) => {
13042
13547
  });
13043
13548
  `;
13044
13549
  String.raw`
13550
+ // Wrapped in an IIFE deliberately. A script's top-level const/let bind into the
13551
+ // context's SHARED global lexical scope (and its function declarations become
13552
+ // globalThis properties), so without this wrapper every bootstrap-local name is
13553
+ // directly referenceable by LLM-authored code run later in the same context:
13554
+ // __RealFunction('...')() walks straight around the codegen block below, and
13555
+ // __cap.applySync(...) / __cap.release() forges or permanently kills stdout
13556
+ // capture. Function scope keeps them unreachable. Note the leak is invisible to
13557
+ // listGlobals(), which reads Object.getOwnPropertyNames(globalThis) and never
13558
+ // saw the lexical bindings - so RESERVED_GLOBAL_NAMES cannot backstop it either.
13559
+ // Anything guest code IS meant to see is assigned onto globalThis explicitly.
13560
+ (function () {
13045
13561
  const __cap = _captureLine;
13046
13562
  const __callTool = _callTool;
13047
13563
  delete globalThis._captureLine;
@@ -13049,28 +13565,68 @@ delete globalThis._callTool;
13049
13565
 
13050
13566
  const HARD_PER_LINE_BYTES = ${5e4};
13051
13567
 
13568
+ // Every intrinsic the formatter below reaches for is captured HERE, while the
13569
+ // context is still pristine. Resolving \`args.map\` / \`.join\` / \`line.slice\`
13570
+ // at CALL time walks a prototype chain the guest owns, so one
13571
+ // \`Array.prototype.join = () => 'X'\` - deliberate, or an innocent polyfill -
13572
+ // forges every stdout line for the rest of the session, and the run still
13573
+ // reports error=null / truncated=false. That is the same integrity failure the
13574
+ // frozen \`console\` below exists to prevent, one level down: freezing the
13575
+ // binding is worthless if the formatter behind it is guest-reachable.
13576
+ const __stringify = JSON.stringify;
13577
+ const __String = String;
13578
+ const __apply = Reflect.apply;
13579
+ const __strSlice = String.prototype.slice;
13580
+
13052
13581
  function __jsonReplacer(_k, v) {
13053
13582
  if (v instanceof Error) return { name: v.name, message: v.message };
13054
13583
  if (typeof v === 'bigint') return v.toString() + 'n';
13055
13584
  return v;
13056
13585
  }
13586
+ // Indexed loop and \`+=\` rather than map/join: string concatenation is an
13587
+ // operator, not a lookup, so there is nothing here for the guest to replace.
13588
+ // What a guest CAN still steer is how its own values render - a \`toJSON\` or
13589
+ // \`toString\` on the object it passed - which is content it already owns, not
13590
+ // the channel.
13057
13591
  function __formatLine(args) {
13058
- const line = args.map(a => {
13059
- if (typeof a === 'string') return a;
13060
- if (a === undefined) return 'undefined';
13061
- if (a === null) return 'null';
13062
- try { return JSON.stringify(a, __jsonReplacer, 2); } catch { return String(a); }
13063
- }).join(' ');
13592
+ let line = '';
13593
+ for (let i = 0; i < args.length; i++) {
13594
+ if (i > 0) line += ' ';
13595
+ const a = args[i];
13596
+ if (typeof a === 'string') { line += a; continue; }
13597
+ if (a === undefined) { line += 'undefined'; continue; }
13598
+ if (a === null) { line += 'null'; continue; }
13599
+ try { line += __stringify(a, __jsonReplacer, 2); } catch { line += __String(a); }
13600
+ }
13064
13601
  return line.length > HARD_PER_LINE_BYTES
13065
- ? line.slice(0, HARD_PER_LINE_BYTES) + ' [...line truncated]'
13602
+ ? __apply(__strSlice, line, [0, HARD_PER_LINE_BYTES]) + ' [...line truncated]'
13066
13603
  : line;
13067
13604
  }
13068
- globalThis.console = {
13605
+ // stdout is the channel the HOST reports back as the run's observation, so its
13606
+ // integrity is ours, not the guest's. A plain assignment left \`console\`
13607
+ // writable and configurable: guest code could set globalThis.console = {log(){}}
13608
+ // (or just reassign console.log) and every later run in the session would come
13609
+ // back with stdout="" or forged lines, error=null, and a clean listGlobals().
13610
+ // Frozen object + non-writable, non-configurable property: the guest's
13611
+ // assignment is a silent no-op in sloppy mode and a TypeError under 'use
13612
+ // strict', and either way capture keeps working.
13613
+ //
13614
+ // The BINDING is what this protects, and the binding is only half of it: a
13615
+ // frozen console whose formatter resolved its intrinsics at call time would
13616
+ // still hand the guest every line. That half is closed above, where
13617
+ // __formatLine captures what it needs.
13618
+ const __console = Object.freeze({
13069
13619
  log: (...a) => __cap.applySync(undefined, [__formatLine(a)], { arguments: { copy: true } }),
13070
13620
  warn: (...a) => __cap.applySync(undefined, [__formatLine(a)], { arguments: { copy: true } }),
13071
13621
  error: (...a) => __cap.applySync(undefined, [__formatLine(a)], { arguments: { copy: true } }),
13072
13622
  info: (...a) => __cap.applySync(undefined, [__formatLine(a)], { arguments: { copy: true } }),
13073
- };
13623
+ });
13624
+ Object.defineProperty(globalThis, 'console', {
13625
+ value: __console,
13626
+ writable: false,
13627
+ configurable: false,
13628
+ enumerable: true,
13629
+ });
13074
13630
 
13075
13631
  // A bare isolate has no structuredClone (it's a host/web API, not a V8
13076
13632
  // intrinsic). The in-process + worker backends expose the *host's* real
@@ -13158,11 +13714,33 @@ for (const __Ctor of [__RealFunction, __AsyncFunction, __GeneratorFunction, __As
13158
13714
  globalThis.eval = __blockCodegen;
13159
13715
  globalThis.Function = __blockCodegen;
13160
13716
 
13717
+ // WebAssembly is removed, not stubbed. Its compile/instantiate promises never
13718
+ // settle inside an isolated-vm isolate (there is no host task runner to drive
13719
+ // them), so \`await WebAssembly.instantiate(...)\` is a one-line way for guest
13720
+ // code to park a run until the host deadline fires - and that deadline kills
13721
+ // the isolate, costing the whole session its sandbox. Deleting it turns that
13722
+ // into an immediate ReferenceError. It is also codegen-from-bytes, so it
13723
+ // belongs on the same side of the line as eval / Function anyway.
13724
+ delete globalThis.WebAssembly;
13725
+
13161
13726
  // Tool-stub registry. Each registered tool becomes a top-level async
13162
13727
  // function that round-trips through the host dispatcher and re-throws on
13163
13728
  // the { ok:false } envelope.
13729
+ //
13730
+ // Assigned to globalThis only so the constructor can lift a Reference to it;
13731
+ // the constructor deletes the global immediately afterwards and calls it
13732
+ // through that Reference forever after. It must NOT stay guest-reachable: a
13733
+ // guest could call __registerTools(['console']) to overwrite the frozen
13734
+ // console binding with a tool stub, or \`delete\` it and make the host's next
13735
+ // setTools() throw.
13736
+ //
13737
+ // Indexed loop, not for..of, deliberately: the host calls this with a copied
13738
+ // array whose iterator comes from the GUEST's Array.prototype, so an
13739
+ // overridden Symbol.iterator would let guest code hang or hijack a host-side
13740
+ // setTools() call. Indexing touches only the copy's own properties.
13164
13741
  globalThis.__registerTools = function (names) {
13165
- for (const name of names) {
13742
+ for (let i = 0; i < names.length; i++) {
13743
+ const name = names[i];
13166
13744
  globalThis[name] = async (...args) => {
13167
13745
  const envJson = await __callTool.apply(
13168
13746
  undefined,
@@ -13175,6 +13753,7 @@ globalThis.__registerTools = function (names) {
13175
13753
  };
13176
13754
  }
13177
13755
  };
13756
+ })();
13178
13757
  `;
13179
13758
  z$1.object({
13180
13759
  reflection: z$1.string().min(1),
@@ -15467,10 +16046,15 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15467
16046
  AnthropicBackend: () => AnthropicBackend,
15468
16047
  AnthropicBatchService: () => AnthropicBatchService,
15469
16048
  AnthropicBedrockBackend: () => AnthropicBedrockBackend,
16049
+ BEDROCK_REQUEST_HANDLER: () => BEDROCK_REQUEST_HANDLER,
15470
16050
  BFLBackend: () => BFLBackend,
15471
16051
  BaseBedrockBackend: () => BaseBedrockBackend,
15472
16052
  ChoiceEndReason: () => ChoiceEndReason,
15473
16053
  ChoiceStatus: () => ChoiceStatus,
16054
+ DEEPSEEK_EFFORT_LEVELS: () => DEEPSEEK_EFFORT_LEVELS,
16055
+ DEEPSEEK_MAX_STOP_SEQUENCES: () => 16,
16056
+ DEEPSEEK_MODELS: () => DEEPSEEK_MODELS,
16057
+ DEEPSEEK_THINKING_TOP_P_FLOOR: () => DEEPSEEK_THINKING_TOP_P_FLOOR,
15474
16058
  DEFAULT_MAX_TOOL_CALLS: () => 10,
15475
16059
  DEFAULT_REALTIME_VOICE_MODEL: () => DEFAULT_REALTIME_VOICE_MODEL,
15476
16060
  DEGENERATE_STREAM_MESSAGE: () => DEGENERATE_STREAM_MESSAGE,
@@ -15478,6 +16062,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15478
16062
  DEPRECATED_MODEL_MAP: () => DEPRECATED_MODEL_MAP,
15479
16063
  DEPRECATED_MODEL_REQUEST_METRIC: () => DEPRECATED_MODEL_REQUEST_METRIC,
15480
16064
  DISPATCHABLE_ADAPTER_FAMILIES: () => DISPATCHABLE_ADAPTER_FAMILIES,
16065
+ DeepSeekBackend: () => DeepSeekBackend,
15481
16066
  DeepSeekBedrockBackend: () => DeepSeekBedrockBackend,
15482
16067
  DispatchModel: () => DispatchModel,
15483
16068
  GeminiBackend: () => GeminiBackend,
@@ -15498,6 +16083,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15498
16083
  UndifferentiatedBedrockBackend: () => UndifferentiatedBedrockBackend,
15499
16084
  UnsupportedAdapterFamilyError: () => UnsupportedAdapterFamilyError,
15500
16085
  XAIBackend: () => XAIBackend,
16086
+ adapterPriceTiers: () => adapterPriceTiers,
15501
16087
  backendForAdapterFamily: () => backendForAdapterFamily,
15502
16088
  buildApiKeyTable: () => buildApiKeyTable,
15503
16089
  buildSupersededIndex: () => buildSupersededIndex,
@@ -15508,6 +16094,10 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15508
16094
  checkStaleModelReferences: () => checkStaleModelReferences,
15509
16095
  classifyModelReference: () => classifyModelReference,
15510
16096
  createDegenerateStreamGuard: () => createDegenerateStreamGuard,
16097
+ deepseekReasoningParams: () => deepseekReasoningParams,
16098
+ deepseekSamplingParams: () => deepseekSamplingParams,
16099
+ deepseekStopSequences: () => deepseekStopSequences,
16100
+ deepseekThinkingEnabled: () => deepseekThinkingEnabled,
15511
16101
  ensureToolPairingIntegrity: () => ensureToolPairingIntegrity,
15512
16102
  extractThinkContent: () => extractThinkContent,
15513
16103
  getAvailableModels: () => getAvailableModels,
@@ -15524,6 +16114,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15524
16114
  logExpiringModels: () => logExpiringModels,
15525
16115
  mergeCatalog: () => mergeCatalog,
15526
16116
  mergeCatalogWithDrops: () => mergeCatalogWithDrops,
16117
+ normalizeToolUseInputs: () => normalizeToolUseInputs,
15527
16118
  reasonsWithinOutputBudget: () => reasonsWithinOutputBudget,
15528
16119
  recordDeprecatedModelRequest: () => recordDeprecatedModelRequest,
15529
16120
  replaceLastToolResultObservationCanonical: () => replaceLastToolResultObservationCanonical,
@@ -15541,6 +16132,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15541
16132
  splitCacheInclusiveInput: () => splitCacheInclusiveInput,
15542
16133
  stripAllToolBlocks: () => stripAllToolBlocks,
15543
16134
  stripToolDependentMessages: () => stripToolDependentMessages,
16135
+ toDeepSeekEffort: () => toDeepSeekEffort,
15544
16136
  toKimiEffort: () => toKimiEffort,
15545
16137
  toProviderEndUserId: () => toProviderEndUserId,
15546
16138
  updateReplacedByOverlay: () => updateReplacedByOverlay
@@ -15898,27 +16490,112 @@ const stripAllToolBlocks = (messages, logger) => {
15898
16490
  return result;
15899
16491
  };
15900
16492
  /**
16493
+ * Restores `input: {}` on any `tool_use` block that reached us without one.
16494
+ *
16495
+ * `MessageContentToolUse.input` is non-optional in the type system, so nothing upstream checks it -
16496
+ * but a message that round-trips through a persistence or serialization layer can lose it. The
16497
+ * known offender is Mongoose's default `minimize`, which deletes empty objects on
16498
+ * `toObject()`/`toJSON()`: a zero-argument tool call (`current_datetime`, `mission_status`, ...)
16499
+ * stores `input: {}` and reads back with the key gone. Anthropic then rejects the whole request
16500
+ * with "messages.N.content.M.tool_use.input: Field required", killing a resumed agent run or a
16501
+ * chat turn that replays history.
16502
+ *
16503
+ * The schema that caused it is fixed at the source (`minimize: false` on AgentExecutionModel), so
16504
+ * this is the last line of defense for every other store that replays blocks verbatim -
16505
+ * `QuestModel.structuredReplies[].content` is the same Mixed-under-default-minimize shape and is
16506
+ * deliberately covered here rather than by widening that hot collection's schema. The cost of a
16507
+ * miss is a hard 400, and `{}` is the only value a zero-argument call could have had.
16508
+ *
16509
+ * Returns the input array unchanged (same reference) when nothing needed repair.
16510
+ */
16511
+ const normalizeToolUseInputs = (messages, logger) => {
16512
+ let repaired = 0;
16513
+ const result = messages.map((message) => {
16514
+ if (!Array.isArray(message.content)) return message;
16515
+ let messageChanged = false;
16516
+ const content = message.content.map((block) => {
16517
+ if (block.type !== "tool_use") return block;
16518
+ const toolUse = block;
16519
+ if (toolUse.input !== null && typeof toolUse.input === "object") return block;
16520
+ repaired++;
16521
+ messageChanged = true;
16522
+ return {
16523
+ ...toolUse,
16524
+ input: {}
16525
+ };
16526
+ });
16527
+ return messageChanged ? {
16528
+ ...message,
16529
+ content
16530
+ } : message;
16531
+ });
16532
+ if (repaired === 0) return messages;
16533
+ logger?.warn(`[Tool Input Repair] Restored empty input on ${repaired} tool_use block(s) that lost it in serialization`);
16534
+ return result;
16535
+ };
16536
+ /**
16537
+ * Anthropic's hard ceiling on `cache_control` markers per request. Exceeding it fails the
16538
+ * WHOLE request with `ValidationException: A maximum of 4 blocks with cache_control may be
16539
+ * provided`, which is non-retryable - so an over-budget request loses the turn outright,
16540
+ * after the user has already waited for it.
16541
+ */
16542
+ const MAX_CACHE_CONTROL_BLOCKS = 4;
16543
+ /**
16544
+ * Does this block already carry a marker? Re-marking one costs no budget.
16545
+ *
16546
+ * Tests the VALUE, not just key presence: a block carrying an explicit
16547
+ * `cache_control: undefined` is not a marker as far as the provider is concerned, and counting
16548
+ * it would spend budget on nothing and drop a breakpoint we could have kept.
16549
+ */
16550
+ function hasMarker(block) {
16551
+ return !!block && typeof block === "object" && !!block.cache_control;
16552
+ }
16553
+ /**
16554
+ * Markers already on the request. Callers upstream attach their own before this runs -
16555
+ * `bedrockBackend/anthropic.ts` marks each system block flagged `cache: true` (the mid-stack
16556
+ * shareable-prefix breakpoint) - so this adapter's budget is whatever they left, not the full four.
16557
+ */
16558
+ function censusMarkers(params) {
16559
+ const tools = Array.isArray(params.tools) ? params.tools.filter(hasMarker).length : 0;
16560
+ const system = Array.isArray(params.system) ? params.system.filter(hasMarker).length : 0;
16561
+ let messages = 0;
16562
+ if (Array.isArray(params.messages)) for (const message of params.messages) {
16563
+ const content = message?.content;
16564
+ if (Array.isArray(content)) messages += content.filter(hasMarker).length;
16565
+ }
16566
+ return {
16567
+ tools,
16568
+ system,
16569
+ messages,
16570
+ total: tools + system + messages
16571
+ };
16572
+ }
16573
+ /**
15901
16574
  * Anthropic-specific caching adapter
15902
16575
  * Adds explicit cache_control markers to content blocks
15903
16576
  */
15904
16577
  var AnthropicCachingAdapter = class {
15905
- applyCaching(apiParams, strategy) {
16578
+ applyCaching(apiParams, strategy, logger) {
15906
16579
  if (!strategy.enableCaching) return apiParams;
15907
16580
  const ttl = strategy.cacheTTL ?? "5m";
15908
16581
  const modifiedParams = { ...apiParams };
15909
- const tools = modifiedParams.tools;
15910
- if (strategy.cacheTools && Array.isArray(tools) && tools.length > 0) {
15911
- const toolsCopy = [...tools];
15912
- const lastTool = toolsCopy[toolsCopy.length - 1];
15913
- toolsCopy[toolsCopy.length - 1] = {
15914
- ...lastTool,
15915
- cache_control: {
15916
- type: "ephemeral",
15917
- ...ttl === "1h" ? { ttl } : {}
15918
- }
15919
- };
15920
- modifiedParams.tools = toolsCopy;
15921
- }
16582
+ const cacheControl = {
16583
+ type: "ephemeral",
16584
+ ...ttl === "1h" ? { ttl } : {}
16585
+ };
16586
+ const inbound = censusMarkers(modifiedParams);
16587
+ let budget = MAX_CACHE_CONTROL_BLOCKS - inbound.total;
16588
+ const dropped = [];
16589
+ /** Claim one marker slot, or record the miss. Re-marking a marked block is free. */
16590
+ const claim = (name, alreadyMarked) => {
16591
+ if (alreadyMarked) return true;
16592
+ if (budget <= 0) {
16593
+ dropped.push(name);
16594
+ return false;
16595
+ }
16596
+ budget -= 1;
16597
+ return true;
16598
+ };
15922
16599
  const systemParam = modifiedParams.system;
15923
16600
  if (strategy.cacheSystemPrompt && systemParam) {
15924
16601
  const systemArray = Array.isArray(systemParam) ? [...systemParam] : [{
@@ -15927,14 +16604,13 @@ var AnthropicCachingAdapter = class {
15927
16604
  }];
15928
16605
  if (systemArray.length > 0) {
15929
16606
  const lastBlock = systemArray[systemArray.length - 1];
15930
- systemArray[systemArray.length - 1] = {
15931
- ...lastBlock,
15932
- cache_control: {
15933
- type: "ephemeral",
15934
- ...ttl === "1h" ? { ttl } : {}
15935
- }
15936
- };
15937
- modifiedParams.system = systemArray;
16607
+ if (claim("system", hasMarker(lastBlock))) {
16608
+ systemArray[systemArray.length - 1] = {
16609
+ ...lastBlock,
16610
+ cache_control: cacheControl
16611
+ };
16612
+ modifiedParams.system = systemArray;
16613
+ }
15938
16614
  }
15939
16615
  }
15940
16616
  const messagesParam = modifiedParams.messages;
@@ -15950,24 +16626,65 @@ var AnthropicCachingAdapter = class {
15950
16626
  text: msgContent
15951
16627
  }];
15952
16628
  else if (Array.isArray(msgContent)) contentArray = [...msgContent];
15953
- else return modifiedParams;
15954
- if (contentArray.length > 0) {
16629
+ if (contentArray && contentArray.length > 0) {
15955
16630
  const lastBlock = contentArray[contentArray.length - 1];
15956
- contentArray[contentArray.length - 1] = {
15957
- ...lastBlock,
15958
- cache_control: {
15959
- type: "ephemeral",
15960
- ...ttl === "1h" ? { ttl } : {}
15961
- }
15962
- };
15963
- messages[anchorIndex] = {
15964
- ...anchorMsg,
15965
- content: contentArray
15966
- };
15967
- modifiedParams.messages = messages;
16631
+ if (claim("history", hasMarker(lastBlock))) {
16632
+ contentArray[contentArray.length - 1] = {
16633
+ ...lastBlock,
16634
+ cache_control: cacheControl
16635
+ };
16636
+ messages[anchorIndex] = {
16637
+ ...anchorMsg,
16638
+ content: contentArray
16639
+ };
16640
+ modifiedParams.messages = messages;
16641
+ }
15968
16642
  }
15969
16643
  }
15970
16644
  }
16645
+ const tools = modifiedParams.tools;
16646
+ if (strategy.cacheTools && Array.isArray(tools) && tools.length > 0) {
16647
+ const toolsCopy = [...tools];
16648
+ const lastTool = toolsCopy[toolsCopy.length - 1];
16649
+ if (claim("tools", hasMarker(lastTool))) {
16650
+ toolsCopy[toolsCopy.length - 1] = {
16651
+ ...lastTool,
16652
+ cache_control: cacheControl
16653
+ };
16654
+ modifiedParams.tools = toolsCopy;
16655
+ }
16656
+ }
16657
+ const outbound = censusMarkers(modifiedParams);
16658
+ const census = {
16659
+ inbound,
16660
+ outbound,
16661
+ limit: MAX_CACHE_CONTROL_BLOCKS
16662
+ };
16663
+ if (outbound.total >= MAX_CACHE_CONTROL_BLOCKS) {
16664
+ const message = "[PromptCache] cache_control census at the ceiling";
16665
+ if (logger) logger.info(message, census);
16666
+ else console.info(message, JSON.stringify(census));
16667
+ } else if (logger) logger.debug("[PromptCache] cache_control census", census);
16668
+ if (outbound.total > MAX_CACHE_CONTROL_BLOCKS) {
16669
+ const message = `[PromptCache] request exceeds the ${MAX_CACHE_CONTROL_BLOCKS}-block cache_control limit on arrival (${outbound.total}); the provider will reject it`;
16670
+ const detail = {
16671
+ inbound,
16672
+ outbound,
16673
+ limit: MAX_CACHE_CONTROL_BLOCKS
16674
+ };
16675
+ if (logger) logger.error(message, detail);
16676
+ else console.error(message, JSON.stringify(detail));
16677
+ } else if (dropped.length > 0) {
16678
+ const message = `[PromptCache] cache_control budget exhausted (limit ${MAX_CACHE_CONTROL_BLOCKS}); skipped breakpoints: ${dropped.join(", ")}`;
16679
+ const detail = {
16680
+ dropped,
16681
+ inbound,
16682
+ outbound,
16683
+ limit: MAX_CACHE_CONTROL_BLOCKS
16684
+ };
16685
+ if (logger) logger.warn(message, detail);
16686
+ else console.warn(message, JSON.stringify(detail));
16687
+ }
15971
16688
  return modifiedParams;
15972
16689
  }
15973
16690
  extractCacheStats(response, model) {
@@ -16131,6 +16848,92 @@ var KimiCachingAdapter = class {
16131
16848
  }
16132
16849
  };
16133
16850
  /**
16851
+ * The cache-inclusive-to-cache-exclusive conversion, shared by every adapter whose
16852
+ * provider reports cached tokens as a SUBSET of the prompt count.
16853
+ *
16854
+ * getTextModelCost expects Anthropic's convention: `inputTokens` counts only uncached
16855
+ * tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
16856
+ * and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
16857
+ * prompt total already CONTAINS the cached tokens - so those adapters must subtract
16858
+ * here before forwarding, or settlement double-bills the cached portion.
16859
+ *
16860
+ * Must stay in sync with the disjoint-fields assumption documented at the settlement
16861
+ * site in ChatCompletionProcess.
16862
+ */
16863
+ /**
16864
+ * Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
16865
+ *
16866
+ * Forwarding the cached count without subtracting double-bills it; forwarding nothing
16867
+ * charges the full input rate on tokens the provider billed at a fraction of it.
16868
+ * Subtracting is the only split that bills what the provider actually charged.
16869
+ *
16870
+ * Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
16871
+ * input count would silently credit the user.
16872
+ */
16873
+ function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
16874
+ if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
16875
+ const cached = Math.min(cacheReadTokens, totalPromptTokens);
16876
+ return {
16877
+ inputTokens: Math.max(0, totalPromptTokens - cached),
16878
+ cacheReadInputTokens: cached
16879
+ };
16880
+ }
16881
+ /**
16882
+ * Cached prompt tokens from a raw provider usage object, across every spelling in use:
16883
+ * OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
16884
+ * Responses API under `input_tokens_details`, Moonshot publishes a flat
16885
+ * `cached_tokens` alongside the OpenAI-shaped nesting, and DeepSeek its own flat
16886
+ * `prompt_cache_hit_tokens`. Reading only one spelling silently bills every cache
16887
+ * hit on the other transports at the full input rate.
16888
+ *
16889
+ * DeepSeek's own spelling leads, because it is the number its invoice is computed
16890
+ * from; the OpenAI-shaped ones it also sends are the fallback for a proxy that
16891
+ * forwards only those.
16892
+ */
16893
+ function cachedTokensFromUsage(usage) {
16894
+ if (!usage) return 0;
16895
+ const candidates = [
16896
+ usage.prompt_cache_hit_tokens,
16897
+ usage.cached_tokens,
16898
+ usage.prompt_tokens_details?.cached_tokens,
16899
+ usage.input_tokens_details?.cached_tokens
16900
+ ];
16901
+ for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
16902
+ return 0;
16903
+ }
16904
+ /**
16905
+ * DeepSeek context caching. Automatic, like Moonshot's and xAI's: no parameter,
16906
+ * no header, no explicit cache-creation call. The adapter exists only to read
16907
+ * the counters back out.
16908
+ * @see https://api-docs.deepseek.com/guides/kv_cache
16909
+ */
16910
+ var DeepSeekCachingAdapter = class {
16911
+ applyCaching(apiParams, _strategy) {
16912
+ return apiParams;
16913
+ }
16914
+ extractCacheStats(response, model) {
16915
+ const usage = response.usage;
16916
+ if (!usage) return void 0;
16917
+ const totalInputTokens = usage.prompt_tokens || 0;
16918
+ const cachedTokens = cachedTokensFromUsage(usage);
16919
+ const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
16920
+ const costSavingsPercent = cacheHitRate * .98;
16921
+ const estimatedLatencyReduction = cacheHitRate * .7;
16922
+ return {
16923
+ provider: ModelBackend.DeepSeek,
16924
+ model,
16925
+ totalInputTokens,
16926
+ cacheReadTokens: cachedTokens,
16927
+ cacheWriteTokens: 0,
16928
+ uncachedTokens: Math.max(0, totalInputTokens - cachedTokens),
16929
+ cacheHitRate,
16930
+ costSavingsPercent,
16931
+ estimatedLatencyReduction,
16932
+ providerMetadata: { automatic: true }
16933
+ };
16934
+ }
16935
+ };
16936
+ /**
16134
16937
  * Helper to log cache statistics in a consistent format across all providers
16135
16938
  */
16136
16939
  function logCacheStats(logger, cacheStats, options) {
@@ -16164,6 +16967,7 @@ const ADAPTERS = {
16164
16967
  [ModelBackend.Bedrock]: new AnthropicCachingAdapter(),
16165
16968
  [ModelBackend.XAI]: new XAICachingAdapter(),
16166
16969
  [ModelBackend.Kimi]: new KimiCachingAdapter(),
16970
+ [ModelBackend.DeepSeek]: new DeepSeekCachingAdapter(),
16167
16971
  [ModelBackend.Ollama]: new NoOpCachingAdapter(),
16168
16972
  [ModelBackend.BFL]: new NoOpCachingAdapter(),
16169
16973
  [ModelBackend.VoyageAI]: new NoOpCachingAdapter(),
@@ -16214,14 +17018,28 @@ const ADAPTIVE_THINKING_MAX_TOKENS_FLOOR = 64e3;
16214
17018
  const THINKING_ANSWER_HEADROOM_TOKENS = 1e3;
16215
17019
  /**
16216
17020
  * Reasoning-inside-the-budget ids that none of the shape checks below can infer.
17021
+ *
16217
17022
  * Bedrock's Kimi always reasons, but it is not Anthropic-adaptive, does not take
16218
17023
  * `reasoning_effort`, and sends plain `max_tokens` - so it looks like an ordinary
16219
17024
  * model at every seam we can inspect. Bedrock copies the monologue inline into
16220
17025
  * `content` (see bedrockBackend/moonshot.ts) and caps output at 16K, so the floor
16221
17026
  * resolves to that entire cap, which is the only value leaving room for an answer
16222
17027
  * after a long trace.
16223
- */
16224
- const REASONS_WITHIN_OUTPUT_BUDGET_IDS = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_THINKING_BEDROCK, ChatModels.KIMI_K2_5_BEDROCK]);
17028
+ *
17029
+ * DeepSeek Flash misses every clause for its own set of reasons: no
17030
+ * `thinkingStyle` (that field is Anthropic's), absent from the OpenAI-only
17031
+ * REASONING_SUPPORTED_MODELS, and DEEPSEEK_PROFILE declares plain `max_tokens`
17032
+ * rather than `max_completion_tokens` because that is the parameter DeepSeek
17033
+ * takes. It reasons on every turn by default at effort 'high', spends those
17034
+ * tokens inside `max_tokens`, and a 4096 budget against a 393K cap is consumed
17035
+ * by the monologue alone: the turn comes back `finish_reason: 'length'` with no
17036
+ * content and deepseekBackend throws.
17037
+ */
17038
+ const REASONS_WITHIN_OUTPUT_BUDGET_IDS = /* @__PURE__ */ new Set([
17039
+ ChatModels.KIMI_K2_THINKING_BEDROCK,
17040
+ ChatModels.KIMI_K2_5_BEDROCK,
17041
+ ChatModels.DEEPSEEK_FLASH
17042
+ ]);
16225
17043
  /**
16226
17044
  * Whether the model spends reasoning tokens inside its output budget on every turn,
16227
17045
  * which is what makes a small budget produce an empty visible reply rather than a
@@ -16737,7 +17555,7 @@ var AnthropicBackend = class {
16737
17555
  supportsTools: true,
16738
17556
  supportsImageVariation: false,
16739
17557
  logoFile: "Anthropic_logo.png",
16740
- rank: 1,
17558
+ rank: 2,
16741
17559
  trainingCutoff: "2024-10-01",
16742
17560
  releaseDate: "2025-05-23",
16743
17561
  deprecationDate: "2026-06-01",
@@ -16760,7 +17578,7 @@ var AnthropicBackend = class {
16760
17578
  supportsTools: true,
16761
17579
  supportsImageVariation: false,
16762
17580
  logoFile: "Anthropic_logo.png",
16763
- rank: 1,
17581
+ rank: 2,
16764
17582
  trainingCutoff: "2025-07-01",
16765
17583
  releaseDate: "2025-09-30",
16766
17584
  description: "Anthropic's most intelligent model in the Claude 4 family. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
@@ -16780,7 +17598,7 @@ var AnthropicBackend = class {
16780
17598
  } },
16781
17599
  supportsVision: true,
16782
17600
  logoFile: "Anthropic_logo.png",
16783
- rank: 1,
17601
+ rank: 3,
16784
17602
  supportsTools: true,
16785
17603
  trainingCutoff: "2025-07-01",
16786
17604
  releaseDate: "2025-10-16",
@@ -16827,7 +17645,7 @@ var AnthropicBackend = class {
16827
17645
  supportsTools: true,
16828
17646
  supportsImageVariation: false,
16829
17647
  logoFile: "Anthropic_logo.png",
16830
- rank: 1,
17648
+ rank: 2,
16831
17649
  trainingCutoff: "2025-10-01",
16832
17650
  releaseDate: "2026-02-19",
16833
17651
  description: "Anthropic's Claude 4.6 Sonnet model. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
@@ -16850,7 +17668,7 @@ var AnthropicBackend = class {
16850
17668
  supportsTools: true,
16851
17669
  supportsImageVariation: false,
16852
17670
  logoFile: "Anthropic_logo.png",
16853
- rank: 0,
17671
+ rank: 1,
16854
17672
  trainingCutoff: "2026-01-01",
16855
17673
  releaseDate: "2026-07-01",
16856
17674
  description: "Anthropic's newest Claude 5 Sonnet model. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
@@ -16943,7 +17761,7 @@ var AnthropicBackend = class {
16943
17761
  } },
16944
17762
  supportsVision: true,
16945
17763
  logoFile: "Anthropic_logo.png",
16946
- rank: 1,
17764
+ rank: 0,
16947
17765
  supportsTools: true,
16948
17766
  trainingCutoff: "2026-01-01",
16949
17767
  releaseDate: "2026-07-01",
@@ -16967,7 +17785,7 @@ var AnthropicBackend = class {
16967
17785
  } },
16968
17786
  supportsVision: true,
16969
17787
  logoFile: "Anthropic_logo.png",
16970
- rank: 1,
17788
+ rank: 0,
16971
17789
  supportsTools: true,
16972
17790
  releaseDate: "2026-07-24",
16973
17791
  description: "Anthropic's latest flagship model. Claude 5 Opus approaches Fable 5 performance at Opus 4.8 pricing, with adaptive extended thinking, coding, and agentic capabilities.",
@@ -17095,7 +17913,7 @@ var AnthropicBackend = class {
17095
17913
  const parts = [this.consolidateSystemMessages(messages), identityReminder].filter(Boolean);
17096
17914
  system = parts.length > 0 ? parts.join("\n") : void 0;
17097
17915
  }
17098
- let filteredMessages = ensureToolPairingIntegrity(this.filterRelevantMessages(cacheStampedMessages), this.logger);
17916
+ let filteredMessages = normalizeToolUseInputs(ensureToolPairingIntegrity(this.filterRelevantMessages(cacheStampedMessages), this.logger), this.logger);
17099
17917
  const countToolBlocks = (msgs) => {
17100
17918
  let useCount = 0;
17101
17919
  let resultCount = 0;
@@ -17195,7 +18013,7 @@ var AnthropicBackend = class {
17195
18013
  } else this.isThinkingEnabled = false;
17196
18014
  const cacheStrategy = options.cacheStrategy;
17197
18015
  if (cacheStrategy?.enableCaching) {
17198
- const cachedParams = getCachingAdapter(ModelBackend.Anthropic).applyCaching(apiParams, cacheStrategy);
18016
+ const cachedParams = getCachingAdapter(ModelBackend.Anthropic).applyCaching(apiParams, cacheStrategy, this.logger);
17199
18017
  Object.assign(apiParams, cachedParams);
17200
18018
  this.logger.debug("[Anthropic] Applying cache control", {
17201
18019
  cacheSystemPrompt: cacheStrategy.cacheSystemPrompt,
@@ -18219,6 +19037,11 @@ const BEDROCK_RETRY_CONFIG = {
18219
19037
  maxAttempts: 6,
18220
19038
  retryMode: "adaptive"
18221
19039
  };
19040
+ const BEDROCK_REQUEST_HANDLER = {
19041
+ requestTimeout: 12e4,
19042
+ sessionTimeout: 13e4,
19043
+ disableConcurrentStreams: true
19044
+ };
18222
19045
  /**
18223
19046
  * Detect cancellation errors so they propagate past tool-error containment to
18224
19047
  * the outer catch (which has dedicated abort handling). Without this, aborts
@@ -18260,7 +19083,8 @@ var BaseBedrockBackend = class {
18260
19083
  };
18261
19084
  this._bedrockRuntime = new BedrockRuntimeClient({
18262
19085
  region: this._options.region,
18263
- ...BEDROCK_RETRY_CONFIG
19086
+ ...BEDROCK_RETRY_CONFIG,
19087
+ requestHandler: BEDROCK_REQUEST_HANDLER
18264
19088
  });
18265
19089
  }
18266
19090
  getRegionForModel(model) {
@@ -18290,12 +19114,31 @@ var BaseBedrockBackend = class {
18290
19114
  takeReasoningBlocks() {
18291
19115
  return [];
18292
19116
  }
19117
+ /**
19118
+ * Whether this adapter's `translateStreamChunk` reports `done: true` ONLY on the provider's
19119
+ * terminal event. When true, complete() treats a stream that produced output but never
19120
+ * reported done as a TRUNCATED response and throws instead of returning the partial text.
19121
+ *
19122
+ * Opt-in rather than the default because "reports done terminally" is a per-adapter contract
19123
+ * the base class cannot infer, and getting it wrong turns every healthy completion into an
19124
+ * error. Three groups exist today:
19125
+ * - terminal-only, so they override this to true: anthropic, deepseek, llama, jurassicTwo
19126
+ * - `done: true` on EVERY content chunk, so the check would be inert: titan, moonshot
19127
+ * (the better fix for those is a stopReason passthrough, as moonshot.ts already does)
19128
+ * - never report done, incl. the test doubles in this directory: left false
19129
+ *
19130
+ * A new streaming backend must opt in deliberately; silence keeps the old behaviour.
19131
+ */
19132
+ get signalsStreamTermination() {
19133
+ return false;
19134
+ }
18293
19135
  updateClientForModel(model) {
18294
19136
  const requiredRegion = this.getRegionForModel(model);
18295
19137
  this._options.region = requiredRegion;
18296
19138
  this._bedrockRuntime = new BedrockRuntimeClient({
18297
19139
  region: this._options.region,
18298
- ...BEDROCK_RETRY_CONFIG
19140
+ ...BEDROCK_RETRY_CONFIG,
19141
+ requestHandler: BEDROCK_REQUEST_HANDLER
18299
19142
  });
18300
19143
  }
18301
19144
  async complete(model, messages, options, callback, toolsUsed = []) {
@@ -18403,9 +19246,11 @@ var BaseBedrockBackend = class {
18403
19246
  if (!response.body) throw new Error("No response body");
18404
19247
  const func = [];
18405
19248
  let emittedTextChars = 0;
19249
+ let sawTerminalEvent = false;
18406
19250
  for await (const streamEvent of response.body) if (streamEvent.chunk?.bytes) {
18407
19251
  const json = new TextDecoder().decode(streamEvent.chunk.bytes);
18408
- const { chunk } = this.translateStreamChunk(model, JSON.parse(json));
19252
+ const { done, chunk } = this.translateStreamChunk(model, JSON.parse(json));
19253
+ sawTerminalEvent ||= done;
18409
19254
  if (chunk?.stopReason) stopReason = chunk.stopReason;
18410
19255
  chunk?.choices?.forEach((choice) => {
18411
19256
  func[choice.index] ||= {};
@@ -18429,6 +19274,7 @@ var BaseBedrockBackend = class {
18429
19274
  await callback(streamedText, buildCompletionInfo());
18430
19275
  }
18431
19276
  if (emittedTextChars === 0 && !func.some((f) => f.name)) throw new Error(`[BaseBedrockBackend] model "${model}" returned an EMPTY response in region ${this._options.region} (no text, no tool call, no output tokens). A "global." cross-region inference profile served from a region that does not host it does exactly this - try the "us." variant, or confirm the model/profile is granted in ${this._options.region}.`);
19277
+ if (this.signalsStreamTermination && !sawTerminalEvent && !options.abortSignal?.aborted) throw new Error(`[BaseBedrockBackend] stream timeout - model "${model}" in region ${this._options.region} ended after ${emittedTextChars} chars without a terminal event, so the response is TRUNCATED. Usually a stalled Bedrock socket cut the stream short; the partial text is withheld deliberately rather than returned as a finished answer.`);
18432
19278
  if (func.some((f) => f.name)) {
18433
19279
  for await (const tool of func) {
18434
19280
  const { id, name, parameters } = tool;
@@ -18733,6 +19579,10 @@ const TEMPERATURE_ONLY_MODELS = [
18733
19579
  ChatModels.CLAUDE_4_6_OPUS_BEDROCK
18734
19580
  ];
18735
19581
  var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19582
+ /** Reports done only on message_stop (anthropic.ts translateStreamChunk), so a missing terminal event means a truncated stream. */
19583
+ get signalsStreamTermination() {
19584
+ return true;
19585
+ }
18736
19586
  isInThinkingBlock = false;
18737
19587
  /**
18738
19588
  * Reasoning blocks of the assistant turn currently being translated, indexed by the
@@ -18780,7 +19630,11 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18780
19630
  if (NO_TEMPERATURE_MODELS.has(model)) return true;
18781
19631
  return !this.getModelInfoList().some((m) => m.id === model) && this._dispatch.for(model)?.thinkingStyle === "adaptive";
18782
19632
  }
18783
- /** Static model info list - synchronous access for getPayload, also used by getModelInfo */
19633
+ /**
19634
+ * Static model info list - synchronous access for getPayload, also used by getModelInfo.
19635
+ * `rank` must match the identically-named entry in anthropicBackend.ts: it is the same
19636
+ * model, so the picker must not show the Bedrock copy above or below its direct twin.
19637
+ */
18784
19638
  getModelInfoList() {
18785
19639
  return [
18786
19640
  {
@@ -18904,7 +19758,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18904
19758
  } },
18905
19759
  supportsVision: true,
18906
19760
  logoFile: "Anthropic_logo.png",
18907
- rank: 0,
19761
+ rank: 1,
18908
19762
  supportsTools: true,
18909
19763
  trainingCutoff: "2025-05-01",
18910
19764
  releaseDate: "2025-05-23",
@@ -18927,7 +19781,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18927
19781
  } },
18928
19782
  supportsVision: true,
18929
19783
  logoFile: "Anthropic_logo.png",
18930
- rank: 0,
19784
+ rank: 1,
18931
19785
  supportsTools: true,
18932
19786
  trainingCutoff: "2025-08-01",
18933
19787
  releaseDate: "2025-08-06",
@@ -18950,7 +19804,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18950
19804
  } },
18951
19805
  supportsVision: true,
18952
19806
  logoFile: "Anthropic_logo.png",
18953
- rank: 1,
19807
+ rank: 2,
18954
19808
  supportsTools: true,
18955
19809
  trainingCutoff: "2025-05-01",
18956
19810
  releaseDate: "2025-05-23",
@@ -18973,7 +19827,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18973
19827
  supportsTools: true,
18974
19828
  supportsImageVariation: false,
18975
19829
  logoFile: "Anthropic_logo.png",
18976
- rank: 1,
19830
+ rank: 2,
18977
19831
  trainingCutoff: "2025-07-01",
18978
19832
  releaseDate: "2025-09-30",
18979
19833
  description: "Anthropic's most intelligent model hosted in AWS Bedrock. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
@@ -18993,7 +19847,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
18993
19847
  } },
18994
19848
  supportsVision: true,
18995
19849
  logoFile: "Anthropic_logo.png",
18996
- rank: 1,
19850
+ rank: 3,
18997
19851
  supportsTools: true,
18998
19852
  trainingCutoff: "2025-07-01",
18999
19853
  releaseDate: "2025-10-16",
@@ -19040,7 +19894,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19040
19894
  supportsTools: true,
19041
19895
  supportsImageVariation: false,
19042
19896
  logoFile: "Anthropic_logo.png",
19043
- rank: 1,
19897
+ rank: 2,
19044
19898
  trainingCutoff: "2025-10-01",
19045
19899
  releaseDate: "2026-02-19",
19046
19900
  description: "Anthropic's Claude 4.6 Sonnet model via AWS Bedrock. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
@@ -19063,7 +19917,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19063
19917
  supportsTools: true,
19064
19918
  supportsImageVariation: false,
19065
19919
  logoFile: "Anthropic_logo.png",
19066
- rank: 0,
19920
+ rank: 1,
19067
19921
  trainingCutoff: "2026-01-01",
19068
19922
  releaseDate: "2026-07-01",
19069
19923
  description: "Anthropic's newest Claude 5 Sonnet model via AWS Bedrock. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
@@ -19084,7 +19938,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19084
19938
  } },
19085
19939
  supportsVision: true,
19086
19940
  logoFile: "Anthropic_logo.png",
19087
- rank: 0,
19941
+ rank: 1,
19088
19942
  supportsTools: true,
19089
19943
  trainingCutoff: "2025-05-01",
19090
19944
  releaseDate: "2026-02-06",
@@ -19108,7 +19962,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19108
19962
  } },
19109
19963
  supportsVision: true,
19110
19964
  logoFile: "Anthropic_logo.png",
19111
- rank: 0,
19965
+ rank: 1,
19112
19966
  supportsTools: true,
19113
19967
  trainingCutoff: "2025-10-01",
19114
19968
  releaseDate: "2026-04-17",
@@ -19132,7 +19986,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19132
19986
  } },
19133
19987
  supportsVision: true,
19134
19988
  logoFile: "Anthropic_logo.png",
19135
- rank: 0,
19989
+ rank: 1,
19136
19990
  supportsTools: true,
19137
19991
  trainingCutoff: "2026-01-01",
19138
19992
  releaseDate: "2026-05-28",
@@ -19495,6 +20349,10 @@ function isReasoningBlock(block) {
19495
20349
  return "reasoningContent" in block;
19496
20350
  }
19497
20351
  var DeepSeekBedrockBackend = class extends BaseBedrockBackend {
20352
+ /** Reports done only on event.messageStop, so a missing terminal event means a truncated stream. */
20353
+ get signalsStreamTermination() {
20354
+ return true;
20355
+ }
19498
20356
  /** Suppresses reasoning/thinking output for summary and title generation calls. */
19499
20357
  isSpecialTask = false;
19500
20358
  /** Tracks whether the stream is currently inside a reasoning span, to emit one <think>/</think> pair per span. */
@@ -19784,7 +20642,7 @@ var JurassicTwoBedrockBackend = class extends BaseBedrockBackend {
19784
20642
  } },
19785
20643
  supportsVision: false,
19786
20644
  logoFile: "AI21Labs.png",
19787
- rank: 50,
20645
+ rank: 51,
19788
20646
  description: "AI21 Labs' balanced Jurassic-2 model offering good performance at moderate cost. Great for everyday tasks and general content generation."
19789
20647
  }];
19790
20648
  }
@@ -19839,6 +20697,10 @@ var JurassicTwoBedrockBackend = class extends BaseBedrockBackend {
19839
20697
  }
19840
20698
  };
19841
20699
  var LlamaBedrockBackend = class extends BaseBedrockBackend {
20700
+ /** Reports done only on response.stop_reason on the terminal chunk, so a missing terminal event means a truncated stream. */
20701
+ get signalsStreamTermination() {
20702
+ return true;
20703
+ }
19842
20704
  async getModelInfo() {
19843
20705
  return [
19844
20706
  {
@@ -20876,7 +21738,7 @@ var GeminiBackend = class {
20876
21738
  supportsVision: true,
20877
21739
  supportsTools: true,
20878
21740
  logoFile: "Google_logo.png",
20879
- rank: 5,
21741
+ rank: 6,
20880
21742
  trainingCutoff: "2025-01-31",
20881
21743
  releaseDate: "2025-11-30",
20882
21744
  description: "Google's Gemini 3 Flash preview for fast, low-latency multimodal understanding, delivering richer visuals and deeper interactivity, built on a foundation of state-of-the-art reasoning."
@@ -20984,7 +21846,7 @@ var GeminiBackend = class {
20984
21846
  rank: 8,
20985
21847
  trainingCutoff: "2025-01-31",
20986
21848
  releaseDate: "2025-06-01",
20987
- deprecationDate: "2026-10-16",
21849
+ deprecationDate: "2026-09-02",
20988
21850
  description: "Google's Gemini 2.5 Flash, offering well-rounded price-performance. Best for large scale processing, low-latency, high volume tasks that require thinking, and agentic use cases"
20989
21851
  },
20990
21852
  {
@@ -21632,112 +22494,38 @@ var GeminiBackend = class {
21632
22494
  }
21633
22495
  };
21634
22496
  /**
21635
- * The cache-inclusive-to-cache-exclusive conversion, shared by every adapter whose
21636
- * provider reports cached tokens as a SUBSET of the prompt count.
22497
+ * Request shaping for DeepSeek's direct API. Kept out of deepseekBackend's
22498
+ * transport for the same reason kimiParams is: every "which parameter does this
22499
+ * id accept" rule is one pure function with a test next to it, rather than a
22500
+ * conditional buried in a 400-line complete().
21637
22501
  *
21638
- * getTextModelCost expects Anthropic's convention: `inputTokens` counts only uncached
21639
- * tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
21640
- * and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
21641
- * prompt total already CONTAINS the cached tokens - so those adapters must subtract
21642
- * here before forwarding, or settlement double-bills the cached portion.
21643
- *
21644
- * Must stay in sync with the disjoint-fields assumption documented at the settlement
21645
- * site in ChatCompletionProcess.
22502
+ * DeepSeek is OpenAI-compatible in envelope. What differs is thinking mode -
22503
+ * on by default, with its own toggle, its own effort vocabulary, and a sampling
22504
+ * group that is IGNORED rather than rejected while it is on.
22505
+ * @see https://api-docs.deepseek.com/guides/thinking_mode
21646
22506
  */
21647
- /**
21648
- * Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
21649
- *
21650
- * Forwarding the cached count without subtracting double-bills it; forwarding nothing
21651
- * charges the full input rate on tokens the provider billed at a fraction of it.
21652
- * Subtracting is the only split that bills what the provider actually charged.
21653
- *
21654
- * Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
21655
- * input count would silently credit the user.
21656
- */
21657
- function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
21658
- if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
21659
- const cached = Math.min(cacheReadTokens, totalPromptTokens);
21660
- return {
21661
- inputTokens: Math.max(0, totalPromptTokens - cached),
21662
- cacheReadInputTokens: cached
21663
- };
21664
- }
21665
- /**
21666
- * Cached prompt tokens from a raw provider usage object, across every spelling in use:
21667
- * OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
21668
- * Responses API under `input_tokens_details`, and Moonshot publishes a flat
21669
- * `cached_tokens` alongside the OpenAI-shaped nesting. Reading only one spelling
21670
- * silently bills every cache hit on the other transports at the full input rate.
21671
- */
21672
- function cachedTokensFromUsage(usage) {
21673
- if (!usage) return 0;
21674
- const candidates = [
21675
- usage.cached_tokens,
21676
- usage.prompt_tokens_details?.cached_tokens,
21677
- usage.input_tokens_details?.cached_tokens
21678
- ];
21679
- for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
21680
- return 0;
21681
- }
21682
- /**
21683
- * Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
21684
- * transport so every "which parameter does this id accept" rule is one pure
21685
- * function with a test, rather than a conditional buried in a 400-line complete().
21686
- *
21687
- * Moonshot is OpenAI-compatible in envelope only. The reasoning controls, the
21688
- * sampling pins, and the max-tokens parameter all differ per model, and sending
21689
- * the wrong one is a 400 rather than a silently ignored field.
21690
- * @see https://platform.kimi.ai/docs/api/chat
21691
- */
21692
- /** Kimi's own effort vocabulary, which is not OpenAI's and not B4M's. */
21693
- const KIMI_EFFORT_LEVELS = [
22507
+ /** DeepSeek's effort vocabulary, which is not OpenAI's and not B4M's. */
22508
+ const DEEPSEEK_EFFORT_LEVELS = [
21694
22509
  "low",
21695
22510
  "high",
21696
22511
  "max"
21697
22512
  ];
21698
22513
  /**
21699
- * Takes `reasoning_effort`. K3 only, and K3 always reasons - there is no way to
21700
- * turn thinking off, so the parameter selects depth, never whether.
22514
+ * Every DeepSeek id this build ships, direct-served. Bedrock-served DeepSeek is
22515
+ * not here. Both the reasoning and the sampling shaper gate on THIS set, so the
22516
+ * two cannot disagree about which ids the rules apply to; a test pins it against
22517
+ * the adapter table and against NO_TEMPERATURE_MODELS.
21701
22518
  */
21702
- const EFFORT_MODELS = /* @__PURE__ */ new Set([ChatModels.KIMI_K3]);
21703
- /** Takes the `thinking` object instead of `reasoning_effort`. */
21704
- const THINKING_MODELS = /* @__PURE__ */ new Set([
21705
- ChatModels.KIMI_K2_7_CODE,
21706
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
21707
- ChatModels.KIMI_K2_6,
21708
- ChatModels.KIMI_K2_5
21709
- ]);
21710
- /**
21711
- * `thinking.type` accepts only 'enabled' on the K2.7 code models - 'disabled' is
21712
- * rejected. So a caller asking for no thinking gets thinking anyway; the
21713
- * alternative is a 400, and the parameter is omitted rather than fought.
21714
- */
21715
- const THINKING_ALWAYS_ON = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED]);
21716
- /**
21717
- * Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
21718
- * function form). Downgraded to 'auto' rather than dropped: a caller that asked
21719
- * for a forced tool still wants tools offered.
21720
- */
21721
- const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
21722
- ChatModels.KIMI_K2_7_CODE,
21723
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
21724
- ChatModels.KIMI_K2_6
21725
- ]);
21726
- /** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
21727
- const KIMI_MODELS = /* @__PURE__ */ new Set([
21728
- ChatModels.KIMI_K3,
21729
- ChatModels.KIMI_K2_7_CODE,
21730
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
21731
- ChatModels.KIMI_K2_6,
21732
- ChatModels.KIMI_K2_5
21733
- ]);
22519
+ const DEEPSEEK_MODELS = /* @__PURE__ */ new Set([ChatModels.DEEPSEEK_FLASH]);
22520
+ /** DeepSeek raises anything below this rather than erroring, so we send what it will use. */
22521
+ const DEEPSEEK_THINKING_TOP_P_FLOOR = .95;
21734
22522
  /**
21735
- * B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
21736
- * rather than to omission because K3 cannot be asked not to think - claiming
21737
- * otherwise by dropping the parameter would silently bill max-effort reasoning
21738
- * (Moonshot's default is 'max').
22523
+ * B4M's six-level effort onto DeepSeek's three. 'none' and 'minimal' map to
22524
+ * 'low' rather than to omission: omitting the parameter leaves DeepSeek's
22525
+ * documented default of 'high', so dropping it on a "least effort" request
22526
+ * would bill more reasoning than was asked for, not less.
21739
22527
  */
21740
- function toKimiEffort(effort) {
22528
+ function toDeepSeekEffort(effort) {
21741
22529
  if (!effort) return void 0;
21742
22530
  switch (effort) {
21743
22531
  case "none":
@@ -21751,48 +22539,66 @@ function toKimiEffort(effort) {
21751
22539
  }
21752
22540
  /**
21753
22541
  * The reasoning parameters for one model, or an empty object when it takes none.
21754
- * Mutually exclusive by construction: no Kimi model accepts both spellings, and
21755
- * sending both is a 400.
22542
+ *
22543
+ * Both spellings are OpenAI-format and independent, unlike Kimi where they are
22544
+ * mutually exclusive: `thinking.type` turns reasoning on or off and
22545
+ * `reasoning_effort` sets its depth. DeepSeek's own example sends both in one
22546
+ * request. Omitting both leaves thinking enabled at effort 'high'.
21756
22547
  */
21757
- function kimiReasoningParams(model, input) {
21758
- if (EFFORT_MODELS.has(model)) {
21759
- const effort = toKimiEffort(input.reasoningEffort);
21760
- return effort ? { reasoning_effort: effort } : {};
21761
- }
21762
- if (THINKING_MODELS.has(model)) {
21763
- if (THINKING_ALWAYS_ON.has(model)) return { thinking: { type: "enabled" } };
21764
- if (input.thinking?.enabled === void 0) return {};
21765
- return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
21766
- }
21767
- return {};
22548
+ function deepseekReasoningParams(model, input = {}) {
22549
+ if (!DEEPSEEK_MODELS.has(model)) return {};
22550
+ const params = {};
22551
+ if (input.thinking?.enabled !== void 0) params.thinking = { type: input.thinking.enabled ? "enabled" : "disabled" };
22552
+ if (input.thinking?.enabled === false) return params;
22553
+ const effort = toDeepSeekEffort(input.reasoningEffort);
22554
+ if (effort) params.reasoning_effort = effort;
22555
+ return params;
21768
22556
  }
21769
22557
  /**
21770
- * Sampling parameters for one model. Moonshot pins temperature (1.0) and top_p
21771
- * (0.95) on every current Kimi and documents them as unmodifiable, so they are
21772
- * omitted rather than sent-and-ignored; NO_TEMPERATURE_MODELS is the shared set
21773
- * the catalog's temperatureMode also lands on.
22558
+ * Whether the turn will reason, which is what the sampling restrictions below
22559
+ * actually hang on. DeepSeek's default is enabled, so only an explicit
22560
+ * `thinking.enabled === false` turns it off.
22561
+ */
22562
+ function deepseekThinkingEnabled(input = {}) {
22563
+ return input.thinking?.enabled !== false;
22564
+ }
22565
+ /**
22566
+ * Sampling parameters for one turn.
21774
22567
  *
21775
- * The penalties and `n` ride the same gate. Moonshot documents the whole sampling
21776
- * group as fixed on these ids, B4M sends penalties on essentially every turn, and
21777
- * an unmodifiable parameter here is a 400 rather than a silently ignored field -
21778
- * so the conservative reading is the safe one. Only the moonshot-v1 family, which
21779
- * this build does not ship, accepts any of them.
22568
+ * In thinking mode - the default - DeepSeek documents temperature,
22569
+ * presence_penalty and frequency_penalty as unsupported. They are accepted and
22570
+ * SILENTLY ignored rather than rejected, which is the worse failure of the two:
22571
+ * a 400 tells you the knob is dead, a no-op does not. They are dropped here so
22572
+ * nothing is sent that cannot take effect.
22573
+ *
22574
+ * Keyed on the TURN's resolved thinking state, not on model id: the restriction
22575
+ * is a property of thinking mode and the caller can turn thinking off, in which
22576
+ * case dropping temperature anyway would reproduce the same silent no-op from
22577
+ * our side of the wire.
22578
+ *
22579
+ * `top_p` does work in thinking mode with a lower bound of 0.95: a smaller value
22580
+ * is raised to it. Sent clamped rather than dropped, so the request states the
22581
+ * value the server will actually apply. The floor is a thinking-mode rule, so it
22582
+ * does not apply once thinking is off.
22583
+ *
22584
+ * `n` is not in DeepSeek's schema in either mode and is never sent.
21780
22585
  */
21781
- function kimiSamplingParams(model, input) {
21782
- if (NO_TEMPERATURE_MODELS.has(model)) return {};
21783
- const params = {};
21784
- if (input.temperature !== void 0) params.temperature = input.temperature;
21785
- if (input.topP !== void 0) params.top_p = input.topP;
21786
- if (input.presencePenalty !== void 0) params.presence_penalty = input.presencePenalty;
21787
- if (input.frequencyPenalty !== void 0) params.frequency_penalty = input.frequencyPenalty;
21788
- if (input.n !== void 0) params.n = input.n;
21789
- return params;
22586
+ function deepseekSamplingParams(model, input, reasoning = {}) {
22587
+ if (!DEEPSEEK_MODELS.has(model) || !deepseekThinkingEnabled(reasoning)) {
22588
+ const passthrough = {};
22589
+ if (input.temperature !== void 0) passthrough.temperature = input.temperature;
22590
+ if (input.topP !== void 0) passthrough.top_p = input.topP;
22591
+ if (input.presencePenalty !== void 0) passthrough.presence_penalty = input.presencePenalty;
22592
+ if (input.frequencyPenalty !== void 0) passthrough.frequency_penalty = input.frequencyPenalty;
22593
+ return passthrough;
22594
+ }
22595
+ if (input.topP === void 0) return {};
22596
+ return { top_p: Math.max(input.topP, DEEPSEEK_THINKING_TOP_P_FLOOR) };
21790
22597
  }
21791
- /** `tool_choice`, downgraded to 'auto' on the ids that reject 'required'. */
21792
- function kimiToolChoice(model, choice) {
21793
- if (choice === void 0) return void 0;
21794
- if (choice === "required" && NO_REQUIRED_TOOL_CHOICE.has(model)) return "auto";
21795
- return choice;
22598
+ /** `stop`, truncated to the 16 sequences DeepSeek accepts. */
22599
+ function deepseekStopSequences(stop) {
22600
+ if (!Array.isArray(stop)) return stop;
22601
+ return stop.length > 16 ? stop.slice(0, 16) : stop;
21796
22602
  }
21797
22603
  /** Type guard: does this message already carry OpenAI-style `tool_calls`? */
21798
22604
  function hasToolCalls(msg) {
@@ -21815,12 +22621,16 @@ function isTextBlock(block) {
21815
22621
  * Messages already in OpenAI format (with `tool_calls` property) pass through unchanged.
21816
22622
  * Messages without tool_use/tool_result content blocks pass through unchanged.
21817
22623
  */
21818
- function convertMessageToOpenAIFormat(msg) {
21819
- if (hasToolCalls(msg)) return [{
21820
- role: "assistant",
21821
- content: null,
21822
- tool_calls: msg.tool_calls
21823
- }];
22624
+ function convertMessageToOpenAIFormat(msg, options = {}) {
22625
+ if (hasToolCalls(msg)) {
22626
+ const reasoningContent = msg.reasoning_content;
22627
+ return [{
22628
+ role: "assistant",
22629
+ content: null,
22630
+ tool_calls: msg.tool_calls,
22631
+ ...options.preserveReasoningContent && typeof reasoningContent === "string" ? { reasoning_content: reasoningContent } : {}
22632
+ }];
22633
+ }
21824
22634
  if (msg.role === "assistant" && Array.isArray(msg.content)) {
21825
22635
  const contentBlocks = msg.content;
21826
22636
  const toolUseBlocks = contentBlocks.filter(isToolUseBlock);
@@ -21858,30 +22668,33 @@ function convertMessageToOpenAIFormat(msg) {
21858
22668
  * Convert an array of IMessages from B4M standard format to OpenAI-compatible format.
21859
22669
  * Returns OpenAIFormattedMessage[] - callers targeting OpenAI SDK types should cast at the boundary.
21860
22670
  */
21861
- function convertMessagesToOpenAIFormat(messages) {
21862
- return messages.flatMap(convertMessageToOpenAIFormat);
22671
+ function convertMessagesToOpenAIFormat(messages, options = {}) {
22672
+ return messages.flatMap((msg) => convertMessageToOpenAIFormat(msg, options));
21863
22673
  }
21864
22674
  /**
21865
- * Moonshot AI's Kimi models, served from their OpenAI-compatible endpoint.
22675
+ * DeepSeek's models, served from their own OpenAI-compatible endpoint.
21866
22676
  *
21867
- * Structurally this is xaiBackend's twin - same OpenAI SDK against a different
22677
+ * Structurally this is kimiBackend's twin - same OpenAI SDK against a different
21868
22678
  * baseURL, same recursive tool loop, same multi-turn token accumulators - and the
21869
- * two must stay in sync on that machinery. Three things genuinely differ:
22679
+ * three OpenAI-compatible backends must stay in sync on that machinery. What
22680
+ * genuinely differs here:
21870
22681
  *
21871
- * 1. `max_tokens` is deprecated upstream in favor of `max_completion_tokens`.
21872
- * 2. Structured output is NATIVE (json_schema), not the best-effort prompt
21873
- * injection xAI needs, so callers get responseFormatMode: 'native'.
21874
- * 3. Reasoning controls are per-model and mutually exclusive; see kimiParams.
22682
+ * 1. The base URL carries NO `/v1` segment; the SDK appends the path itself.
22683
+ * 2. Thinking is on by default and its sampling restrictions are SILENT no-ops
22684
+ * rather than 400s; see deepseekParams.
22685
+ * 3. The prior turn's `reasoning_content` has to be replayed on the assistant
22686
+ * tool-call message whenever the request carries `tools`, which is the
22687
+ * opposite of the usual provider rule. See pushToolMessages.
21875
22688
  *
21876
- * @see https://platform.kimi.ai/docs/api/chat
22689
+ * @see https://api-docs.deepseek.com/api/create-chat-completion
21877
22690
  */
21878
- var KimiBackend = class {
21879
- _baseUrl = "https://api.moonshot.ai/v1";
22691
+ var DeepSeekBackend = class {
22692
+ _baseUrl = "https://api.deepseek.com";
21880
22693
  _api;
21881
22694
  logger;
21882
22695
  currentModel = "";
21883
22696
  constructor(apiKey, logger) {
21884
- if (!apiKey) throw new Error("Moonshot API key is required");
22697
+ if (!apiKey) throw new Error("DeepSeek API key is required");
21885
22698
  this._api = new OpenAI({
21886
22699
  apiKey,
21887
22700
  baseURL: this._baseUrl
@@ -21891,117 +22704,35 @@ var KimiBackend = class {
21891
22704
  /**
21892
22705
  * Seed listing. Post-registry this is the fallback tier, not the source of
21893
22706
  * truth: the catalog overlays context window, limits, lifecycle and price on
21894
- * top of these rows, and discovery keeps them current without a deploy. What
21895
- * cannot come from a feed - and so has to live here - is the reasoning and
21896
- * dispatch shape each id needs.
22707
+ * top of these rows. DeepSeek's own GET /models returns id/object/owned_by and
22708
+ * nothing else, so everything below has to live here.
22709
+ *
22710
+ * Prices are the PEAK rates. Off-peak (outside 01:00-04:00 and 06:00-10:00 UTC,
22711
+ * Mon-Fri) is exactly half, and ModelInfo.pricing is keyed by context tier with
22712
+ * no time dimension to express that in - so the rate that never under-bills is
22713
+ * the one recorded.
21897
22714
  */
21898
22715
  async getModelInfo() {
21899
- return [
21900
- {
21901
- id: ChatModels.KIMI_K3,
21902
- type: "text",
21903
- name: "Kimi K3",
21904
- backend: ModelBackend.Kimi,
21905
- contextWindow: 1048576,
21906
- max_tokens: 131072,
21907
- can_stream: true,
21908
- pricing: { 1048576: {
21909
- input: 3 / 1e6,
21910
- output: 15 / 1e6,
21911
- cache_read: .3 / 1e6
21912
- } },
21913
- can_think: true,
21914
- supportsVision: true,
21915
- supportsTools: true,
21916
- supportsImageVariation: false,
21917
- releaseDate: "2026-07-16",
21918
- description: "Moonshot's Kimi K3 flagship. 1M context with native vision, tool use, and selectable reasoning effort (low/high/max). Always reasons - effort sets depth, not whether."
21919
- },
21920
- {
21921
- id: ChatModels.KIMI_K2_7_CODE,
21922
- type: "text",
21923
- name: "Kimi K2.7 Code",
21924
- backend: ModelBackend.Kimi,
21925
- contextWindow: 262144,
21926
- max_tokens: 131072,
21927
- can_stream: true,
21928
- pricing: { 262144: {
21929
- input: .95 / 1e6,
21930
- output: 4 / 1e6,
21931
- cache_read: .19 / 1e6
21932
- } },
21933
- can_think: true,
21934
- supportsVision: true,
21935
- supportsTools: true,
21936
- supportsImageVariation: false,
21937
- releaseDate: "2026-06-12",
21938
- trainingCutoff: "2025-01-01",
21939
- description: "Moonshot's coding-focused Kimi, tuned for long-horizon repository work with less overthinking. Thinking cannot be disabled."
21940
- },
21941
- {
21942
- id: ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
21943
- type: "text",
21944
- name: "Kimi K2.7 Code (High Speed)",
21945
- backend: ModelBackend.Kimi,
21946
- contextWindow: 262144,
21947
- max_tokens: 131072,
21948
- can_stream: true,
21949
- pricing: { 262144: {
21950
- input: 1.9 / 1e6,
21951
- output: 8 / 1e6,
21952
- cache_read: .38 / 1e6
21953
- } },
21954
- can_think: true,
21955
- supportsVision: true,
21956
- supportsTools: true,
21957
- supportsImageVariation: false,
21958
- releaseDate: "2026-06-12",
21959
- trainingCutoff: "2025-01-01",
21960
- description: "Kimi K2.7 Code served at 180-260 tokens/s for latency-sensitive work. Identical capabilities to K2.7 Code at twice the price."
21961
- },
21962
- {
21963
- id: ChatModels.KIMI_K2_6,
21964
- type: "text",
21965
- name: "Kimi K2.6",
21966
- backend: ModelBackend.Kimi,
21967
- contextWindow: 262144,
21968
- max_tokens: 131072,
21969
- can_stream: true,
21970
- pricing: { 262144: {
21971
- input: .95 / 1e6,
21972
- output: 4 / 1e6,
21973
- cache_read: .16 / 1e6
21974
- } },
21975
- can_think: true,
21976
- supportsVision: true,
21977
- supportsTools: true,
21978
- supportsImageVariation: false,
21979
- releaseDate: "2026-04-21",
21980
- trainingCutoff: "2025-01-01",
21981
- description: "Moonshot's multimodal workhorse for agent loops, coding, and visual context. Thinking can be turned off on this one, unlike the K2.7 code models."
21982
- },
21983
- {
21984
- id: ChatModels.KIMI_K2_5,
21985
- type: "text",
21986
- name: "Kimi K2.5",
21987
- backend: ModelBackend.Kimi,
21988
- contextWindow: 262144,
21989
- max_tokens: 131072,
21990
- can_stream: true,
21991
- pricing: { 262144: {
21992
- input: .6 / 1e6,
21993
- output: 3 / 1e6,
21994
- cache_read: .1 / 1e6
21995
- } },
21996
- can_think: true,
21997
- supportsVision: true,
21998
- supportsTools: true,
21999
- supportsImageVariation: false,
22000
- releaseDate: "2026-01-01",
22001
- trainingCutoff: "2025-01-01",
22002
- description: "The previous-generation Kimi, still the cheapest of the family. Superseded by K2.6 on quality at a modest price increase."
22003
- }
22004
- ];
22716
+ return [{
22717
+ id: ChatModels.DEEPSEEK_FLASH,
22718
+ type: "text",
22719
+ name: "DeepSeek Flash",
22720
+ backend: ModelBackend.DeepSeek,
22721
+ contextWindow: 1e6,
22722
+ max_tokens: 393216,
22723
+ can_stream: true,
22724
+ pricing: { 1e6: {
22725
+ input: .3 / 1e6,
22726
+ output: 1.2 / 1e6,
22727
+ cache_read: .006 / 1e6
22728
+ } },
22729
+ can_think: true,
22730
+ supportsVision: true,
22731
+ supportsTools: true,
22732
+ supportsImageVariation: false,
22733
+ releaseDate: "2026-08-13",
22734
+ description: "DeepSeek's V4.1-Flash. 1M context with native vision, tool use, and selectable reasoning effort (low/high/max). Always reasons unless thinking is turned off."
22735
+ }];
22005
22736
  }
22006
22737
  async complete(model, messages, options, callback, toolsUsed = []) {
22007
22738
  this.currentModel = model;
@@ -22011,7 +22742,7 @@ var KimiBackend = class {
22011
22742
  const accumCacheReadTokens = options._internal?.accumCacheReadTokens ?? 0;
22012
22743
  const maxToolCalls = options._internal?.maxToolCalls ?? 10;
22013
22744
  if (toolCallCount >= maxToolCalls && options.tools?.length) {
22014
- this.logger.warn(`⚠️ Max tool calls limit (${maxToolCalls}) reached. Disabling tools to prevent infinite loops.`);
22745
+ this.logger.warn(`Max tool calls limit (${maxToolCalls}) reached. Disabling tools to prevent infinite loops.`);
22015
22746
  await this.complete(model, stripToolDependentMessages(messages), {
22016
22747
  ...options,
22017
22748
  tools: void 0,
@@ -22021,53 +22752,740 @@ var KimiBackend = class {
22021
22752
  }
22022
22753
  const rawTools = options.tools;
22023
22754
  options.tools = Array.isArray(rawTools) ? rawTools : rawTools ? [rawTools] : void 0;
22024
- const useStreaming = options.stream && (!options.n || options.n === 1);
22755
+ if ((options.n ?? 1) > 1) this.logger.warn(`DeepSeek has no 'n' parameter; ignoring the request for ${options.n} choices.`);
22756
+ const useStreaming = Boolean(options.stream);
22757
+ const reasoning = {
22758
+ thinking: options.thinking,
22759
+ reasoningEffort: options.reasoningEffort
22760
+ };
22761
+ const messagesWithFormat = injectJsonSchemaInstruction(messages, options.responseFormat);
22762
+ const bestEffortFormat = isBestEffortJsonSchema(options.responseFormat);
22025
22763
  const parameters = {
22026
22764
  model,
22027
- messages: this.formatMessages(messages)
22765
+ messages: this.formatMessages(messagesWithFormat)
22028
22766
  };
22029
22767
  Object.assign(parameters, {
22030
- ...kimiSamplingParams(model, {
22768
+ ...deepseekSamplingParams(model, {
22031
22769
  temperature: options.temperature,
22032
22770
  topP: options.topP,
22033
22771
  presencePenalty: options.presencePenalty,
22034
- frequencyPenalty: options.frequencyPenalty,
22035
- n: options.n
22036
- }),
22037
- ...kimiReasoningParams(model, {
22038
- thinking: options.thinking,
22039
- reasoningEffort: options.reasoningEffort
22040
- }),
22041
- stop: options.stop,
22772
+ frequencyPenalty: options.frequencyPenalty
22773
+ }, reasoning),
22774
+ ...deepseekReasoningParams(model, reasoning),
22775
+ stop: deepseekStopSequences(options.stop),
22042
22776
  stream: useStreaming,
22043
- max_completion_tokens: options.maxTokens,
22777
+ max_tokens: options.maxTokens,
22044
22778
  ...useStreaming && { stream_options: { include_usage: true } }
22045
22779
  });
22046
22780
  if (options.tools?.length) {
22047
22781
  parameters.tools = this.formatTools(options.tools);
22048
- const choice = kimiToolChoice(model, options.tool_choice);
22049
- if (choice !== void 0) parameters.tool_choice = choice;
22782
+ if (options.tool_choice !== void 0) parameters.tool_choice = options.tool_choice;
22050
22783
  }
22051
- if (options.responseFormat?.type === "json_schema") {
22052
- const rf = options.responseFormat;
22053
- parameters.response_format = {
22054
- type: "json_schema",
22055
- json_schema: {
22056
- name: rf.json_schema.name,
22057
- ...rf.json_schema.description ? { description: rf.json_schema.description } : {},
22058
- schema: rf.json_schema.schema,
22059
- ...rf.json_schema.strict !== void 0 ? { strict: rf.json_schema.strict } : { strict: true }
22060
- }
22061
- };
22062
- } else if (options.responseFormat?.type === "text") parameters.response_format = { type: "text" };
22063
- const nativeFormat = options.responseFormat?.type === "json_schema";
22784
+ if (options.responseFormat?.type === "json_schema") parameters.response_format = { type: "json_object" };
22785
+ else if (options.responseFormat?.type === "text") parameters.response_format = { type: "text" };
22064
22786
  const cacheStrategy = options.cacheStrategy;
22065
22787
  const response = await this._api.chat.completions.create(parameters, { signal: options.abortSignal });
22066
22788
  let inputTokens = 0;
22067
22789
  let outputTokens = 0;
22068
22790
  if (!(response instanceof Stream)) {
22069
22791
  const streamedText = [];
22070
- if (!response.choices || response.choices.length === 0) throw new Error("No choices returned from the Moonshot API");
22792
+ if (!response.choices || response.choices.length === 0) throw new Error("No choices returned from the DeepSeek API");
22793
+ const turnCacheReadTokens = cachedTokensFromUsage(response.usage);
22794
+ for (const c of response.choices) {
22795
+ if (!c.message) continue;
22796
+ const reasoningContent = c.message.reasoning_content;
22797
+ if (c.message.tool_calls && c.message.tool_calls.length > 0) {
22798
+ for (const toolCall of c.message.tool_calls) {
22799
+ if (toolCall.type !== "function") continue;
22800
+ if (toolCall.function.arguments) toolsUsed.push({
22801
+ name: toolCall.function.name,
22802
+ arguments: toolCall.function.arguments,
22803
+ id: toolCall.id
22804
+ });
22805
+ }
22806
+ if (options.executeTools !== false) {
22807
+ const resolvedTools = [];
22808
+ for (const toolCall of c.message.tool_calls) {
22809
+ if (toolCall.type !== "function" || !toolCall.function.arguments) continue;
22810
+ const toolFn = options.tools?.find((t) => t.toolSchema.name === toolCall.function.name)?.toolFn;
22811
+ if (!toolFn) continue;
22812
+ try {
22813
+ const parsedParams = JSON.parse(toolCall.function.arguments);
22814
+ resolvedTools.push({
22815
+ id: toolCall.id,
22816
+ name: toolCall.function.name,
22817
+ parameters: toolCall.function.arguments,
22818
+ parsedParams,
22819
+ toolFn
22820
+ });
22821
+ } catch {
22822
+ this.logger.warn(`JSON parse error for ${toolCall.function.name} arguments`);
22823
+ const entry = toolsUsed.find((t) => t.name === toolCall.function.name && t.id === toolCall.id);
22824
+ if (entry) entry.arguments = "{}";
22825
+ recordToolResult(toolsUsed, {
22826
+ id: toolCall.id,
22827
+ name: toolCall.function.name
22828
+ }, "Error: Tool arguments were malformed and could not be parsed.", false);
22829
+ }
22830
+ }
22831
+ const parallelEnabled = options.parallelToolExecution !== false;
22832
+ this.logger.debug("[Tool Execution] Executing tools (DeepSeek non-streaming)", {
22833
+ mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
22834
+ toolNames: resolvedTools.map((t) => t.name)
22835
+ });
22836
+ const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
22837
+ return {
22838
+ id,
22839
+ name,
22840
+ parameters: toolParams,
22841
+ result: await toolFn(parsedParams)
22842
+ };
22843
+ }), {
22844
+ parallel: parallelEnabled,
22845
+ maxConcurrency: options.maxParallelTools
22846
+ })).map((outcome, i) => outcome.ok ? {
22847
+ ok: true,
22848
+ ...outcome.result
22849
+ } : {
22850
+ ok: false,
22851
+ id: resolvedTools[i].id,
22852
+ name: resolvedTools[i].name,
22853
+ parameters: resolvedTools[i].parameters,
22854
+ error: outcome.error
22855
+ });
22856
+ let turnReasoning = reasoningContent;
22857
+ for (const outcome of outcomes) {
22858
+ if (outcome.ok) {
22859
+ const resultStr = outcome.result.toString();
22860
+ recordToolResult(toolsUsed, {
22861
+ id: outcome.id,
22862
+ name: outcome.name
22863
+ }, resultStr, true);
22864
+ this.pushToolMessages(messages, {
22865
+ id: outcome.id,
22866
+ name: outcome.name,
22867
+ parameters: outcome.parameters
22868
+ }, resultStr, turnReasoning ? [turnReasoning] : void 0);
22869
+ } else {
22870
+ if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
22871
+ const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
22872
+ const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
22873
+ recordToolResult(toolsUsed, {
22874
+ id: outcome.id,
22875
+ name: outcome.name
22876
+ }, observation, false);
22877
+ this.pushToolMessages(messages, {
22878
+ id: outcome.id,
22879
+ name: outcome.name,
22880
+ parameters: outcome.parameters
22881
+ }, observation, turnReasoning ? [turnReasoning] : void 0);
22882
+ }
22883
+ turnReasoning = void 0;
22884
+ }
22885
+ await this.complete(model, messages, {
22886
+ ...options,
22887
+ _internal: {
22888
+ ...options._internal,
22889
+ toolCallCount: toolCallCount + 1,
22890
+ accumInputTokens: accumInputTokens + (response.usage?.prompt_tokens || 0),
22891
+ accumOutputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22892
+ accumCacheReadTokens: accumCacheReadTokens + turnCacheReadTokens
22893
+ }
22894
+ }, callback, toolsUsed);
22895
+ return;
22896
+ } else {
22897
+ this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
22898
+ await callback([null], {
22899
+ ...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), accumCacheReadTokens + turnCacheReadTokens),
22900
+ outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22901
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
22902
+ });
22903
+ return;
22904
+ }
22905
+ } else {
22906
+ const content = c.message.content || "";
22907
+ streamedText[c.index] = reasoningContent ? `<think>${reasoningContent}</think>${content}` : content;
22908
+ }
22909
+ }
22910
+ if (streamedText.every((text) => !text) && toolsUsed.length === 0) {
22911
+ const finish = response.choices[0]?.finish_reason;
22912
+ throw new Error(finish === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${finish ?? "unknown"}).`);
22913
+ }
22914
+ let cacheStats;
22915
+ if (cacheStrategy?.enableCaching && response.usage) {
22916
+ cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats(response, model);
22917
+ if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: false });
22918
+ }
22919
+ const finishReason = normalizeOpenAIFinishReason(response.choices[0]?.finish_reason);
22920
+ const totalCacheReadTokens = accumCacheReadTokens + turnCacheReadTokens;
22921
+ await callback(streamedText, {
22922
+ ...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), totalCacheReadTokens),
22923
+ outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22924
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
22925
+ cacheStats,
22926
+ ...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
22927
+ ...finishReason ? { stopReason: finishReason } : {}
22928
+ });
22929
+ return;
22930
+ }
22931
+ const func = [];
22932
+ let isInThinkingBlock = false;
22933
+ let streamedReasoning = "";
22934
+ let cachedTokensFromStream = 0;
22935
+ let streamFinishReason;
22936
+ let sawAnyText = false;
22937
+ for await (const chunk of response) {
22938
+ const streamedText = [];
22939
+ if (chunk.usage) {
22940
+ inputTokens = Math.max(inputTokens, chunk.usage?.prompt_tokens || 0);
22941
+ outputTokens += chunk.usage?.completion_tokens || 0;
22942
+ const chunkCached = cachedTokensFromUsage(chunk.usage);
22943
+ if (chunkCached > 0) cachedTokensFromStream = chunkCached;
22944
+ }
22945
+ chunk?.choices.forEach((c) => {
22946
+ if (c.finish_reason) streamFinishReason = c.finish_reason;
22947
+ const deltaReasoning = c.delta.reasoning_content;
22948
+ if (deltaReasoning) {
22949
+ streamedReasoning += deltaReasoning;
22950
+ if (!isInThinkingBlock) {
22951
+ isInThinkingBlock = true;
22952
+ streamedText[c.index] = "<think>" + deltaReasoning;
22953
+ } else streamedText[c.index] = deltaReasoning;
22954
+ if (!c.delta.content) return;
22955
+ }
22956
+ if (isInThinkingBlock && c.delta.content) {
22957
+ isInThinkingBlock = false;
22958
+ streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
22959
+ return;
22960
+ }
22961
+ c.delta.tool_calls?.map((tool) => {
22962
+ func[tool.index] ||= {};
22963
+ func[tool.index].name ||= tool.function?.name;
22964
+ func[tool.index].id ||= tool.id;
22965
+ func[tool.index].parameters ??= "";
22966
+ func[tool.index].parameters += tool.function?.arguments || "";
22967
+ });
22968
+ if (func.length > 0) return;
22969
+ streamedText[c.index] = c.delta.content || "";
22970
+ });
22971
+ if (streamedText.some((t) => t)) sawAnyText = true;
22972
+ const normalizedFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
22973
+ await callback(streamedText, {
22974
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
22975
+ outputTokens: accumOutputTokens + outputTokens,
22976
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
22977
+ ...normalizedFinishReason ? { stopReason: normalizedFinishReason } : {}
22978
+ });
22979
+ }
22980
+ if (isInThinkingBlock) {
22981
+ await callback(["</think>"], {
22982
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
22983
+ outputTokens: accumOutputTokens + outputTokens,
22984
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
22985
+ });
22986
+ isInThinkingBlock = false;
22987
+ }
22988
+ if (!sawAnyText && func.length === 0 && toolsUsed.length === 0) throw new Error(streamFinishReason === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${streamFinishReason ?? "unknown"}).`);
22989
+ let cacheStats;
22990
+ if (cacheStrategy?.enableCaching && inputTokens > 0) {
22991
+ cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats({ usage: {
22992
+ prompt_tokens: inputTokens,
22993
+ completion_tokens: outputTokens,
22994
+ prompt_cache_hit_tokens: cachedTokensFromStream
22995
+ } }, model);
22996
+ if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: true });
22997
+ }
22998
+ if ((cacheStats || bestEffortFormat) && func.length === 0) {
22999
+ const terminalFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
23000
+ await callback([""], {
23001
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
23002
+ outputTokens: accumOutputTokens + outputTokens,
23003
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
23004
+ ...cacheStats ? { cacheStats } : {},
23005
+ ...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
23006
+ ...terminalFinishReason ? { stopReason: terminalFinishReason } : {}
23007
+ });
23008
+ }
23009
+ if (func.length > 0) {
23010
+ for await (const tool of func) {
23011
+ const { name, parameters: toolParams, id } = tool;
23012
+ if (name) toolsUsed.push({
23013
+ name,
23014
+ arguments: toolParams || "{}",
23015
+ id
23016
+ });
23017
+ }
23018
+ if (options.executeTools !== false) {
23019
+ const resolvedTools = [];
23020
+ for (const tool of func) {
23021
+ const { id, name } = tool;
23022
+ if (!id || !name) continue;
23023
+ const toolParams = tool.parameters || "{}";
23024
+ const toolFn = options.tools?.find((t) => t.toolSchema.name === name)?.toolFn;
23025
+ if (!toolFn) continue;
23026
+ try {
23027
+ const parsedParams = JSON.parse(toolParams);
23028
+ resolvedTools.push({
23029
+ id,
23030
+ name,
23031
+ parameters: toolParams,
23032
+ parsedParams,
23033
+ toolFn
23034
+ });
23035
+ } catch {
23036
+ this.logger.warn(`JSON parse error for ${name} arguments (streaming)`);
23037
+ const entry = toolsUsed.find((t) => t.name === name && t.id === id);
23038
+ if (entry) entry.arguments = "{}";
23039
+ recordToolResult(toolsUsed, {
23040
+ id,
23041
+ name
23042
+ }, "Error: Tool arguments were malformed and could not be parsed.", false);
23043
+ }
23044
+ }
23045
+ const parallelEnabled = options.parallelToolExecution !== false;
23046
+ this.logger.debug("[Tool Execution] Executing tools (DeepSeek streaming)", {
23047
+ mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
23048
+ toolNames: resolvedTools.map((t) => t.name)
23049
+ });
23050
+ const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
23051
+ return {
23052
+ id,
23053
+ name,
23054
+ parameters: toolParams,
23055
+ result: await toolFn(parsedParams)
23056
+ };
23057
+ }), {
23058
+ parallel: parallelEnabled,
23059
+ maxConcurrency: options.maxParallelTools
23060
+ })).map((outcome, i) => outcome.ok ? {
23061
+ ok: true,
23062
+ ...outcome.result
23063
+ } : {
23064
+ ok: false,
23065
+ id: resolvedTools[i].id,
23066
+ name: resolvedTools[i].name,
23067
+ parameters: resolvedTools[i].parameters,
23068
+ error: outcome.error
23069
+ });
23070
+ let turnReasoning = streamedReasoning || void 0;
23071
+ for (const outcome of outcomes) {
23072
+ if (outcome.ok) {
23073
+ const resultStr = outcome.result.toString();
23074
+ recordToolResult(toolsUsed, {
23075
+ id: outcome.id,
23076
+ name: outcome.name
23077
+ }, resultStr, true);
23078
+ this.pushToolMessages(messages, {
23079
+ id: outcome.id,
23080
+ name: outcome.name,
23081
+ parameters: outcome.parameters
23082
+ }, resultStr, turnReasoning ? [turnReasoning] : void 0);
23083
+ } else {
23084
+ if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
23085
+ const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
23086
+ const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
23087
+ recordToolResult(toolsUsed, {
23088
+ id: outcome.id,
23089
+ name: outcome.name
23090
+ }, observation, false);
23091
+ this.pushToolMessages(messages, {
23092
+ id: outcome.id,
23093
+ name: outcome.name,
23094
+ parameters: outcome.parameters
23095
+ }, observation, turnReasoning ? [turnReasoning] : void 0);
23096
+ }
23097
+ turnReasoning = void 0;
23098
+ }
23099
+ await this.complete(model, messages, {
23100
+ ...options,
23101
+ _internal: {
23102
+ ...options._internal,
23103
+ toolCallCount: toolCallCount + 1,
23104
+ accumInputTokens: accumInputTokens + inputTokens,
23105
+ accumOutputTokens: accumOutputTokens + outputTokens,
23106
+ accumCacheReadTokens: accumCacheReadTokens + cachedTokensFromStream
23107
+ }
23108
+ }, callback, toolsUsed);
23109
+ } else {
23110
+ this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
23111
+ await callback([null], {
23112
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
23113
+ outputTokens: accumOutputTokens + outputTokens,
23114
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
23115
+ ...cacheStats ? { cacheStats } : {}
23116
+ });
23117
+ }
23118
+ }
23119
+ }
23120
+ formatMessages(messages) {
23121
+ return convertMessagesToOpenAIFormat(messages, { preserveReasoningContent: true });
23122
+ }
23123
+ formatTools(tools = []) {
23124
+ return tools.map((tool) => ({
23125
+ type: "function",
23126
+ function: tool.toolSchema
23127
+ }));
23128
+ }
23129
+ /**
23130
+ * `thinkingBlocks` carries the turn's `reasoning_content` as a single string
23131
+ * entry. DeepSeek inverts the usual rule: when a request carries `tools`, the
23132
+ * prior turn's monologue MUST be replayed on the assistant tool-call message or
23133
+ * reasoning continuity breaks across the loop. formatMessages opts into the
23134
+ * converter's `preserveReasoningContent` for exactly this path; every other
23135
+ * target strips it, because this array is shared with the fallback hop.
23136
+ */
23137
+ pushToolMessages(messages, tool, result, thinkingBlocks) {
23138
+ const reasoningContent = typeof thinkingBlocks?.[0] === "string" ? thinkingBlocks[0] : void 0;
23139
+ messages.push({
23140
+ content: null,
23141
+ role: "assistant",
23142
+ ...reasoningContent ? { reasoning_content: reasoningContent } : {},
23143
+ tool_calls: [{
23144
+ id: tool.id,
23145
+ type: "function",
23146
+ function: {
23147
+ name: tool.name,
23148
+ arguments: tool.parameters
23149
+ }
23150
+ }]
23151
+ });
23152
+ messages.push({
23153
+ role: "tool",
23154
+ content: JSON.stringify({ result }),
23155
+ tool_call_id: tool.id
23156
+ });
23157
+ }
23158
+ replaceLastToolResultObservation(messages, toolCallId, newObservation) {
23159
+ replaceLastToolResultObservationOpenAI(messages, toolCallId, newObservation);
23160
+ }
23161
+ getLatestToolCallId(messages, toolName) {
23162
+ return getLatestToolCallIdOpenAI(messages, toolName);
23163
+ }
23164
+ };
23165
+ /**
23166
+ * Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
23167
+ * transport so every "which parameter does this id accept" rule is one pure
23168
+ * function with a test, rather than a conditional buried in a 400-line complete().
23169
+ *
23170
+ * Moonshot is OpenAI-compatible in envelope only. The reasoning controls, the
23171
+ * sampling pins, and the max-tokens parameter all differ per model, and sending
23172
+ * the wrong one is a 400 rather than a silently ignored field.
23173
+ * @see https://platform.kimi.ai/docs/api/chat
23174
+ */
23175
+ /** Kimi's own effort vocabulary, which is not OpenAI's and not B4M's. */
23176
+ const KIMI_EFFORT_LEVELS = [
23177
+ "low",
23178
+ "high",
23179
+ "max"
23180
+ ];
23181
+ /**
23182
+ * Takes `reasoning_effort`. K3 only, and K3 always reasons - there is no way to
23183
+ * turn thinking off, so the parameter selects depth, never whether.
23184
+ */
23185
+ const EFFORT_MODELS = /* @__PURE__ */ new Set([ChatModels.KIMI_K3]);
23186
+ /** Takes the `thinking` object instead of `reasoning_effort`. */
23187
+ const THINKING_MODELS = /* @__PURE__ */ new Set([
23188
+ ChatModels.KIMI_K2_7_CODE,
23189
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23190
+ ChatModels.KIMI_K2_6,
23191
+ ChatModels.KIMI_K2_5
23192
+ ]);
23193
+ /**
23194
+ * `thinking.type` accepts only 'enabled' on the K2.7 code models - 'disabled' is
23195
+ * rejected. So a caller asking for no thinking gets thinking anyway; the
23196
+ * alternative is a 400, and the parameter is omitted rather than fought.
23197
+ */
23198
+ const THINKING_ALWAYS_ON = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED]);
23199
+ /**
23200
+ * Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
23201
+ * function form). Downgraded to 'auto' rather than dropped: a caller that asked
23202
+ * for a forced tool still wants tools offered.
23203
+ */
23204
+ const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
23205
+ ChatModels.KIMI_K2_7_CODE,
23206
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23207
+ ChatModels.KIMI_K2_6
23208
+ ]);
23209
+ /** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
23210
+ const KIMI_MODELS = /* @__PURE__ */ new Set([
23211
+ ChatModels.KIMI_K3,
23212
+ ChatModels.KIMI_K2_7_CODE,
23213
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23214
+ ChatModels.KIMI_K2_6,
23215
+ ChatModels.KIMI_K2_5
23216
+ ]);
23217
+ /**
23218
+ * B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
23219
+ * rather than to omission because K3 cannot be asked not to think - claiming
23220
+ * otherwise by dropping the parameter would silently bill max-effort reasoning
23221
+ * (Moonshot's default is 'max').
23222
+ */
23223
+ function toKimiEffort(effort) {
23224
+ if (!effort) return void 0;
23225
+ switch (effort) {
23226
+ case "none":
23227
+ case "minimal":
23228
+ case "low": return "low";
23229
+ case "medium":
23230
+ case "high": return "high";
23231
+ case "xhigh": return "max";
23232
+ default: return;
23233
+ }
23234
+ }
23235
+ /**
23236
+ * The reasoning parameters for one model, or an empty object when it takes none.
23237
+ * Mutually exclusive by construction: no Kimi model accepts both spellings, and
23238
+ * sending both is a 400.
23239
+ */
23240
+ function kimiReasoningParams(model, input) {
23241
+ if (EFFORT_MODELS.has(model)) {
23242
+ const effort = toKimiEffort(input.reasoningEffort);
23243
+ return effort ? { reasoning_effort: effort } : {};
23244
+ }
23245
+ if (THINKING_MODELS.has(model)) {
23246
+ if (THINKING_ALWAYS_ON.has(model)) return { thinking: { type: "enabled" } };
23247
+ if (input.thinking?.enabled === void 0) return {};
23248
+ return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
23249
+ }
23250
+ return {};
23251
+ }
23252
+ /**
23253
+ * Sampling parameters for one model. Moonshot pins temperature (1.0) and top_p
23254
+ * (0.95) on every current Kimi and documents them as unmodifiable, so they are
23255
+ * omitted rather than sent-and-ignored; NO_TEMPERATURE_MODELS is the shared set
23256
+ * the catalog's temperatureMode also lands on.
23257
+ *
23258
+ * The penalties and `n` ride the same gate. Moonshot documents the whole sampling
23259
+ * group as fixed on these ids, B4M sends penalties on essentially every turn, and
23260
+ * an unmodifiable parameter here is a 400 rather than a silently ignored field -
23261
+ * so the conservative reading is the safe one. Only the moonshot-v1 family, which
23262
+ * this build does not ship, accepts any of them.
23263
+ */
23264
+ function kimiSamplingParams(model, input) {
23265
+ if (NO_TEMPERATURE_MODELS.has(model)) return {};
23266
+ const params = {};
23267
+ if (input.temperature !== void 0) params.temperature = input.temperature;
23268
+ if (input.topP !== void 0) params.top_p = input.topP;
23269
+ if (input.presencePenalty !== void 0) params.presence_penalty = input.presencePenalty;
23270
+ if (input.frequencyPenalty !== void 0) params.frequency_penalty = input.frequencyPenalty;
23271
+ if (input.n !== void 0) params.n = input.n;
23272
+ return params;
23273
+ }
23274
+ /** `tool_choice`, downgraded to 'auto' on the ids that reject 'required'. */
23275
+ function kimiToolChoice(model, choice) {
23276
+ if (choice === void 0) return void 0;
23277
+ if (choice === "required" && NO_REQUIRED_TOOL_CHOICE.has(model)) return "auto";
23278
+ return choice;
23279
+ }
23280
+ /**
23281
+ * Moonshot AI's Kimi models, served from their OpenAI-compatible endpoint.
23282
+ *
23283
+ * Structurally this is xaiBackend's twin - same OpenAI SDK against a different
23284
+ * baseURL, same recursive tool loop, same multi-turn token accumulators - and the
23285
+ * two must stay in sync on that machinery. Three things genuinely differ:
23286
+ *
23287
+ * 1. `max_tokens` is deprecated upstream in favor of `max_completion_tokens`.
23288
+ * 2. Structured output is NATIVE (json_schema), not the best-effort prompt
23289
+ * injection xAI needs, so callers get responseFormatMode: 'native'.
23290
+ * 3. Reasoning controls are per-model and mutually exclusive; see kimiParams.
23291
+ *
23292
+ * @see https://platform.kimi.ai/docs/api/chat
23293
+ */
23294
+ var KimiBackend = class {
23295
+ _baseUrl = "https://api.moonshot.ai/v1";
23296
+ _api;
23297
+ logger;
23298
+ currentModel = "";
23299
+ constructor(apiKey, logger) {
23300
+ if (!apiKey) throw new Error("Moonshot API key is required");
23301
+ this._api = new OpenAI({
23302
+ apiKey,
23303
+ baseURL: this._baseUrl
23304
+ });
23305
+ this.logger = logger ?? new Logger();
23306
+ }
23307
+ /**
23308
+ * Seed listing. Post-registry this is the fallback tier, not the source of
23309
+ * truth: the catalog overlays context window, limits, lifecycle and price on
23310
+ * top of these rows, and discovery keeps them current without a deploy. What
23311
+ * cannot come from a feed - and so has to live here - is the reasoning and
23312
+ * dispatch shape each id needs.
23313
+ */
23314
+ async getModelInfo() {
23315
+ return [
23316
+ {
23317
+ id: ChatModels.KIMI_K3,
23318
+ type: "text",
23319
+ name: "Kimi K3",
23320
+ backend: ModelBackend.Kimi,
23321
+ contextWindow: 1048576,
23322
+ max_tokens: 131072,
23323
+ can_stream: true,
23324
+ pricing: { 1048576: {
23325
+ input: 3 / 1e6,
23326
+ output: 15 / 1e6,
23327
+ cache_read: .3 / 1e6
23328
+ } },
23329
+ can_think: true,
23330
+ supportsVision: true,
23331
+ supportsTools: true,
23332
+ supportsImageVariation: false,
23333
+ releaseDate: "2026-07-16",
23334
+ description: "Moonshot's Kimi K3 flagship. 1M context with native vision, tool use, and selectable reasoning effort (low/high/max). Always reasons - effort sets depth, not whether."
23335
+ },
23336
+ {
23337
+ id: ChatModels.KIMI_K2_7_CODE,
23338
+ type: "text",
23339
+ name: "Kimi K2.7 Code",
23340
+ backend: ModelBackend.Kimi,
23341
+ contextWindow: 262144,
23342
+ max_tokens: 131072,
23343
+ can_stream: true,
23344
+ pricing: { 262144: {
23345
+ input: .95 / 1e6,
23346
+ output: 4 / 1e6,
23347
+ cache_read: .19 / 1e6
23348
+ } },
23349
+ can_think: true,
23350
+ supportsVision: true,
23351
+ supportsTools: true,
23352
+ supportsImageVariation: false,
23353
+ releaseDate: "2026-06-12",
23354
+ trainingCutoff: "2025-01-01",
23355
+ description: "Moonshot's coding-focused Kimi, tuned for long-horizon repository work with less overthinking. Thinking cannot be disabled."
23356
+ },
23357
+ {
23358
+ id: ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23359
+ type: "text",
23360
+ name: "Kimi K2.7 Code (High Speed)",
23361
+ backend: ModelBackend.Kimi,
23362
+ contextWindow: 262144,
23363
+ max_tokens: 131072,
23364
+ can_stream: true,
23365
+ pricing: { 262144: {
23366
+ input: 1.9 / 1e6,
23367
+ output: 8 / 1e6,
23368
+ cache_read: .38 / 1e6
23369
+ } },
23370
+ can_think: true,
23371
+ supportsVision: true,
23372
+ supportsTools: true,
23373
+ supportsImageVariation: false,
23374
+ releaseDate: "2026-06-12",
23375
+ trainingCutoff: "2025-01-01",
23376
+ description: "Kimi K2.7 Code served at 180-260 tokens/s for latency-sensitive work. Identical capabilities to K2.7 Code at twice the price."
23377
+ },
23378
+ {
23379
+ id: ChatModels.KIMI_K2_6,
23380
+ type: "text",
23381
+ name: "Kimi K2.6",
23382
+ backend: ModelBackend.Kimi,
23383
+ contextWindow: 262144,
23384
+ max_tokens: 131072,
23385
+ can_stream: true,
23386
+ pricing: { 262144: {
23387
+ input: .95 / 1e6,
23388
+ output: 4 / 1e6,
23389
+ cache_read: .16 / 1e6
23390
+ } },
23391
+ can_think: true,
23392
+ supportsVision: true,
23393
+ supportsTools: true,
23394
+ supportsImageVariation: false,
23395
+ releaseDate: "2026-04-21",
23396
+ trainingCutoff: "2025-01-01",
23397
+ description: "Moonshot's multimodal workhorse for agent loops, coding, and visual context. Thinking can be turned off on this one, unlike the K2.7 code models."
23398
+ },
23399
+ {
23400
+ id: ChatModels.KIMI_K2_5,
23401
+ type: "text",
23402
+ name: "Kimi K2.5",
23403
+ backend: ModelBackend.Kimi,
23404
+ contextWindow: 262144,
23405
+ max_tokens: 131072,
23406
+ can_stream: true,
23407
+ pricing: { 262144: {
23408
+ input: .6 / 1e6,
23409
+ output: 3 / 1e6,
23410
+ cache_read: .1 / 1e6
23411
+ } },
23412
+ can_think: true,
23413
+ supportsVision: true,
23414
+ supportsTools: true,
23415
+ supportsImageVariation: false,
23416
+ releaseDate: "2026-01-01",
23417
+ trainingCutoff: "2025-01-01",
23418
+ deprecationDate: "2026-08-31",
23419
+ replacedBy: ChatModels.KIMI_K2_6,
23420
+ description: "The previous-generation Kimi, still the cheapest of the family. Superseded by K2.6 on quality at a modest price increase."
23421
+ }
23422
+ ];
23423
+ }
23424
+ async complete(model, messages, options, callback, toolsUsed = []) {
23425
+ this.currentModel = model;
23426
+ const toolCallCount = options._internal?.toolCallCount ?? 0;
23427
+ const accumInputTokens = options._internal?.accumInputTokens ?? 0;
23428
+ const accumOutputTokens = options._internal?.accumOutputTokens ?? 0;
23429
+ const accumCacheReadTokens = options._internal?.accumCacheReadTokens ?? 0;
23430
+ const maxToolCalls = options._internal?.maxToolCalls ?? 10;
23431
+ if (toolCallCount >= maxToolCalls && options.tools?.length) {
23432
+ this.logger.warn(`⚠️ Max tool calls limit (${maxToolCalls}) reached. Disabling tools to prevent infinite loops.`);
23433
+ await this.complete(model, stripToolDependentMessages(messages), {
23434
+ ...options,
23435
+ tools: void 0,
23436
+ _internal: options._internal
23437
+ }, callback, toolsUsed);
23438
+ return;
23439
+ }
23440
+ const rawTools = options.tools;
23441
+ options.tools = Array.isArray(rawTools) ? rawTools : rawTools ? [rawTools] : void 0;
23442
+ const useStreaming = options.stream && (!options.n || options.n === 1);
23443
+ const parameters = {
23444
+ model,
23445
+ messages: this.formatMessages(messages)
23446
+ };
23447
+ Object.assign(parameters, {
23448
+ ...kimiSamplingParams(model, {
23449
+ temperature: options.temperature,
23450
+ topP: options.topP,
23451
+ presencePenalty: options.presencePenalty,
23452
+ frequencyPenalty: options.frequencyPenalty,
23453
+ n: options.n
23454
+ }),
23455
+ ...kimiReasoningParams(model, {
23456
+ thinking: options.thinking,
23457
+ reasoningEffort: options.reasoningEffort
23458
+ }),
23459
+ stop: options.stop,
23460
+ stream: useStreaming,
23461
+ max_completion_tokens: options.maxTokens,
23462
+ ...useStreaming && { stream_options: { include_usage: true } }
23463
+ });
23464
+ if (options.tools?.length) {
23465
+ parameters.tools = this.formatTools(options.tools);
23466
+ const choice = kimiToolChoice(model, options.tool_choice);
23467
+ if (choice !== void 0) parameters.tool_choice = choice;
23468
+ }
23469
+ if (options.responseFormat?.type === "json_schema") {
23470
+ const rf = options.responseFormat;
23471
+ parameters.response_format = {
23472
+ type: "json_schema",
23473
+ json_schema: {
23474
+ name: rf.json_schema.name,
23475
+ ...rf.json_schema.description ? { description: rf.json_schema.description } : {},
23476
+ schema: rf.json_schema.schema,
23477
+ ...rf.json_schema.strict !== void 0 ? { strict: rf.json_schema.strict } : { strict: true }
23478
+ }
23479
+ };
23480
+ } else if (options.responseFormat?.type === "text") parameters.response_format = { type: "text" };
23481
+ const nativeFormat = options.responseFormat?.type === "json_schema";
23482
+ const cacheStrategy = options.cacheStrategy;
23483
+ const response = await this._api.chat.completions.create(parameters, { signal: options.abortSignal });
23484
+ let inputTokens = 0;
23485
+ let outputTokens = 0;
23486
+ if (!(response instanceof Stream)) {
23487
+ const streamedText = [];
23488
+ if (!response.choices || response.choices.length === 0) throw new Error("No choices returned from the Moonshot API");
22071
23489
  const turnCacheReadTokens = cachedTokensFromUsage(response.usage);
22072
23490
  for (const c of response.choices) {
22073
23491
  if (!c.message) continue;
@@ -22217,16 +23635,17 @@ var KimiBackend = class {
22217
23635
  }
22218
23636
  chunk?.choices.forEach((c) => {
22219
23637
  if (c.finish_reason) streamFinishReason = c.finish_reason;
22220
- if (c.delta.reasoning_content) {
23638
+ const deltaReasoning = c.delta.reasoning_content;
23639
+ if (deltaReasoning) {
22221
23640
  if (!isInThinkingBlock) {
22222
23641
  isInThinkingBlock = true;
22223
- streamedText[c.index] = "<think>" + c.delta.reasoning_content;
22224
- } else streamedText[c.index] = c.delta.reasoning_content;
22225
- return;
23642
+ streamedText[c.index] = "<think>" + deltaReasoning;
23643
+ } else streamedText[c.index] = deltaReasoning;
23644
+ if (!c.delta.content) return;
22226
23645
  }
22227
- if (isInThinkingBlock && c.delta.content && !c.delta.reasoning_content) {
23646
+ if (isInThinkingBlock && c.delta.content) {
22228
23647
  isInThinkingBlock = false;
22229
- streamedText[c.index] = "</think>" + (c.delta.content || "");
23648
+ streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
22230
23649
  return;
22231
23650
  }
22232
23651
  c.delta.tool_calls?.map((tool) => {
@@ -23137,7 +24556,7 @@ var OpenAIBackend = class {
23137
24556
  supportsTools: true,
23138
24557
  supportsImageVariation: false,
23139
24558
  logoFile: "OpenAI_Logo.svg",
23140
- rank: 0,
24559
+ rank: 4,
23141
24560
  trainingCutoff: "2024-06-01",
23142
24561
  description: "Reliable for general-purpose text generation and analysis with a standard context window, suitable for a wide range of applications."
23143
24562
  },
@@ -23158,7 +24577,7 @@ var OpenAIBackend = class {
23158
24577
  supportsTools: true,
23159
24578
  supportsImageVariation: false,
23160
24579
  logoFile: "OpenAI_Logo.svg",
23161
- rank: 0,
24580
+ rank: 4,
23162
24581
  trainingCutoff: "2024-06-01",
23163
24582
  description: "OpenAI's balanced GPT-4.1 model offering optimal price-performance ratio. Ideal for tasks requiring intelligence and cost efficiency."
23164
24583
  },
@@ -23179,7 +24598,7 @@ var OpenAIBackend = class {
23179
24598
  supportsTools: true,
23180
24599
  supportsImageVariation: false,
23181
24600
  logoFile: "OpenAI_Logo.svg",
23182
- rank: 0,
24601
+ rank: 4,
23183
24602
  trainingCutoff: "2024-06-01",
23184
24603
  deprecationDate: "2026-10-23",
23185
24604
  description: "Designed for high-volume, low-cost processing with rapid response times, ideal for budget-conscious applications."
@@ -23243,7 +24662,7 @@ var OpenAIBackend = class {
23243
24662
  supportsTools: true,
23244
24663
  supportsImageVariation: false,
23245
24664
  logoFile: "OpenAI_Logo.svg",
23246
- rank: 0,
24665
+ rank: 3,
23247
24666
  trainingCutoff: "2024-06-01",
23248
24667
  deprecationDate: "2026-12-11",
23249
24668
  description: "OpenAI's O3 reasoning model with broad capabilities and up-to-date training data. Superseded by O4 Mini for most use cases.",
@@ -23415,7 +24834,7 @@ var OpenAIBackend = class {
23415
24834
  supportsImageVariation: false,
23416
24835
  supportsTools: true,
23417
24836
  logoFile: "OpenAI_Logo.svg",
23418
- rank: 1,
24837
+ rank: 2,
23419
24838
  trainingCutoff: "2026-01-01",
23420
24839
  releaseDate: "2026-06-23",
23421
24840
  description: "GPT-5.6 Luna - the fast, cost-efficient GPT-5.6 variant. Great for high-volume workloads that still need solid reasoning, vision, and tool use."
@@ -23481,7 +24900,7 @@ var OpenAIBackend = class {
23481
24900
  supportsImageVariation: false,
23482
24901
  supportsTools: true,
23483
24902
  logoFile: "OpenAI_Logo.svg",
23484
- rank: 1,
24903
+ rank: 2,
23485
24904
  trainingCutoff: "2025-08-31",
23486
24905
  releaseDate: "2026-03-17",
23487
24906
  description: "Compact GPT-5.4 variant balancing strong performance with lower cost. Great for everyday tasks needing solid reasoning and vision."
@@ -23503,7 +24922,7 @@ var OpenAIBackend = class {
23503
24922
  supportsImageVariation: false,
23504
24923
  supportsTools: true,
23505
24924
  logoFile: "OpenAI_Logo.svg",
23506
- rank: 1,
24925
+ rank: 2,
23507
24926
  trainingCutoff: "2025-08-31",
23508
24927
  releaseDate: "2026-03-17",
23509
24928
  description: "Ultra-lightweight GPT-5.4 model optimized for speed and cost efficiency. Ideal for high-volume workloads and quick interactions."
@@ -25376,6 +26795,10 @@ function backendForAdapterFamily(family, ctx) {
25376
26795
  const key = keyOrThrow(apiKeyTable.kimi, "Moonshot");
25377
26796
  return key ? new KimiBackend(key, logger) : null;
25378
26797
  }
26798
+ case "deepseek": {
26799
+ const key = keyOrThrow(apiKeyTable.deepseek, "DeepSeek");
26800
+ return key ? new DeepSeekBackend(key, logger) : null;
26801
+ }
25379
26802
  case "bfl": return new BFLBackend(keyOrThrow(apiKeyTable.bfl, "BFL") ?? "demo-key");
25380
26803
  case "local-image": {
25381
26804
  const baseUrl = keyOrThrow(apiKeyTable["local-image"], "Local image");
@@ -25416,6 +26839,7 @@ function buildApiKeyTable(keys) {
25416
26839
  [ModelBackend.Ollama]: keys.ollama || void 0,
25417
26840
  [ModelBackend.XAI]: keys.xai || void 0,
25418
26841
  [ModelBackend.Kimi]: keys.kimi || void 0,
26842
+ [ModelBackend.DeepSeek]: keys.deepseek || void 0,
25419
26843
  [ModelBackend.VoyageAI]: keys.voyageai || void 0,
25420
26844
  [ModelBackend.LocalImage]: keys.imageGen || void 0,
25421
26845
  [ModelBackend.Bedrock]: void 0,
@@ -25442,6 +26866,7 @@ const KEYED_LISTING_BACKENDS = [
25442
26866
  ModelBackend.BFL,
25443
26867
  ModelBackend.XAI,
25444
26868
  ModelBackend.Kimi,
26869
+ ModelBackend.DeepSeek,
25445
26870
  ModelBackend.LocalImage
25446
26871
  ];
25447
26872
  /**
@@ -25499,6 +26924,7 @@ const DISPATCHABLE_ADAPTER_FAMILIES = [
25499
26924
  "gemini",
25500
26925
  "xai",
25501
26926
  "kimi",
26927
+ "deepseek",
25502
26928
  "ollama",
25503
26929
  "bfl",
25504
26930
  "local-image",
@@ -25576,6 +27002,15 @@ function mergeCatalogWithDrops(seedModels, rows, ctx) {
25576
27002
  for (const [modelId, bucket] of rowsByModel) {
25577
27003
  if (seeded.has(modelId)) continue;
25578
27004
  const { draft } = mergeRows(bucket, null);
27005
+ const status = draft.lifecycle?.status;
27006
+ const lifecycleReason = inactiveLifecycleReason(status);
27007
+ if (lifecycleReason) {
27008
+ dropped.push({
27009
+ modelId,
27010
+ reason: lifecycleReason
27011
+ });
27012
+ continue;
27013
+ }
25579
27014
  const parsed = asRenderableRecord(draft);
25580
27015
  if ("reason" in parsed) {
25581
27016
  dropped.push({
@@ -25692,10 +27127,19 @@ function asRenderableRecord(draft) {
25692
27127
  if (typeof draft.type !== "string" || !isRenderableModelType(draft.type)) return { reason: `unsupported model type "${String(draft.type)}"` };
25693
27128
  return { record: draft };
25694
27129
  }
27130
+ /**
27131
+ * Why a lifecycle status is not invocable, or null when it is "active". Shared
27132
+ * between invocabilityBlocker and the catalog-only tier's pre-parse check, so
27133
+ * both agree on the exact wording.
27134
+ */
27135
+ function inactiveLifecycleReason(status) {
27136
+ if (status !== "active") return `lifecycle status "${status ?? "unset"}" is not invocable`;
27137
+ return null;
27138
+ }
25695
27139
  /** Why a catalog-only record is metadata-only, or null when it is invocable. */
25696
27140
  function invocabilityBlocker(record) {
25697
- const status = record.lifecycle?.status;
25698
- if (status !== "active") return `lifecycle status "${status ?? "unset"}" is not invocable`;
27141
+ const lifecycleReason = inactiveLifecycleReason(record.lifecycle?.status);
27142
+ if (lifecycleReason) return lifecycleReason;
25699
27143
  if (!record.adapterFamily) return "no adapterFamily";
25700
27144
  if (!DISPATCHABLE_ADAPTER_FAMILIES.includes(record.adapterFamily)) return `adapterFamily "${record.adapterFamily}" is not dispatchable by this build`;
25701
27145
  if (!record.dispatchProfile) return "no dispatchProfile";
@@ -25949,12 +27393,14 @@ const DEPRECATED_MODEL_MAP = {
25949
27393
  "claude-3-haiku-20240307": "claude-haiku-4-5-20251001",
25950
27394
  "gpt-5-chat-latest": "gpt-5.5",
25951
27395
  "gpt-5.1-chat-latest": "gpt-5.5",
27396
+ "gemini-2.5-flash": "gemini-3.1-flash-lite",
25952
27397
  "grok-3": "grok-4.5",
25953
27398
  "grok-3-fast": "grok-4.5",
25954
27399
  "grok-2-1212": "grok-4.5",
25955
27400
  "grok-2-vision-1212": "grok-4.5",
25956
27401
  "grok-beta": "grok-4.5",
25957
27402
  "grok-vision-beta": "grok-4.5",
27403
+ "kimi-k2.5": "kimi-k2.6",
25958
27404
  "grok-3-mini-fast": "grok-3-mini"
25959
27405
  };
25960
27406
  /**
@@ -26079,6 +27525,58 @@ var UndifferentiatedBedrockBackend = class extends BaseBedrockBackend {
26079
27525
  }
26080
27526
  };
26081
27527
  /**
27528
+ * The prices this build ships in code, keyed by model id.
27529
+ *
27530
+ * Same provenance as packages/database's modelPrices.seed.json - the adapter
27531
+ * `getModelInfo()` literals - reachable without a database, which is what the
27532
+ * price planner needs: a model's FIRST discovery-written row has no row in force
27533
+ * to carry the rates no feed publishes from, and a tier that reaches
27534
+ * getTextModelCost without `cache_read` settles cached reads at
27535
+ * input * CACHE_READ_MULTIPLIER. On DeepSeek Flash that default is 0.03/1M
27536
+ * against a real 0.006/1M. MUST STAY IN SYNC with collectStaticTextModels in
27537
+ * packages/database/src/seeds/generateModelPriceSeed.ts: both lists are "every
27538
+ * backend whose getModelInfo() is a static table", and Ollama is absent from
27539
+ * both because its listing is a live server call.
27540
+ */
27541
+ const STATIC_PRICE_BACKENDS = () => [
27542
+ new OpenAIBackend("price-literal"),
27543
+ new AnthropicBackend("price-literal"),
27544
+ new UndifferentiatedBedrockBackend(),
27545
+ new GeminiBackend("price-literal"),
27546
+ new XAIBackend("price-literal"),
27547
+ new KimiBackend("price-literal"),
27548
+ new DeepSeekBackend("price-literal"),
27549
+ new AWSBackend()
27550
+ ];
27551
+ let cached;
27552
+ /**
27553
+ * The lowest-threshold tier of each priced text model's adapter literal.
27554
+ *
27555
+ * Lowest tier on purpose: this is a last-resort carry for rates no feed
27556
+ * publishes (cache and audio), and those do not vary by context bracket in any
27557
+ * literal we ship, while the threshold keys of a discovered ladder need not
27558
+ * match the literal's. Memoized - the tables are static, and the planner runs
27559
+ * once per convergence pass.
27560
+ */
27561
+ async function adapterPriceTiers() {
27562
+ cached ??= collect();
27563
+ return cached;
27564
+ }
27565
+ async function collect() {
27566
+ const tables = await Promise.all(STATIC_PRICE_BACKENDS().map((backend) => backend.getModelInfo()));
27567
+ const tiers = /* @__PURE__ */ new Map();
27568
+ for (const model of tables.flat()) {
27569
+ if (model.type !== "text" || model.freeToRun) continue;
27570
+ const tier = lowestTier(model);
27571
+ if (tier) tiers.set(String(model.id), tier);
27572
+ }
27573
+ return tiers;
27574
+ }
27575
+ function lowestTier(model) {
27576
+ const thresholds = Object.keys(model.pricing).map(Number).filter((threshold) => Number.isFinite(threshold)).sort((a, b) => a - b);
27577
+ return thresholds.length > 0 ? model.pricing[thresholds[0]] : void 0;
27578
+ }
27579
+ /**
26082
27580
  * The dispatch group for a family whose request builder shapes its payload from
26083
27581
  * the provider's own contract and reads nothing out of the profile (Bedrock,
26084
27582
  * Gemini, xAI, Ollama, and the image/speech backends). Promotion still requires
@@ -26119,6 +27617,15 @@ const KIMI_PROFILE = {
26119
27617
  maxTokensParam: "max_completion_tokens",
26120
27618
  toolTransport: "chat"
26121
27619
  };
27620
+ /**
27621
+ * DeepSeek direct. Its own constant rather than PROVIDER_NATIVE_PROFILE because
27622
+ * tools ride Chat Completions rather than a provider-native field, which is what
27623
+ * deepseekBackend sends; the token parameter is still `max_tokens`.
27624
+ */
27625
+ const DEEPSEEK_PROFILE = {
27626
+ maxTokensParam: "max_tokens",
27627
+ toolTransport: "chat"
27628
+ };
26122
27629
  /** Backends whose family is the backend, with a request shape this build fixes. */
26123
27630
  const FAMILY_BY_BACKEND = {
26124
27631
  [ModelBackend.Anthropic]: "anthropic-messages",
@@ -26160,6 +27667,10 @@ function resolveDispatchForRecord(record) {
26160
27667
  adapterFamily: "kimi",
26161
27668
  dispatchProfile: KIMI_PROFILE
26162
27669
  };
27670
+ if (record.backend === ModelBackend.DeepSeek) return {
27671
+ adapterFamily: "deepseek",
27672
+ dispatchProfile: DEEPSEEK_PROFILE
27673
+ };
26163
27674
  const adapterFamily = FAMILY_BY_BACKEND[record.backend];
26164
27675
  return adapterFamily ? {
26165
27676
  adapterFamily,
@@ -26279,7 +27790,11 @@ var AnthropicBatchService = class AnthropicBatchService {
26279
27790
  reply: msg.content.filter((b) => b.type === "text").map((b) => b.text).join(""),
26280
27791
  tokenUsage: {
26281
27792
  inputTokens: msg.usage?.input_tokens ?? 0,
26282
- outputTokens: msg.usage?.output_tokens ?? 0
27793
+ outputTokens: msg.usage?.output_tokens ?? 0,
27794
+ cacheReadInputTokens: msg.usage?.cache_read_input_tokens ?? void 0,
27795
+ cacheCreationInputTokens: msg.usage?.cache_creation_input_tokens ?? void 0,
27796
+ cacheWrite5mInputTokens: msg.usage?.cache_creation?.ephemeral_5m_input_tokens ?? void 0,
27797
+ cacheWrite1hInputTokens: msg.usage?.cache_creation?.ephemeral_1h_input_tokens ?? void 0
26283
27798
  }
26284
27799
  };
26285
27800
  }
@@ -26519,6 +28034,10 @@ function getLlmByModel(apiKeyTable, options) {
26519
28034
  if (apiKeyTable.kimi === "expired") throw new Error("Moonshot API key is expired");
26520
28035
  backend = apiKeyTable.kimi ? new KimiBackend(apiKeyTable.kimi, logger) : null;
26521
28036
  break;
28037
+ case "deepseek":
28038
+ if (apiKeyTable.deepseek === "expired") throw new Error("DeepSeek API key is expired");
28039
+ backend = apiKeyTable.deepseek ? new DeepSeekBackend(apiKeyTable.deepseek, logger) : null;
28040
+ break;
26522
28041
  case "aws":
26523
28042
  backend = new AWSBackend();
26524
28043
  break;
@@ -26613,6 +28132,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
26613
28132
  const bflKey = resolveListingKey(ModelBackend.BFL, gateCtx);
26614
28133
  const xaiKey = resolveListingKey(ModelBackend.XAI, gateCtx);
26615
28134
  const kimiKey = resolveListingKey(ModelBackend.Kimi, gateCtx);
28135
+ const deepseekKey = resolveListingKey(ModelBackend.DeepSeek, gateCtx);
26616
28136
  const localImageBaseUrl = resolveListingKey(ModelBackend.LocalImage, gateCtx);
26617
28137
  const backends = {
26618
28138
  [ModelBackend.OpenAI]: openaiKey ? new OpenAIBackend(openaiKey) : null,
@@ -26623,6 +28143,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
26623
28143
  [ModelBackend.BFL]: bflKey ? new BFLBackend(bflKey) : null,
26624
28144
  [ModelBackend.XAI]: xaiKey ? new XAIBackend(xaiKey) : null,
26625
28145
  [ModelBackend.Kimi]: kimiKey ? new KimiBackend(kimiKey) : null,
28146
+ [ModelBackend.DeepSeek]: deepseekKey ? new DeepSeekBackend(deepseekKey) : null,
26626
28147
  [ModelBackend.AWS]: isBackendUsable(ModelBackend.AWS, gateCtx) ? new AWSBackend() : null,
26627
28148
  [ModelBackend.LocalImage]: localImageBaseUrl ? new LocalImageBackend(localImageBaseUrl, Logger.globalInstance) : null
26628
28149
  };
@@ -27522,6 +29043,109 @@ async function createSseBackend(input, deps = defaultSseTransportDeps) {
27522
29043
  }
27523
29044
  //#endregion
27524
29045
  //#region ../../b4m-core/mcp/dist/index.mjs
29046
+ /**
29047
+ * Environment construction for the MCP stdio child process.
29048
+ *
29049
+ * The child is spawned from a process that also holds platform credentials - provider API keys,
29050
+ * database URIs, signing secrets - so what it inherits is a trust decision, not a convenience.
29051
+ * Two rules follow:
29052
+ *
29053
+ * 1. The child environment is built from an allowlist, never spread from `process.env`. The base
29054
+ * layer is the MCP SDK's own `getDefaultEnvironment()`, which the stdio transport merges
29055
+ * underneath whatever we pass (PATH, HOME, SHELL, TERM, USER on POSIX; the equivalent set on
29056
+ * Windows). Everything above that base comes from the table below.
29057
+ * 2. A stored variable is provider data, never runtime configuration. The child is a Node
29058
+ * process, so a key like NODE_OPTIONS is applied by the runtime before a single line of
29059
+ * server code loads: `--require /tmp/x.js` would turn a credential field into arbitrary code
29060
+ * execution inside the child. Those keys are refused rather than dropped quietly.
29061
+ *
29062
+ * MUST STAY IN SYNC with the `process.env` reads under each server directory
29063
+ * (`github/config.ts`, `notion/config.ts`, `atlassian/config.ts`, `linkedin/index.ts`). A
29064
+ * variable a server reads but this table omits arrives `undefined`, so add it here in the same
29065
+ * change. `childEnv.test.ts` pins that both ways.
29066
+ */
29067
+ const MCP_SERVER_ENV_KEYS = {
29068
+ [McpServerName.LinkedIn]: ["LINKEDIN_ACCESS_TOKEN", "COMPANY_NAME"],
29069
+ [McpServerName.Github]: ["GITHUB_ACCESS_TOKEN"],
29070
+ [McpServerName.Atlassian]: [
29071
+ "ATLASSIAN_ACCESS_TOKEN",
29072
+ "ATLASSIAN_CLOUD_ID",
29073
+ "ATLASSIAN_SITE_URL"
29074
+ ],
29075
+ [McpServerName.Notion]: [
29076
+ "NOTION_ACCESS_TOKEN",
29077
+ "NOTION_WORKSPACE_ID",
29078
+ "NOTION_WRITE_ENABLED",
29079
+ "NOTION_ROOT_PAGE_ID",
29080
+ "NOTION_ACCESS_MODE",
29081
+ "NOTION_ALLOWED_PAGES",
29082
+ "NOTION_EXCLUDED_PAGE_IDS",
29083
+ "NOTION_DEBUG"
29084
+ ]
29085
+ };
29086
+ /**
29087
+ * Keys that make the runtime execute caller-chosen code before the server's entry point runs:
29088
+ * NODE_OPTIONS can `--require` a file, the loader variables preload a shared object, and
29089
+ * ELECTRON_RUN_AS_NODE changes what the binary is. Matching is case-insensitive because Windows
29090
+ * environment names are.
29091
+ */
29092
+ const CODE_INJECTING_ENV_KEY_PATTERNS = [
29093
+ /^NODE_/i,
29094
+ /^ELECTRON_RUN_AS_NODE$/i,
29095
+ /^LD_/i,
29096
+ /^DYLD_/i
29097
+ ];
29098
+ /**
29099
+ * Keys that steer where the child resolves things rather than what it executes: npm_* redirects
29100
+ * package resolution, PATH decides which binary a bare command name finds, and the proxy
29101
+ * variables redirect outbound traffic.
29102
+ */
29103
+ const RESOLUTION_STEERING_ENV_KEY_PATTERNS = [
29104
+ /^npm_/i,
29105
+ /^PATH$/i,
29106
+ /^PATHEXT$/i,
29107
+ /^(HTTP|HTTPS|ALL|NO|FTP)_PROXY$/i,
29108
+ /^GLOBAL_AGENT_/i
29109
+ ];
29110
+ [...CODE_INJECTING_ENV_KEY_PATTERNS, ...RESOLUTION_STEERING_ENV_KEY_PATTERNS];
29111
+ const matchesAny = (patterns, key) => {
29112
+ const normalized = key.trim();
29113
+ return patterns.some((pattern) => pattern.test(normalized));
29114
+ };
29115
+ /** True when `key` would have the runtime load caller-chosen code before the server starts. */
29116
+ function isCodeInjectingMcpEnvKey(key) {
29117
+ return matchesAny(CODE_INJECTING_ENV_KEY_PATTERNS, key);
29118
+ }
29119
+ /**
29120
+ * Build the environment for a stdio MCP child.
29121
+ *
29122
+ * A bundled server gets exactly its declared variables - the allowlist decides, and the denylist
29123
+ * above is never consulted.
29124
+ *
29125
+ * A caller-defined command has no declared contract to check against, so it gets everything
29126
+ * except the code-injecting keys. Only the `b4m` CLI config reaches this branch, and that file
29127
+ * already lets its owner set `command` and `args` to any binary - so withholding PATH or a proxy
29128
+ * variable from them protects nobody while breaking a wrapper script or a corporate proxy, and
29129
+ * the warning that says so goes to a stderr the TUI hides. The code-injecting half stays because
29130
+ * an env-only `--require` is the one lever that is easy to set by accident.
29131
+ */
29132
+ function buildMcpChildEnv({ serverName, envVariables, hasCustomCommand = false }) {
29133
+ const declaredKeys = hasCustomCommand ? void 0 : MCP_SERVER_ENV_KEYS[serverName];
29134
+ const isAllowed = declaredKeys ? (key) => declaredKeys.includes(key) : (key) => !isCodeInjectingMcpEnvKey(key);
29135
+ const env = {};
29136
+ const droppedKeys = [];
29137
+ for (const { key, value } of envVariables) {
29138
+ if (!isAllowed(key)) {
29139
+ droppedKeys.push(key);
29140
+ continue;
29141
+ }
29142
+ env[key] = value;
29143
+ }
29144
+ return {
29145
+ env,
29146
+ droppedKeys
29147
+ };
29148
+ }
27525
29149
  var MCPClient = class {
27526
29150
  mcp;
27527
29151
  transport = null;
@@ -27565,14 +29189,11 @@ var MCPClient = class {
27565
29189
  }));
27566
29190
  return;
27567
29191
  }
27568
- const envVarsObject = this.envVariables.reduce((acc, env) => ({
27569
- ...acc,
27570
- [env.key]: env.value
27571
- }), {});
27572
29192
  let command;
27573
29193
  let args;
27574
- if (this.customCommand && this.customCommand.trim() !== "") {
27575
- command = this.customCommand;
29194
+ const customCommand = this.customCommand?.trim() ? this.customCommand : void 0;
29195
+ if (customCommand) {
29196
+ command = customCommand;
27576
29197
  args = this.customArgs ?? [];
27577
29198
  } else {
27578
29199
  const moduleDir = path.dirname(fileURLToPath(import.meta.url));
@@ -27589,13 +29210,16 @@ var MCPClient = class {
27589
29210
  console.log(`[MCP] Using server: ${this.serverName} at ${serverScriptPath}`);
27590
29211
  }
27591
29212
  const stderrMode = this.suppressStderr ? "ignore" : this.onStderrLine ? "pipe" : void 0;
29213
+ const { env, droppedKeys } = buildMcpChildEnv({
29214
+ serverName: this.serverName,
29215
+ envVariables: this.envVariables,
29216
+ hasCustomCommand: Boolean(customCommand)
29217
+ });
29218
+ if (droppedKeys.length > 0) console.warn(`[MCP] Withheld ${droppedKeys.length} undeclared env variable(s) from ${this.serverName}: ${droppedKeys.join(", ")}`);
27592
29219
  const transportConfig = {
27593
29220
  command,
27594
29221
  args,
27595
- env: {
27596
- ...Object.fromEntries(Object.entries(process.env).filter((entry) => entry[1] !== void 0)),
27597
- ...envVarsObject
27598
- },
29222
+ env,
27599
29223
  ...stderrMode && { stderr: stderrMode }
27600
29224
  };
27601
29225
  const stdioTransport = new StdioClientTransport(transportConfig);
@@ -28023,7 +29647,7 @@ const MODEL_ALIASES = {
28023
29647
  "o4-mini": ChatModels.O4_MINI,
28024
29648
  gemini: ChatModels.GEMINI_2_5_PRO,
28025
29649
  "gemini-pro": ChatModels.GEMINI_2_5_PRO,
28026
- "gemini-flash": ChatModels.GEMINI_2_5_FLASH,
29650
+ "gemini-flash": ChatModels.GEMINI_3_5_FLASH,
28027
29651
  "gemini-flash-lite": ChatModels.GEMINI_2_5_FLASH_LITE,
28028
29652
  "gemini-3": ChatModels.GEMINI_3_PRO_PREVIEW,
28029
29653
  "gemini-3-pro": ChatModels.GEMINI_3_PRO_PREVIEW,
@@ -28042,7 +29666,7 @@ const MODEL_ALIASES = {
28042
29666
  "grok-3-mini-fast": ChatModels.GROK_3_MINI_FAST,
28043
29667
  "grok-2": ChatModels.GROK_2,
28044
29668
  "grok-2-vision": ChatModels.GROK_2_VISION,
28045
- deepseek: ChatModels.DEEPSEEK_R1,
29669
+ deepseek: ChatModels.DEEPSEEK_FLASH,
28046
29670
  "deepseek-r1": ChatModels.DEEPSEEK_R1,
28047
29671
  llama: ChatModels.LLAMA3_LOCAL,
28048
29672
  llama3: ChatModels.LLAMA3_LOCAL,
@@ -28458,7 +30082,7 @@ function buildFilenameMarkerRegex(markers) {
28458
30082
  * of the best-effort DB pre-filter. Fail-closed by design.
28459
30083
  */
28460
30084
  function isRetrievalExcluded(file, opts) {
28461
- const stalledByConvergence = isConvergencePausedNote(file.notes) || isChunkRebuildPending(file.chunkRebuildRequestedAt);
30085
+ const stalledByConvergence = isChunkStalledFile(file) || isChunkRebuildPending(file.chunkRebuildRequestedAt);
28462
30086
  if (opts.vectorizedOnly && !file.vectorized && !stalledByConvergence) return true;
28463
30087
  const re = buildFilenameMarkerRegex(opts.excludeFilenameMarkers);
28464
30088
  return !!re && re.test((file.fileName ?? "").toLowerCase());
@@ -28891,8 +30515,11 @@ function effectiveContextWindow(modelInfo) {
28891
30515
  * budget below - and they must not drift apart.
28892
30516
  *
28893
30517
  * The static catalog tables are held to the positive-budget property by
28894
- * modelCatalogInputBudget.test.ts, and a discovered claim that would break it for a TEXT row is
28895
- * refused in modelDiscoveryService/catalogWrite.
30518
+ * modelCatalogInputBudget.test.ts. A discovered claim is guarded in two places, one per direction:
30519
+ * modelDiscoveryService/catalogWrite refuses a TEXT row whose output cap starves its own window,
30520
+ * and the docs parsers refuse a window or an output cap past MAX_PLAUSIBLE_TOKENS
30521
+ * (modelDiscoveryService/sources/openaiDocs.ts) - an overstated window is not a non-positive
30522
+ * budget, so catalogWrite would never see it, and no aggregator may correct a provider's figure.
28896
30523
  *
28897
30524
  * The buffer figure is imported rather than redeclared here: common owns it, and two copies of the
28898
30525
  * same number is the drift that made it a shared export in the first place.
@@ -29042,6 +30669,18 @@ function attachedContentBudgetsAgree(maxSafeInputTokens, systemPromptReserve) {
29042
30669
  var AdminSettingsCache = class AdminSettingsCache {
29043
30670
  cache = /* @__PURE__ */ new Map();
29044
30671
  individualCache = /* @__PURE__ */ new Map();
30672
+ /**
30673
+ * Every call through this field is optional-chained (`this.logger.debug?.()`).
30674
+ *
30675
+ * A cache must not throw because it could not log, and this one is exposed to that: it is a
30676
+ * process-wide singleton created with whichever logger happens to reach `getSettingsCache` first.
30677
+ * What each caller then does with a throw varies, and it is mostly NOT a degrade-to-defaults
30678
+ * guard: `getSettingsByNames` has none at all, the scoped resolver guards one layer out in
30679
+ * `resolveAll`, and `resolveSpendLevers` deliberately rethrows to halt spend. So a logger missing
30680
+ * a quieter level could surface as a silent wrong VALUE, as an unhandled rejection, or as a hard
30681
+ * fail-closed, depending on who asked. `ScopedSettingsCache` is built by the same factory pair
30682
+ * and still has one unguarded call - the same hazard, not a solved one.
30683
+ */
29045
30684
  logger;
29046
30685
  cleanupInterval = null;
29047
30686
  maxCacheSize = 1e3;
@@ -29057,13 +30696,13 @@ var AdminSettingsCache = class AdminSettingsCache {
29057
30696
  */
29058
30697
  startCleanupTimer() {
29059
30698
  if (process.env.NODE_ENV !== "production" || process.env.VERCEL || process.env.AWS_LAMBDA_FUNCTION_NAME) {
29060
- this.logger.debug("Skipping cleanup timer in serverless environment");
30699
+ this.logger.debug?.("Skipping cleanup timer in serverless environment");
29061
30700
  return;
29062
30701
  }
29063
30702
  this.cleanupInterval = setInterval(() => {
29064
30703
  this.performCleanup();
29065
30704
  }, AdminSettingsCache.CLEANUP_INTERVAL);
29066
- this.logger.debug("Started cache cleanup timer");
30705
+ this.logger.debug?.("Started cache cleanup timer");
29067
30706
  }
29068
30707
  /**
29069
30708
  * Stop cleanup timer (for graceful shutdown)
@@ -29072,7 +30711,7 @@ var AdminSettingsCache = class AdminSettingsCache {
29072
30711
  if (this.cleanupInterval) {
29073
30712
  clearInterval(this.cleanupInterval);
29074
30713
  this.cleanupInterval = null;
29075
- this.logger.debug("Stopped cache cleanup timer");
30714
+ this.logger.debug?.("Stopped cache cleanup timer");
29076
30715
  }
29077
30716
  }
29078
30717
  /**
@@ -29097,9 +30736,9 @@ var AdminSettingsCache = class AdminSettingsCache {
29097
30736
  this.individualCache.delete(entries[i][0]);
29098
30737
  removedCount++;
29099
30738
  }
29100
- this.logger.warn(`Emergency cache cleanup: removed ${toRemove} entries due to size limit`);
30739
+ this.logger.warn?.(`Emergency cache cleanup: removed ${toRemove} entries due to size limit`);
29101
30740
  }
29102
- if (removedCount > 0) this.logger.debug(`Cache cleanup removed ${removedCount} expired entries (${beforeSize} → ${this.cache.size + this.individualCache.size})`);
30741
+ if (removedCount > 0) this.logger.debug?.(`Cache cleanup removed ${removedCount} expired entries (${beforeSize} → ${this.cache.size + this.individualCache.size})`);
29103
30742
  }
29104
30743
  /**
29105
30744
  * Get TTL based on environment
@@ -29120,18 +30759,18 @@ var AdminSettingsCache = class AdminSettingsCache {
29120
30759
  const cacheKey = "all_settings";
29121
30760
  const cached = this.cache.get(cacheKey);
29122
30761
  if (cached && this.isValid(cached.timestamp, cached.ttl)) {
29123
- this.logger.debug("📦 Admin settings cache HIT");
30762
+ this.logger.debug?.("📦 Admin settings cache HIT");
29124
30763
  return cached.data;
29125
30764
  }
29126
30765
  if (cached) this.cache.delete(cacheKey);
29127
- this.logger.debug("🔍 Admin settings cache MISS - fetching from database");
30766
+ this.logger.debug?.("🔍 Admin settings cache MISS - fetching from database");
29128
30767
  const fetchStart = Date.now();
29129
30768
  const settingsMap = (await db.adminSettings.findAll()).reduce((out, s) => {
29130
30769
  out[s.settingName] = s.settingValue;
29131
30770
  return out;
29132
30771
  }, {});
29133
30772
  const fetchTime = Date.now() - fetchStart;
29134
- this.logger.info(`📦 Cached ${Object.keys(settingsMap).length} admin settings in ${fetchTime}ms`);
30773
+ this.logger.info?.(`📦 Cached ${Object.keys(settingsMap).length} admin settings in ${fetchTime}ms`);
29135
30774
  const ttl = this.getTTL();
29136
30775
  this.cache.set(cacheKey, {
29137
30776
  data: settingsMap,
@@ -29153,15 +30792,15 @@ var AdminSettingsCache = class AdminSettingsCache {
29153
30792
  async getSettingByName(settingName, db) {
29154
30793
  const cached = this.individualCache.get(settingName);
29155
30794
  if (cached && this.isValid(cached.timestamp, cached.ttl)) {
29156
- this.logger.debug(`📦 Individual setting '${settingName}' cache HIT`);
30795
+ this.logger.debug?.(`📦 Individual setting '${settingName}' cache HIT`);
29157
30796
  return cached.value;
29158
30797
  }
29159
30798
  if (cached) this.individualCache.delete(settingName);
29160
- this.logger.debug(`🔍 Individual setting '${settingName}' cache MISS - fetching from database`);
30799
+ this.logger.debug?.(`🔍 Individual setting '${settingName}' cache MISS - fetching from database`);
29161
30800
  const fetchStart = Date.now();
29162
30801
  const value = (await db.adminSettings.findBySettingName(settingName))?.settingValue ?? null;
29163
30802
  const fetchTime = Date.now() - fetchStart;
29164
- this.logger.debug(`📦 Cached individual setting '${settingName}' in ${fetchTime}ms`);
30803
+ this.logger.debug?.(`📦 Cached individual setting '${settingName}' in ${fetchTime}ms`);
29165
30804
  this.individualCache.set(settingName, {
29166
30805
  value,
29167
30806
  timestamp: Date.now(),
@@ -29185,11 +30824,11 @@ var AdminSettingsCache = class AdminSettingsCache {
29185
30824
  }
29186
30825
  }
29187
30826
  if (uncachedSettings.length > 0) {
29188
- this.logger.debug(`🔍 Batch fetching ${uncachedSettings.length} uncached settings: ${uncachedSettings.join(", ")}`);
30827
+ this.logger.debug?.(`🔍 Batch fetching ${uncachedSettings.length} uncached settings: ${uncachedSettings.join(", ")}`);
29189
30828
  const fetchStart = Date.now();
29190
30829
  const settings = await db.adminSettings.findBySettingNames(uncachedSettings);
29191
30830
  const fetchTime = Date.now() - fetchStart;
29192
- this.logger.debug(`📦 Batch fetched ${settings.length} settings in ${fetchTime}ms`);
30831
+ this.logger.debug?.(`📦 Batch fetched ${settings.length} settings in ${fetchTime}ms`);
29193
30832
  const ttl = this.getTTL();
29194
30833
  settings.forEach((setting) => {
29195
30834
  result[setting.settingName] = setting.settingValue;
@@ -29207,7 +30846,7 @@ var AdminSettingsCache = class AdminSettingsCache {
29207
30846
  });
29208
30847
  });
29209
30848
  }
29210
- this.logger.debug(`📦 Returned ${Object.keys(result).length} settings (${settingNames.length - uncachedSettings.length} from cache, ${uncachedSettings.length} from DB)`);
30849
+ this.logger.debug?.(`📦 Returned ${Object.keys(result).length} settings (${settingNames.length - uncachedSettings.length} from cache, ${uncachedSettings.length} from DB)`);
29211
30850
  return result;
29212
30851
  }
29213
30852
  /**
@@ -29216,7 +30855,7 @@ var AdminSettingsCache = class AdminSettingsCache {
29216
30855
  invalidateSetting(settingName) {
29217
30856
  this.individualCache.delete(settingName);
29218
30857
  this.cache.delete("all_settings");
29219
- this.logger.info(`🗑️ Invalidated cache for setting: ${settingName}`);
30858
+ this.logger.info?.(`🗑️ Invalidated cache for setting: ${settingName}`);
29220
30859
  }
29221
30860
  /**
29222
30861
  * Invalidate all cached admin settings
@@ -29224,7 +30863,7 @@ var AdminSettingsCache = class AdminSettingsCache {
29224
30863
  invalidateAll() {
29225
30864
  this.cache.clear();
29226
30865
  this.individualCache.clear();
29227
- this.logger.info("🗑️ Invalidated all admin settings cache");
30866
+ this.logger.info?.("🗑️ Invalidated all admin settings cache");
29228
30867
  }
29229
30868
  /**
29230
30869
  * Get cache statistics for monitoring
@@ -29257,16 +30896,16 @@ var AdminSettingsCache = class AdminSettingsCache {
29257
30896
  * Warm up the cache by fetching all settings
29258
30897
  */
29259
30898
  async warmUp(db) {
29260
- this.logger.info("🔥 Warming up admin settings cache...");
30899
+ this.logger.info?.("🔥 Warming up admin settings cache...");
29261
30900
  await this.getSettingsMap(db);
29262
- this.logger.info("✅ Admin settings cache warmed up");
30901
+ this.logger.info?.("✅ Admin settings cache warmed up");
29263
30902
  }
29264
30903
  /**
29265
30904
  * Graceful shutdown - cleanup timers
29266
30905
  */
29267
30906
  shutdown() {
29268
30907
  this.stopCleanupTimer();
29269
- this.logger.info("🛑 Admin settings cache shutdown complete");
30908
+ this.logger.info?.("🛑 Admin settings cache shutdown complete");
29270
30909
  }
29271
30910
  };
29272
30911
  /** Address of one cached override, shared by the cache and its callers so lookups are consistent. */
@@ -29615,6 +31254,17 @@ const getFileContent = async (fabFile, { storage, logger }) => {
29615
31254
  }
29616
31255
  return content;
29617
31256
  };
31257
+ /**
31258
+ * Content hash for per-lake FabFile dedup (`findByContentHashesInDataLake`). Shared by every
31259
+ * ingest path that needs to hash bytes before creating a FabFile - the Slack attachment path
31260
+ * (raw downloaded buffer) and the URL/link path (`fetchAndParseURL`'s extracted `textContent`) -
31261
+ * so at least the HASHING ITSELF cannot drift between two copies of the same algorithm.
31262
+ *
31263
+ * This does NOT make `contentHash` one hash domain: the two callers feed it different inputs
31264
+ * (raw bytes vs. extracted text), so the same document added once as an attachment and once as a
31265
+ * link produces two different hashes and is not caught as a duplicate by this field.
31266
+ */
31267
+ const computeContentHash = (content) => createHash$1("sha256").update(content).digest("hex");
29618
31268
  /** The next 1-based version number given the existing (possibly absent) version history. */
29619
31269
  const nextVersionNumber = (versions) => {
29620
31270
  if (!versions || versions.length === 0) return 1;
@@ -29825,19 +31475,6 @@ const EDITABLE_IMAGE_KEY_RE = /\.(jpe?g|png|webp|gif)$/i;
29825
31475
  const PREVIEW_CHUNK = 700;
29826
31476
  const CHARS_PER_TOKEN = 3.5;
29827
31477
  /**
29828
- * Chunks per attached file that cosine retrieval feeds to the model. Three starved small embedders: a
29829
- * chunk is the embedding model's context window less a 20% buffer (see SmartChunker), so three chunks
29830
- * is roughly 69k chars on an 8192-token embedder but only 4.3k on a 512-token one, which answers a
29831
- * question about a 200-row table from 43 rows without saying so.
29832
- *
29833
- * 10 is borrowed from rankChunksForFiles' topK default, but note the two caps differ in shape: that
29834
- * one is global across every file in the search, this one is PER FILE, so a multi-file attachment can
29835
- * yield more chunks here. What bounds the payload is the per-file character budget applied to these
29836
- * results (maxChars in processFabFilesServer), not this count - and that budget now derives from the
29837
- * model's input window rather than its output limit; see attachedContentExtractionBudget.
29838
- */
29839
- const COSINE_SEARCH_TOP_K = 10;
29840
- /**
29841
31478
  * How much of one attached file the cosine scan will read, and in what size pages.
29842
31479
  *
29843
31480
  * Module constants rather than admin settings: unlike a data lake, an attachment is one file the
@@ -30249,14 +31886,28 @@ async function fetchAgentConversationHistory(session, questCount, { db }) {
30249
31886
  return acc;
30250
31887
  }, new Array());
30251
31888
  }
30252
- async function fetchAndConvertFabFiles(fabFileIds, { scope }, { db, storage }) {
30253
- const fabFiles = await db.fabfiles.getAccessibleFiles(fabFileIds, scope);
30254
- return await Promise.all(fabFiles.map(async (file) => {
31889
+ /**
31890
+ * Resolves attachment ids to documents, and reports the ones it could NOT resolve. The missing set
31891
+ * is the point: `getAccessibleFiles` applies a permission scope and simply omits what it rejects, so
31892
+ * an id dropped by the scope filter or by a delete/upload race used to leave no trace anywhere - the
31893
+ * turn ran as though the file had never been attached (#2228). Callers report `missingIds` through
31894
+ * the same channel as the per-file notices rather than inferring the drop from a shorter array.
31895
+ */
31896
+ async function fetchAndConvertFabFiles(fabFileIds, { scope, lakeAccess }, { db, storage, logger }) {
31897
+ const fabFiles = await db.fabfiles.getAccessibleFiles(fabFileIds, scope, lakeAccess);
31898
+ const files = await Promise.all(fabFiles.map(async (file) => {
30255
31899
  return {
30256
31900
  ...file,
30257
31901
  userId: file.userId.toString()
30258
31902
  };
30259
31903
  }));
31904
+ const returnedIds = new Set(files.map((file) => String(file.id)));
31905
+ const missingIds = Array.from(new Set(fabFileIds)).filter((id) => !returnedIds.has(String(id)));
31906
+ if (missingIds.length > 0) logger?.warn(`[fetchAndConvertFabFiles] ${missingIds.length} of ${fabFileIds.length} requested file id(s) were not returned by getAccessibleFiles and contribute nothing to this turn: ${missingIds.join(", ")}`);
31907
+ return {
31908
+ files,
31909
+ missingIds
31910
+ };
30260
31911
  }
30261
31912
  async function getCachedSignedUrl(filePath, storage, db) {
30262
31913
  const key = `cachedSignedUrl:${filePath}`;
@@ -30458,7 +32109,7 @@ async function cosineSearch(file, userPromptVector, { db, logger }) {
30458
32109
  for (const chunk of usable) {
30459
32110
  const position = scanned;
30460
32111
  scanned++;
30461
- if (head.length < COSINE_SEARCH_TOP_K) head.push({
32112
+ if (head.length < 10) head.push({
30462
32113
  chunkId: chunk.id,
30463
32114
  content: chunk.text,
30464
32115
  score: 0
@@ -30479,9 +32130,9 @@ async function cosineSearch(file, userPromptVector, { db, logger }) {
30479
32130
  position
30480
32131
  });
30481
32132
  }
30482
- if (ranked.length > COSINE_SEARCH_TOP_K) {
32133
+ if (ranked.length > 10) {
30483
32134
  ranked.sort(compareRankedChunks);
30484
- ranked.length = COSINE_SEARCH_TOP_K;
32135
+ ranked.length = 10;
30485
32136
  }
30486
32137
  if (!moreExist) break;
30487
32138
  }
@@ -30503,14 +32154,14 @@ const noopResize = async (imageBuffer) => imageBuffer;
30503
32154
  async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, attachedContentTokenBudget, modelInfo, sendStatusUpdate, { logger, storage, db, resizeImageForModel = noopResize }, progressCallback) {
30504
32155
  if (!fabFiles || fabFiles.length === 0) return {
30505
32156
  userMessages: [],
30506
- errorMessages: [],
32157
+ fileNotices: [],
30507
32158
  deliveredFileIds: [],
30508
32159
  fullyDeliveredFileIds: []
30509
32160
  };
30510
32161
  const fileProcessingStartTime = Date.now();
30511
32162
  let systemContent = "";
30512
32163
  const userMessages = [];
30513
- const errorMessages = [];
32164
+ const fileNotices = [];
30514
32165
  const deliveredFileIds = /* @__PURE__ */ new Set();
30515
32166
  const fullyDeliveredFileIds = /* @__PURE__ */ new Set();
30516
32167
  const contextFiles = [];
@@ -30541,11 +32192,25 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30541
32192
  try {
30542
32193
  if (isAudioMimeType(file.mimeType)) {
30543
32194
  logger.warn(`[processFabFilesServer] Skipping audio file ${file.fileName} — audio is not attachable to an LLM.`);
32195
+ fileNotices.push({
32196
+ fabFileId: file.id,
32197
+ fileName: file.fileName,
32198
+ band: "audio",
32199
+ message: `"${noticeFileName(file.fileName)}" is an audio file and was not sent: no model accepts audio as input.`,
32200
+ delivered: false
32201
+ });
30544
32202
  return;
30545
32203
  }
30546
32204
  if (supportsVision && isImageAttachment(file.mimeType)) {
30547
32205
  if (!isImageServeable(file)) {
30548
32206
  logger.warn(`[processFabFilesServer] Skipping image file ${file.fileName} — held pending moderation or blocked (#9776 Q2b).`);
32207
+ fileNotices.push({
32208
+ fabFileId: file.id,
32209
+ fileName: file.fileName,
32210
+ band: "image_not_serveable",
32211
+ message: `Image "${noticeFileName(file.fileName)}" was not sent: it is held pending moderation or has been blocked.`,
32212
+ delivered: false
32213
+ });
30549
32214
  return;
30550
32215
  }
30551
32216
  sendStatusUpdate(`Processing image file ${file.fileName}...`);
@@ -30554,7 +32219,8 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30554
32219
  switch (modelInfo?.backend) {
30555
32220
  case ModelBackend.OpenAI:
30556
32221
  case ModelBackend.XAI:
30557
- case ModelBackend.Kimi: {
32222
+ case ModelBackend.Kimi:
32223
+ case ModelBackend.DeepSeek: {
30558
32224
  const openaiImageBuffer = await storage.download(file.filePath);
30559
32225
  const { mime: openaiMimeType } = await getFileType(openaiImageBuffer, file.fileName, file.mimeType);
30560
32226
  const openaiBase64 = openaiImageBuffer.toString("base64");
@@ -30581,9 +32247,12 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30581
32247
  const errorMsg = `⚠️ Image "${file.fileName}" (${fileSizeMB.toFixed(1)}MB) is too large for ${backendName}. Max: ${MAX_IMAGE_SIZE_MB}MB. Please delete this file and re-upload to auto-resize.`;
30582
32248
  logger.warn(errorMsg);
30583
32249
  await sendStatusUpdate(errorMsg);
30584
- errorMessages.push({
30585
- role: "error",
30586
- content: errorMsg
32250
+ fileNotices.push({
32251
+ fabFileId: file.id,
32252
+ fileName: file.fileName,
32253
+ band: "image_too_large",
32254
+ message: errorMsg,
32255
+ delivered: false
30587
32256
  });
30588
32257
  return;
30589
32258
  }
@@ -30613,9 +32282,12 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30613
32282
  const errorMsg = `⚠️ Image "${file.fileName}" (${encodedMB}MB encoded) is too large for ${modelInfo.name}. Max ~3MB. Please delete this file and re-upload a smaller image.`;
30614
32283
  logger.warn(errorMsg);
30615
32284
  await sendStatusUpdate(errorMsg);
30616
- errorMessages.push({
30617
- role: "error",
30618
- content: errorMsg
32285
+ fileNotices.push({
32286
+ fabFileId: file.id,
32287
+ fileName: file.fileName,
32288
+ band: "image_too_large",
32289
+ message: errorMsg,
32290
+ delivered: false
30619
32291
  });
30620
32292
  return;
30621
32293
  }
@@ -30629,7 +32301,16 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30629
32301
  });
30630
32302
  delivered = true;
30631
32303
  fullyDelivered = true;
30632
- } else logger.warn(`Vision support for the model ${modelInfo.id} is not implemented. Skipping image processing.`);
32304
+ } else {
32305
+ logger.warn(`Vision support for the model ${modelInfo.id} is not implemented. Skipping image processing.`);
32306
+ fileNotices.push({
32307
+ fabFileId: file.id,
32308
+ fileName: file.fileName,
32309
+ band: "vision_unsupported",
32310
+ message: `Image "${noticeFileName(file.fileName)}" was not sent: image input is not implemented for ${modelInfo.name ?? modelInfo.id}.`,
32311
+ delivered: false
32312
+ });
32313
+ }
30633
32314
  break;
30634
32315
  case ModelBackend.Ollama: {
30635
32316
  const imageBuffer = await resizeImageForModel(await storage.download(file.filePath), void 0, logger);
@@ -30658,10 +32339,26 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30658
32339
  fullyDelivered = true;
30659
32340
  break;
30660
32341
  }
30661
- default: logger.error(`Unsupported backend for model ${modelInfo.id} backend ${modelInfo?.backend ?? "undefined"}`);
32342
+ default:
32343
+ logger.error(`Unsupported backend for model ${modelInfo.id} backend ${modelInfo?.backend ?? "undefined"}`);
32344
+ fileNotices.push({
32345
+ fabFileId: file.id,
32346
+ fileName: file.fileName,
32347
+ band: "unsupported_backend",
32348
+ message: `Image "${noticeFileName(file.fileName)}" was not sent: this model's backend does not accept image attachments.`,
32349
+ delivered: false
32350
+ });
30662
32351
  }
30663
- } else if (!supportsVision && isImageAttachment(file.mimeType)) logger.warn(`File ${file.fileName} is an image but model does not support vision. Skipping...`);
30664
- else {
32352
+ } else if (!supportsVision && isImageAttachment(file.mimeType)) {
32353
+ logger.warn(`File ${file.fileName} is an image but model does not support vision. Skipping...`);
32354
+ fileNotices.push({
32355
+ fabFileId: file.id,
32356
+ fileName: file.fileName,
32357
+ band: "vision_unsupported",
32358
+ message: `Image "${noticeFileName(file.fileName)}" was not sent: ${modelInfo?.name ?? modelInfo?.id ?? "this model"} cannot read images.`,
32359
+ delivered: false
32360
+ });
32361
+ } else {
30665
32362
  const embeddingModel = file.embeddingModel ?? OpenAIEmbeddingModel.TEXT_EMBEDDING_ADA_002;
30666
32363
  const userVector = userVectorPrompt[embeddingModel];
30667
32364
  const canCosineSearch = file.vectorized && !!userVector && userVector.length > 0;
@@ -30737,9 +32434,12 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30737
32434
  const originalFileSize = fabContent.length;
30738
32435
  fabContent = fabContent.substring(0, finalMaxFileSize ?? PREVIEW_CHUNK) + CONTENT_TRUNCATION_NOTICE;
30739
32436
  errorMsg = `Knowledge in the workbench with the fileName ${file.fileName} is ${originalFileSize} long which exceeds ${finalMaxFileSize}. ` + (canCosineSearch ? "None of its vectorized chunks could be searched with this turn's embedding model, so it was sent as raw text and truncated. Re-vectorize it under the current embedding model, or select a model with a higher context window." : "Vectorize your large file or select a model with higher context window.");
30740
- errorMessages.push({
30741
- role: "error",
30742
- content: errorMsg
32437
+ fileNotices.push({
32438
+ fabFileId: file.id,
32439
+ fileName: file.fileName,
32440
+ band: "truncated",
32441
+ message: `"${noticeFileName(file.fileName)}" was too large to send whole; only the first ${Math.floor(finalMaxFileSize)} characters of ${originalFileSize} reached this conversation.`,
32442
+ delivered: true
30743
32443
  });
30744
32444
  } else errorMsg = null;
30745
32445
  delivered = true;
@@ -30753,19 +32453,41 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30753
32453
  error: errorMsg
30754
32454
  });
30755
32455
  } catch (e) {
30756
- if (e instanceof BadRequestError && e.message.includes("Unsupported file type")) logger.warn(`Unsupported file type: ${file.fileName}`);
30757
- else if (isAxiosError(e) && e.response?.status === 404) {
32456
+ if (e instanceof BadRequestError && e.message.includes("Unsupported file type")) {
32457
+ logger.warn(`Unsupported file type: ${file.fileName}`);
32458
+ fileNotices.push({
32459
+ fabFileId: file.id,
32460
+ fileName: file.fileName,
32461
+ band: "unsupported_type",
32462
+ message: `"${noticeFileName(file.fileName)}" was not sent: its file type (${file.mimeType}) cannot be read as text.`,
32463
+ delivered: false
32464
+ });
32465
+ } else if (isAxiosError(e) && e.response?.status === 404) {
30758
32466
  await sendStatusUpdate(`Skipping file ${file.fileName}. File might be corrupted or deleted`);
30759
32467
  await db.fabfiles.update({
30760
32468
  id: file.id,
30761
32469
  error: "This file appears to be corrupted or may have been deleted. Please try uploading the file again."
30762
32470
  });
32471
+ fileNotices.push({
32472
+ fabFileId: file.id,
32473
+ fileName: file.fileName,
32474
+ band: "read_failed",
32475
+ message: `"${noticeFileName(file.fileName)}" could not be read and was not sent: it appears to be corrupted or deleted. Try uploading it again.`,
32476
+ delivered: false
32477
+ });
30763
32478
  } else if (e instanceof CorruptedFileError) {
30764
32479
  await sendStatusUpdate(`Skipping corrupted file ${file.fileName}. Please try re-uploading`);
30765
32480
  await db.fabfiles.update({
30766
32481
  id: file.id,
30767
32482
  error: e.message
30768
32483
  });
32484
+ fileNotices.push({
32485
+ fabFileId: file.id,
32486
+ fileName: file.fileName,
32487
+ band: "read_failed",
32488
+ message: `"${noticeFileName(file.fileName)}" could not be read and was not sent: ${e.message}`,
32489
+ delivered: false
32490
+ });
30769
32491
  } else {
30770
32492
  logger.updateMetadata({ filePath: file.filePath });
30771
32493
  throw e;
@@ -30785,6 +32507,18 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30785
32507
  processedFiles++;
30786
32508
  if (progressCallback) await progressCallback(processedFiles, totalFiles);
30787
32509
  }))));
32510
+ const noticedFileIds = new Set(fileNotices.map((notice) => notice.fabFileId));
32511
+ for (const file of fabFiles) {
32512
+ if (deliveredFileIds.has(file.id) || noticedFileIds.has(file.id)) continue;
32513
+ logger.warn(`[processFabFilesServer] "${file.fileName}" (${file.id}) contributed no content and produced no notice; reporting it as undelivered.`);
32514
+ fileNotices.push({
32515
+ fabFileId: file.id,
32516
+ fileName: file.fileName,
32517
+ band: "no_readable_content",
32518
+ message: `"${noticeFileName(file.fileName)}" was not sent: no readable content could be extracted from it.`,
32519
+ delivered: false
32520
+ });
32521
+ }
30788
32522
  if (imageContent.length > 0) userMessages.push({
30789
32523
  role: "user",
30790
32524
  content: imageContent
@@ -30812,7 +32546,7 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
30812
32546
  logger.info(`📁 File processing completed in ${fileProcessingTime}ms for ${fabFiles.length} files`);
30813
32547
  return {
30814
32548
  userMessages,
30815
- errorMessages,
32549
+ fileNotices,
30816
32550
  deliveredFileIds: Array.from(deliveredFileIds),
30817
32551
  fullyDeliveredFileIds: Array.from(fullyDeliveredFileIds)
30818
32552
  };
@@ -31337,6 +33071,7 @@ var llm_exports = /* @__PURE__ */ __exportAll({
31337
33071
  ATTACHED_CONTENT_EXTRACTION_SHARE: () => ATTACHED_CONTENT_EXTRACTION_SHARE,
31338
33072
  ATTACHMENT_DELIVERED_NOTICE: () => ATTACHMENT_DELIVERED_NOTICE,
31339
33073
  BUILDER_INJECTED_BLOCK_IDS: () => BUILDER_INJECTED_BLOCK_IDS,
33074
+ COSINE_SEARCH_TOP_K: () => 10,
31340
33075
  DEFAULT_OUTPUT_MAX_TOKENS: () => DEFAULT_OUTPUT_MAX_TOKENS,
31341
33076
  EXTRACTION_SYSTEM_RESERVE_MAX_SHARE: () => EXTRACTION_SYSTEM_RESERVE_MAX_SHARE,
31342
33077
  FORMAT_PROMPT_PRIORITY: () => 60,
@@ -32955,6 +34690,7 @@ const OPENAI_IMAGE_CLIENT_OPTS = {
32955
34690
  maxRetries: 0
32956
34691
  };
32957
34692
  const ALTERNATIVE_IMAGE_MODELS = "Flux Pro, Flux Dev, or Grok";
34693
+ const truncatePromptForLog = (prompt) => prompt.length > 100 ? `${prompt.slice(0, 100)}...` : prompt;
32958
34694
  /**
32959
34695
  * Builds a user-friendly error when OpenAI's safety system blocks an image
32960
34696
  * request, guiding the user to rephrase or switch to an alternative model.
@@ -32982,6 +34718,53 @@ function buildModerationBlockedError(error) {
32982
34718
 
32983
34719
  Tip: Switch to an alternative model with different content policies — e.g. ${ALTERNATIVE_IMAGE_MODELS} — which may accept this prompt.\n\nIf you believe this is an error, you can report it to OpenAI with request ID: ${requestId}`);
32984
34720
  }
34721
+ /**
34722
+ * Splits a WIDTHxHEIGHT size into its two edges, or null when the value is not a
34723
+ * pair of non-zero numbers (e.g. 'auto', '', 'wide'). Null means "not a custom
34724
+ * resolution" rather than "invalid": generate() has always left such values
34725
+ * untouched, and that behaviour is preserved.
34726
+ */
34727
+ function parseSizeEdges(size) {
34728
+ if (typeof size !== "string") return null;
34729
+ const [width, height] = size.split("x").map(Number);
34730
+ if (!width || !height) return null;
34731
+ return {
34732
+ width,
34733
+ height
34734
+ };
34735
+ }
34736
+ /**
34737
+ * True when a custom gpt-image-2 resolution meets OpenAI's documented limits.
34738
+ * gpt-image-2 accepts any resolution satisfying these, not only the presets in
34739
+ * OPENAI_GPT_IMAGE_2_IMAGE_SIZES, so a flat preset check would reject valid
34740
+ * custom sizes. Must stay the single source of this rule for generate() and edit().
34741
+ */
34742
+ function satisfiesGptImage2Constraints({ width, height }) {
34743
+ const { maxEdge, minTotalPixels, maxTotalPixels, edgeMultiple, maxAspectRatio } = IMAGE_SIZE_CONSTRAINTS.GPT_IMAGE_2.constraints;
34744
+ const longEdge = Math.max(width, height);
34745
+ const shortEdge = Math.min(width, height);
34746
+ const totalPixels = width * height;
34747
+ return longEdge <= maxEdge && width % edgeMultiple === 0 && height % edgeMultiple === 0 && longEdge / shortEdge <= maxAspectRatio && totalPixels >= minTotalPixels && totalPixels <= maxTotalPixels;
34748
+ }
34749
+ /**
34750
+ * True when `size` may be forwarded to images.edit for `model`. gpt-image-2 takes
34751
+ * its presets (including 'auto') or any custom WIDTHxHEIGHT meeting the same
34752
+ * constraints generate() enforces; the gpt-image-1 family is limited to its three
34753
+ * fixed sizes. An unsupported size is dropped by the caller so OpenAI applies its
34754
+ * own default instead of rejecting the whole request with a 400.
34755
+ *
34756
+ * GPT-Image tiers only: dall-e-2 has its own size list and passes size through
34757
+ * untouched, so do not route that model here.
34758
+ */
34759
+ function isSupportedEditSize(model, size) {
34760
+ if (typeof size !== "string") return false;
34761
+ if (isGPTImage2Model(model)) {
34762
+ if (OPENAI_GPT_IMAGE_2_IMAGE_SIZES.includes(size)) return true;
34763
+ const edges = parseSizeEdges(size);
34764
+ return edges !== null && satisfiesGptImage2Constraints(edges);
34765
+ }
34766
+ return OPENAI_GPT_IMAGE_1_IMAGE_SIZES.includes(size);
34767
+ }
32985
34768
  var OpenAIImageService = class extends AIImageService {
32986
34769
  async generate(prompt, options) {
32987
34770
  const openai = new OpenAI({
@@ -33006,16 +34789,11 @@ var OpenAIImageService = class extends AIImageService {
33006
34789
  }
33007
34790
  if (isGPTImage2Model(options.model)) {
33008
34791
  if (openaiOptions.size && openaiOptions.size !== "auto") {
33009
- const [w, h] = openaiOptions.size.split("x").map(Number);
33010
- if (w && h) {
33011
- const maxEdge = Math.max(w, h);
33012
- const minEdge = Math.min(w, h);
33013
- const totalPixels = w * h;
33014
- if (maxEdge > 3840 || w % 16 !== 0 || h % 16 !== 0 || maxEdge / minEdge > 3 || totalPixels < 655360 || totalPixels > 8294400) {
33015
- const originalSize = openaiOptions.size;
33016
- openaiOptions.size = "1024x1024";
33017
- parameterWarnings.push(`Size '${originalSize}' violates gpt-image-2 constraints, changed to '1024x1024'`);
33018
- }
34792
+ const edges = parseSizeEdges(openaiOptions.size);
34793
+ if (edges && !satisfiesGptImage2Constraints(edges)) {
34794
+ const originalSize = openaiOptions.size;
34795
+ openaiOptions.size = "1024x1024";
34796
+ parameterWarnings.push(`Size '${originalSize}' violates gpt-image-2 constraints, changed to '1024x1024'`);
33019
34797
  }
33020
34798
  } else if (!openaiOptions.size) openaiOptions.size = "auto";
33021
34799
  } else {
@@ -33068,6 +34846,10 @@ var OpenAIImageService = class extends AIImageService {
33068
34846
  const imageFile = new File([pngBuffer], "image.png", { type: "image/png" });
33069
34847
  if (isGPTImageModel(options.model)) {
33070
34848
  const editModel = options.model || ImageModels.GPT_IMAGE_2;
34849
+ this.logger.log("OpenAI image generation request (edit endpoint, image-to-image):", {
34850
+ model: editModel,
34851
+ prompt: truncatePromptForLog(prompt)
34852
+ });
33071
34853
  result = await openai.images.edit({
33072
34854
  model: editModel,
33073
34855
  image: [imageFile],
@@ -33075,20 +34857,31 @@ var OpenAIImageService = class extends AIImageService {
33075
34857
  });
33076
34858
  } else {
33077
34859
  const { style, quality, model, ...opts } = openaiOptions;
34860
+ const variationSize = [
34861
+ "256x256",
34862
+ "512x512",
34863
+ "1024x1024"
34864
+ ].find((s) => s === openaiOptions.size);
34865
+ this.logger.log("OpenAI image generation request (variation endpoint):", {
34866
+ ...opts,
34867
+ size: variationSize
34868
+ });
33078
34869
  result = await openai.images.createVariation({
33079
34870
  ...opts,
33080
34871
  image: imageFile,
33081
- size: [
33082
- "256x256",
33083
- "512x512",
33084
- "1024x1024"
33085
- ].find((s) => s === openaiOptions.size)
34872
+ size: variationSize
33086
34873
  });
33087
34874
  }
33088
- } else result = await openai.images.generate({
33089
- prompt,
33090
- ...openaiOptions
33091
- });
34875
+ } else {
34876
+ this.logger.log("OpenAI image generation request:", {
34877
+ prompt: truncatePromptForLog(prompt),
34878
+ ...openaiOptions
34879
+ });
34880
+ result = await openai.images.generate({
34881
+ prompt,
34882
+ ...openaiOptions
34883
+ });
34884
+ }
33092
34885
  images = this.imageResponseToUrl(result);
33093
34886
  return images;
33094
34887
  } catch (error) {
@@ -33145,10 +34938,21 @@ var OpenAIImageService = class extends AIImageService {
33145
34938
  Logger.globalInstance.debug(`[DEBUG] ⚠️ Edit endpoint doesn't support ${model}, defaulting to gpt-image-2`);
33146
34939
  editModel = ImageModels.GPT_IMAGE_2;
33147
34940
  }
34941
+ const forwardSize = isSupportedEditSize(editModel, size);
34942
+ this.logger.log("OpenAI image edit request:", {
34943
+ model: editModel,
34944
+ prompt: truncatePromptForLog(prompt),
34945
+ hasMask: !!maskFile,
34946
+ n,
34947
+ size,
34948
+ response_format
34949
+ });
33148
34950
  const response = await openai.images.edit(isGPTImageModel(editModel) ? {
33149
34951
  model: editModel,
33150
34952
  image: [imageFile],
33151
- prompt
34953
+ prompt,
34954
+ ...forwardSize ? { size } : {},
34955
+ ...maskFile ? { mask: maskFile } : {}
33152
34956
  } : {
33153
34957
  model: editModel,
33154
34958
  image: imageFile,
@@ -35175,6 +36979,21 @@ var TiktokenTokenizer = class {
35175
36979
  return Array.from(encoder.encode_ordinary(text));
35176
36980
  }
35177
36981
  /**
36982
+ * Decode token ids back to text through the same encoder encodeTokens used, so an
36983
+ * encode -> slice -> decode round trip yields real text rather than the ids themselves.
36984
+ * @param tokens - Token ids, typically a slice of an encodeTokens result
36985
+ * @param modelId - Model ID to determine encoding (must match the one used to encode)
36986
+ * @returns Promise<string> - The decoded text
36987
+ *
36988
+ * tiktoken's wasm decode() hands back raw UTF-8 bytes. A slice that ends mid-character therefore
36989
+ * decodes to a trailing U+FFFD; callers that sliced are expected to trim it.
36990
+ */
36991
+ async decodeTokens(tokens, modelId, logger) {
36992
+ if (this.isShuttingDown) throw new Error("TiktokenTokenizer is shutting down");
36993
+ const encoder = await this.getEncoder(modelId, logger);
36994
+ return new TextDecoder().decode(encoder.decode(new Uint32Array(tokens)));
36995
+ }
36996
+ /**
35178
36997
  * Returns a lightweight ITokenizer proxy that delegates WASM encoder operations
35179
36998
  * to this instance (preserving the shared encoder cache) but routes log output
35180
36999
  * through the provided logger. Useful for attaching per-request context (e.g.
@@ -35183,7 +37002,8 @@ var TiktokenTokenizer = class {
35183
37002
  withLogger(logger) {
35184
37003
  return {
35185
37004
  countTokens: (text, modelId) => this.countTokens(text, modelId, logger),
35186
- encodeTokens: (text, modelId) => this.encodeTokens(text, modelId, logger)
37005
+ encodeTokens: (text, modelId) => this.encodeTokens(text, modelId, logger),
37006
+ decodeTokens: (tokens, modelId) => this.decodeTokens(tokens, modelId, logger)
35187
37007
  };
35188
37008
  }
35189
37009
  /**
@@ -35643,6 +37463,7 @@ __reExport(/* @__PURE__ */ __exportAll({
35643
37463
  BaseStorage: () => BaseStorage,
35644
37464
  BedrockEmbeddingService: () => BedrockEmbeddingService,
35645
37465
  CONTENT_TYPE_BY_FORMAT: () => CONTENT_TYPE_BY_FORMAT,
37466
+ COSINE_SEARCH_TOP_K: () => 10,
35646
37467
  CacheKeys: () => CacheKeys,
35647
37468
  ChunkSchema: () => ChunkSchema,
35648
37469
  CircuitBreaker: () => CircuitBreaker,
@@ -35737,6 +37558,7 @@ __reExport(/* @__PURE__ */ __exportAll({
35737
37558
  checkStorageLimit: () => checkStorageLimit,
35738
37559
  checkStorageLimitForFile: () => checkStorageLimitForFile,
35739
37560
  cleanMermaidSyntax: () => cleanMermaidSyntax,
37561
+ computeContentHash: () => computeContentHash,
35740
37562
  computeCosineSimilarity: () => computeCosineSimilarity,
35741
37563
  computeVerbatimTokenBudget: () => computeVerbatimTokenBudget,
35742
37564
  convertCodeBlocksToArtifacts: () => convertCodeBlocksToArtifacts,
@@ -35808,7 +37630,9 @@ __reExport(/* @__PURE__ */ __exportAll({
35808
37630
  registerLambdaErrorHandlers: () => registerLambdaErrorHandlers,
35809
37631
  registerProcessErrorHandlers: () => registerProcessErrorHandlers,
35810
37632
  registrableDomain: () => registrableDomain,
37633
+ reservationOutputTokens: () => reservationOutputTokens,
35811
37634
  resolveEmbeddingConfig: () => resolveEmbeddingConfig,
37635
+ resolveEmbeddingWithKeylessFallback: () => resolveEmbeddingWithKeylessFallback,
35812
37636
  resolveSupportedMimeType: () => resolveSupportedMimeType,
35813
37637
  safeInputWindow: () => safeInputWindow,
35814
37638
  scopedOverrideKey: () => scopedOverrideKey,