@bike4mind/cli 0.21.0 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { $ as SupportedFabFileMimeTypes, $t as buildRateLimitLogEntry, A as HTTPError, At as isRenderableModelType, B as OPENAI_GPT_IMAGE_1_IMAGE_SIZES, Bt as resolveHistoryFetchLimit, Ct as isGeminiModelId, D as FIXED_TEMPERATURE_MODELS, Dt as isModelAccessible, E as FIELD_GROUP_OF, Et as isMediaModelType, F as MODEL_INFO_FIELD_GROUP_OF, Ft as isZodError, G as PermissionDeniedError, Gt as usdToCredits, H as OllamaEmbeddingModel, Ht as settingsMap, I as McpServerName, It as mapMimeTypeToArtifactType, J as REFUSAL_FALLBACK_MODELS, K as REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS, Kt as usdToCreditsStochastic, L as ModelBackend, Lt as obfuscateApiKey, M as IMAGE_SIZE_CONSTRAINTS, Mt as isSupportedFabFileMimeType, N as ImageModels, Nt as isUnlimitedHistory, O as FORMAT_PROMPT_TEMPLATE, Ot as isModelDeprecated, P as InternalServerError, Pt as isUserInitiatedAbort, Q as SpeechToTextModels, R as NO_TEMPERATURE_MODELS, Rt as parseEmbeddingRateLimitHeaders, S as CorruptedFileError, St as isGPTImageModel, Tt as isImageServeable, U as OpenAIEmbeddingModel, Ut as toModelInfo, V as OPENAI_GPT_IMAGE_2_IMAGE_SIZES, Vt as secureParameters, Wt as toModelRecord, Y as RESPONSES_API_TOOL_MODELS, _ as BadRequestError, _t as isChunkRebuildPending, at as VideoModels, bt as isFieldGroup, ct as applyModelPriceCatalog, dt as defaultEmbeddingModelForEnv, en as extractSnippetMeta, et as TTS_MAX_INPUT_CHARS, ft as getMcpProviderMetadata, g as BFL_SAFETY_TOLERANCE, gt as isAudioMimeType, h as BEDROCK_NO_PROMPT_CACHING_MODELS, ht as hasUsableLimits, it as VIDEO_SIZE_CONSTRAINTS, j as HttpStatus, jt as isRetryableError, k as ForbiddenError, kt as isPlaceholderApiKey, lt as calculateRetryDelay, m as ApiKeyType, mt as getRetryAfterMs, n as logger, nn as parseRateLimitHeaders, nt as UnauthorizedError, ot as VoyageAIEmbeddingModel, p as ARTIFACT_ATTRS_PATTERN, pt as getQuestErrorCode, q as REASONING_SUPPORTED_MODELS, qt as withRetry, rt as UnprocessableEntityError, st as WORK_ITEM_STATUSES, tn as isNearLimit, tt as TooManyRequestsError, ut as dayjsConfig_default, v as BedrockEmbeddingModel, vt as isChunkStalledFile, w as DEFAULT_UNKNOWN_CONTEXT_WINDOW, wt as isImageAttachment, x as ChatModels, xt as isGPTImage2Model, y as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, z as NotFoundError, zt as reservationOutputTokens } from "./ConfigStore-CoY0l0gr.mjs";
2
+ import { $ as SupportedFabFileMimeTypes, A as HTTPError, At as isPlaceholderApiKey, B as OPENAI_GPT_IMAGE_1_IMAGE_SIZES, Bt as reservationOutputTokens, Ct as isGPTImageModel, D as FIXED_TEMPERATURE_MODELS, Dt as isMediaModelType, E as FIELD_GROUP_OF, Et as isImageServeable, F as MODEL_INFO_FIELD_GROUP_OF, Ft as isUserInitiatedAbort, G as PermissionDeniedError, Gt as toModelRecord, H as OllamaEmbeddingModel, Ht as secureParameters, I as McpServerName, It as isZodError, J as REFUSAL_FALLBACK_MODELS, Jt as withRetry, K as REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS, Kt as usdToCredits, L as ModelBackend, Lt as mapMimeTypeToArtifactType, M as IMAGE_SIZE_CONSTRAINTS, Mt as isRetryableError, N as ImageModels, Nt as isSupportedFabFileMimeType, O as FORMAT_PROMPT_TEMPLATE, Ot as isModelAccessible, P as InternalServerError, Pt as isUnlimitedHistory, Q as SpeechToTextModels, R as NO_TEMPERATURE_MODELS, Rt as obfuscateApiKey, S as CorruptedFileError, St as isGPTImage2Model, Tt as isImageAttachment, U as OpenAIEmbeddingModel, Ut as settingsMap, V as OPENAI_GPT_IMAGE_2_IMAGE_SIZES, Vt as resolveHistoryFetchLimit, Wt as toModelInfo, Y as RESPONSES_API_TOOL_MODELS, _ as BadRequestError, _t as isAudioMimeType, at as VideoModels, ct as applyModelPriceCatalog, dt as defaultEmbeddingModelForEnv, en as buildRateLimitLogEntry, et as TTS_MAX_INPUT_CHARS, ft as getMcpProviderMetadata, g as BFL_SAFETY_TOLERANCE, gt as hasUsableLimits, h as BEDROCK_NO_PROMPT_CACHING_MODELS, ht as hasKeylessCloudEmbedder, it as VIDEO_SIZE_CONSTRAINTS, j as HttpStatus, jt as isRenderableModelType, k as ForbiddenError, kt as isModelDeprecated, lt as calculateRetryDelay, m as ApiKeyType, mt as getRetryAfterMs, n as logger, nn as isNearLimit, nt as UnauthorizedError, ot as VoyageAIEmbeddingModel, p as ARTIFACT_ATTRS_PATTERN, pt as getQuestErrorCode, q as REASONING_SUPPORTED_MODELS, qt as usdToCreditsStochastic, rn as parseRateLimitHeaders, rt as UnprocessableEntityError, st as WORK_ITEM_STATUSES, tn as extractSnippetMeta, tt as TooManyRequestsError, ut as dayjsConfig_default, v as BedrockEmbeddingModel, vt as isChunkRebuildPending, w as DEFAULT_UNKNOWN_CONTEXT_WINDOW, wt as isGeminiModelId, x as ChatModels, xt as isFieldGroup, y as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, yt as isChunkStalledFile, z as NotFoundError, zt as parseEmbeddingRateLimitHeaders } from "./ConfigStore-8_0WsN5r.mjs";
3
3
  import { n as isPathAllowed, t as assertPathAllowed } from "./pathValidation-D8tjkQXE-1HwvsuYT.mjs";
4
4
  import { n as isTerminalShellStatus, t as getShellSessionManager } from "./ShellSessionManager-6o8KZzl1-vrbPAUTq.mjs";
5
5
  import { execFile, execFileSync, spawn } from "child_process";
@@ -21,6 +21,7 @@ import * as turndownPluginGfm from "@joplin/turndown-plugin-gfm";
21
21
  import * as cheerio from "cheerio";
22
22
  import FirecrawlDefault, { FirecrawlError } from "@mendable/firecrawl-js";
23
23
  import { lookup } from "node:dns/promises";
24
+ import mongoose, { isObjectIdOrHexString } from "mongoose";
24
25
  import random from "lodash/random.js";
25
26
  import sum from "lodash/sum.js";
26
27
  import times from "lodash/times.js";
@@ -48,7 +49,6 @@ import { NodeHttpHandler } from "@smithy/node-http-handler";
48
49
  import "@opensearch-project/opensearch";
49
50
  import "@aws-sdk/credential-provider-node";
50
51
  import "@opensearch-project/opensearch/aws-v3";
51
- import mongoose from "mongoose";
52
52
  import { parse } from "shell-quote";
53
53
  import { homedir as homedir$1 } from "node:os";
54
54
  import { EventEmitter } from "events";
@@ -921,6 +921,7 @@ const DEMO_KEY_MAP = {
921
921
  [ApiKeyType.gemini]: "geminiDemoKey",
922
922
  [ApiKeyType.xai]: "xaiApiKey",
923
923
  [ApiKeyType.kimi]: "moonshotApiKey",
924
+ [ApiKeyType.deepseek]: "deepseekApiKey",
924
925
  [ApiKeyType.bfl]: "bflApiKey",
925
926
  [ApiKeyType.voyageai]: "voyageApiKey",
926
927
  [ApiKeyType.elevenlabs]: "elevenLabsServerApiKey"
@@ -978,6 +979,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
978
979
  ApiKeyType.bfl,
979
980
  ApiKeyType.xai,
980
981
  ApiKeyType.kimi,
982
+ ApiKeyType.deepseek,
981
983
  ApiKeyType.voyageai
982
984
  ], adapters) : Promise.resolve([]), adapters.getSettingsByNames([
983
985
  "openaiDemoKey",
@@ -986,6 +988,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
986
988
  "bflApiKey",
987
989
  "xaiApiKey",
988
990
  "moonshotApiKey",
991
+ "deepseekApiKey",
989
992
  "voyageApiKey",
990
993
  "ollamaBackend",
991
994
  "EnableOllama"
@@ -998,6 +1001,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
998
1001
  const bflUserKey = userKeyMap.get(ApiKeyType.bfl) || null;
999
1002
  const xaiUserKey = userKeyMap.get(ApiKeyType.xai) || null;
1000
1003
  const kimiUserKey = userKeyMap.get(ApiKeyType.kimi) || null;
1004
+ const deepseekUserKey = userKeyMap.get(ApiKeyType.deepseek) || null;
1001
1005
  const voyageaiUserKey = userKeyMap.get(ApiKeyType.voyageai) || null;
1002
1006
  const openaiDemoKey = adminSettings["openaiDemoKey"];
1003
1007
  const anthropicDemoKey = adminSettings["anthropicDemoKey"];
@@ -1005,6 +1009,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
1005
1009
  const bflDemoKey = adminSettings["bflApiKey"];
1006
1010
  const xaiDemoKey = adminSettings["xaiApiKey"];
1007
1011
  const kimiDemoKey = adminSettings["moonshotApiKey"];
1012
+ const deepseekDemoKey = adminSettings["deepseekApiKey"];
1008
1013
  const voyageaiDemoKey = adminSettings["voyageApiKey"];
1009
1014
  const ollamaBackend = adminSettings["ollamaBackend"];
1010
1015
  const enableOllama = adminSettings["EnableOllama"];
@@ -1023,6 +1028,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
1023
1028
  bfl: keyOrExpired(bflUserKey) || bflDemoKey || null,
1024
1029
  xai: keyOrExpired(xaiUserKey) || xaiDemoKey || envKey("XAI_API_KEY"),
1025
1030
  kimi: keyOrExpired(kimiUserKey) || kimiDemoKey || envKey("MOONSHOT_API_KEY"),
1031
+ deepseek: keyOrExpired(deepseekUserKey) || deepseekDemoKey || envKey("DEEPSEEK_API_KEY"),
1026
1032
  voyageai: keyOrExpired(voyageaiUserKey) || voyageaiDemoKey || null,
1027
1033
  ollama: (ollamaEnabled ? ollamaBackend || null : null) || envKey("OLLAMA_BASE_URL"),
1028
1034
  imageGen: envKey("IMAGE_GEN_BASE_URL")
@@ -2088,7 +2094,7 @@ const webSearchTool = {
2088
2094
  })
2089
2095
  };
2090
2096
  //#endregion
2091
- //#region ../../b4m-core/services/dist/toolGenerators-hk-Robqc.mjs
2097
+ //#region ../../b4m-core/services/dist/toolGenerators-DGjRmthM.mjs
2092
2098
  const diceRoll = async (parameters) => {
2093
2099
  if (!parameters?.sides || !parameters?.times) throw new Error("Tool dice roll: Missing required parameters");
2094
2100
  return sum(times(parameters.times, () => random(1, parameters.sides))).toString();
@@ -2656,6 +2662,27 @@ const promptEnhancementTool = {
2656
2662
  }
2657
2663
  })
2658
2664
  };
2665
+ /**
2666
+ * Is this value shaped like something Mongoose can cast to an `_id`?
2667
+ *
2668
+ * Tool arguments are composed by the model out of conversation text and reach us as unvalidated
2669
+ * JSON, so an id parameter routinely holds something that is not an id at all - a filename token,
2670
+ * an arXiv number, a bare integer. Mongoose casts `_id` and throws a CastError on those, which a
2671
+ * generic catch upstream then reports as an outage rather than the bad argument it is (#2530).
2672
+ * Call this before handing a model-supplied id to `findById` and answer a false the same way the
2673
+ * surface answers a genuinely missing row.
2674
+ *
2675
+ * `isObjectIdOrHexString`, not `isValidObjectId`: the latter also accepts a number and casts it to
2676
+ * a fabricated id, and a model emitting `{"file_id": 12}` gives us exactly that despite the
2677
+ * `string` type. Same choice, same reason, as `usableObjectIds` in @bike4mind/db-core, which is
2678
+ * the array-shaped version of this check.
2679
+ *
2680
+ * NOT usable for artifact ids (`artifact_<...>`), which are matched on a string `id` field rather
2681
+ * than `_id` - see `createArtifactId` in @bike4mind/common.
2682
+ */
2683
+ function isObjectIdShaped(id) {
2684
+ return isObjectIdOrHexString(id);
2685
+ }
2659
2686
  let _showUserQuestion = null;
2660
2687
  /**
2661
2688
  * Inject the CLI callback that displays the question UI.
@@ -4268,7 +4295,7 @@ const latticeAddEntityTool = {
4268
4295
  createdAt: /* @__PURE__ */ new Date(),
4269
4296
  updatedAt: /* @__PURE__ */ new Date()
4270
4297
  };
4271
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4298
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4272
4299
  const model = await context.db.latticeModels.findById(modelId);
4273
4300
  if (model && model.userId === context.userId) {
4274
4301
  const existingIndex = model.data.entities.findIndex((e) => e.id === entityId);
@@ -4412,7 +4439,7 @@ const latticeSetValueTool = {
4412
4439
  else if (rawValue.toLowerCase() === "true") value = true;
4413
4440
  else if (rawValue.toLowerCase() === "false") value = false;
4414
4441
  const entityId = entityName.toLowerCase().replace(/\s+/g, "_");
4415
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4442
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4416
4443
  const model = await context.db.latticeModels.findById(modelId);
4417
4444
  if (model && model.userId === context.userId) {
4418
4445
  const entity = model.data.entities.find((e) => e.id === entityId || e.name === entityName);
@@ -4547,7 +4574,7 @@ const latticeCreateRuleTool = {
4547
4574
  };
4548
4575
  const outputEntityId = parsedRule.outputEntity.toLowerCase().replace(/\s+/g, "_");
4549
4576
  let entityCreatedMessage = "";
4550
- if (context.db.latticeModels && modelId && /^[a-f0-9]{24}$/.test(modelId)) try {
4577
+ if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
4551
4578
  const model = await context.db.latticeModels.findById(modelId);
4552
4579
  if (model && model.userId === context.userId) {
4553
4580
  if (!model.data.entities.some((e) => e.id === outputEntityId || e.name.toLowerCase() === parsedRule.outputEntity.toLowerCase()) && parsedRule.outputEntity !== "unknown") {
@@ -7230,7 +7257,25 @@ const getProviderFromModel = (modelName) => {
7230
7257
  * ("OpenAI rejected the embedding request") instead of the actionable missing-credential path.
7231
7258
  */
7232
7259
  const EXPIRED_KEY_SENTINEL = "expired";
7233
- const usableKey = (value) => value && value !== EXPIRED_KEY_SENTINEL ? value : null;
7260
+ /**
7261
+ * A placeholder is rejected for the same reason the sentinel is, and the two sibling answers to
7262
+ * "is this key usable" both already do it (modelDiscoveryService/credentials.ts,
7263
+ * toolAvailability.ts, and defaultEmbeddingModelForEnv's own key test). Keeping a placeholder here
7264
+ * would report `missing: null`, so the keyless fallback would never fire and EmbeddingFactory would
7265
+ * then throw on the placeholder itself - the PR's headline case failing silently rather than
7266
+ * substituting. `.trim()` because a whitespace-only value is no key either.
7267
+ */
7268
+ const usableKey = (value) => {
7269
+ const trimmed = value?.trim();
7270
+ if (!trimmed || trimmed === EXPIRED_KEY_SENTINEL || isPlaceholderApiKey(trimmed)) return null;
7271
+ return trimmed;
7272
+ };
7273
+ /**
7274
+ * The slot is missing because THIS CALLER's key expired, not because the deployment holds none.
7275
+ * Bedrock has no credential and Ollama's base URL carries no expiry, so only the two keyed cloud
7276
+ * providers can be in this state. See the keyless-fallback doc comment for why it matters.
7277
+ */
7278
+ const isExpiredCallerKey = (missing, keyTable) => missing === "openai" && keyTable?.openai === EXPIRED_KEY_SENTINEL || missing === "voyageai" && keyTable?.voyageai === EXPIRED_KEY_SENTINEL;
7234
7279
  /**
7235
7280
  * Map an embedding provider plus the caller's resolved key table to the config
7236
7281
  * `EmbeddingFactory` expects, and report which credential is missing if any.
@@ -7250,6 +7295,13 @@ const usableKey = (value) => value && value !== EXPIRED_KEY_SENTINEL ? value : n
7250
7295
  *
7251
7296
  * Adding a provider means editing this function and its table test, not auditing
7252
7297
  * every call site.
7298
+ *
7299
+ * `keyTable` is always an ANSWER about the caller's credentials, never a failure channel. `null` /
7300
+ * `undefined` mean "resolved: this caller holds none", and both this function and the keyless
7301
+ * fallback below act on that - substituting the keyless embedder is a real decision with a real
7302
+ * vector space attached. A caller whose own key lookup THREW must therefore not pass the failure in
7303
+ * here; it has to report unknown instead, or an unavailable Mongo becomes a confident Titan on a
7304
+ * fully keyed production stage.
7253
7305
  */
7254
7306
  function resolveEmbeddingConfig(provider, keyTable) {
7255
7307
  switch (provider) {
@@ -7273,19 +7325,78 @@ function resolveEmbeddingConfig(provider, keyTable) {
7273
7325
  missing: "voyageai"
7274
7326
  };
7275
7327
  }
7276
- case ModelBackend.Ollama: return keyTable?.ollama ? {
7277
- config: { ollamaBaseUrl: keyTable.ollama },
7278
- missing: null
7279
- } : {
7280
- config: {},
7281
- missing: "ollama"
7282
- };
7328
+ case ModelBackend.Ollama: {
7329
+ const baseUrl = keyTable?.ollama?.trim();
7330
+ return baseUrl ? {
7331
+ config: { ollamaBaseUrl: baseUrl },
7332
+ missing: null
7333
+ } : {
7334
+ config: {},
7335
+ missing: "ollama"
7336
+ };
7337
+ }
7283
7338
  case ModelBackend.Bedrock: return {
7284
7339
  config: {},
7285
7340
  missing: null
7286
7341
  };
7287
7342
  }
7288
7343
  }
7344
+ /**
7345
+ * Resolve a config for `model`, falling back to keyless Bedrock when this deployment holds no
7346
+ * credential for the provider `model` needs but can reach Bedrock with its own AWS role.
7347
+ *
7348
+ * WHY THIS EXISTS HERE and not in `defaultEmbeddingModelForEnv`: "does this deployment have a
7349
+ * cloud embedding key" is unanswerable from process.env on a hosted stage - an SST secret arrives
7350
+ * as a linked Resource, so OPENAI_API_KEY is absent on production exactly as it is on a preview.
7351
+ * The key table passed in here is the first point that actually knows, which is why the decision
7352
+ * belongs at this seam.
7353
+ *
7354
+ * Related to but NOT the same as EmbeddingFactory.getDefaultEmbeddingModel, which ranks providers
7355
+ * from scratch (OpenAI > VoyageAI > Ollama > Bedrock). This keeps the model the admin asked for
7356
+ * whenever it is reachable and only substitutes the keyless one otherwise - so a deployment
7357
+ * holding only a Voyage key still falls back to Bedrock here, where the factory would pick
7358
+ * voyage-3. Deliberate: this is a reachability backstop, not a second opinion on the setting.
7359
+ *
7360
+ * ONLY FOR CALLERS FREE TO CHOOSE THE MODEL - i.e. the model came from the `defaultEmbeddingModel`
7361
+ * admin setting. A caller that must hit one specific vector space MUST keep using
7362
+ * `resolveEmbeddingConfig` and fail, because a fallback there would silently compare or write
7363
+ * across incompatible spaces:
7364
+ * - V2 mementos are pinned to MEMENTO_EMBEDDING_MODEL at 512 truncated dims (see embedding.ts);
7365
+ * - V1 mementos (mementoEmbedding.ts, getRelevantMementos.ts) read the admin default and so LOOK
7366
+ * free to choose, but neither live write path stamps `Memento.embeddingModel` - only the
7367
+ * reembedMementos backfill does. Their vectors are ranked by in-process cosine with no width
7368
+ * guard and no Atlas index, so a substitution here would drop 1024-dim vectors into a field
7369
+ * holding 1536-dim ones with nothing recording which is which, and nothing able to tell them
7370
+ * apart afterwards. Stamping V1 is the prerequisite for including it, not this helper.
7371
+ * - alternateModelAnn embeds one query per model bucket to match each chunk's recorded stamp.
7372
+ *
7373
+ * Returns the model actually used, so callers stamp what they embedded with rather than what they
7374
+ * asked for - that is what keeps `fabFileChunk`'s recorded `embeddingModel` honest.
7375
+ *
7376
+ * TWO credential states are deliberately NOT treated as "this deployment is keyless":
7377
+ * - `missing: 'ollama'` - a self-host that set no OLLAMA_BASE_URL has no AWS role either, and
7378
+ * OPENAI_KEY_MISSING_MESSAGE naming OPENAI_API_KEY / OLLAMA_BASE_URL is the actionable error
7379
+ * there. `hasKeylessCloudEmbedder()` already excludes self-host; this is belt-and-braces.
7380
+ * - an EXPIRED caller key. `getEffectiveLLMApiKeys` returns the `'expired'` sentinel instead of
7381
+ * falling through to the platform demo key, deliberately, so the user is told their key
7382
+ * expired rather than silently moved onto the platform's (see the reasoning in the
7383
+ * reactivate-collateral-deactivated-api-keys migration). `usableKey` normalizes that to null
7384
+ * for the CREDENTIAL check, which is right - but read as "this deployment holds no key" it
7385
+ * would substitute Titan for that one caller on keyed production, querying a vector space the
7386
+ * corpus was never written in. The deployment's own key state is unchanged by one expiry, so
7387
+ * the requested model is returned and the actionable expired-key error stands.
7388
+ */
7389
+ function resolveEmbeddingWithKeylessFallback(model, keyTable) {
7390
+ const resolved = resolveEmbeddingConfig(getProviderFromModel(model), keyTable);
7391
+ if (!resolved.missing || resolved.missing === "ollama" || isExpiredCallerKey(resolved.missing, keyTable) || !hasKeylessCloudEmbedder()) return {
7392
+ ...resolved,
7393
+ model
7394
+ };
7395
+ return {
7396
+ ...resolveEmbeddingConfig(ModelBackend.Bedrock, null),
7397
+ model: BedrockEmbeddingModel.TITAN_TEXT_EMBEDDINGS_V2
7398
+ };
7399
+ }
7289
7400
  const ChunkSchema = z$1.object({
7290
7401
  text: z$1.string(),
7291
7402
  tokenCount: z$1.number()
@@ -10099,16 +10210,16 @@ function parseSettingsHooks(settingsJson) {
10099
10210
  return null;
10100
10211
  }
10101
10212
  }
10102
- let cached;
10213
+ let cached$1;
10103
10214
  /**
10104
10215
  * Lazily-built process-hook singleton from `B4M_SETTINGS_JSON`. Returns null when
10105
10216
  * no hooks are configured, so call sites can `void getProcessHooks()?.fireStop()`.
10106
10217
  */
10107
10218
  function getProcessHooks() {
10108
- if (cached !== void 0) return cached;
10219
+ if (cached$1 !== void 0) return cached$1;
10109
10220
  const hooks = parseSettingsHooks(process.env.B4M_SETTINGS_JSON);
10110
- cached = hooks ? new ProcessHooks(hooks) : null;
10111
- return cached;
10221
+ cached$1 = hooks ? new ProcessHooks(hooks) : null;
10222
+ return cached$1;
10112
10223
  }
10113
10224
  //#endregion
10114
10225
  //#region src/agents/interactionModeClamp.ts
@@ -15940,6 +16051,10 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15940
16051
  BaseBedrockBackend: () => BaseBedrockBackend,
15941
16052
  ChoiceEndReason: () => ChoiceEndReason,
15942
16053
  ChoiceStatus: () => ChoiceStatus,
16054
+ DEEPSEEK_EFFORT_LEVELS: () => DEEPSEEK_EFFORT_LEVELS,
16055
+ DEEPSEEK_MAX_STOP_SEQUENCES: () => 16,
16056
+ DEEPSEEK_MODELS: () => DEEPSEEK_MODELS,
16057
+ DEEPSEEK_THINKING_TOP_P_FLOOR: () => DEEPSEEK_THINKING_TOP_P_FLOOR,
15943
16058
  DEFAULT_MAX_TOOL_CALLS: () => 10,
15944
16059
  DEFAULT_REALTIME_VOICE_MODEL: () => DEFAULT_REALTIME_VOICE_MODEL,
15945
16060
  DEGENERATE_STREAM_MESSAGE: () => DEGENERATE_STREAM_MESSAGE,
@@ -15947,6 +16062,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15947
16062
  DEPRECATED_MODEL_MAP: () => DEPRECATED_MODEL_MAP,
15948
16063
  DEPRECATED_MODEL_REQUEST_METRIC: () => DEPRECATED_MODEL_REQUEST_METRIC,
15949
16064
  DISPATCHABLE_ADAPTER_FAMILIES: () => DISPATCHABLE_ADAPTER_FAMILIES,
16065
+ DeepSeekBackend: () => DeepSeekBackend,
15950
16066
  DeepSeekBedrockBackend: () => DeepSeekBedrockBackend,
15951
16067
  DispatchModel: () => DispatchModel,
15952
16068
  GeminiBackend: () => GeminiBackend,
@@ -15967,6 +16083,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15967
16083
  UndifferentiatedBedrockBackend: () => UndifferentiatedBedrockBackend,
15968
16084
  UnsupportedAdapterFamilyError: () => UnsupportedAdapterFamilyError,
15969
16085
  XAIBackend: () => XAIBackend,
16086
+ adapterPriceTiers: () => adapterPriceTiers,
15970
16087
  backendForAdapterFamily: () => backendForAdapterFamily,
15971
16088
  buildApiKeyTable: () => buildApiKeyTable,
15972
16089
  buildSupersededIndex: () => buildSupersededIndex,
@@ -15977,6 +16094,10 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
15977
16094
  checkStaleModelReferences: () => checkStaleModelReferences,
15978
16095
  classifyModelReference: () => classifyModelReference,
15979
16096
  createDegenerateStreamGuard: () => createDegenerateStreamGuard,
16097
+ deepseekReasoningParams: () => deepseekReasoningParams,
16098
+ deepseekSamplingParams: () => deepseekSamplingParams,
16099
+ deepseekStopSequences: () => deepseekStopSequences,
16100
+ deepseekThinkingEnabled: () => deepseekThinkingEnabled,
15980
16101
  ensureToolPairingIntegrity: () => ensureToolPairingIntegrity,
15981
16102
  extractThinkContent: () => extractThinkContent,
15982
16103
  getAvailableModels: () => getAvailableModels,
@@ -16011,6 +16132,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
16011
16132
  splitCacheInclusiveInput: () => splitCacheInclusiveInput,
16012
16133
  stripAllToolBlocks: () => stripAllToolBlocks,
16013
16134
  stripToolDependentMessages: () => stripToolDependentMessages,
16135
+ toDeepSeekEffort: () => toDeepSeekEffort,
16014
16136
  toKimiEffort: () => toKimiEffort,
16015
16137
  toProviderEndUserId: () => toProviderEndUserId,
16016
16138
  updateReplacedByOverlay: () => updateReplacedByOverlay
@@ -16726,6 +16848,92 @@ var KimiCachingAdapter = class {
16726
16848
  }
16727
16849
  };
16728
16850
  /**
16851
+ * The cache-inclusive-to-cache-exclusive conversion, shared by every adapter whose
16852
+ * provider reports cached tokens as a SUBSET of the prompt count.
16853
+ *
16854
+ * getTextModelCost expects Anthropic's convention: `inputTokens` counts only uncached
16855
+ * tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
16856
+ * and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
16857
+ * prompt total already CONTAINS the cached tokens - so those adapters must subtract
16858
+ * here before forwarding, or settlement double-bills the cached portion.
16859
+ *
16860
+ * Must stay in sync with the disjoint-fields assumption documented at the settlement
16861
+ * site in ChatCompletionProcess.
16862
+ */
16863
+ /**
16864
+ * Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
16865
+ *
16866
+ * Forwarding the cached count without subtracting double-bills it; forwarding nothing
16867
+ * charges the full input rate on tokens the provider billed at a fraction of it.
16868
+ * Subtracting is the only split that bills what the provider actually charged.
16869
+ *
16870
+ * Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
16871
+ * input count would silently credit the user.
16872
+ */
16873
+ function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
16874
+ if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
16875
+ const cached = Math.min(cacheReadTokens, totalPromptTokens);
16876
+ return {
16877
+ inputTokens: Math.max(0, totalPromptTokens - cached),
16878
+ cacheReadInputTokens: cached
16879
+ };
16880
+ }
16881
+ /**
16882
+ * Cached prompt tokens from a raw provider usage object, across every spelling in use:
16883
+ * OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
16884
+ * Responses API under `input_tokens_details`, Moonshot publishes a flat
16885
+ * `cached_tokens` alongside the OpenAI-shaped nesting, and DeepSeek its own flat
16886
+ * `prompt_cache_hit_tokens`. Reading only one spelling silently bills every cache
16887
+ * hit on the other transports at the full input rate.
16888
+ *
16889
+ * DeepSeek's own spelling leads, because it is the number its invoice is computed
16890
+ * from; the OpenAI-shaped ones it also sends are the fallback for a proxy that
16891
+ * forwards only those.
16892
+ */
16893
+ function cachedTokensFromUsage(usage) {
16894
+ if (!usage) return 0;
16895
+ const candidates = [
16896
+ usage.prompt_cache_hit_tokens,
16897
+ usage.cached_tokens,
16898
+ usage.prompt_tokens_details?.cached_tokens,
16899
+ usage.input_tokens_details?.cached_tokens
16900
+ ];
16901
+ for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
16902
+ return 0;
16903
+ }
16904
+ /**
16905
+ * DeepSeek context caching. Automatic, like Moonshot's and xAI's: no parameter,
16906
+ * no header, no explicit cache-creation call. The adapter exists only to read
16907
+ * the counters back out.
16908
+ * @see https://api-docs.deepseek.com/guides/kv_cache
16909
+ */
16910
+ var DeepSeekCachingAdapter = class {
16911
+ applyCaching(apiParams, _strategy) {
16912
+ return apiParams;
16913
+ }
16914
+ extractCacheStats(response, model) {
16915
+ const usage = response.usage;
16916
+ if (!usage) return void 0;
16917
+ const totalInputTokens = usage.prompt_tokens || 0;
16918
+ const cachedTokens = cachedTokensFromUsage(usage);
16919
+ const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
16920
+ const costSavingsPercent = cacheHitRate * .98;
16921
+ const estimatedLatencyReduction = cacheHitRate * .7;
16922
+ return {
16923
+ provider: ModelBackend.DeepSeek,
16924
+ model,
16925
+ totalInputTokens,
16926
+ cacheReadTokens: cachedTokens,
16927
+ cacheWriteTokens: 0,
16928
+ uncachedTokens: Math.max(0, totalInputTokens - cachedTokens),
16929
+ cacheHitRate,
16930
+ costSavingsPercent,
16931
+ estimatedLatencyReduction,
16932
+ providerMetadata: { automatic: true }
16933
+ };
16934
+ }
16935
+ };
16936
+ /**
16729
16937
  * Helper to log cache statistics in a consistent format across all providers
16730
16938
  */
16731
16939
  function logCacheStats(logger, cacheStats, options) {
@@ -16759,6 +16967,7 @@ const ADAPTERS = {
16759
16967
  [ModelBackend.Bedrock]: new AnthropicCachingAdapter(),
16760
16968
  [ModelBackend.XAI]: new XAICachingAdapter(),
16761
16969
  [ModelBackend.Kimi]: new KimiCachingAdapter(),
16970
+ [ModelBackend.DeepSeek]: new DeepSeekCachingAdapter(),
16762
16971
  [ModelBackend.Ollama]: new NoOpCachingAdapter(),
16763
16972
  [ModelBackend.BFL]: new NoOpCachingAdapter(),
16764
16973
  [ModelBackend.VoyageAI]: new NoOpCachingAdapter(),
@@ -16809,14 +17018,28 @@ const ADAPTIVE_THINKING_MAX_TOKENS_FLOOR = 64e3;
16809
17018
  const THINKING_ANSWER_HEADROOM_TOKENS = 1e3;
16810
17019
  /**
16811
17020
  * Reasoning-inside-the-budget ids that none of the shape checks below can infer.
17021
+ *
16812
17022
  * Bedrock's Kimi always reasons, but it is not Anthropic-adaptive, does not take
16813
17023
  * `reasoning_effort`, and sends plain `max_tokens` - so it looks like an ordinary
16814
17024
  * model at every seam we can inspect. Bedrock copies the monologue inline into
16815
17025
  * `content` (see bedrockBackend/moonshot.ts) and caps output at 16K, so the floor
16816
17026
  * resolves to that entire cap, which is the only value leaving room for an answer
16817
17027
  * after a long trace.
16818
- */
16819
- const REASONS_WITHIN_OUTPUT_BUDGET_IDS = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_THINKING_BEDROCK, ChatModels.KIMI_K2_5_BEDROCK]);
17028
+ *
17029
+ * DeepSeek Flash misses every clause for its own set of reasons: no
17030
+ * `thinkingStyle` (that field is Anthropic's), absent from the OpenAI-only
17031
+ * REASONING_SUPPORTED_MODELS, and DEEPSEEK_PROFILE declares plain `max_tokens`
17032
+ * rather than `max_completion_tokens` because that is the parameter DeepSeek
17033
+ * takes. It reasons on every turn by default at effort 'high', spends those
17034
+ * tokens inside `max_tokens`, and a 4096 budget against a 393K cap is consumed
17035
+ * by the monologue alone: the turn comes back `finish_reason: 'length'` with no
17036
+ * content and deepseekBackend throws.
17037
+ */
17038
+ const REASONS_WITHIN_OUTPUT_BUDGET_IDS = /* @__PURE__ */ new Set([
17039
+ ChatModels.KIMI_K2_THINKING_BEDROCK,
17040
+ ChatModels.KIMI_K2_5_BEDROCK,
17041
+ ChatModels.DEEPSEEK_FLASH
17042
+ ]);
16820
17043
  /**
16821
17044
  * Whether the model spends reasoning tokens inside its output budget on every turn,
16822
17045
  * which is what makes a small budget produce an empty visible reply rather than a
@@ -17332,7 +17555,7 @@ var AnthropicBackend = class {
17332
17555
  supportsTools: true,
17333
17556
  supportsImageVariation: false,
17334
17557
  logoFile: "Anthropic_logo.png",
17335
- rank: 1,
17558
+ rank: 2,
17336
17559
  trainingCutoff: "2024-10-01",
17337
17560
  releaseDate: "2025-05-23",
17338
17561
  deprecationDate: "2026-06-01",
@@ -17355,7 +17578,7 @@ var AnthropicBackend = class {
17355
17578
  supportsTools: true,
17356
17579
  supportsImageVariation: false,
17357
17580
  logoFile: "Anthropic_logo.png",
17358
- rank: 1,
17581
+ rank: 2,
17359
17582
  trainingCutoff: "2025-07-01",
17360
17583
  releaseDate: "2025-09-30",
17361
17584
  description: "Anthropic's most intelligent model in the Claude 4 family. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
@@ -17375,7 +17598,7 @@ var AnthropicBackend = class {
17375
17598
  } },
17376
17599
  supportsVision: true,
17377
17600
  logoFile: "Anthropic_logo.png",
17378
- rank: 1,
17601
+ rank: 3,
17379
17602
  supportsTools: true,
17380
17603
  trainingCutoff: "2025-07-01",
17381
17604
  releaseDate: "2025-10-16",
@@ -17422,7 +17645,7 @@ var AnthropicBackend = class {
17422
17645
  supportsTools: true,
17423
17646
  supportsImageVariation: false,
17424
17647
  logoFile: "Anthropic_logo.png",
17425
- rank: 1,
17648
+ rank: 2,
17426
17649
  trainingCutoff: "2025-10-01",
17427
17650
  releaseDate: "2026-02-19",
17428
17651
  description: "Anthropic's Claude 4.6 Sonnet model. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
@@ -17445,7 +17668,7 @@ var AnthropicBackend = class {
17445
17668
  supportsTools: true,
17446
17669
  supportsImageVariation: false,
17447
17670
  logoFile: "Anthropic_logo.png",
17448
- rank: 0,
17671
+ rank: 1,
17449
17672
  trainingCutoff: "2026-01-01",
17450
17673
  releaseDate: "2026-07-01",
17451
17674
  description: "Anthropic's newest Claude 5 Sonnet model. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
@@ -17538,7 +17761,7 @@ var AnthropicBackend = class {
17538
17761
  } },
17539
17762
  supportsVision: true,
17540
17763
  logoFile: "Anthropic_logo.png",
17541
- rank: 1,
17764
+ rank: 0,
17542
17765
  supportsTools: true,
17543
17766
  trainingCutoff: "2026-01-01",
17544
17767
  releaseDate: "2026-07-01",
@@ -17562,7 +17785,7 @@ var AnthropicBackend = class {
17562
17785
  } },
17563
17786
  supportsVision: true,
17564
17787
  logoFile: "Anthropic_logo.png",
17565
- rank: 1,
17788
+ rank: 0,
17566
17789
  supportsTools: true,
17567
17790
  releaseDate: "2026-07-24",
17568
17791
  description: "Anthropic's latest flagship model. Claude 5 Opus approaches Fable 5 performance at Opus 4.8 pricing, with adaptive extended thinking, coding, and agentic capabilities.",
@@ -19407,7 +19630,11 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19407
19630
  if (NO_TEMPERATURE_MODELS.has(model)) return true;
19408
19631
  return !this.getModelInfoList().some((m) => m.id === model) && this._dispatch.for(model)?.thinkingStyle === "adaptive";
19409
19632
  }
19410
- /** Static model info list - synchronous access for getPayload, also used by getModelInfo */
19633
+ /**
19634
+ * Static model info list - synchronous access for getPayload, also used by getModelInfo.
19635
+ * `rank` must match the identically-named entry in anthropicBackend.ts: it is the same
19636
+ * model, so the picker must not show the Bedrock copy above or below its direct twin.
19637
+ */
19411
19638
  getModelInfoList() {
19412
19639
  return [
19413
19640
  {
@@ -19531,7 +19758,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19531
19758
  } },
19532
19759
  supportsVision: true,
19533
19760
  logoFile: "Anthropic_logo.png",
19534
- rank: 0,
19761
+ rank: 1,
19535
19762
  supportsTools: true,
19536
19763
  trainingCutoff: "2025-05-01",
19537
19764
  releaseDate: "2025-05-23",
@@ -19554,7 +19781,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19554
19781
  } },
19555
19782
  supportsVision: true,
19556
19783
  logoFile: "Anthropic_logo.png",
19557
- rank: 0,
19784
+ rank: 1,
19558
19785
  supportsTools: true,
19559
19786
  trainingCutoff: "2025-08-01",
19560
19787
  releaseDate: "2025-08-06",
@@ -19577,7 +19804,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19577
19804
  } },
19578
19805
  supportsVision: true,
19579
19806
  logoFile: "Anthropic_logo.png",
19580
- rank: 1,
19807
+ rank: 2,
19581
19808
  supportsTools: true,
19582
19809
  trainingCutoff: "2025-05-01",
19583
19810
  releaseDate: "2025-05-23",
@@ -19600,7 +19827,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19600
19827
  supportsTools: true,
19601
19828
  supportsImageVariation: false,
19602
19829
  logoFile: "Anthropic_logo.png",
19603
- rank: 1,
19830
+ rank: 2,
19604
19831
  trainingCutoff: "2025-07-01",
19605
19832
  releaseDate: "2025-09-30",
19606
19833
  description: "Anthropic's most intelligent model hosted in AWS Bedrock. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
@@ -19620,7 +19847,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19620
19847
  } },
19621
19848
  supportsVision: true,
19622
19849
  logoFile: "Anthropic_logo.png",
19623
- rank: 1,
19850
+ rank: 3,
19624
19851
  supportsTools: true,
19625
19852
  trainingCutoff: "2025-07-01",
19626
19853
  releaseDate: "2025-10-16",
@@ -19667,7 +19894,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19667
19894
  supportsTools: true,
19668
19895
  supportsImageVariation: false,
19669
19896
  logoFile: "Anthropic_logo.png",
19670
- rank: 1,
19897
+ rank: 2,
19671
19898
  trainingCutoff: "2025-10-01",
19672
19899
  releaseDate: "2026-02-19",
19673
19900
  description: "Anthropic's Claude 4.6 Sonnet model via AWS Bedrock. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
@@ -19690,7 +19917,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19690
19917
  supportsTools: true,
19691
19918
  supportsImageVariation: false,
19692
19919
  logoFile: "Anthropic_logo.png",
19693
- rank: 0,
19920
+ rank: 1,
19694
19921
  trainingCutoff: "2026-01-01",
19695
19922
  releaseDate: "2026-07-01",
19696
19923
  description: "Anthropic's newest Claude 5 Sonnet model via AWS Bedrock. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
@@ -19711,7 +19938,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19711
19938
  } },
19712
19939
  supportsVision: true,
19713
19940
  logoFile: "Anthropic_logo.png",
19714
- rank: 0,
19941
+ rank: 1,
19715
19942
  supportsTools: true,
19716
19943
  trainingCutoff: "2025-05-01",
19717
19944
  releaseDate: "2026-02-06",
@@ -19735,7 +19962,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19735
19962
  } },
19736
19963
  supportsVision: true,
19737
19964
  logoFile: "Anthropic_logo.png",
19738
- rank: 0,
19965
+ rank: 1,
19739
19966
  supportsTools: true,
19740
19967
  trainingCutoff: "2025-10-01",
19741
19968
  releaseDate: "2026-04-17",
@@ -19759,7 +19986,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
19759
19986
  } },
19760
19987
  supportsVision: true,
19761
19988
  logoFile: "Anthropic_logo.png",
19762
- rank: 0,
19989
+ rank: 1,
19763
19990
  supportsTools: true,
19764
19991
  trainingCutoff: "2026-01-01",
19765
19992
  releaseDate: "2026-05-28",
@@ -20415,7 +20642,7 @@ var JurassicTwoBedrockBackend = class extends BaseBedrockBackend {
20415
20642
  } },
20416
20643
  supportsVision: false,
20417
20644
  logoFile: "AI21Labs.png",
20418
- rank: 50,
20645
+ rank: 51,
20419
20646
  description: "AI21 Labs' balanced Jurassic-2 model offering good performance at moderate cost. Great for everyday tasks and general content generation."
20420
20647
  }];
20421
20648
  }
@@ -21511,7 +21738,7 @@ var GeminiBackend = class {
21511
21738
  supportsVision: true,
21512
21739
  supportsTools: true,
21513
21740
  logoFile: "Google_logo.png",
21514
- rank: 5,
21741
+ rank: 6,
21515
21742
  trainingCutoff: "2025-01-31",
21516
21743
  releaseDate: "2025-11-30",
21517
21744
  description: "Google's Gemini 3 Flash preview for fast, low-latency multimodal understanding, delivering richer visuals and deeper interactivity, built on a foundation of state-of-the-art reasoning."
@@ -22267,112 +22494,38 @@ var GeminiBackend = class {
22267
22494
  }
22268
22495
  };
22269
22496
  /**
22270
- * The cache-inclusive-to-cache-exclusive conversion, shared by every adapter whose
22271
- * provider reports cached tokens as a SUBSET of the prompt count.
22272
- *
22273
- * getTextModelCost expects Anthropic's convention: `inputTokens` counts only uncached
22274
- * tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
22275
- * and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
22276
- * prompt total already CONTAINS the cached tokens - so those adapters must subtract
22277
- * here before forwarding, or settlement double-bills the cached portion.
22278
- *
22279
- * Must stay in sync with the disjoint-fields assumption documented at the settlement
22280
- * site in ChatCompletionProcess.
22281
- */
22282
- /**
22283
- * Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
22284
- *
22285
- * Forwarding the cached count without subtracting double-bills it; forwarding nothing
22286
- * charges the full input rate on tokens the provider billed at a fraction of it.
22287
- * Subtracting is the only split that bills what the provider actually charged.
22288
- *
22289
- * Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
22290
- * input count would silently credit the user.
22291
- */
22292
- function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
22293
- if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
22294
- const cached = Math.min(cacheReadTokens, totalPromptTokens);
22295
- return {
22296
- inputTokens: Math.max(0, totalPromptTokens - cached),
22297
- cacheReadInputTokens: cached
22298
- };
22299
- }
22300
- /**
22301
- * Cached prompt tokens from a raw provider usage object, across every spelling in use:
22302
- * OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
22303
- * Responses API under `input_tokens_details`, and Moonshot publishes a flat
22304
- * `cached_tokens` alongside the OpenAI-shaped nesting. Reading only one spelling
22305
- * silently bills every cache hit on the other transports at the full input rate.
22306
- */
22307
- function cachedTokensFromUsage(usage) {
22308
- if (!usage) return 0;
22309
- const candidates = [
22310
- usage.cached_tokens,
22311
- usage.prompt_tokens_details?.cached_tokens,
22312
- usage.input_tokens_details?.cached_tokens
22313
- ];
22314
- for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
22315
- return 0;
22316
- }
22317
- /**
22318
- * Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
22319
- * transport so every "which parameter does this id accept" rule is one pure
22320
- * function with a test, rather than a conditional buried in a 400-line complete().
22497
+ * Request shaping for DeepSeek's direct API. Kept out of deepseekBackend's
22498
+ * transport for the same reason kimiParams is: every "which parameter does this
22499
+ * id accept" rule is one pure function with a test next to it, rather than a
22500
+ * conditional buried in a 400-line complete().
22321
22501
  *
22322
- * Moonshot is OpenAI-compatible in envelope only. The reasoning controls, the
22323
- * sampling pins, and the max-tokens parameter all differ per model, and sending
22324
- * the wrong one is a 400 rather than a silently ignored field.
22325
- * @see https://platform.kimi.ai/docs/api/chat
22502
+ * DeepSeek is OpenAI-compatible in envelope. What differs is thinking mode -
22503
+ * on by default, with its own toggle, its own effort vocabulary, and a sampling
22504
+ * group that is IGNORED rather than rejected while it is on.
22505
+ * @see https://api-docs.deepseek.com/guides/thinking_mode
22326
22506
  */
22327
- /** Kimi's own effort vocabulary, which is not OpenAI's and not B4M's. */
22328
- const KIMI_EFFORT_LEVELS = [
22507
+ /** DeepSeek's effort vocabulary, which is not OpenAI's and not B4M's. */
22508
+ const DEEPSEEK_EFFORT_LEVELS = [
22329
22509
  "low",
22330
22510
  "high",
22331
22511
  "max"
22332
22512
  ];
22333
22513
  /**
22334
- * Takes `reasoning_effort`. K3 only, and K3 always reasons - there is no way to
22335
- * turn thinking off, so the parameter selects depth, never whether.
22514
+ * Every DeepSeek id this build ships, direct-served. Bedrock-served DeepSeek is
22515
+ * not here. Both the reasoning and the sampling shaper gate on THIS set, so the
22516
+ * two cannot disagree about which ids the rules apply to; a test pins it against
22517
+ * the adapter table and against NO_TEMPERATURE_MODELS.
22336
22518
  */
22337
- const EFFORT_MODELS = /* @__PURE__ */ new Set([ChatModels.KIMI_K3]);
22338
- /** Takes the `thinking` object instead of `reasoning_effort`. */
22339
- const THINKING_MODELS = /* @__PURE__ */ new Set([
22340
- ChatModels.KIMI_K2_7_CODE,
22341
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
22342
- ChatModels.KIMI_K2_6,
22343
- ChatModels.KIMI_K2_5
22344
- ]);
22519
+ const DEEPSEEK_MODELS = /* @__PURE__ */ new Set([ChatModels.DEEPSEEK_FLASH]);
22520
+ /** DeepSeek raises anything below this rather than erroring, so we send what it will use. */
22521
+ const DEEPSEEK_THINKING_TOP_P_FLOOR = .95;
22345
22522
  /**
22346
- * `thinking.type` accepts only 'enabled' on the K2.7 code models - 'disabled' is
22347
- * rejected. So a caller asking for no thinking gets thinking anyway; the
22348
- * alternative is a 400, and the parameter is omitted rather than fought.
22523
+ * B4M's six-level effort onto DeepSeek's three. 'none' and 'minimal' map to
22524
+ * 'low' rather than to omission: omitting the parameter leaves DeepSeek's
22525
+ * documented default of 'high', so dropping it on a "least effort" request
22526
+ * would bill more reasoning than was asked for, not less.
22349
22527
  */
22350
- const THINKING_ALWAYS_ON = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED]);
22351
- /**
22352
- * Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
22353
- * function form). Downgraded to 'auto' rather than dropped: a caller that asked
22354
- * for a forced tool still wants tools offered.
22355
- */
22356
- const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
22357
- ChatModels.KIMI_K2_7_CODE,
22358
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
22359
- ChatModels.KIMI_K2_6
22360
- ]);
22361
- /** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
22362
- const KIMI_MODELS = /* @__PURE__ */ new Set([
22363
- ChatModels.KIMI_K3,
22364
- ChatModels.KIMI_K2_7_CODE,
22365
- ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
22366
- ChatModels.KIMI_K2_6,
22367
- ChatModels.KIMI_K2_5
22368
- ]);
22369
- /**
22370
- * B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
22371
- * rather than to omission because K3 cannot be asked not to think - claiming
22372
- * otherwise by dropping the parameter would silently bill max-effort reasoning
22373
- * (Moonshot's default is 'max').
22374
- */
22375
- function toKimiEffort(effort) {
22528
+ function toDeepSeekEffort(effort) {
22376
22529
  if (!effort) return void 0;
22377
22530
  switch (effort) {
22378
22531
  case "none":
@@ -22386,48 +22539,66 @@ function toKimiEffort(effort) {
22386
22539
  }
22387
22540
  /**
22388
22541
  * The reasoning parameters for one model, or an empty object when it takes none.
22389
- * Mutually exclusive by construction: no Kimi model accepts both spellings, and
22390
- * sending both is a 400.
22542
+ *
22543
+ * Both spellings are OpenAI-format and independent, unlike Kimi where they are
22544
+ * mutually exclusive: `thinking.type` turns reasoning on or off and
22545
+ * `reasoning_effort` sets its depth. DeepSeek's own example sends both in one
22546
+ * request. Omitting both leaves thinking enabled at effort 'high'.
22391
22547
  */
22392
- function kimiReasoningParams(model, input) {
22393
- if (EFFORT_MODELS.has(model)) {
22394
- const effort = toKimiEffort(input.reasoningEffort);
22395
- return effort ? { reasoning_effort: effort } : {};
22396
- }
22397
- if (THINKING_MODELS.has(model)) {
22398
- if (THINKING_ALWAYS_ON.has(model)) return { thinking: { type: "enabled" } };
22399
- if (input.thinking?.enabled === void 0) return {};
22400
- return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
22401
- }
22402
- return {};
22548
+ function deepseekReasoningParams(model, input = {}) {
22549
+ if (!DEEPSEEK_MODELS.has(model)) return {};
22550
+ const params = {};
22551
+ if (input.thinking?.enabled !== void 0) params.thinking = { type: input.thinking.enabled ? "enabled" : "disabled" };
22552
+ if (input.thinking?.enabled === false) return params;
22553
+ const effort = toDeepSeekEffort(input.reasoningEffort);
22554
+ if (effort) params.reasoning_effort = effort;
22555
+ return params;
22403
22556
  }
22404
22557
  /**
22405
- * Sampling parameters for one model. Moonshot pins temperature (1.0) and top_p
22406
- * (0.95) on every current Kimi and documents them as unmodifiable, so they are
22407
- * omitted rather than sent-and-ignored; NO_TEMPERATURE_MODELS is the shared set
22408
- * the catalog's temperatureMode also lands on.
22558
+ * Whether the turn will reason, which is what the sampling restrictions below
22559
+ * actually hang on. DeepSeek's default is enabled, so only an explicit
22560
+ * `thinking.enabled === false` turns it off.
22561
+ */
22562
+ function deepseekThinkingEnabled(input = {}) {
22563
+ return input.thinking?.enabled !== false;
22564
+ }
22565
+ /**
22566
+ * Sampling parameters for one turn.
22409
22567
  *
22410
- * The penalties and `n` ride the same gate. Moonshot documents the whole sampling
22411
- * group as fixed on these ids, B4M sends penalties on essentially every turn, and
22412
- * an unmodifiable parameter here is a 400 rather than a silently ignored field -
22413
- * so the conservative reading is the safe one. Only the moonshot-v1 family, which
22414
- * this build does not ship, accepts any of them.
22568
+ * In thinking mode - the default - DeepSeek documents temperature,
22569
+ * presence_penalty and frequency_penalty as unsupported. They are accepted and
22570
+ * SILENTLY ignored rather than rejected, which is the worse failure of the two:
22571
+ * a 400 tells you the knob is dead, a no-op does not. They are dropped here so
22572
+ * nothing is sent that cannot take effect.
22573
+ *
22574
+ * Keyed on the TURN's resolved thinking state, not on model id: the restriction
22575
+ * is a property of thinking mode and the caller can turn thinking off, in which
22576
+ * case dropping temperature anyway would reproduce the same silent no-op from
22577
+ * our side of the wire.
22578
+ *
22579
+ * `top_p` does work in thinking mode with a lower bound of 0.95: a smaller value
22580
+ * is raised to it. Sent clamped rather than dropped, so the request states the
22581
+ * value the server will actually apply. The floor is a thinking-mode rule, so it
22582
+ * does not apply once thinking is off.
22583
+ *
22584
+ * `n` is not in DeepSeek's schema in either mode and is never sent.
22415
22585
  */
22416
- function kimiSamplingParams(model, input) {
22417
- if (NO_TEMPERATURE_MODELS.has(model)) return {};
22418
- const params = {};
22419
- if (input.temperature !== void 0) params.temperature = input.temperature;
22420
- if (input.topP !== void 0) params.top_p = input.topP;
22421
- if (input.presencePenalty !== void 0) params.presence_penalty = input.presencePenalty;
22422
- if (input.frequencyPenalty !== void 0) params.frequency_penalty = input.frequencyPenalty;
22423
- if (input.n !== void 0) params.n = input.n;
22424
- return params;
22586
+ function deepseekSamplingParams(model, input, reasoning = {}) {
22587
+ if (!DEEPSEEK_MODELS.has(model) || !deepseekThinkingEnabled(reasoning)) {
22588
+ const passthrough = {};
22589
+ if (input.temperature !== void 0) passthrough.temperature = input.temperature;
22590
+ if (input.topP !== void 0) passthrough.top_p = input.topP;
22591
+ if (input.presencePenalty !== void 0) passthrough.presence_penalty = input.presencePenalty;
22592
+ if (input.frequencyPenalty !== void 0) passthrough.frequency_penalty = input.frequencyPenalty;
22593
+ return passthrough;
22594
+ }
22595
+ if (input.topP === void 0) return {};
22596
+ return { top_p: Math.max(input.topP, DEEPSEEK_THINKING_TOP_P_FLOOR) };
22425
22597
  }
22426
- /** `tool_choice`, downgraded to 'auto' on the ids that reject 'required'. */
22427
- function kimiToolChoice(model, choice) {
22428
- if (choice === void 0) return void 0;
22429
- if (choice === "required" && NO_REQUIRED_TOOL_CHOICE.has(model)) return "auto";
22430
- return choice;
22598
+ /** `stop`, truncated to the 16 sequences DeepSeek accepts. */
22599
+ function deepseekStopSequences(stop) {
22600
+ if (!Array.isArray(stop)) return stop;
22601
+ return stop.length > 16 ? stop.slice(0, 16) : stop;
22431
22602
  }
22432
22603
  /** Type guard: does this message already carry OpenAI-style `tool_calls`? */
22433
22604
  function hasToolCalls(msg) {
@@ -22450,12 +22621,16 @@ function isTextBlock(block) {
22450
22621
  * Messages already in OpenAI format (with `tool_calls` property) pass through unchanged.
22451
22622
  * Messages without tool_use/tool_result content blocks pass through unchanged.
22452
22623
  */
22453
- function convertMessageToOpenAIFormat(msg) {
22454
- if (hasToolCalls(msg)) return [{
22455
- role: "assistant",
22456
- content: null,
22457
- tool_calls: msg.tool_calls
22458
- }];
22624
+ function convertMessageToOpenAIFormat(msg, options = {}) {
22625
+ if (hasToolCalls(msg)) {
22626
+ const reasoningContent = msg.reasoning_content;
22627
+ return [{
22628
+ role: "assistant",
22629
+ content: null,
22630
+ tool_calls: msg.tool_calls,
22631
+ ...options.preserveReasoningContent && typeof reasoningContent === "string" ? { reasoning_content: reasoningContent } : {}
22632
+ }];
22633
+ }
22459
22634
  if (msg.role === "assistant" && Array.isArray(msg.content)) {
22460
22635
  const contentBlocks = msg.content;
22461
22636
  const toolUseBlocks = contentBlocks.filter(isToolUseBlock);
@@ -22493,8 +22668,614 @@ function convertMessageToOpenAIFormat(msg) {
22493
22668
  * Convert an array of IMessages from B4M standard format to OpenAI-compatible format.
22494
22669
  * Returns OpenAIFormattedMessage[] - callers targeting OpenAI SDK types should cast at the boundary.
22495
22670
  */
22496
- function convertMessagesToOpenAIFormat(messages) {
22497
- return messages.flatMap(convertMessageToOpenAIFormat);
22671
+ function convertMessagesToOpenAIFormat(messages, options = {}) {
22672
+ return messages.flatMap((msg) => convertMessageToOpenAIFormat(msg, options));
22673
+ }
22674
+ /**
22675
+ * DeepSeek's models, served from their own OpenAI-compatible endpoint.
22676
+ *
22677
+ * Structurally this is kimiBackend's twin - same OpenAI SDK against a different
22678
+ * baseURL, same recursive tool loop, same multi-turn token accumulators - and the
22679
+ * three OpenAI-compatible backends must stay in sync on that machinery. What
22680
+ * genuinely differs here:
22681
+ *
22682
+ * 1. The base URL carries NO `/v1` segment; the SDK appends the path itself.
22683
+ * 2. Thinking is on by default and its sampling restrictions are SILENT no-ops
22684
+ * rather than 400s; see deepseekParams.
22685
+ * 3. The prior turn's `reasoning_content` has to be replayed on the assistant
22686
+ * tool-call message whenever the request carries `tools`, which is the
22687
+ * opposite of the usual provider rule. See pushToolMessages.
22688
+ *
22689
+ * @see https://api-docs.deepseek.com/api/create-chat-completion
22690
+ */
22691
+ var DeepSeekBackend = class {
22692
+ _baseUrl = "https://api.deepseek.com";
22693
+ _api;
22694
+ logger;
22695
+ currentModel = "";
22696
+ constructor(apiKey, logger) {
22697
+ if (!apiKey) throw new Error("DeepSeek API key is required");
22698
+ this._api = new OpenAI({
22699
+ apiKey,
22700
+ baseURL: this._baseUrl
22701
+ });
22702
+ this.logger = logger ?? new Logger();
22703
+ }
22704
+ /**
22705
+ * Seed listing. Post-registry this is the fallback tier, not the source of
22706
+ * truth: the catalog overlays context window, limits, lifecycle and price on
22707
+ * top of these rows. DeepSeek's own GET /models returns id/object/owned_by and
22708
+ * nothing else, so everything below has to live here.
22709
+ *
22710
+ * Prices are the PEAK rates. Off-peak (outside 01:00-04:00 and 06:00-10:00 UTC,
22711
+ * Mon-Fri) is exactly half, and ModelInfo.pricing is keyed by context tier with
22712
+ * no time dimension to express that in - so the rate that never under-bills is
22713
+ * the one recorded.
22714
+ */
22715
+ async getModelInfo() {
22716
+ return [{
22717
+ id: ChatModels.DEEPSEEK_FLASH,
22718
+ type: "text",
22719
+ name: "DeepSeek Flash",
22720
+ backend: ModelBackend.DeepSeek,
22721
+ contextWindow: 1e6,
22722
+ max_tokens: 393216,
22723
+ can_stream: true,
22724
+ pricing: { 1e6: {
22725
+ input: .3 / 1e6,
22726
+ output: 1.2 / 1e6,
22727
+ cache_read: .006 / 1e6
22728
+ } },
22729
+ can_think: true,
22730
+ supportsVision: true,
22731
+ supportsTools: true,
22732
+ supportsImageVariation: false,
22733
+ releaseDate: "2026-08-13",
22734
+ description: "DeepSeek's V4.1-Flash. 1M context with native vision, tool use, and selectable reasoning effort (low/high/max). Always reasons unless thinking is turned off."
22735
+ }];
22736
+ }
22737
+ async complete(model, messages, options, callback, toolsUsed = []) {
22738
+ this.currentModel = model;
22739
+ const toolCallCount = options._internal?.toolCallCount ?? 0;
22740
+ const accumInputTokens = options._internal?.accumInputTokens ?? 0;
22741
+ const accumOutputTokens = options._internal?.accumOutputTokens ?? 0;
22742
+ const accumCacheReadTokens = options._internal?.accumCacheReadTokens ?? 0;
22743
+ const maxToolCalls = options._internal?.maxToolCalls ?? 10;
22744
+ if (toolCallCount >= maxToolCalls && options.tools?.length) {
22745
+ this.logger.warn(`Max tool calls limit (${maxToolCalls}) reached. Disabling tools to prevent infinite loops.`);
22746
+ await this.complete(model, stripToolDependentMessages(messages), {
22747
+ ...options,
22748
+ tools: void 0,
22749
+ _internal: options._internal
22750
+ }, callback, toolsUsed);
22751
+ return;
22752
+ }
22753
+ const rawTools = options.tools;
22754
+ options.tools = Array.isArray(rawTools) ? rawTools : rawTools ? [rawTools] : void 0;
22755
+ if ((options.n ?? 1) > 1) this.logger.warn(`DeepSeek has no 'n' parameter; ignoring the request for ${options.n} choices.`);
22756
+ const useStreaming = Boolean(options.stream);
22757
+ const reasoning = {
22758
+ thinking: options.thinking,
22759
+ reasoningEffort: options.reasoningEffort
22760
+ };
22761
+ const messagesWithFormat = injectJsonSchemaInstruction(messages, options.responseFormat);
22762
+ const bestEffortFormat = isBestEffortJsonSchema(options.responseFormat);
22763
+ const parameters = {
22764
+ model,
22765
+ messages: this.formatMessages(messagesWithFormat)
22766
+ };
22767
+ Object.assign(parameters, {
22768
+ ...deepseekSamplingParams(model, {
22769
+ temperature: options.temperature,
22770
+ topP: options.topP,
22771
+ presencePenalty: options.presencePenalty,
22772
+ frequencyPenalty: options.frequencyPenalty
22773
+ }, reasoning),
22774
+ ...deepseekReasoningParams(model, reasoning),
22775
+ stop: deepseekStopSequences(options.stop),
22776
+ stream: useStreaming,
22777
+ max_tokens: options.maxTokens,
22778
+ ...useStreaming && { stream_options: { include_usage: true } }
22779
+ });
22780
+ if (options.tools?.length) {
22781
+ parameters.tools = this.formatTools(options.tools);
22782
+ if (options.tool_choice !== void 0) parameters.tool_choice = options.tool_choice;
22783
+ }
22784
+ if (options.responseFormat?.type === "json_schema") parameters.response_format = { type: "json_object" };
22785
+ else if (options.responseFormat?.type === "text") parameters.response_format = { type: "text" };
22786
+ const cacheStrategy = options.cacheStrategy;
22787
+ const response = await this._api.chat.completions.create(parameters, { signal: options.abortSignal });
22788
+ let inputTokens = 0;
22789
+ let outputTokens = 0;
22790
+ if (!(response instanceof Stream)) {
22791
+ const streamedText = [];
22792
+ if (!response.choices || response.choices.length === 0) throw new Error("No choices returned from the DeepSeek API");
22793
+ const turnCacheReadTokens = cachedTokensFromUsage(response.usage);
22794
+ for (const c of response.choices) {
22795
+ if (!c.message) continue;
22796
+ const reasoningContent = c.message.reasoning_content;
22797
+ if (c.message.tool_calls && c.message.tool_calls.length > 0) {
22798
+ for (const toolCall of c.message.tool_calls) {
22799
+ if (toolCall.type !== "function") continue;
22800
+ if (toolCall.function.arguments) toolsUsed.push({
22801
+ name: toolCall.function.name,
22802
+ arguments: toolCall.function.arguments,
22803
+ id: toolCall.id
22804
+ });
22805
+ }
22806
+ if (options.executeTools !== false) {
22807
+ const resolvedTools = [];
22808
+ for (const toolCall of c.message.tool_calls) {
22809
+ if (toolCall.type !== "function" || !toolCall.function.arguments) continue;
22810
+ const toolFn = options.tools?.find((t) => t.toolSchema.name === toolCall.function.name)?.toolFn;
22811
+ if (!toolFn) continue;
22812
+ try {
22813
+ const parsedParams = JSON.parse(toolCall.function.arguments);
22814
+ resolvedTools.push({
22815
+ id: toolCall.id,
22816
+ name: toolCall.function.name,
22817
+ parameters: toolCall.function.arguments,
22818
+ parsedParams,
22819
+ toolFn
22820
+ });
22821
+ } catch {
22822
+ this.logger.warn(`JSON parse error for ${toolCall.function.name} arguments`);
22823
+ const entry = toolsUsed.find((t) => t.name === toolCall.function.name && t.id === toolCall.id);
22824
+ if (entry) entry.arguments = "{}";
22825
+ recordToolResult(toolsUsed, {
22826
+ id: toolCall.id,
22827
+ name: toolCall.function.name
22828
+ }, "Error: Tool arguments were malformed and could not be parsed.", false);
22829
+ }
22830
+ }
22831
+ const parallelEnabled = options.parallelToolExecution !== false;
22832
+ this.logger.debug("[Tool Execution] Executing tools (DeepSeek non-streaming)", {
22833
+ mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
22834
+ toolNames: resolvedTools.map((t) => t.name)
22835
+ });
22836
+ const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
22837
+ return {
22838
+ id,
22839
+ name,
22840
+ parameters: toolParams,
22841
+ result: await toolFn(parsedParams)
22842
+ };
22843
+ }), {
22844
+ parallel: parallelEnabled,
22845
+ maxConcurrency: options.maxParallelTools
22846
+ })).map((outcome, i) => outcome.ok ? {
22847
+ ok: true,
22848
+ ...outcome.result
22849
+ } : {
22850
+ ok: false,
22851
+ id: resolvedTools[i].id,
22852
+ name: resolvedTools[i].name,
22853
+ parameters: resolvedTools[i].parameters,
22854
+ error: outcome.error
22855
+ });
22856
+ let turnReasoning = reasoningContent;
22857
+ for (const outcome of outcomes) {
22858
+ if (outcome.ok) {
22859
+ const resultStr = outcome.result.toString();
22860
+ recordToolResult(toolsUsed, {
22861
+ id: outcome.id,
22862
+ name: outcome.name
22863
+ }, resultStr, true);
22864
+ this.pushToolMessages(messages, {
22865
+ id: outcome.id,
22866
+ name: outcome.name,
22867
+ parameters: outcome.parameters
22868
+ }, resultStr, turnReasoning ? [turnReasoning] : void 0);
22869
+ } else {
22870
+ if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
22871
+ const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
22872
+ const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
22873
+ recordToolResult(toolsUsed, {
22874
+ id: outcome.id,
22875
+ name: outcome.name
22876
+ }, observation, false);
22877
+ this.pushToolMessages(messages, {
22878
+ id: outcome.id,
22879
+ name: outcome.name,
22880
+ parameters: outcome.parameters
22881
+ }, observation, turnReasoning ? [turnReasoning] : void 0);
22882
+ }
22883
+ turnReasoning = void 0;
22884
+ }
22885
+ await this.complete(model, messages, {
22886
+ ...options,
22887
+ _internal: {
22888
+ ...options._internal,
22889
+ toolCallCount: toolCallCount + 1,
22890
+ accumInputTokens: accumInputTokens + (response.usage?.prompt_tokens || 0),
22891
+ accumOutputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22892
+ accumCacheReadTokens: accumCacheReadTokens + turnCacheReadTokens
22893
+ }
22894
+ }, callback, toolsUsed);
22895
+ return;
22896
+ } else {
22897
+ this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
22898
+ await callback([null], {
22899
+ ...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), accumCacheReadTokens + turnCacheReadTokens),
22900
+ outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22901
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
22902
+ });
22903
+ return;
22904
+ }
22905
+ } else {
22906
+ const content = c.message.content || "";
22907
+ streamedText[c.index] = reasoningContent ? `<think>${reasoningContent}</think>${content}` : content;
22908
+ }
22909
+ }
22910
+ if (streamedText.every((text) => !text) && toolsUsed.length === 0) {
22911
+ const finish = response.choices[0]?.finish_reason;
22912
+ throw new Error(finish === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${finish ?? "unknown"}).`);
22913
+ }
22914
+ let cacheStats;
22915
+ if (cacheStrategy?.enableCaching && response.usage) {
22916
+ cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats(response, model);
22917
+ if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: false });
22918
+ }
22919
+ const finishReason = normalizeOpenAIFinishReason(response.choices[0]?.finish_reason);
22920
+ const totalCacheReadTokens = accumCacheReadTokens + turnCacheReadTokens;
22921
+ await callback(streamedText, {
22922
+ ...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), totalCacheReadTokens),
22923
+ outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
22924
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
22925
+ cacheStats,
22926
+ ...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
22927
+ ...finishReason ? { stopReason: finishReason } : {}
22928
+ });
22929
+ return;
22930
+ }
22931
+ const func = [];
22932
+ let isInThinkingBlock = false;
22933
+ let streamedReasoning = "";
22934
+ let cachedTokensFromStream = 0;
22935
+ let streamFinishReason;
22936
+ let sawAnyText = false;
22937
+ for await (const chunk of response) {
22938
+ const streamedText = [];
22939
+ if (chunk.usage) {
22940
+ inputTokens = Math.max(inputTokens, chunk.usage?.prompt_tokens || 0);
22941
+ outputTokens += chunk.usage?.completion_tokens || 0;
22942
+ const chunkCached = cachedTokensFromUsage(chunk.usage);
22943
+ if (chunkCached > 0) cachedTokensFromStream = chunkCached;
22944
+ }
22945
+ chunk?.choices.forEach((c) => {
22946
+ if (c.finish_reason) streamFinishReason = c.finish_reason;
22947
+ const deltaReasoning = c.delta.reasoning_content;
22948
+ if (deltaReasoning) {
22949
+ streamedReasoning += deltaReasoning;
22950
+ if (!isInThinkingBlock) {
22951
+ isInThinkingBlock = true;
22952
+ streamedText[c.index] = "<think>" + deltaReasoning;
22953
+ } else streamedText[c.index] = deltaReasoning;
22954
+ if (!c.delta.content) return;
22955
+ }
22956
+ if (isInThinkingBlock && c.delta.content) {
22957
+ isInThinkingBlock = false;
22958
+ streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
22959
+ return;
22960
+ }
22961
+ c.delta.tool_calls?.map((tool) => {
22962
+ func[tool.index] ||= {};
22963
+ func[tool.index].name ||= tool.function?.name;
22964
+ func[tool.index].id ||= tool.id;
22965
+ func[tool.index].parameters ??= "";
22966
+ func[tool.index].parameters += tool.function?.arguments || "";
22967
+ });
22968
+ if (func.length > 0) return;
22969
+ streamedText[c.index] = c.delta.content || "";
22970
+ });
22971
+ if (streamedText.some((t) => t)) sawAnyText = true;
22972
+ const normalizedFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
22973
+ await callback(streamedText, {
22974
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
22975
+ outputTokens: accumOutputTokens + outputTokens,
22976
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
22977
+ ...normalizedFinishReason ? { stopReason: normalizedFinishReason } : {}
22978
+ });
22979
+ }
22980
+ if (isInThinkingBlock) {
22981
+ await callback(["</think>"], {
22982
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
22983
+ outputTokens: accumOutputTokens + outputTokens,
22984
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
22985
+ });
22986
+ isInThinkingBlock = false;
22987
+ }
22988
+ if (!sawAnyText && func.length === 0 && toolsUsed.length === 0) throw new Error(streamFinishReason === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${streamFinishReason ?? "unknown"}).`);
22989
+ let cacheStats;
22990
+ if (cacheStrategy?.enableCaching && inputTokens > 0) {
22991
+ cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats({ usage: {
22992
+ prompt_tokens: inputTokens,
22993
+ completion_tokens: outputTokens,
22994
+ prompt_cache_hit_tokens: cachedTokensFromStream
22995
+ } }, model);
22996
+ if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: true });
22997
+ }
22998
+ if ((cacheStats || bestEffortFormat) && func.length === 0) {
22999
+ const terminalFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
23000
+ await callback([""], {
23001
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
23002
+ outputTokens: accumOutputTokens + outputTokens,
23003
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
23004
+ ...cacheStats ? { cacheStats } : {},
23005
+ ...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
23006
+ ...terminalFinishReason ? { stopReason: terminalFinishReason } : {}
23007
+ });
23008
+ }
23009
+ if (func.length > 0) {
23010
+ for await (const tool of func) {
23011
+ const { name, parameters: toolParams, id } = tool;
23012
+ if (name) toolsUsed.push({
23013
+ name,
23014
+ arguments: toolParams || "{}",
23015
+ id
23016
+ });
23017
+ }
23018
+ if (options.executeTools !== false) {
23019
+ const resolvedTools = [];
23020
+ for (const tool of func) {
23021
+ const { id, name } = tool;
23022
+ if (!id || !name) continue;
23023
+ const toolParams = tool.parameters || "{}";
23024
+ const toolFn = options.tools?.find((t) => t.toolSchema.name === name)?.toolFn;
23025
+ if (!toolFn) continue;
23026
+ try {
23027
+ const parsedParams = JSON.parse(toolParams);
23028
+ resolvedTools.push({
23029
+ id,
23030
+ name,
23031
+ parameters: toolParams,
23032
+ parsedParams,
23033
+ toolFn
23034
+ });
23035
+ } catch {
23036
+ this.logger.warn(`JSON parse error for ${name} arguments (streaming)`);
23037
+ const entry = toolsUsed.find((t) => t.name === name && t.id === id);
23038
+ if (entry) entry.arguments = "{}";
23039
+ recordToolResult(toolsUsed, {
23040
+ id,
23041
+ name
23042
+ }, "Error: Tool arguments were malformed and could not be parsed.", false);
23043
+ }
23044
+ }
23045
+ const parallelEnabled = options.parallelToolExecution !== false;
23046
+ this.logger.debug("[Tool Execution] Executing tools (DeepSeek streaming)", {
23047
+ mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
23048
+ toolNames: resolvedTools.map((t) => t.name)
23049
+ });
23050
+ const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
23051
+ return {
23052
+ id,
23053
+ name,
23054
+ parameters: toolParams,
23055
+ result: await toolFn(parsedParams)
23056
+ };
23057
+ }), {
23058
+ parallel: parallelEnabled,
23059
+ maxConcurrency: options.maxParallelTools
23060
+ })).map((outcome, i) => outcome.ok ? {
23061
+ ok: true,
23062
+ ...outcome.result
23063
+ } : {
23064
+ ok: false,
23065
+ id: resolvedTools[i].id,
23066
+ name: resolvedTools[i].name,
23067
+ parameters: resolvedTools[i].parameters,
23068
+ error: outcome.error
23069
+ });
23070
+ let turnReasoning = streamedReasoning || void 0;
23071
+ for (const outcome of outcomes) {
23072
+ if (outcome.ok) {
23073
+ const resultStr = outcome.result.toString();
23074
+ recordToolResult(toolsUsed, {
23075
+ id: outcome.id,
23076
+ name: outcome.name
23077
+ }, resultStr, true);
23078
+ this.pushToolMessages(messages, {
23079
+ id: outcome.id,
23080
+ name: outcome.name,
23081
+ parameters: outcome.parameters
23082
+ }, resultStr, turnReasoning ? [turnReasoning] : void 0);
23083
+ } else {
23084
+ if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
23085
+ const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
23086
+ const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
23087
+ recordToolResult(toolsUsed, {
23088
+ id: outcome.id,
23089
+ name: outcome.name
23090
+ }, observation, false);
23091
+ this.pushToolMessages(messages, {
23092
+ id: outcome.id,
23093
+ name: outcome.name,
23094
+ parameters: outcome.parameters
23095
+ }, observation, turnReasoning ? [turnReasoning] : void 0);
23096
+ }
23097
+ turnReasoning = void 0;
23098
+ }
23099
+ await this.complete(model, messages, {
23100
+ ...options,
23101
+ _internal: {
23102
+ ...options._internal,
23103
+ toolCallCount: toolCallCount + 1,
23104
+ accumInputTokens: accumInputTokens + inputTokens,
23105
+ accumOutputTokens: accumOutputTokens + outputTokens,
23106
+ accumCacheReadTokens: accumCacheReadTokens + cachedTokensFromStream
23107
+ }
23108
+ }, callback, toolsUsed);
23109
+ } else {
23110
+ this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
23111
+ await callback([null], {
23112
+ ...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
23113
+ outputTokens: accumOutputTokens + outputTokens,
23114
+ toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
23115
+ ...cacheStats ? { cacheStats } : {}
23116
+ });
23117
+ }
23118
+ }
23119
+ }
23120
+ formatMessages(messages) {
23121
+ return convertMessagesToOpenAIFormat(messages, { preserveReasoningContent: true });
23122
+ }
23123
+ formatTools(tools = []) {
23124
+ return tools.map((tool) => ({
23125
+ type: "function",
23126
+ function: tool.toolSchema
23127
+ }));
23128
+ }
23129
+ /**
23130
+ * `thinkingBlocks` carries the turn's `reasoning_content` as a single string
23131
+ * entry. DeepSeek inverts the usual rule: when a request carries `tools`, the
23132
+ * prior turn's monologue MUST be replayed on the assistant tool-call message or
23133
+ * reasoning continuity breaks across the loop. formatMessages opts into the
23134
+ * converter's `preserveReasoningContent` for exactly this path; every other
23135
+ * target strips it, because this array is shared with the fallback hop.
23136
+ */
23137
+ pushToolMessages(messages, tool, result, thinkingBlocks) {
23138
+ const reasoningContent = typeof thinkingBlocks?.[0] === "string" ? thinkingBlocks[0] : void 0;
23139
+ messages.push({
23140
+ content: null,
23141
+ role: "assistant",
23142
+ ...reasoningContent ? { reasoning_content: reasoningContent } : {},
23143
+ tool_calls: [{
23144
+ id: tool.id,
23145
+ type: "function",
23146
+ function: {
23147
+ name: tool.name,
23148
+ arguments: tool.parameters
23149
+ }
23150
+ }]
23151
+ });
23152
+ messages.push({
23153
+ role: "tool",
23154
+ content: JSON.stringify({ result }),
23155
+ tool_call_id: tool.id
23156
+ });
23157
+ }
23158
+ replaceLastToolResultObservation(messages, toolCallId, newObservation) {
23159
+ replaceLastToolResultObservationOpenAI(messages, toolCallId, newObservation);
23160
+ }
23161
+ getLatestToolCallId(messages, toolName) {
23162
+ return getLatestToolCallIdOpenAI(messages, toolName);
23163
+ }
23164
+ };
23165
+ /**
23166
+ * Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
23167
+ * transport so every "which parameter does this id accept" rule is one pure
23168
+ * function with a test, rather than a conditional buried in a 400-line complete().
23169
+ *
23170
+ * Moonshot is OpenAI-compatible in envelope only. The reasoning controls, the
23171
+ * sampling pins, and the max-tokens parameter all differ per model, and sending
23172
+ * the wrong one is a 400 rather than a silently ignored field.
23173
+ * @see https://platform.kimi.ai/docs/api/chat
23174
+ */
23175
+ /** Kimi's own effort vocabulary, which is not OpenAI's and not B4M's. */
23176
+ const KIMI_EFFORT_LEVELS = [
23177
+ "low",
23178
+ "high",
23179
+ "max"
23180
+ ];
23181
+ /**
23182
+ * Takes `reasoning_effort`. K3 only, and K3 always reasons - there is no way to
23183
+ * turn thinking off, so the parameter selects depth, never whether.
23184
+ */
23185
+ const EFFORT_MODELS = /* @__PURE__ */ new Set([ChatModels.KIMI_K3]);
23186
+ /** Takes the `thinking` object instead of `reasoning_effort`. */
23187
+ const THINKING_MODELS = /* @__PURE__ */ new Set([
23188
+ ChatModels.KIMI_K2_7_CODE,
23189
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23190
+ ChatModels.KIMI_K2_6,
23191
+ ChatModels.KIMI_K2_5
23192
+ ]);
23193
+ /**
23194
+ * `thinking.type` accepts only 'enabled' on the K2.7 code models - 'disabled' is
23195
+ * rejected. So a caller asking for no thinking gets thinking anyway; the
23196
+ * alternative is a 400, and the parameter is omitted rather than fought.
23197
+ */
23198
+ const THINKING_ALWAYS_ON = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED]);
23199
+ /**
23200
+ * Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
23201
+ * function form). Downgraded to 'auto' rather than dropped: a caller that asked
23202
+ * for a forced tool still wants tools offered.
23203
+ */
23204
+ const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
23205
+ ChatModels.KIMI_K2_7_CODE,
23206
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23207
+ ChatModels.KIMI_K2_6
23208
+ ]);
23209
+ /** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
23210
+ const KIMI_MODELS = /* @__PURE__ */ new Set([
23211
+ ChatModels.KIMI_K3,
23212
+ ChatModels.KIMI_K2_7_CODE,
23213
+ ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
23214
+ ChatModels.KIMI_K2_6,
23215
+ ChatModels.KIMI_K2_5
23216
+ ]);
23217
+ /**
23218
+ * B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
23219
+ * rather than to omission because K3 cannot be asked not to think - claiming
23220
+ * otherwise by dropping the parameter would silently bill max-effort reasoning
23221
+ * (Moonshot's default is 'max').
23222
+ */
23223
+ function toKimiEffort(effort) {
23224
+ if (!effort) return void 0;
23225
+ switch (effort) {
23226
+ case "none":
23227
+ case "minimal":
23228
+ case "low": return "low";
23229
+ case "medium":
23230
+ case "high": return "high";
23231
+ case "xhigh": return "max";
23232
+ default: return;
23233
+ }
23234
+ }
23235
+ /**
23236
+ * The reasoning parameters for one model, or an empty object when it takes none.
23237
+ * Mutually exclusive by construction: no Kimi model accepts both spellings, and
23238
+ * sending both is a 400.
23239
+ */
23240
+ function kimiReasoningParams(model, input) {
23241
+ if (EFFORT_MODELS.has(model)) {
23242
+ const effort = toKimiEffort(input.reasoningEffort);
23243
+ return effort ? { reasoning_effort: effort } : {};
23244
+ }
23245
+ if (THINKING_MODELS.has(model)) {
23246
+ if (THINKING_ALWAYS_ON.has(model)) return { thinking: { type: "enabled" } };
23247
+ if (input.thinking?.enabled === void 0) return {};
23248
+ return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
23249
+ }
23250
+ return {};
23251
+ }
23252
+ /**
23253
+ * Sampling parameters for one model. Moonshot pins temperature (1.0) and top_p
23254
+ * (0.95) on every current Kimi and documents them as unmodifiable, so they are
23255
+ * omitted rather than sent-and-ignored; NO_TEMPERATURE_MODELS is the shared set
23256
+ * the catalog's temperatureMode also lands on.
23257
+ *
23258
+ * The penalties and `n` ride the same gate. Moonshot documents the whole sampling
23259
+ * group as fixed on these ids, B4M sends penalties on essentially every turn, and
23260
+ * an unmodifiable parameter here is a 400 rather than a silently ignored field -
23261
+ * so the conservative reading is the safe one. Only the moonshot-v1 family, which
23262
+ * this build does not ship, accepts any of them.
23263
+ */
23264
+ function kimiSamplingParams(model, input) {
23265
+ if (NO_TEMPERATURE_MODELS.has(model)) return {};
23266
+ const params = {};
23267
+ if (input.temperature !== void 0) params.temperature = input.temperature;
23268
+ if (input.topP !== void 0) params.top_p = input.topP;
23269
+ if (input.presencePenalty !== void 0) params.presence_penalty = input.presencePenalty;
23270
+ if (input.frequencyPenalty !== void 0) params.frequency_penalty = input.frequencyPenalty;
23271
+ if (input.n !== void 0) params.n = input.n;
23272
+ return params;
23273
+ }
23274
+ /** `tool_choice`, downgraded to 'auto' on the ids that reject 'required'. */
23275
+ function kimiToolChoice(model, choice) {
23276
+ if (choice === void 0) return void 0;
23277
+ if (choice === "required" && NO_REQUIRED_TOOL_CHOICE.has(model)) return "auto";
23278
+ return choice;
22498
23279
  }
22499
23280
  /**
22500
23281
  * Moonshot AI's Kimi models, served from their OpenAI-compatible endpoint.
@@ -22634,6 +23415,8 @@ var KimiBackend = class {
22634
23415
  supportsImageVariation: false,
22635
23416
  releaseDate: "2026-01-01",
22636
23417
  trainingCutoff: "2025-01-01",
23418
+ deprecationDate: "2026-08-31",
23419
+ replacedBy: ChatModels.KIMI_K2_6,
22637
23420
  description: "The previous-generation Kimi, still the cheapest of the family. Superseded by K2.6 on quality at a modest price increase."
22638
23421
  }
22639
23422
  ];
@@ -22852,16 +23635,17 @@ var KimiBackend = class {
22852
23635
  }
22853
23636
  chunk?.choices.forEach((c) => {
22854
23637
  if (c.finish_reason) streamFinishReason = c.finish_reason;
22855
- if (c.delta.reasoning_content) {
23638
+ const deltaReasoning = c.delta.reasoning_content;
23639
+ if (deltaReasoning) {
22856
23640
  if (!isInThinkingBlock) {
22857
23641
  isInThinkingBlock = true;
22858
- streamedText[c.index] = "<think>" + c.delta.reasoning_content;
22859
- } else streamedText[c.index] = c.delta.reasoning_content;
22860
- return;
23642
+ streamedText[c.index] = "<think>" + deltaReasoning;
23643
+ } else streamedText[c.index] = deltaReasoning;
23644
+ if (!c.delta.content) return;
22861
23645
  }
22862
- if (isInThinkingBlock && c.delta.content && !c.delta.reasoning_content) {
23646
+ if (isInThinkingBlock && c.delta.content) {
22863
23647
  isInThinkingBlock = false;
22864
- streamedText[c.index] = "</think>" + (c.delta.content || "");
23648
+ streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
22865
23649
  return;
22866
23650
  }
22867
23651
  c.delta.tool_calls?.map((tool) => {
@@ -23772,7 +24556,7 @@ var OpenAIBackend = class {
23772
24556
  supportsTools: true,
23773
24557
  supportsImageVariation: false,
23774
24558
  logoFile: "OpenAI_Logo.svg",
23775
- rank: 0,
24559
+ rank: 4,
23776
24560
  trainingCutoff: "2024-06-01",
23777
24561
  description: "Reliable for general-purpose text generation and analysis with a standard context window, suitable for a wide range of applications."
23778
24562
  },
@@ -23793,7 +24577,7 @@ var OpenAIBackend = class {
23793
24577
  supportsTools: true,
23794
24578
  supportsImageVariation: false,
23795
24579
  logoFile: "OpenAI_Logo.svg",
23796
- rank: 0,
24580
+ rank: 4,
23797
24581
  trainingCutoff: "2024-06-01",
23798
24582
  description: "OpenAI's balanced GPT-4.1 model offering optimal price-performance ratio. Ideal for tasks requiring intelligence and cost efficiency."
23799
24583
  },
@@ -23814,7 +24598,7 @@ var OpenAIBackend = class {
23814
24598
  supportsTools: true,
23815
24599
  supportsImageVariation: false,
23816
24600
  logoFile: "OpenAI_Logo.svg",
23817
- rank: 0,
24601
+ rank: 4,
23818
24602
  trainingCutoff: "2024-06-01",
23819
24603
  deprecationDate: "2026-10-23",
23820
24604
  description: "Designed for high-volume, low-cost processing with rapid response times, ideal for budget-conscious applications."
@@ -23878,7 +24662,7 @@ var OpenAIBackend = class {
23878
24662
  supportsTools: true,
23879
24663
  supportsImageVariation: false,
23880
24664
  logoFile: "OpenAI_Logo.svg",
23881
- rank: 0,
24665
+ rank: 3,
23882
24666
  trainingCutoff: "2024-06-01",
23883
24667
  deprecationDate: "2026-12-11",
23884
24668
  description: "OpenAI's O3 reasoning model with broad capabilities and up-to-date training data. Superseded by O4 Mini for most use cases.",
@@ -24050,7 +24834,7 @@ var OpenAIBackend = class {
24050
24834
  supportsImageVariation: false,
24051
24835
  supportsTools: true,
24052
24836
  logoFile: "OpenAI_Logo.svg",
24053
- rank: 1,
24837
+ rank: 2,
24054
24838
  trainingCutoff: "2026-01-01",
24055
24839
  releaseDate: "2026-06-23",
24056
24840
  description: "GPT-5.6 Luna - the fast, cost-efficient GPT-5.6 variant. Great for high-volume workloads that still need solid reasoning, vision, and tool use."
@@ -24116,7 +24900,7 @@ var OpenAIBackend = class {
24116
24900
  supportsImageVariation: false,
24117
24901
  supportsTools: true,
24118
24902
  logoFile: "OpenAI_Logo.svg",
24119
- rank: 1,
24903
+ rank: 2,
24120
24904
  trainingCutoff: "2025-08-31",
24121
24905
  releaseDate: "2026-03-17",
24122
24906
  description: "Compact GPT-5.4 variant balancing strong performance with lower cost. Great for everyday tasks needing solid reasoning and vision."
@@ -24138,7 +24922,7 @@ var OpenAIBackend = class {
24138
24922
  supportsImageVariation: false,
24139
24923
  supportsTools: true,
24140
24924
  logoFile: "OpenAI_Logo.svg",
24141
- rank: 1,
24925
+ rank: 2,
24142
24926
  trainingCutoff: "2025-08-31",
24143
24927
  releaseDate: "2026-03-17",
24144
24928
  description: "Ultra-lightweight GPT-5.4 model optimized for speed and cost efficiency. Ideal for high-volume workloads and quick interactions."
@@ -26011,6 +26795,10 @@ function backendForAdapterFamily(family, ctx) {
26011
26795
  const key = keyOrThrow(apiKeyTable.kimi, "Moonshot");
26012
26796
  return key ? new KimiBackend(key, logger) : null;
26013
26797
  }
26798
+ case "deepseek": {
26799
+ const key = keyOrThrow(apiKeyTable.deepseek, "DeepSeek");
26800
+ return key ? new DeepSeekBackend(key, logger) : null;
26801
+ }
26014
26802
  case "bfl": return new BFLBackend(keyOrThrow(apiKeyTable.bfl, "BFL") ?? "demo-key");
26015
26803
  case "local-image": {
26016
26804
  const baseUrl = keyOrThrow(apiKeyTable["local-image"], "Local image");
@@ -26051,6 +26839,7 @@ function buildApiKeyTable(keys) {
26051
26839
  [ModelBackend.Ollama]: keys.ollama || void 0,
26052
26840
  [ModelBackend.XAI]: keys.xai || void 0,
26053
26841
  [ModelBackend.Kimi]: keys.kimi || void 0,
26842
+ [ModelBackend.DeepSeek]: keys.deepseek || void 0,
26054
26843
  [ModelBackend.VoyageAI]: keys.voyageai || void 0,
26055
26844
  [ModelBackend.LocalImage]: keys.imageGen || void 0,
26056
26845
  [ModelBackend.Bedrock]: void 0,
@@ -26077,6 +26866,7 @@ const KEYED_LISTING_BACKENDS = [
26077
26866
  ModelBackend.BFL,
26078
26867
  ModelBackend.XAI,
26079
26868
  ModelBackend.Kimi,
26869
+ ModelBackend.DeepSeek,
26080
26870
  ModelBackend.LocalImage
26081
26871
  ];
26082
26872
  /**
@@ -26134,6 +26924,7 @@ const DISPATCHABLE_ADAPTER_FAMILIES = [
26134
26924
  "gemini",
26135
26925
  "xai",
26136
26926
  "kimi",
26927
+ "deepseek",
26137
26928
  "ollama",
26138
26929
  "bfl",
26139
26930
  "local-image",
@@ -26211,6 +27002,15 @@ function mergeCatalogWithDrops(seedModels, rows, ctx) {
26211
27002
  for (const [modelId, bucket] of rowsByModel) {
26212
27003
  if (seeded.has(modelId)) continue;
26213
27004
  const { draft } = mergeRows(bucket, null);
27005
+ const status = draft.lifecycle?.status;
27006
+ const lifecycleReason = inactiveLifecycleReason(status);
27007
+ if (lifecycleReason) {
27008
+ dropped.push({
27009
+ modelId,
27010
+ reason: lifecycleReason
27011
+ });
27012
+ continue;
27013
+ }
26214
27014
  const parsed = asRenderableRecord(draft);
26215
27015
  if ("reason" in parsed) {
26216
27016
  dropped.push({
@@ -26327,10 +27127,19 @@ function asRenderableRecord(draft) {
26327
27127
  if (typeof draft.type !== "string" || !isRenderableModelType(draft.type)) return { reason: `unsupported model type "${String(draft.type)}"` };
26328
27128
  return { record: draft };
26329
27129
  }
27130
+ /**
27131
+ * Why a lifecycle status is not invocable, or null when it is "active". Shared
27132
+ * between invocabilityBlocker and the catalog-only tier's pre-parse check, so
27133
+ * both agree on the exact wording.
27134
+ */
27135
+ function inactiveLifecycleReason(status) {
27136
+ if (status !== "active") return `lifecycle status "${status ?? "unset"}" is not invocable`;
27137
+ return null;
27138
+ }
26330
27139
  /** Why a catalog-only record is metadata-only, or null when it is invocable. */
26331
27140
  function invocabilityBlocker(record) {
26332
- const status = record.lifecycle?.status;
26333
- if (status !== "active") return `lifecycle status "${status ?? "unset"}" is not invocable`;
27141
+ const lifecycleReason = inactiveLifecycleReason(record.lifecycle?.status);
27142
+ if (lifecycleReason) return lifecycleReason;
26334
27143
  if (!record.adapterFamily) return "no adapterFamily";
26335
27144
  if (!DISPATCHABLE_ADAPTER_FAMILIES.includes(record.adapterFamily)) return `adapterFamily "${record.adapterFamily}" is not dispatchable by this build`;
26336
27145
  if (!record.dispatchProfile) return "no dispatchProfile";
@@ -26591,6 +27400,7 @@ const DEPRECATED_MODEL_MAP = {
26591
27400
  "grok-2-vision-1212": "grok-4.5",
26592
27401
  "grok-beta": "grok-4.5",
26593
27402
  "grok-vision-beta": "grok-4.5",
27403
+ "kimi-k2.5": "kimi-k2.6",
26594
27404
  "grok-3-mini-fast": "grok-3-mini"
26595
27405
  };
26596
27406
  /**
@@ -26715,6 +27525,58 @@ var UndifferentiatedBedrockBackend = class extends BaseBedrockBackend {
26715
27525
  }
26716
27526
  };
26717
27527
  /**
27528
+ * The prices this build ships in code, keyed by model id.
27529
+ *
27530
+ * Same provenance as packages/database's modelPrices.seed.json - the adapter
27531
+ * `getModelInfo()` literals - reachable without a database, which is what the
27532
+ * price planner needs: a model's FIRST discovery-written row has no row in force
27533
+ * to carry the rates no feed publishes from, and a tier that reaches
27534
+ * getTextModelCost without `cache_read` settles cached reads at
27535
+ * input * CACHE_READ_MULTIPLIER. On DeepSeek Flash that default is 0.03/1M
27536
+ * against a real 0.006/1M. MUST STAY IN SYNC with collectStaticTextModels in
27537
+ * packages/database/src/seeds/generateModelPriceSeed.ts: both lists are "every
27538
+ * backend whose getModelInfo() is a static table", and Ollama is absent from
27539
+ * both because its listing is a live server call.
27540
+ */
27541
+ const STATIC_PRICE_BACKENDS = () => [
27542
+ new OpenAIBackend("price-literal"),
27543
+ new AnthropicBackend("price-literal"),
27544
+ new UndifferentiatedBedrockBackend(),
27545
+ new GeminiBackend("price-literal"),
27546
+ new XAIBackend("price-literal"),
27547
+ new KimiBackend("price-literal"),
27548
+ new DeepSeekBackend("price-literal"),
27549
+ new AWSBackend()
27550
+ ];
27551
+ let cached;
27552
+ /**
27553
+ * The lowest-threshold tier of each priced text model's adapter literal.
27554
+ *
27555
+ * Lowest tier on purpose: this is a last-resort carry for rates no feed
27556
+ * publishes (cache and audio), and those do not vary by context bracket in any
27557
+ * literal we ship, while the threshold keys of a discovered ladder need not
27558
+ * match the literal's. Memoized - the tables are static, and the planner runs
27559
+ * once per convergence pass.
27560
+ */
27561
+ async function adapterPriceTiers() {
27562
+ cached ??= collect();
27563
+ return cached;
27564
+ }
27565
+ async function collect() {
27566
+ const tables = await Promise.all(STATIC_PRICE_BACKENDS().map((backend) => backend.getModelInfo()));
27567
+ const tiers = /* @__PURE__ */ new Map();
27568
+ for (const model of tables.flat()) {
27569
+ if (model.type !== "text" || model.freeToRun) continue;
27570
+ const tier = lowestTier(model);
27571
+ if (tier) tiers.set(String(model.id), tier);
27572
+ }
27573
+ return tiers;
27574
+ }
27575
+ function lowestTier(model) {
27576
+ const thresholds = Object.keys(model.pricing).map(Number).filter((threshold) => Number.isFinite(threshold)).sort((a, b) => a - b);
27577
+ return thresholds.length > 0 ? model.pricing[thresholds[0]] : void 0;
27578
+ }
27579
+ /**
26718
27580
  * The dispatch group for a family whose request builder shapes its payload from
26719
27581
  * the provider's own contract and reads nothing out of the profile (Bedrock,
26720
27582
  * Gemini, xAI, Ollama, and the image/speech backends). Promotion still requires
@@ -26755,6 +27617,15 @@ const KIMI_PROFILE = {
26755
27617
  maxTokensParam: "max_completion_tokens",
26756
27618
  toolTransport: "chat"
26757
27619
  };
27620
+ /**
27621
+ * DeepSeek direct. Its own constant rather than PROVIDER_NATIVE_PROFILE because
27622
+ * tools ride Chat Completions rather than a provider-native field, which is what
27623
+ * deepseekBackend sends; the token parameter is still `max_tokens`.
27624
+ */
27625
+ const DEEPSEEK_PROFILE = {
27626
+ maxTokensParam: "max_tokens",
27627
+ toolTransport: "chat"
27628
+ };
26758
27629
  /** Backends whose family is the backend, with a request shape this build fixes. */
26759
27630
  const FAMILY_BY_BACKEND = {
26760
27631
  [ModelBackend.Anthropic]: "anthropic-messages",
@@ -26796,6 +27667,10 @@ function resolveDispatchForRecord(record) {
26796
27667
  adapterFamily: "kimi",
26797
27668
  dispatchProfile: KIMI_PROFILE
26798
27669
  };
27670
+ if (record.backend === ModelBackend.DeepSeek) return {
27671
+ adapterFamily: "deepseek",
27672
+ dispatchProfile: DEEPSEEK_PROFILE
27673
+ };
26799
27674
  const adapterFamily = FAMILY_BY_BACKEND[record.backend];
26800
27675
  return adapterFamily ? {
26801
27676
  adapterFamily,
@@ -26915,7 +27790,11 @@ var AnthropicBatchService = class AnthropicBatchService {
26915
27790
  reply: msg.content.filter((b) => b.type === "text").map((b) => b.text).join(""),
26916
27791
  tokenUsage: {
26917
27792
  inputTokens: msg.usage?.input_tokens ?? 0,
26918
- outputTokens: msg.usage?.output_tokens ?? 0
27793
+ outputTokens: msg.usage?.output_tokens ?? 0,
27794
+ cacheReadInputTokens: msg.usage?.cache_read_input_tokens ?? void 0,
27795
+ cacheCreationInputTokens: msg.usage?.cache_creation_input_tokens ?? void 0,
27796
+ cacheWrite5mInputTokens: msg.usage?.cache_creation?.ephemeral_5m_input_tokens ?? void 0,
27797
+ cacheWrite1hInputTokens: msg.usage?.cache_creation?.ephemeral_1h_input_tokens ?? void 0
26919
27798
  }
26920
27799
  };
26921
27800
  }
@@ -27155,6 +28034,10 @@ function getLlmByModel(apiKeyTable, options) {
27155
28034
  if (apiKeyTable.kimi === "expired") throw new Error("Moonshot API key is expired");
27156
28035
  backend = apiKeyTable.kimi ? new KimiBackend(apiKeyTable.kimi, logger) : null;
27157
28036
  break;
28037
+ case "deepseek":
28038
+ if (apiKeyTable.deepseek === "expired") throw new Error("DeepSeek API key is expired");
28039
+ backend = apiKeyTable.deepseek ? new DeepSeekBackend(apiKeyTable.deepseek, logger) : null;
28040
+ break;
27158
28041
  case "aws":
27159
28042
  backend = new AWSBackend();
27160
28043
  break;
@@ -27249,6 +28132,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
27249
28132
  const bflKey = resolveListingKey(ModelBackend.BFL, gateCtx);
27250
28133
  const xaiKey = resolveListingKey(ModelBackend.XAI, gateCtx);
27251
28134
  const kimiKey = resolveListingKey(ModelBackend.Kimi, gateCtx);
28135
+ const deepseekKey = resolveListingKey(ModelBackend.DeepSeek, gateCtx);
27252
28136
  const localImageBaseUrl = resolveListingKey(ModelBackend.LocalImage, gateCtx);
27253
28137
  const backends = {
27254
28138
  [ModelBackend.OpenAI]: openaiKey ? new OpenAIBackend(openaiKey) : null,
@@ -27259,6 +28143,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
27259
28143
  [ModelBackend.BFL]: bflKey ? new BFLBackend(bflKey) : null,
27260
28144
  [ModelBackend.XAI]: xaiKey ? new XAIBackend(xaiKey) : null,
27261
28145
  [ModelBackend.Kimi]: kimiKey ? new KimiBackend(kimiKey) : null,
28146
+ [ModelBackend.DeepSeek]: deepseekKey ? new DeepSeekBackend(deepseekKey) : null,
27262
28147
  [ModelBackend.AWS]: isBackendUsable(ModelBackend.AWS, gateCtx) ? new AWSBackend() : null,
27263
28148
  [ModelBackend.LocalImage]: localImageBaseUrl ? new LocalImageBackend(localImageBaseUrl, Logger.globalInstance) : null
27264
28149
  };
@@ -28781,7 +29666,7 @@ const MODEL_ALIASES = {
28781
29666
  "grok-3-mini-fast": ChatModels.GROK_3_MINI_FAST,
28782
29667
  "grok-2": ChatModels.GROK_2,
28783
29668
  "grok-2-vision": ChatModels.GROK_2_VISION,
28784
- deepseek: ChatModels.DEEPSEEK_R1,
29669
+ deepseek: ChatModels.DEEPSEEK_FLASH,
28785
29670
  "deepseek-r1": ChatModels.DEEPSEEK_R1,
28786
29671
  llama: ChatModels.LLAMA3_LOCAL,
28787
29672
  llama3: ChatModels.LLAMA3_LOCAL,
@@ -29630,8 +30515,11 @@ function effectiveContextWindow(modelInfo) {
29630
30515
  * budget below - and they must not drift apart.
29631
30516
  *
29632
30517
  * The static catalog tables are held to the positive-budget property by
29633
- * modelCatalogInputBudget.test.ts, and a discovered claim that would break it for a TEXT row is
29634
- * refused in modelDiscoveryService/catalogWrite.
30518
+ * modelCatalogInputBudget.test.ts. A discovered claim is guarded in two places, one per direction:
30519
+ * modelDiscoveryService/catalogWrite refuses a TEXT row whose output cap starves its own window,
30520
+ * and the docs parsers refuse a window or an output cap past MAX_PLAUSIBLE_TOKENS
30521
+ * (modelDiscoveryService/sources/openaiDocs.ts) - an overstated window is not a non-positive
30522
+ * budget, so catalogWrite would never see it, and no aggregator may correct a provider's figure.
29635
30523
  *
29636
30524
  * The buffer figure is imported rather than redeclared here: common owns it, and two copies of the
29637
30525
  * same number is the drift that made it a shared export in the first place.
@@ -31331,7 +32219,8 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
31331
32219
  switch (modelInfo?.backend) {
31332
32220
  case ModelBackend.OpenAI:
31333
32221
  case ModelBackend.XAI:
31334
- case ModelBackend.Kimi: {
32222
+ case ModelBackend.Kimi:
32223
+ case ModelBackend.DeepSeek: {
31335
32224
  const openaiImageBuffer = await storage.download(file.filePath);
31336
32225
  const { mime: openaiMimeType } = await getFileType(openaiImageBuffer, file.fileName, file.mimeType);
31337
32226
  const openaiBase64 = openaiImageBuffer.toString("base64");
@@ -36743,6 +37632,7 @@ __reExport(/* @__PURE__ */ __exportAll({
36743
37632
  registrableDomain: () => registrableDomain,
36744
37633
  reservationOutputTokens: () => reservationOutputTokens,
36745
37634
  resolveEmbeddingConfig: () => resolveEmbeddingConfig,
37635
+ resolveEmbeddingWithKeylessFallback: () => resolveEmbeddingWithKeylessFallback,
36746
37636
  resolveSupportedMimeType: () => resolveSupportedMimeType,
36747
37637
  safeInputWindow: () => safeInputWindow,
36748
37638
  scopedOverrideKey: () => scopedOverrideKey,