@bike4mind/cli 0.21.0 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{AgentHistoryStore-ucv6xXVT.mjs → AgentHistoryStore-7zQUOxkA.mjs} +1090 -200
- package/dist/{ApiClient-Ut1MXtOs.mjs → ApiClient-XdY9O3nz.mjs} +2 -2
- package/dist/{ConfigStore-CoY0l0gr.mjs → ConfigStore-8_0WsN5r.mjs} +210 -22
- package/dist/{buildAgent-jZhBReAr.mjs → buildAgent-CvPRH2n-.mjs} +2 -2
- package/dist/commands/acpCommand.mjs +4 -4
- package/dist/commands/apiCommand.mjs +1 -1
- package/dist/commands/doctorCommand.mjs +1 -1
- package/dist/commands/envCommand.mjs +1 -1
- package/dist/commands/headlessCommand.mjs +3 -3
- package/dist/commands/mcpCommand.mjs +3 -3
- package/dist/commands/pluginCommand.mjs +1 -1
- package/dist/commands/updateCommand.mjs +1 -1
- package/dist/index.mjs +5 -5
- package/dist/{package-CfIETbXd.mjs → package-OZdXqn0e.mjs} +1 -1
- package/dist/{serve-C9UDR5px.mjs → serve-CEqSZwl6.mjs} +2 -2
- package/package.json +8 -8
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { $ as SupportedFabFileMimeTypes,
|
|
2
|
+
import { $ as SupportedFabFileMimeTypes, A as HTTPError, At as isPlaceholderApiKey, B as OPENAI_GPT_IMAGE_1_IMAGE_SIZES, Bt as reservationOutputTokens, Ct as isGPTImageModel, D as FIXED_TEMPERATURE_MODELS, Dt as isMediaModelType, E as FIELD_GROUP_OF, Et as isImageServeable, F as MODEL_INFO_FIELD_GROUP_OF, Ft as isUserInitiatedAbort, G as PermissionDeniedError, Gt as toModelRecord, H as OllamaEmbeddingModel, Ht as secureParameters, I as McpServerName, It as isZodError, J as REFUSAL_FALLBACK_MODELS, Jt as withRetry, K as REASONING_EFFORT_INCOMPATIBLE_WITH_TOOLS_MODELS, Kt as usdToCredits, L as ModelBackend, Lt as mapMimeTypeToArtifactType, M as IMAGE_SIZE_CONSTRAINTS, Mt as isRetryableError, N as ImageModels, Nt as isSupportedFabFileMimeType, O as FORMAT_PROMPT_TEMPLATE, Ot as isModelAccessible, P as InternalServerError, Pt as isUnlimitedHistory, Q as SpeechToTextModels, R as NO_TEMPERATURE_MODELS, Rt as obfuscateApiKey, S as CorruptedFileError, St as isGPTImage2Model, Tt as isImageAttachment, U as OpenAIEmbeddingModel, Ut as settingsMap, V as OPENAI_GPT_IMAGE_2_IMAGE_SIZES, Vt as resolveHistoryFetchLimit, Wt as toModelInfo, Y as RESPONSES_API_TOOL_MODELS, _ as BadRequestError, _t as isAudioMimeType, at as VideoModels, ct as applyModelPriceCatalog, dt as defaultEmbeddingModelForEnv, en as buildRateLimitLogEntry, et as TTS_MAX_INPUT_CHARS, ft as getMcpProviderMetadata, g as BFL_SAFETY_TOLERANCE, gt as hasUsableLimits, h as BEDROCK_NO_PROMPT_CACHING_MODELS, ht as hasKeylessCloudEmbedder, it as VIDEO_SIZE_CONSTRAINTS, j as HttpStatus, jt as isRenderableModelType, k as ForbiddenError, kt as isModelDeprecated, lt as calculateRetryDelay, m as ApiKeyType, mt as getRetryAfterMs, n as logger, nn as isNearLimit, nt as UnauthorizedError, ot as VoyageAIEmbeddingModel, p as ARTIFACT_ATTRS_PATTERN, pt as getQuestErrorCode, q as REASONING_SUPPORTED_MODELS, qt as usdToCreditsStochastic, rn as parseRateLimitHeaders, rt as UnprocessableEntityError, st as WORK_ITEM_STATUSES, tn as extractSnippetMeta, tt as TooManyRequestsError, ut as dayjsConfig_default, v as BedrockEmbeddingModel, vt as isChunkRebuildPending, w as DEFAULT_UNKNOWN_CONTEXT_WINDOW, wt as isGeminiModelId, x as ChatModels, xt as isFieldGroup, y as CONTEXT_WINDOW_SAFETY_BUFFER_TOKENS, yt as isChunkStalledFile, z as NotFoundError, zt as parseEmbeddingRateLimitHeaders } from "./ConfigStore-8_0WsN5r.mjs";
|
|
3
3
|
import { n as isPathAllowed, t as assertPathAllowed } from "./pathValidation-D8tjkQXE-1HwvsuYT.mjs";
|
|
4
4
|
import { n as isTerminalShellStatus, t as getShellSessionManager } from "./ShellSessionManager-6o8KZzl1-vrbPAUTq.mjs";
|
|
5
5
|
import { execFile, execFileSync, spawn } from "child_process";
|
|
@@ -21,6 +21,7 @@ import * as turndownPluginGfm from "@joplin/turndown-plugin-gfm";
|
|
|
21
21
|
import * as cheerio from "cheerio";
|
|
22
22
|
import FirecrawlDefault, { FirecrawlError } from "@mendable/firecrawl-js";
|
|
23
23
|
import { lookup } from "node:dns/promises";
|
|
24
|
+
import mongoose, { isObjectIdOrHexString } from "mongoose";
|
|
24
25
|
import random from "lodash/random.js";
|
|
25
26
|
import sum from "lodash/sum.js";
|
|
26
27
|
import times from "lodash/times.js";
|
|
@@ -48,7 +49,6 @@ import { NodeHttpHandler } from "@smithy/node-http-handler";
|
|
|
48
49
|
import "@opensearch-project/opensearch";
|
|
49
50
|
import "@aws-sdk/credential-provider-node";
|
|
50
51
|
import "@opensearch-project/opensearch/aws-v3";
|
|
51
|
-
import mongoose from "mongoose";
|
|
52
52
|
import { parse } from "shell-quote";
|
|
53
53
|
import { homedir as homedir$1 } from "node:os";
|
|
54
54
|
import { EventEmitter } from "events";
|
|
@@ -921,6 +921,7 @@ const DEMO_KEY_MAP = {
|
|
|
921
921
|
[ApiKeyType.gemini]: "geminiDemoKey",
|
|
922
922
|
[ApiKeyType.xai]: "xaiApiKey",
|
|
923
923
|
[ApiKeyType.kimi]: "moonshotApiKey",
|
|
924
|
+
[ApiKeyType.deepseek]: "deepseekApiKey",
|
|
924
925
|
[ApiKeyType.bfl]: "bflApiKey",
|
|
925
926
|
[ApiKeyType.voyageai]: "voyageApiKey",
|
|
926
927
|
[ApiKeyType.elevenlabs]: "elevenLabsServerApiKey"
|
|
@@ -978,6 +979,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
|
|
|
978
979
|
ApiKeyType.bfl,
|
|
979
980
|
ApiKeyType.xai,
|
|
980
981
|
ApiKeyType.kimi,
|
|
982
|
+
ApiKeyType.deepseek,
|
|
981
983
|
ApiKeyType.voyageai
|
|
982
984
|
], adapters) : Promise.resolve([]), adapters.getSettingsByNames([
|
|
983
985
|
"openaiDemoKey",
|
|
@@ -986,6 +988,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
|
|
|
986
988
|
"bflApiKey",
|
|
987
989
|
"xaiApiKey",
|
|
988
990
|
"moonshotApiKey",
|
|
991
|
+
"deepseekApiKey",
|
|
989
992
|
"voyageApiKey",
|
|
990
993
|
"ollamaBackend",
|
|
991
994
|
"EnableOllama"
|
|
@@ -998,6 +1001,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
|
|
|
998
1001
|
const bflUserKey = userKeyMap.get(ApiKeyType.bfl) || null;
|
|
999
1002
|
const xaiUserKey = userKeyMap.get(ApiKeyType.xai) || null;
|
|
1000
1003
|
const kimiUserKey = userKeyMap.get(ApiKeyType.kimi) || null;
|
|
1004
|
+
const deepseekUserKey = userKeyMap.get(ApiKeyType.deepseek) || null;
|
|
1001
1005
|
const voyageaiUserKey = userKeyMap.get(ApiKeyType.voyageai) || null;
|
|
1002
1006
|
const openaiDemoKey = adminSettings["openaiDemoKey"];
|
|
1003
1007
|
const anthropicDemoKey = adminSettings["anthropicDemoKey"];
|
|
@@ -1005,6 +1009,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
|
|
|
1005
1009
|
const bflDemoKey = adminSettings["bflApiKey"];
|
|
1006
1010
|
const xaiDemoKey = adminSettings["xaiApiKey"];
|
|
1007
1011
|
const kimiDemoKey = adminSettings["moonshotApiKey"];
|
|
1012
|
+
const deepseekDemoKey = adminSettings["deepseekApiKey"];
|
|
1008
1013
|
const voyageaiDemoKey = adminSettings["voyageApiKey"];
|
|
1009
1014
|
const ollamaBackend = adminSettings["ollamaBackend"];
|
|
1010
1015
|
const enableOllama = adminSettings["EnableOllama"];
|
|
@@ -1023,6 +1028,7 @@ const getEffectiveLLMApiKeys = async (userId, adapters, options) => {
|
|
|
1023
1028
|
bfl: keyOrExpired(bflUserKey) || bflDemoKey || null,
|
|
1024
1029
|
xai: keyOrExpired(xaiUserKey) || xaiDemoKey || envKey("XAI_API_KEY"),
|
|
1025
1030
|
kimi: keyOrExpired(kimiUserKey) || kimiDemoKey || envKey("MOONSHOT_API_KEY"),
|
|
1031
|
+
deepseek: keyOrExpired(deepseekUserKey) || deepseekDemoKey || envKey("DEEPSEEK_API_KEY"),
|
|
1026
1032
|
voyageai: keyOrExpired(voyageaiUserKey) || voyageaiDemoKey || null,
|
|
1027
1033
|
ollama: (ollamaEnabled ? ollamaBackend || null : null) || envKey("OLLAMA_BASE_URL"),
|
|
1028
1034
|
imageGen: envKey("IMAGE_GEN_BASE_URL")
|
|
@@ -2088,7 +2094,7 @@ const webSearchTool = {
|
|
|
2088
2094
|
})
|
|
2089
2095
|
};
|
|
2090
2096
|
//#endregion
|
|
2091
|
-
//#region ../../b4m-core/services/dist/toolGenerators-
|
|
2097
|
+
//#region ../../b4m-core/services/dist/toolGenerators-DGjRmthM.mjs
|
|
2092
2098
|
const diceRoll = async (parameters) => {
|
|
2093
2099
|
if (!parameters?.sides || !parameters?.times) throw new Error("Tool dice roll: Missing required parameters");
|
|
2094
2100
|
return sum(times(parameters.times, () => random(1, parameters.sides))).toString();
|
|
@@ -2656,6 +2662,27 @@ const promptEnhancementTool = {
|
|
|
2656
2662
|
}
|
|
2657
2663
|
})
|
|
2658
2664
|
};
|
|
2665
|
+
/**
|
|
2666
|
+
* Is this value shaped like something Mongoose can cast to an `_id`?
|
|
2667
|
+
*
|
|
2668
|
+
* Tool arguments are composed by the model out of conversation text and reach us as unvalidated
|
|
2669
|
+
* JSON, so an id parameter routinely holds something that is not an id at all - a filename token,
|
|
2670
|
+
* an arXiv number, a bare integer. Mongoose casts `_id` and throws a CastError on those, which a
|
|
2671
|
+
* generic catch upstream then reports as an outage rather than the bad argument it is (#2530).
|
|
2672
|
+
* Call this before handing a model-supplied id to `findById` and answer a false the same way the
|
|
2673
|
+
* surface answers a genuinely missing row.
|
|
2674
|
+
*
|
|
2675
|
+
* `isObjectIdOrHexString`, not `isValidObjectId`: the latter also accepts a number and casts it to
|
|
2676
|
+
* a fabricated id, and a model emitting `{"file_id": 12}` gives us exactly that despite the
|
|
2677
|
+
* `string` type. Same choice, same reason, as `usableObjectIds` in @bike4mind/db-core, which is
|
|
2678
|
+
* the array-shaped version of this check.
|
|
2679
|
+
*
|
|
2680
|
+
* NOT usable for artifact ids (`artifact_<...>`), which are matched on a string `id` field rather
|
|
2681
|
+
* than `_id` - see `createArtifactId` in @bike4mind/common.
|
|
2682
|
+
*/
|
|
2683
|
+
function isObjectIdShaped(id) {
|
|
2684
|
+
return isObjectIdOrHexString(id);
|
|
2685
|
+
}
|
|
2659
2686
|
let _showUserQuestion = null;
|
|
2660
2687
|
/**
|
|
2661
2688
|
* Inject the CLI callback that displays the question UI.
|
|
@@ -4268,7 +4295,7 @@ const latticeAddEntityTool = {
|
|
|
4268
4295
|
createdAt: /* @__PURE__ */ new Date(),
|
|
4269
4296
|
updatedAt: /* @__PURE__ */ new Date()
|
|
4270
4297
|
};
|
|
4271
|
-
if (context.db.latticeModels && modelId &&
|
|
4298
|
+
if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
|
|
4272
4299
|
const model = await context.db.latticeModels.findById(modelId);
|
|
4273
4300
|
if (model && model.userId === context.userId) {
|
|
4274
4301
|
const existingIndex = model.data.entities.findIndex((e) => e.id === entityId);
|
|
@@ -4412,7 +4439,7 @@ const latticeSetValueTool = {
|
|
|
4412
4439
|
else if (rawValue.toLowerCase() === "true") value = true;
|
|
4413
4440
|
else if (rawValue.toLowerCase() === "false") value = false;
|
|
4414
4441
|
const entityId = entityName.toLowerCase().replace(/\s+/g, "_");
|
|
4415
|
-
if (context.db.latticeModels && modelId &&
|
|
4442
|
+
if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
|
|
4416
4443
|
const model = await context.db.latticeModels.findById(modelId);
|
|
4417
4444
|
if (model && model.userId === context.userId) {
|
|
4418
4445
|
const entity = model.data.entities.find((e) => e.id === entityId || e.name === entityName);
|
|
@@ -4547,7 +4574,7 @@ const latticeCreateRuleTool = {
|
|
|
4547
4574
|
};
|
|
4548
4575
|
const outputEntityId = parsedRule.outputEntity.toLowerCase().replace(/\s+/g, "_");
|
|
4549
4576
|
let entityCreatedMessage = "";
|
|
4550
|
-
if (context.db.latticeModels && modelId &&
|
|
4577
|
+
if (context.db.latticeModels && modelId && isObjectIdShaped(modelId)) try {
|
|
4551
4578
|
const model = await context.db.latticeModels.findById(modelId);
|
|
4552
4579
|
if (model && model.userId === context.userId) {
|
|
4553
4580
|
if (!model.data.entities.some((e) => e.id === outputEntityId || e.name.toLowerCase() === parsedRule.outputEntity.toLowerCase()) && parsedRule.outputEntity !== "unknown") {
|
|
@@ -7230,7 +7257,25 @@ const getProviderFromModel = (modelName) => {
|
|
|
7230
7257
|
* ("OpenAI rejected the embedding request") instead of the actionable missing-credential path.
|
|
7231
7258
|
*/
|
|
7232
7259
|
const EXPIRED_KEY_SENTINEL = "expired";
|
|
7233
|
-
|
|
7260
|
+
/**
|
|
7261
|
+
* A placeholder is rejected for the same reason the sentinel is, and the two sibling answers to
|
|
7262
|
+
* "is this key usable" both already do it (modelDiscoveryService/credentials.ts,
|
|
7263
|
+
* toolAvailability.ts, and defaultEmbeddingModelForEnv's own key test). Keeping a placeholder here
|
|
7264
|
+
* would report `missing: null`, so the keyless fallback would never fire and EmbeddingFactory would
|
|
7265
|
+
* then throw on the placeholder itself - the PR's headline case failing silently rather than
|
|
7266
|
+
* substituting. `.trim()` because a whitespace-only value is no key either.
|
|
7267
|
+
*/
|
|
7268
|
+
const usableKey = (value) => {
|
|
7269
|
+
const trimmed = value?.trim();
|
|
7270
|
+
if (!trimmed || trimmed === EXPIRED_KEY_SENTINEL || isPlaceholderApiKey(trimmed)) return null;
|
|
7271
|
+
return trimmed;
|
|
7272
|
+
};
|
|
7273
|
+
/**
|
|
7274
|
+
* The slot is missing because THIS CALLER's key expired, not because the deployment holds none.
|
|
7275
|
+
* Bedrock has no credential and Ollama's base URL carries no expiry, so only the two keyed cloud
|
|
7276
|
+
* providers can be in this state. See the keyless-fallback doc comment for why it matters.
|
|
7277
|
+
*/
|
|
7278
|
+
const isExpiredCallerKey = (missing, keyTable) => missing === "openai" && keyTable?.openai === EXPIRED_KEY_SENTINEL || missing === "voyageai" && keyTable?.voyageai === EXPIRED_KEY_SENTINEL;
|
|
7234
7279
|
/**
|
|
7235
7280
|
* Map an embedding provider plus the caller's resolved key table to the config
|
|
7236
7281
|
* `EmbeddingFactory` expects, and report which credential is missing if any.
|
|
@@ -7250,6 +7295,13 @@ const usableKey = (value) => value && value !== EXPIRED_KEY_SENTINEL ? value : n
|
|
|
7250
7295
|
*
|
|
7251
7296
|
* Adding a provider means editing this function and its table test, not auditing
|
|
7252
7297
|
* every call site.
|
|
7298
|
+
*
|
|
7299
|
+
* `keyTable` is always an ANSWER about the caller's credentials, never a failure channel. `null` /
|
|
7300
|
+
* `undefined` mean "resolved: this caller holds none", and both this function and the keyless
|
|
7301
|
+
* fallback below act on that - substituting the keyless embedder is a real decision with a real
|
|
7302
|
+
* vector space attached. A caller whose own key lookup THREW must therefore not pass the failure in
|
|
7303
|
+
* here; it has to report unknown instead, or an unavailable Mongo becomes a confident Titan on a
|
|
7304
|
+
* fully keyed production stage.
|
|
7253
7305
|
*/
|
|
7254
7306
|
function resolveEmbeddingConfig(provider, keyTable) {
|
|
7255
7307
|
switch (provider) {
|
|
@@ -7273,19 +7325,78 @@ function resolveEmbeddingConfig(provider, keyTable) {
|
|
|
7273
7325
|
missing: "voyageai"
|
|
7274
7326
|
};
|
|
7275
7327
|
}
|
|
7276
|
-
case ModelBackend.Ollama:
|
|
7277
|
-
|
|
7278
|
-
|
|
7279
|
-
|
|
7280
|
-
|
|
7281
|
-
|
|
7282
|
-
|
|
7328
|
+
case ModelBackend.Ollama: {
|
|
7329
|
+
const baseUrl = keyTable?.ollama?.trim();
|
|
7330
|
+
return baseUrl ? {
|
|
7331
|
+
config: { ollamaBaseUrl: baseUrl },
|
|
7332
|
+
missing: null
|
|
7333
|
+
} : {
|
|
7334
|
+
config: {},
|
|
7335
|
+
missing: "ollama"
|
|
7336
|
+
};
|
|
7337
|
+
}
|
|
7283
7338
|
case ModelBackend.Bedrock: return {
|
|
7284
7339
|
config: {},
|
|
7285
7340
|
missing: null
|
|
7286
7341
|
};
|
|
7287
7342
|
}
|
|
7288
7343
|
}
|
|
7344
|
+
/**
|
|
7345
|
+
* Resolve a config for `model`, falling back to keyless Bedrock when this deployment holds no
|
|
7346
|
+
* credential for the provider `model` needs but can reach Bedrock with its own AWS role.
|
|
7347
|
+
*
|
|
7348
|
+
* WHY THIS EXISTS HERE and not in `defaultEmbeddingModelForEnv`: "does this deployment have a
|
|
7349
|
+
* cloud embedding key" is unanswerable from process.env on a hosted stage - an SST secret arrives
|
|
7350
|
+
* as a linked Resource, so OPENAI_API_KEY is absent on production exactly as it is on a preview.
|
|
7351
|
+
* The key table passed in here is the first point that actually knows, which is why the decision
|
|
7352
|
+
* belongs at this seam.
|
|
7353
|
+
*
|
|
7354
|
+
* Related to but NOT the same as EmbeddingFactory.getDefaultEmbeddingModel, which ranks providers
|
|
7355
|
+
* from scratch (OpenAI > VoyageAI > Ollama > Bedrock). This keeps the model the admin asked for
|
|
7356
|
+
* whenever it is reachable and only substitutes the keyless one otherwise - so a deployment
|
|
7357
|
+
* holding only a Voyage key still falls back to Bedrock here, where the factory would pick
|
|
7358
|
+
* voyage-3. Deliberate: this is a reachability backstop, not a second opinion on the setting.
|
|
7359
|
+
*
|
|
7360
|
+
* ONLY FOR CALLERS FREE TO CHOOSE THE MODEL - i.e. the model came from the `defaultEmbeddingModel`
|
|
7361
|
+
* admin setting. A caller that must hit one specific vector space MUST keep using
|
|
7362
|
+
* `resolveEmbeddingConfig` and fail, because a fallback there would silently compare or write
|
|
7363
|
+
* across incompatible spaces:
|
|
7364
|
+
* - V2 mementos are pinned to MEMENTO_EMBEDDING_MODEL at 512 truncated dims (see embedding.ts);
|
|
7365
|
+
* - V1 mementos (mementoEmbedding.ts, getRelevantMementos.ts) read the admin default and so LOOK
|
|
7366
|
+
* free to choose, but neither live write path stamps `Memento.embeddingModel` - only the
|
|
7367
|
+
* reembedMementos backfill does. Their vectors are ranked by in-process cosine with no width
|
|
7368
|
+
* guard and no Atlas index, so a substitution here would drop 1024-dim vectors into a field
|
|
7369
|
+
* holding 1536-dim ones with nothing recording which is which, and nothing able to tell them
|
|
7370
|
+
* apart afterwards. Stamping V1 is the prerequisite for including it, not this helper.
|
|
7371
|
+
* - alternateModelAnn embeds one query per model bucket to match each chunk's recorded stamp.
|
|
7372
|
+
*
|
|
7373
|
+
* Returns the model actually used, so callers stamp what they embedded with rather than what they
|
|
7374
|
+
* asked for - that is what keeps `fabFileChunk`'s recorded `embeddingModel` honest.
|
|
7375
|
+
*
|
|
7376
|
+
* TWO credential states are deliberately NOT treated as "this deployment is keyless":
|
|
7377
|
+
* - `missing: 'ollama'` - a self-host that set no OLLAMA_BASE_URL has no AWS role either, and
|
|
7378
|
+
* OPENAI_KEY_MISSING_MESSAGE naming OPENAI_API_KEY / OLLAMA_BASE_URL is the actionable error
|
|
7379
|
+
* there. `hasKeylessCloudEmbedder()` already excludes self-host; this is belt-and-braces.
|
|
7380
|
+
* - an EXPIRED caller key. `getEffectiveLLMApiKeys` returns the `'expired'` sentinel instead of
|
|
7381
|
+
* falling through to the platform demo key, deliberately, so the user is told their key
|
|
7382
|
+
* expired rather than silently moved onto the platform's (see the reasoning in the
|
|
7383
|
+
* reactivate-collateral-deactivated-api-keys migration). `usableKey` normalizes that to null
|
|
7384
|
+
* for the CREDENTIAL check, which is right - but read as "this deployment holds no key" it
|
|
7385
|
+
* would substitute Titan for that one caller on keyed production, querying a vector space the
|
|
7386
|
+
* corpus was never written in. The deployment's own key state is unchanged by one expiry, so
|
|
7387
|
+
* the requested model is returned and the actionable expired-key error stands.
|
|
7388
|
+
*/
|
|
7389
|
+
function resolveEmbeddingWithKeylessFallback(model, keyTable) {
|
|
7390
|
+
const resolved = resolveEmbeddingConfig(getProviderFromModel(model), keyTable);
|
|
7391
|
+
if (!resolved.missing || resolved.missing === "ollama" || isExpiredCallerKey(resolved.missing, keyTable) || !hasKeylessCloudEmbedder()) return {
|
|
7392
|
+
...resolved,
|
|
7393
|
+
model
|
|
7394
|
+
};
|
|
7395
|
+
return {
|
|
7396
|
+
...resolveEmbeddingConfig(ModelBackend.Bedrock, null),
|
|
7397
|
+
model: BedrockEmbeddingModel.TITAN_TEXT_EMBEDDINGS_V2
|
|
7398
|
+
};
|
|
7399
|
+
}
|
|
7289
7400
|
const ChunkSchema = z$1.object({
|
|
7290
7401
|
text: z$1.string(),
|
|
7291
7402
|
tokenCount: z$1.number()
|
|
@@ -10099,16 +10210,16 @@ function parseSettingsHooks(settingsJson) {
|
|
|
10099
10210
|
return null;
|
|
10100
10211
|
}
|
|
10101
10212
|
}
|
|
10102
|
-
let cached;
|
|
10213
|
+
let cached$1;
|
|
10103
10214
|
/**
|
|
10104
10215
|
* Lazily-built process-hook singleton from `B4M_SETTINGS_JSON`. Returns null when
|
|
10105
10216
|
* no hooks are configured, so call sites can `void getProcessHooks()?.fireStop()`.
|
|
10106
10217
|
*/
|
|
10107
10218
|
function getProcessHooks() {
|
|
10108
|
-
if (cached !== void 0) return cached;
|
|
10219
|
+
if (cached$1 !== void 0) return cached$1;
|
|
10109
10220
|
const hooks = parseSettingsHooks(process.env.B4M_SETTINGS_JSON);
|
|
10110
|
-
cached = hooks ? new ProcessHooks(hooks) : null;
|
|
10111
|
-
return cached;
|
|
10221
|
+
cached$1 = hooks ? new ProcessHooks(hooks) : null;
|
|
10222
|
+
return cached$1;
|
|
10112
10223
|
}
|
|
10113
10224
|
//#endregion
|
|
10114
10225
|
//#region src/agents/interactionModeClamp.ts
|
|
@@ -15940,6 +16051,10 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
|
|
|
15940
16051
|
BaseBedrockBackend: () => BaseBedrockBackend,
|
|
15941
16052
|
ChoiceEndReason: () => ChoiceEndReason,
|
|
15942
16053
|
ChoiceStatus: () => ChoiceStatus,
|
|
16054
|
+
DEEPSEEK_EFFORT_LEVELS: () => DEEPSEEK_EFFORT_LEVELS,
|
|
16055
|
+
DEEPSEEK_MAX_STOP_SEQUENCES: () => 16,
|
|
16056
|
+
DEEPSEEK_MODELS: () => DEEPSEEK_MODELS,
|
|
16057
|
+
DEEPSEEK_THINKING_TOP_P_FLOOR: () => DEEPSEEK_THINKING_TOP_P_FLOOR,
|
|
15943
16058
|
DEFAULT_MAX_TOOL_CALLS: () => 10,
|
|
15944
16059
|
DEFAULT_REALTIME_VOICE_MODEL: () => DEFAULT_REALTIME_VOICE_MODEL,
|
|
15945
16060
|
DEGENERATE_STREAM_MESSAGE: () => DEGENERATE_STREAM_MESSAGE,
|
|
@@ -15947,6 +16062,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
|
|
|
15947
16062
|
DEPRECATED_MODEL_MAP: () => DEPRECATED_MODEL_MAP,
|
|
15948
16063
|
DEPRECATED_MODEL_REQUEST_METRIC: () => DEPRECATED_MODEL_REQUEST_METRIC,
|
|
15949
16064
|
DISPATCHABLE_ADAPTER_FAMILIES: () => DISPATCHABLE_ADAPTER_FAMILIES,
|
|
16065
|
+
DeepSeekBackend: () => DeepSeekBackend,
|
|
15950
16066
|
DeepSeekBedrockBackend: () => DeepSeekBedrockBackend,
|
|
15951
16067
|
DispatchModel: () => DispatchModel,
|
|
15952
16068
|
GeminiBackend: () => GeminiBackend,
|
|
@@ -15967,6 +16083,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
|
|
|
15967
16083
|
UndifferentiatedBedrockBackend: () => UndifferentiatedBedrockBackend,
|
|
15968
16084
|
UnsupportedAdapterFamilyError: () => UnsupportedAdapterFamilyError,
|
|
15969
16085
|
XAIBackend: () => XAIBackend,
|
|
16086
|
+
adapterPriceTiers: () => adapterPriceTiers,
|
|
15970
16087
|
backendForAdapterFamily: () => backendForAdapterFamily,
|
|
15971
16088
|
buildApiKeyTable: () => buildApiKeyTable,
|
|
15972
16089
|
buildSupersededIndex: () => buildSupersededIndex,
|
|
@@ -15977,6 +16094,10 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
|
|
|
15977
16094
|
checkStaleModelReferences: () => checkStaleModelReferences,
|
|
15978
16095
|
classifyModelReference: () => classifyModelReference,
|
|
15979
16096
|
createDegenerateStreamGuard: () => createDegenerateStreamGuard,
|
|
16097
|
+
deepseekReasoningParams: () => deepseekReasoningParams,
|
|
16098
|
+
deepseekSamplingParams: () => deepseekSamplingParams,
|
|
16099
|
+
deepseekStopSequences: () => deepseekStopSequences,
|
|
16100
|
+
deepseekThinkingEnabled: () => deepseekThinkingEnabled,
|
|
15980
16101
|
ensureToolPairingIntegrity: () => ensureToolPairingIntegrity,
|
|
15981
16102
|
extractThinkContent: () => extractThinkContent,
|
|
15982
16103
|
getAvailableModels: () => getAvailableModels,
|
|
@@ -16011,6 +16132,7 @@ var dist_exports = /* @__PURE__ */ __exportAll$3({
|
|
|
16011
16132
|
splitCacheInclusiveInput: () => splitCacheInclusiveInput,
|
|
16012
16133
|
stripAllToolBlocks: () => stripAllToolBlocks,
|
|
16013
16134
|
stripToolDependentMessages: () => stripToolDependentMessages,
|
|
16135
|
+
toDeepSeekEffort: () => toDeepSeekEffort,
|
|
16014
16136
|
toKimiEffort: () => toKimiEffort,
|
|
16015
16137
|
toProviderEndUserId: () => toProviderEndUserId,
|
|
16016
16138
|
updateReplacedByOverlay: () => updateReplacedByOverlay
|
|
@@ -16726,6 +16848,92 @@ var KimiCachingAdapter = class {
|
|
|
16726
16848
|
}
|
|
16727
16849
|
};
|
|
16728
16850
|
/**
|
|
16851
|
+
* The cache-inclusive-to-cache-exclusive conversion, shared by every adapter whose
|
|
16852
|
+
* provider reports cached tokens as a SUBSET of the prompt count.
|
|
16853
|
+
*
|
|
16854
|
+
* getTextModelCost expects Anthropic's convention: `inputTokens` counts only uncached
|
|
16855
|
+
* tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
|
|
16856
|
+
* and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
|
|
16857
|
+
* prompt total already CONTAINS the cached tokens - so those adapters must subtract
|
|
16858
|
+
* here before forwarding, or settlement double-bills the cached portion.
|
|
16859
|
+
*
|
|
16860
|
+
* Must stay in sync with the disjoint-fields assumption documented at the settlement
|
|
16861
|
+
* site in ChatCompletionProcess.
|
|
16862
|
+
*/
|
|
16863
|
+
/**
|
|
16864
|
+
* Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
|
|
16865
|
+
*
|
|
16866
|
+
* Forwarding the cached count without subtracting double-bills it; forwarding nothing
|
|
16867
|
+
* charges the full input rate on tokens the provider billed at a fraction of it.
|
|
16868
|
+
* Subtracting is the only split that bills what the provider actually charged.
|
|
16869
|
+
*
|
|
16870
|
+
* Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
|
|
16871
|
+
* input count would silently credit the user.
|
|
16872
|
+
*/
|
|
16873
|
+
function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
|
|
16874
|
+
if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
|
|
16875
|
+
const cached = Math.min(cacheReadTokens, totalPromptTokens);
|
|
16876
|
+
return {
|
|
16877
|
+
inputTokens: Math.max(0, totalPromptTokens - cached),
|
|
16878
|
+
cacheReadInputTokens: cached
|
|
16879
|
+
};
|
|
16880
|
+
}
|
|
16881
|
+
/**
|
|
16882
|
+
* Cached prompt tokens from a raw provider usage object, across every spelling in use:
|
|
16883
|
+
* OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
|
|
16884
|
+
* Responses API under `input_tokens_details`, Moonshot publishes a flat
|
|
16885
|
+
* `cached_tokens` alongside the OpenAI-shaped nesting, and DeepSeek its own flat
|
|
16886
|
+
* `prompt_cache_hit_tokens`. Reading only one spelling silently bills every cache
|
|
16887
|
+
* hit on the other transports at the full input rate.
|
|
16888
|
+
*
|
|
16889
|
+
* DeepSeek's own spelling leads, because it is the number its invoice is computed
|
|
16890
|
+
* from; the OpenAI-shaped ones it also sends are the fallback for a proxy that
|
|
16891
|
+
* forwards only those.
|
|
16892
|
+
*/
|
|
16893
|
+
function cachedTokensFromUsage(usage) {
|
|
16894
|
+
if (!usage) return 0;
|
|
16895
|
+
const candidates = [
|
|
16896
|
+
usage.prompt_cache_hit_tokens,
|
|
16897
|
+
usage.cached_tokens,
|
|
16898
|
+
usage.prompt_tokens_details?.cached_tokens,
|
|
16899
|
+
usage.input_tokens_details?.cached_tokens
|
|
16900
|
+
];
|
|
16901
|
+
for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
|
|
16902
|
+
return 0;
|
|
16903
|
+
}
|
|
16904
|
+
/**
|
|
16905
|
+
* DeepSeek context caching. Automatic, like Moonshot's and xAI's: no parameter,
|
|
16906
|
+
* no header, no explicit cache-creation call. The adapter exists only to read
|
|
16907
|
+
* the counters back out.
|
|
16908
|
+
* @see https://api-docs.deepseek.com/guides/kv_cache
|
|
16909
|
+
*/
|
|
16910
|
+
var DeepSeekCachingAdapter = class {
|
|
16911
|
+
applyCaching(apiParams, _strategy) {
|
|
16912
|
+
return apiParams;
|
|
16913
|
+
}
|
|
16914
|
+
extractCacheStats(response, model) {
|
|
16915
|
+
const usage = response.usage;
|
|
16916
|
+
if (!usage) return void 0;
|
|
16917
|
+
const totalInputTokens = usage.prompt_tokens || 0;
|
|
16918
|
+
const cachedTokens = cachedTokensFromUsage(usage);
|
|
16919
|
+
const cacheHitRate = totalInputTokens > 0 ? cachedTokens / totalInputTokens * 100 : 0;
|
|
16920
|
+
const costSavingsPercent = cacheHitRate * .98;
|
|
16921
|
+
const estimatedLatencyReduction = cacheHitRate * .7;
|
|
16922
|
+
return {
|
|
16923
|
+
provider: ModelBackend.DeepSeek,
|
|
16924
|
+
model,
|
|
16925
|
+
totalInputTokens,
|
|
16926
|
+
cacheReadTokens: cachedTokens,
|
|
16927
|
+
cacheWriteTokens: 0,
|
|
16928
|
+
uncachedTokens: Math.max(0, totalInputTokens - cachedTokens),
|
|
16929
|
+
cacheHitRate,
|
|
16930
|
+
costSavingsPercent,
|
|
16931
|
+
estimatedLatencyReduction,
|
|
16932
|
+
providerMetadata: { automatic: true }
|
|
16933
|
+
};
|
|
16934
|
+
}
|
|
16935
|
+
};
|
|
16936
|
+
/**
|
|
16729
16937
|
* Helper to log cache statistics in a consistent format across all providers
|
|
16730
16938
|
*/
|
|
16731
16939
|
function logCacheStats(logger, cacheStats, options) {
|
|
@@ -16759,6 +16967,7 @@ const ADAPTERS = {
|
|
|
16759
16967
|
[ModelBackend.Bedrock]: new AnthropicCachingAdapter(),
|
|
16760
16968
|
[ModelBackend.XAI]: new XAICachingAdapter(),
|
|
16761
16969
|
[ModelBackend.Kimi]: new KimiCachingAdapter(),
|
|
16970
|
+
[ModelBackend.DeepSeek]: new DeepSeekCachingAdapter(),
|
|
16762
16971
|
[ModelBackend.Ollama]: new NoOpCachingAdapter(),
|
|
16763
16972
|
[ModelBackend.BFL]: new NoOpCachingAdapter(),
|
|
16764
16973
|
[ModelBackend.VoyageAI]: new NoOpCachingAdapter(),
|
|
@@ -16809,14 +17018,28 @@ const ADAPTIVE_THINKING_MAX_TOKENS_FLOOR = 64e3;
|
|
|
16809
17018
|
const THINKING_ANSWER_HEADROOM_TOKENS = 1e3;
|
|
16810
17019
|
/**
|
|
16811
17020
|
* Reasoning-inside-the-budget ids that none of the shape checks below can infer.
|
|
17021
|
+
*
|
|
16812
17022
|
* Bedrock's Kimi always reasons, but it is not Anthropic-adaptive, does not take
|
|
16813
17023
|
* `reasoning_effort`, and sends plain `max_tokens` - so it looks like an ordinary
|
|
16814
17024
|
* model at every seam we can inspect. Bedrock copies the monologue inline into
|
|
16815
17025
|
* `content` (see bedrockBackend/moonshot.ts) and caps output at 16K, so the floor
|
|
16816
17026
|
* resolves to that entire cap, which is the only value leaving room for an answer
|
|
16817
17027
|
* after a long trace.
|
|
16818
|
-
|
|
16819
|
-
|
|
17028
|
+
*
|
|
17029
|
+
* DeepSeek Flash misses every clause for its own set of reasons: no
|
|
17030
|
+
* `thinkingStyle` (that field is Anthropic's), absent from the OpenAI-only
|
|
17031
|
+
* REASONING_SUPPORTED_MODELS, and DEEPSEEK_PROFILE declares plain `max_tokens`
|
|
17032
|
+
* rather than `max_completion_tokens` because that is the parameter DeepSeek
|
|
17033
|
+
* takes. It reasons on every turn by default at effort 'high', spends those
|
|
17034
|
+
* tokens inside `max_tokens`, and a 4096 budget against a 393K cap is consumed
|
|
17035
|
+
* by the monologue alone: the turn comes back `finish_reason: 'length'` with no
|
|
17036
|
+
* content and deepseekBackend throws.
|
|
17037
|
+
*/
|
|
17038
|
+
const REASONS_WITHIN_OUTPUT_BUDGET_IDS = /* @__PURE__ */ new Set([
|
|
17039
|
+
ChatModels.KIMI_K2_THINKING_BEDROCK,
|
|
17040
|
+
ChatModels.KIMI_K2_5_BEDROCK,
|
|
17041
|
+
ChatModels.DEEPSEEK_FLASH
|
|
17042
|
+
]);
|
|
16820
17043
|
/**
|
|
16821
17044
|
* Whether the model spends reasoning tokens inside its output budget on every turn,
|
|
16822
17045
|
* which is what makes a small budget produce an empty visible reply rather than a
|
|
@@ -17332,7 +17555,7 @@ var AnthropicBackend = class {
|
|
|
17332
17555
|
supportsTools: true,
|
|
17333
17556
|
supportsImageVariation: false,
|
|
17334
17557
|
logoFile: "Anthropic_logo.png",
|
|
17335
|
-
rank:
|
|
17558
|
+
rank: 2,
|
|
17336
17559
|
trainingCutoff: "2024-10-01",
|
|
17337
17560
|
releaseDate: "2025-05-23",
|
|
17338
17561
|
deprecationDate: "2026-06-01",
|
|
@@ -17355,7 +17578,7 @@ var AnthropicBackend = class {
|
|
|
17355
17578
|
supportsTools: true,
|
|
17356
17579
|
supportsImageVariation: false,
|
|
17357
17580
|
logoFile: "Anthropic_logo.png",
|
|
17358
|
-
rank:
|
|
17581
|
+
rank: 2,
|
|
17359
17582
|
trainingCutoff: "2025-07-01",
|
|
17360
17583
|
releaseDate: "2025-09-30",
|
|
17361
17584
|
description: "Anthropic's most intelligent model in the Claude 4 family. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
|
|
@@ -17375,7 +17598,7 @@ var AnthropicBackend = class {
|
|
|
17375
17598
|
} },
|
|
17376
17599
|
supportsVision: true,
|
|
17377
17600
|
logoFile: "Anthropic_logo.png",
|
|
17378
|
-
rank:
|
|
17601
|
+
rank: 3,
|
|
17379
17602
|
supportsTools: true,
|
|
17380
17603
|
trainingCutoff: "2025-07-01",
|
|
17381
17604
|
releaseDate: "2025-10-16",
|
|
@@ -17422,7 +17645,7 @@ var AnthropicBackend = class {
|
|
|
17422
17645
|
supportsTools: true,
|
|
17423
17646
|
supportsImageVariation: false,
|
|
17424
17647
|
logoFile: "Anthropic_logo.png",
|
|
17425
|
-
rank:
|
|
17648
|
+
rank: 2,
|
|
17426
17649
|
trainingCutoff: "2025-10-01",
|
|
17427
17650
|
releaseDate: "2026-02-19",
|
|
17428
17651
|
description: "Anthropic's Claude 4.6 Sonnet model. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
|
|
@@ -17445,7 +17668,7 @@ var AnthropicBackend = class {
|
|
|
17445
17668
|
supportsTools: true,
|
|
17446
17669
|
supportsImageVariation: false,
|
|
17447
17670
|
logoFile: "Anthropic_logo.png",
|
|
17448
|
-
rank:
|
|
17671
|
+
rank: 1,
|
|
17449
17672
|
trainingCutoff: "2026-01-01",
|
|
17450
17673
|
releaseDate: "2026-07-01",
|
|
17451
17674
|
description: "Anthropic's newest Claude 5 Sonnet model. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
|
|
@@ -17538,7 +17761,7 @@ var AnthropicBackend = class {
|
|
|
17538
17761
|
} },
|
|
17539
17762
|
supportsVision: true,
|
|
17540
17763
|
logoFile: "Anthropic_logo.png",
|
|
17541
|
-
rank:
|
|
17764
|
+
rank: 0,
|
|
17542
17765
|
supportsTools: true,
|
|
17543
17766
|
trainingCutoff: "2026-01-01",
|
|
17544
17767
|
releaseDate: "2026-07-01",
|
|
@@ -17562,7 +17785,7 @@ var AnthropicBackend = class {
|
|
|
17562
17785
|
} },
|
|
17563
17786
|
supportsVision: true,
|
|
17564
17787
|
logoFile: "Anthropic_logo.png",
|
|
17565
|
-
rank:
|
|
17788
|
+
rank: 0,
|
|
17566
17789
|
supportsTools: true,
|
|
17567
17790
|
releaseDate: "2026-07-24",
|
|
17568
17791
|
description: "Anthropic's latest flagship model. Claude 5 Opus approaches Fable 5 performance at Opus 4.8 pricing, with adaptive extended thinking, coding, and agentic capabilities.",
|
|
@@ -19407,7 +19630,11 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19407
19630
|
if (NO_TEMPERATURE_MODELS.has(model)) return true;
|
|
19408
19631
|
return !this.getModelInfoList().some((m) => m.id === model) && this._dispatch.for(model)?.thinkingStyle === "adaptive";
|
|
19409
19632
|
}
|
|
19410
|
-
/**
|
|
19633
|
+
/**
|
|
19634
|
+
* Static model info list - synchronous access for getPayload, also used by getModelInfo.
|
|
19635
|
+
* `rank` must match the identically-named entry in anthropicBackend.ts: it is the same
|
|
19636
|
+
* model, so the picker must not show the Bedrock copy above or below its direct twin.
|
|
19637
|
+
*/
|
|
19411
19638
|
getModelInfoList() {
|
|
19412
19639
|
return [
|
|
19413
19640
|
{
|
|
@@ -19531,7 +19758,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19531
19758
|
} },
|
|
19532
19759
|
supportsVision: true,
|
|
19533
19760
|
logoFile: "Anthropic_logo.png",
|
|
19534
|
-
rank:
|
|
19761
|
+
rank: 1,
|
|
19535
19762
|
supportsTools: true,
|
|
19536
19763
|
trainingCutoff: "2025-05-01",
|
|
19537
19764
|
releaseDate: "2025-05-23",
|
|
@@ -19554,7 +19781,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19554
19781
|
} },
|
|
19555
19782
|
supportsVision: true,
|
|
19556
19783
|
logoFile: "Anthropic_logo.png",
|
|
19557
|
-
rank:
|
|
19784
|
+
rank: 1,
|
|
19558
19785
|
supportsTools: true,
|
|
19559
19786
|
trainingCutoff: "2025-08-01",
|
|
19560
19787
|
releaseDate: "2025-08-06",
|
|
@@ -19577,7 +19804,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19577
19804
|
} },
|
|
19578
19805
|
supportsVision: true,
|
|
19579
19806
|
logoFile: "Anthropic_logo.png",
|
|
19580
|
-
rank:
|
|
19807
|
+
rank: 2,
|
|
19581
19808
|
supportsTools: true,
|
|
19582
19809
|
trainingCutoff: "2025-05-01",
|
|
19583
19810
|
releaseDate: "2025-05-23",
|
|
@@ -19600,7 +19827,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19600
19827
|
supportsTools: true,
|
|
19601
19828
|
supportsImageVariation: false,
|
|
19602
19829
|
logoFile: "Anthropic_logo.png",
|
|
19603
|
-
rank:
|
|
19830
|
+
rank: 2,
|
|
19604
19831
|
trainingCutoff: "2025-07-01",
|
|
19605
19832
|
releaseDate: "2025-09-30",
|
|
19606
19833
|
description: "Anthropic's most intelligent model hosted in AWS Bedrock. Delivers exceptional performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency. Ideal for production workloads requiring both power and reliability."
|
|
@@ -19620,7 +19847,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19620
19847
|
} },
|
|
19621
19848
|
supportsVision: true,
|
|
19622
19849
|
logoFile: "Anthropic_logo.png",
|
|
19623
|
-
rank:
|
|
19850
|
+
rank: 3,
|
|
19624
19851
|
supportsTools: true,
|
|
19625
19852
|
trainingCutoff: "2025-07-01",
|
|
19626
19853
|
releaseDate: "2025-10-16",
|
|
@@ -19667,7 +19894,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19667
19894
|
supportsTools: true,
|
|
19668
19895
|
supportsImageVariation: false,
|
|
19669
19896
|
logoFile: "Anthropic_logo.png",
|
|
19670
|
-
rank:
|
|
19897
|
+
rank: 2,
|
|
19671
19898
|
trainingCutoff: "2025-10-01",
|
|
19672
19899
|
releaseDate: "2026-02-19",
|
|
19673
19900
|
description: "Anthropic's Claude 4.6 Sonnet model via AWS Bedrock. Delivers enhanced performance across coding, analysis, and complex reasoning tasks with improved speed and efficiency."
|
|
@@ -19690,7 +19917,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19690
19917
|
supportsTools: true,
|
|
19691
19918
|
supportsImageVariation: false,
|
|
19692
19919
|
logoFile: "Anthropic_logo.png",
|
|
19693
|
-
rank:
|
|
19920
|
+
rank: 1,
|
|
19694
19921
|
trainingCutoff: "2026-01-01",
|
|
19695
19922
|
releaseDate: "2026-07-01",
|
|
19696
19923
|
description: "Anthropic's newest Claude 5 Sonnet model via AWS Bedrock. Near-Opus quality on coding and agentic work at Sonnet cost, with adaptive extended thinking and a 1M-token context window."
|
|
@@ -19711,7 +19938,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19711
19938
|
} },
|
|
19712
19939
|
supportsVision: true,
|
|
19713
19940
|
logoFile: "Anthropic_logo.png",
|
|
19714
|
-
rank:
|
|
19941
|
+
rank: 1,
|
|
19715
19942
|
supportsTools: true,
|
|
19716
19943
|
trainingCutoff: "2025-05-01",
|
|
19717
19944
|
releaseDate: "2026-02-06",
|
|
@@ -19735,7 +19962,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19735
19962
|
} },
|
|
19736
19963
|
supportsVision: true,
|
|
19737
19964
|
logoFile: "Anthropic_logo.png",
|
|
19738
|
-
rank:
|
|
19965
|
+
rank: 1,
|
|
19739
19966
|
supportsTools: true,
|
|
19740
19967
|
trainingCutoff: "2025-10-01",
|
|
19741
19968
|
releaseDate: "2026-04-17",
|
|
@@ -19759,7 +19986,7 @@ var AnthropicBedrockBackend = class extends BaseBedrockBackend {
|
|
|
19759
19986
|
} },
|
|
19760
19987
|
supportsVision: true,
|
|
19761
19988
|
logoFile: "Anthropic_logo.png",
|
|
19762
|
-
rank:
|
|
19989
|
+
rank: 1,
|
|
19763
19990
|
supportsTools: true,
|
|
19764
19991
|
trainingCutoff: "2026-01-01",
|
|
19765
19992
|
releaseDate: "2026-05-28",
|
|
@@ -20415,7 +20642,7 @@ var JurassicTwoBedrockBackend = class extends BaseBedrockBackend {
|
|
|
20415
20642
|
} },
|
|
20416
20643
|
supportsVision: false,
|
|
20417
20644
|
logoFile: "AI21Labs.png",
|
|
20418
|
-
rank:
|
|
20645
|
+
rank: 51,
|
|
20419
20646
|
description: "AI21 Labs' balanced Jurassic-2 model offering good performance at moderate cost. Great for everyday tasks and general content generation."
|
|
20420
20647
|
}];
|
|
20421
20648
|
}
|
|
@@ -21511,7 +21738,7 @@ var GeminiBackend = class {
|
|
|
21511
21738
|
supportsVision: true,
|
|
21512
21739
|
supportsTools: true,
|
|
21513
21740
|
logoFile: "Google_logo.png",
|
|
21514
|
-
rank:
|
|
21741
|
+
rank: 6,
|
|
21515
21742
|
trainingCutoff: "2025-01-31",
|
|
21516
21743
|
releaseDate: "2025-11-30",
|
|
21517
21744
|
description: "Google's Gemini 3 Flash preview for fast, low-latency multimodal understanding, delivering richer visuals and deeper interactivity, built on a foundation of state-of-the-art reasoning."
|
|
@@ -22267,112 +22494,38 @@ var GeminiBackend = class {
|
|
|
22267
22494
|
}
|
|
22268
22495
|
};
|
|
22269
22496
|
/**
|
|
22270
|
-
*
|
|
22271
|
-
*
|
|
22272
|
-
*
|
|
22273
|
-
*
|
|
22274
|
-
* tokens and cache reads bill separately at their own (much cheaper) rate. Anthropic
|
|
22275
|
-
* and Claude-on-Bedrock deliver that natively. OpenAI and Moonshot do not - their
|
|
22276
|
-
* prompt total already CONTAINS the cached tokens - so those adapters must subtract
|
|
22277
|
-
* here before forwarding, or settlement double-bills the cached portion.
|
|
22278
|
-
*
|
|
22279
|
-
* Must stay in sync with the disjoint-fields assumption documented at the settlement
|
|
22280
|
-
* site in ChatCompletionProcess.
|
|
22281
|
-
*/
|
|
22282
|
-
/**
|
|
22283
|
-
* Split a cache-INCLUSIVE prompt total into the disjoint pair CompletionInfo carries.
|
|
22284
|
-
*
|
|
22285
|
-
* Forwarding the cached count without subtracting double-bills it; forwarding nothing
|
|
22286
|
-
* charges the full input rate on tokens the provider billed at a fraction of it.
|
|
22287
|
-
* Subtracting is the only split that bills what the provider actually charged.
|
|
22288
|
-
*
|
|
22289
|
-
* Clamped at zero: if a feed ever reports more cached than prompt tokens, a negative
|
|
22290
|
-
* input count would silently credit the user.
|
|
22291
|
-
*/
|
|
22292
|
-
function splitCacheInclusiveInput(totalPromptTokens, cacheReadTokens) {
|
|
22293
|
-
if (cacheReadTokens <= 0) return { inputTokens: totalPromptTokens };
|
|
22294
|
-
const cached = Math.min(cacheReadTokens, totalPromptTokens);
|
|
22295
|
-
return {
|
|
22296
|
-
inputTokens: Math.max(0, totalPromptTokens - cached),
|
|
22297
|
-
cacheReadInputTokens: cached
|
|
22298
|
-
};
|
|
22299
|
-
}
|
|
22300
|
-
/**
|
|
22301
|
-
* Cached prompt tokens from a raw provider usage object, across every spelling in use:
|
|
22302
|
-
* OpenAI Chat Completions nests them under `prompt_tokens_details`, the OpenAI
|
|
22303
|
-
* Responses API under `input_tokens_details`, and Moonshot publishes a flat
|
|
22304
|
-
* `cached_tokens` alongside the OpenAI-shaped nesting. Reading only one spelling
|
|
22305
|
-
* silently bills every cache hit on the other transports at the full input rate.
|
|
22306
|
-
*/
|
|
22307
|
-
function cachedTokensFromUsage(usage) {
|
|
22308
|
-
if (!usage) return 0;
|
|
22309
|
-
const candidates = [
|
|
22310
|
-
usage.cached_tokens,
|
|
22311
|
-
usage.prompt_tokens_details?.cached_tokens,
|
|
22312
|
-
usage.input_tokens_details?.cached_tokens
|
|
22313
|
-
];
|
|
22314
|
-
for (const value of candidates) if (typeof value === "number" && Number.isFinite(value) && value > 0) return value;
|
|
22315
|
-
return 0;
|
|
22316
|
-
}
|
|
22317
|
-
/**
|
|
22318
|
-
* Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
|
|
22319
|
-
* transport so every "which parameter does this id accept" rule is one pure
|
|
22320
|
-
* function with a test, rather than a conditional buried in a 400-line complete().
|
|
22497
|
+
* Request shaping for DeepSeek's direct API. Kept out of deepseekBackend's
|
|
22498
|
+
* transport for the same reason kimiParams is: every "which parameter does this
|
|
22499
|
+
* id accept" rule is one pure function with a test next to it, rather than a
|
|
22500
|
+
* conditional buried in a 400-line complete().
|
|
22321
22501
|
*
|
|
22322
|
-
*
|
|
22323
|
-
*
|
|
22324
|
-
*
|
|
22325
|
-
* @see https://
|
|
22502
|
+
* DeepSeek is OpenAI-compatible in envelope. What differs is thinking mode -
|
|
22503
|
+
* on by default, with its own toggle, its own effort vocabulary, and a sampling
|
|
22504
|
+
* group that is IGNORED rather than rejected while it is on.
|
|
22505
|
+
* @see https://api-docs.deepseek.com/guides/thinking_mode
|
|
22326
22506
|
*/
|
|
22327
|
-
/**
|
|
22328
|
-
const
|
|
22507
|
+
/** DeepSeek's effort vocabulary, which is not OpenAI's and not B4M's. */
|
|
22508
|
+
const DEEPSEEK_EFFORT_LEVELS = [
|
|
22329
22509
|
"low",
|
|
22330
22510
|
"high",
|
|
22331
22511
|
"max"
|
|
22332
22512
|
];
|
|
22333
22513
|
/**
|
|
22334
|
-
*
|
|
22335
|
-
*
|
|
22514
|
+
* Every DeepSeek id this build ships, direct-served. Bedrock-served DeepSeek is
|
|
22515
|
+
* not here. Both the reasoning and the sampling shaper gate on THIS set, so the
|
|
22516
|
+
* two cannot disagree about which ids the rules apply to; a test pins it against
|
|
22517
|
+
* the adapter table and against NO_TEMPERATURE_MODELS.
|
|
22336
22518
|
*/
|
|
22337
|
-
const
|
|
22338
|
-
/**
|
|
22339
|
-
const
|
|
22340
|
-
ChatModels.KIMI_K2_7_CODE,
|
|
22341
|
-
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
22342
|
-
ChatModels.KIMI_K2_6,
|
|
22343
|
-
ChatModels.KIMI_K2_5
|
|
22344
|
-
]);
|
|
22519
|
+
const DEEPSEEK_MODELS = /* @__PURE__ */ new Set([ChatModels.DEEPSEEK_FLASH]);
|
|
22520
|
+
/** DeepSeek raises anything below this rather than erroring, so we send what it will use. */
|
|
22521
|
+
const DEEPSEEK_THINKING_TOP_P_FLOOR = .95;
|
|
22345
22522
|
/**
|
|
22346
|
-
*
|
|
22347
|
-
*
|
|
22348
|
-
*
|
|
22523
|
+
* B4M's six-level effort onto DeepSeek's three. 'none' and 'minimal' map to
|
|
22524
|
+
* 'low' rather than to omission: omitting the parameter leaves DeepSeek's
|
|
22525
|
+
* documented default of 'high', so dropping it on a "least effort" request
|
|
22526
|
+
* would bill more reasoning than was asked for, not less.
|
|
22349
22527
|
*/
|
|
22350
|
-
|
|
22351
|
-
/**
|
|
22352
|
-
* Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
|
|
22353
|
-
* function form). Downgraded to 'auto' rather than dropped: a caller that asked
|
|
22354
|
-
* for a forced tool still wants tools offered.
|
|
22355
|
-
*/
|
|
22356
|
-
const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
|
|
22357
|
-
ChatModels.KIMI_K2_7_CODE,
|
|
22358
|
-
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
22359
|
-
ChatModels.KIMI_K2_6
|
|
22360
|
-
]);
|
|
22361
|
-
/** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
|
|
22362
|
-
const KIMI_MODELS = /* @__PURE__ */ new Set([
|
|
22363
|
-
ChatModels.KIMI_K3,
|
|
22364
|
-
ChatModels.KIMI_K2_7_CODE,
|
|
22365
|
-
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
22366
|
-
ChatModels.KIMI_K2_6,
|
|
22367
|
-
ChatModels.KIMI_K2_5
|
|
22368
|
-
]);
|
|
22369
|
-
/**
|
|
22370
|
-
* B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
|
|
22371
|
-
* rather than to omission because K3 cannot be asked not to think - claiming
|
|
22372
|
-
* otherwise by dropping the parameter would silently bill max-effort reasoning
|
|
22373
|
-
* (Moonshot's default is 'max').
|
|
22374
|
-
*/
|
|
22375
|
-
function toKimiEffort(effort) {
|
|
22528
|
+
function toDeepSeekEffort(effort) {
|
|
22376
22529
|
if (!effort) return void 0;
|
|
22377
22530
|
switch (effort) {
|
|
22378
22531
|
case "none":
|
|
@@ -22386,48 +22539,66 @@ function toKimiEffort(effort) {
|
|
|
22386
22539
|
}
|
|
22387
22540
|
/**
|
|
22388
22541
|
* The reasoning parameters for one model, or an empty object when it takes none.
|
|
22389
|
-
*
|
|
22390
|
-
*
|
|
22542
|
+
*
|
|
22543
|
+
* Both spellings are OpenAI-format and independent, unlike Kimi where they are
|
|
22544
|
+
* mutually exclusive: `thinking.type` turns reasoning on or off and
|
|
22545
|
+
* `reasoning_effort` sets its depth. DeepSeek's own example sends both in one
|
|
22546
|
+
* request. Omitting both leaves thinking enabled at effort 'high'.
|
|
22391
22547
|
*/
|
|
22392
|
-
function
|
|
22393
|
-
if (
|
|
22394
|
-
|
|
22395
|
-
|
|
22396
|
-
|
|
22397
|
-
|
|
22398
|
-
|
|
22399
|
-
|
|
22400
|
-
return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
|
|
22401
|
-
}
|
|
22402
|
-
return {};
|
|
22548
|
+
function deepseekReasoningParams(model, input = {}) {
|
|
22549
|
+
if (!DEEPSEEK_MODELS.has(model)) return {};
|
|
22550
|
+
const params = {};
|
|
22551
|
+
if (input.thinking?.enabled !== void 0) params.thinking = { type: input.thinking.enabled ? "enabled" : "disabled" };
|
|
22552
|
+
if (input.thinking?.enabled === false) return params;
|
|
22553
|
+
const effort = toDeepSeekEffort(input.reasoningEffort);
|
|
22554
|
+
if (effort) params.reasoning_effort = effort;
|
|
22555
|
+
return params;
|
|
22403
22556
|
}
|
|
22404
22557
|
/**
|
|
22405
|
-
*
|
|
22406
|
-
*
|
|
22407
|
-
*
|
|
22408
|
-
|
|
22558
|
+
* Whether the turn will reason, which is what the sampling restrictions below
|
|
22559
|
+
* actually hang on. DeepSeek's default is enabled, so only an explicit
|
|
22560
|
+
* `thinking.enabled === false` turns it off.
|
|
22561
|
+
*/
|
|
22562
|
+
function deepseekThinkingEnabled(input = {}) {
|
|
22563
|
+
return input.thinking?.enabled !== false;
|
|
22564
|
+
}
|
|
22565
|
+
/**
|
|
22566
|
+
* Sampling parameters for one turn.
|
|
22409
22567
|
*
|
|
22410
|
-
*
|
|
22411
|
-
*
|
|
22412
|
-
*
|
|
22413
|
-
*
|
|
22414
|
-
*
|
|
22568
|
+
* In thinking mode - the default - DeepSeek documents temperature,
|
|
22569
|
+
* presence_penalty and frequency_penalty as unsupported. They are accepted and
|
|
22570
|
+
* SILENTLY ignored rather than rejected, which is the worse failure of the two:
|
|
22571
|
+
* a 400 tells you the knob is dead, a no-op does not. They are dropped here so
|
|
22572
|
+
* nothing is sent that cannot take effect.
|
|
22573
|
+
*
|
|
22574
|
+
* Keyed on the TURN's resolved thinking state, not on model id: the restriction
|
|
22575
|
+
* is a property of thinking mode and the caller can turn thinking off, in which
|
|
22576
|
+
* case dropping temperature anyway would reproduce the same silent no-op from
|
|
22577
|
+
* our side of the wire.
|
|
22578
|
+
*
|
|
22579
|
+
* `top_p` does work in thinking mode with a lower bound of 0.95: a smaller value
|
|
22580
|
+
* is raised to it. Sent clamped rather than dropped, so the request states the
|
|
22581
|
+
* value the server will actually apply. The floor is a thinking-mode rule, so it
|
|
22582
|
+
* does not apply once thinking is off.
|
|
22583
|
+
*
|
|
22584
|
+
* `n` is not in DeepSeek's schema in either mode and is never sent.
|
|
22415
22585
|
*/
|
|
22416
|
-
function
|
|
22417
|
-
if (
|
|
22418
|
-
|
|
22419
|
-
|
|
22420
|
-
|
|
22421
|
-
|
|
22422
|
-
|
|
22423
|
-
|
|
22424
|
-
|
|
22586
|
+
function deepseekSamplingParams(model, input, reasoning = {}) {
|
|
22587
|
+
if (!DEEPSEEK_MODELS.has(model) || !deepseekThinkingEnabled(reasoning)) {
|
|
22588
|
+
const passthrough = {};
|
|
22589
|
+
if (input.temperature !== void 0) passthrough.temperature = input.temperature;
|
|
22590
|
+
if (input.topP !== void 0) passthrough.top_p = input.topP;
|
|
22591
|
+
if (input.presencePenalty !== void 0) passthrough.presence_penalty = input.presencePenalty;
|
|
22592
|
+
if (input.frequencyPenalty !== void 0) passthrough.frequency_penalty = input.frequencyPenalty;
|
|
22593
|
+
return passthrough;
|
|
22594
|
+
}
|
|
22595
|
+
if (input.topP === void 0) return {};
|
|
22596
|
+
return { top_p: Math.max(input.topP, DEEPSEEK_THINKING_TOP_P_FLOOR) };
|
|
22425
22597
|
}
|
|
22426
|
-
/** `
|
|
22427
|
-
function
|
|
22428
|
-
if (
|
|
22429
|
-
|
|
22430
|
-
return choice;
|
|
22598
|
+
/** `stop`, truncated to the 16 sequences DeepSeek accepts. */
|
|
22599
|
+
function deepseekStopSequences(stop) {
|
|
22600
|
+
if (!Array.isArray(stop)) return stop;
|
|
22601
|
+
return stop.length > 16 ? stop.slice(0, 16) : stop;
|
|
22431
22602
|
}
|
|
22432
22603
|
/** Type guard: does this message already carry OpenAI-style `tool_calls`? */
|
|
22433
22604
|
function hasToolCalls(msg) {
|
|
@@ -22450,12 +22621,16 @@ function isTextBlock(block) {
|
|
|
22450
22621
|
* Messages already in OpenAI format (with `tool_calls` property) pass through unchanged.
|
|
22451
22622
|
* Messages without tool_use/tool_result content blocks pass through unchanged.
|
|
22452
22623
|
*/
|
|
22453
|
-
function convertMessageToOpenAIFormat(msg) {
|
|
22454
|
-
if (hasToolCalls(msg))
|
|
22455
|
-
|
|
22456
|
-
|
|
22457
|
-
|
|
22458
|
-
|
|
22624
|
+
function convertMessageToOpenAIFormat(msg, options = {}) {
|
|
22625
|
+
if (hasToolCalls(msg)) {
|
|
22626
|
+
const reasoningContent = msg.reasoning_content;
|
|
22627
|
+
return [{
|
|
22628
|
+
role: "assistant",
|
|
22629
|
+
content: null,
|
|
22630
|
+
tool_calls: msg.tool_calls,
|
|
22631
|
+
...options.preserveReasoningContent && typeof reasoningContent === "string" ? { reasoning_content: reasoningContent } : {}
|
|
22632
|
+
}];
|
|
22633
|
+
}
|
|
22459
22634
|
if (msg.role === "assistant" && Array.isArray(msg.content)) {
|
|
22460
22635
|
const contentBlocks = msg.content;
|
|
22461
22636
|
const toolUseBlocks = contentBlocks.filter(isToolUseBlock);
|
|
@@ -22493,8 +22668,614 @@ function convertMessageToOpenAIFormat(msg) {
|
|
|
22493
22668
|
* Convert an array of IMessages from B4M standard format to OpenAI-compatible format.
|
|
22494
22669
|
* Returns OpenAIFormattedMessage[] - callers targeting OpenAI SDK types should cast at the boundary.
|
|
22495
22670
|
*/
|
|
22496
|
-
function convertMessagesToOpenAIFormat(messages) {
|
|
22497
|
-
return messages.flatMap(convertMessageToOpenAIFormat);
|
|
22671
|
+
function convertMessagesToOpenAIFormat(messages, options = {}) {
|
|
22672
|
+
return messages.flatMap((msg) => convertMessageToOpenAIFormat(msg, options));
|
|
22673
|
+
}
|
|
22674
|
+
/**
|
|
22675
|
+
* DeepSeek's models, served from their own OpenAI-compatible endpoint.
|
|
22676
|
+
*
|
|
22677
|
+
* Structurally this is kimiBackend's twin - same OpenAI SDK against a different
|
|
22678
|
+
* baseURL, same recursive tool loop, same multi-turn token accumulators - and the
|
|
22679
|
+
* three OpenAI-compatible backends must stay in sync on that machinery. What
|
|
22680
|
+
* genuinely differs here:
|
|
22681
|
+
*
|
|
22682
|
+
* 1. The base URL carries NO `/v1` segment; the SDK appends the path itself.
|
|
22683
|
+
* 2. Thinking is on by default and its sampling restrictions are SILENT no-ops
|
|
22684
|
+
* rather than 400s; see deepseekParams.
|
|
22685
|
+
* 3. The prior turn's `reasoning_content` has to be replayed on the assistant
|
|
22686
|
+
* tool-call message whenever the request carries `tools`, which is the
|
|
22687
|
+
* opposite of the usual provider rule. See pushToolMessages.
|
|
22688
|
+
*
|
|
22689
|
+
* @see https://api-docs.deepseek.com/api/create-chat-completion
|
|
22690
|
+
*/
|
|
22691
|
+
var DeepSeekBackend = class {
|
|
22692
|
+
_baseUrl = "https://api.deepseek.com";
|
|
22693
|
+
_api;
|
|
22694
|
+
logger;
|
|
22695
|
+
currentModel = "";
|
|
22696
|
+
constructor(apiKey, logger) {
|
|
22697
|
+
if (!apiKey) throw new Error("DeepSeek API key is required");
|
|
22698
|
+
this._api = new OpenAI({
|
|
22699
|
+
apiKey,
|
|
22700
|
+
baseURL: this._baseUrl
|
|
22701
|
+
});
|
|
22702
|
+
this.logger = logger ?? new Logger();
|
|
22703
|
+
}
|
|
22704
|
+
/**
|
|
22705
|
+
* Seed listing. Post-registry this is the fallback tier, not the source of
|
|
22706
|
+
* truth: the catalog overlays context window, limits, lifecycle and price on
|
|
22707
|
+
* top of these rows. DeepSeek's own GET /models returns id/object/owned_by and
|
|
22708
|
+
* nothing else, so everything below has to live here.
|
|
22709
|
+
*
|
|
22710
|
+
* Prices are the PEAK rates. Off-peak (outside 01:00-04:00 and 06:00-10:00 UTC,
|
|
22711
|
+
* Mon-Fri) is exactly half, and ModelInfo.pricing is keyed by context tier with
|
|
22712
|
+
* no time dimension to express that in - so the rate that never under-bills is
|
|
22713
|
+
* the one recorded.
|
|
22714
|
+
*/
|
|
22715
|
+
async getModelInfo() {
|
|
22716
|
+
return [{
|
|
22717
|
+
id: ChatModels.DEEPSEEK_FLASH,
|
|
22718
|
+
type: "text",
|
|
22719
|
+
name: "DeepSeek Flash",
|
|
22720
|
+
backend: ModelBackend.DeepSeek,
|
|
22721
|
+
contextWindow: 1e6,
|
|
22722
|
+
max_tokens: 393216,
|
|
22723
|
+
can_stream: true,
|
|
22724
|
+
pricing: { 1e6: {
|
|
22725
|
+
input: .3 / 1e6,
|
|
22726
|
+
output: 1.2 / 1e6,
|
|
22727
|
+
cache_read: .006 / 1e6
|
|
22728
|
+
} },
|
|
22729
|
+
can_think: true,
|
|
22730
|
+
supportsVision: true,
|
|
22731
|
+
supportsTools: true,
|
|
22732
|
+
supportsImageVariation: false,
|
|
22733
|
+
releaseDate: "2026-08-13",
|
|
22734
|
+
description: "DeepSeek's V4.1-Flash. 1M context with native vision, tool use, and selectable reasoning effort (low/high/max). Always reasons unless thinking is turned off."
|
|
22735
|
+
}];
|
|
22736
|
+
}
|
|
22737
|
+
async complete(model, messages, options, callback, toolsUsed = []) {
|
|
22738
|
+
this.currentModel = model;
|
|
22739
|
+
const toolCallCount = options._internal?.toolCallCount ?? 0;
|
|
22740
|
+
const accumInputTokens = options._internal?.accumInputTokens ?? 0;
|
|
22741
|
+
const accumOutputTokens = options._internal?.accumOutputTokens ?? 0;
|
|
22742
|
+
const accumCacheReadTokens = options._internal?.accumCacheReadTokens ?? 0;
|
|
22743
|
+
const maxToolCalls = options._internal?.maxToolCalls ?? 10;
|
|
22744
|
+
if (toolCallCount >= maxToolCalls && options.tools?.length) {
|
|
22745
|
+
this.logger.warn(`Max tool calls limit (${maxToolCalls}) reached. Disabling tools to prevent infinite loops.`);
|
|
22746
|
+
await this.complete(model, stripToolDependentMessages(messages), {
|
|
22747
|
+
...options,
|
|
22748
|
+
tools: void 0,
|
|
22749
|
+
_internal: options._internal
|
|
22750
|
+
}, callback, toolsUsed);
|
|
22751
|
+
return;
|
|
22752
|
+
}
|
|
22753
|
+
const rawTools = options.tools;
|
|
22754
|
+
options.tools = Array.isArray(rawTools) ? rawTools : rawTools ? [rawTools] : void 0;
|
|
22755
|
+
if ((options.n ?? 1) > 1) this.logger.warn(`DeepSeek has no 'n' parameter; ignoring the request for ${options.n} choices.`);
|
|
22756
|
+
const useStreaming = Boolean(options.stream);
|
|
22757
|
+
const reasoning = {
|
|
22758
|
+
thinking: options.thinking,
|
|
22759
|
+
reasoningEffort: options.reasoningEffort
|
|
22760
|
+
};
|
|
22761
|
+
const messagesWithFormat = injectJsonSchemaInstruction(messages, options.responseFormat);
|
|
22762
|
+
const bestEffortFormat = isBestEffortJsonSchema(options.responseFormat);
|
|
22763
|
+
const parameters = {
|
|
22764
|
+
model,
|
|
22765
|
+
messages: this.formatMessages(messagesWithFormat)
|
|
22766
|
+
};
|
|
22767
|
+
Object.assign(parameters, {
|
|
22768
|
+
...deepseekSamplingParams(model, {
|
|
22769
|
+
temperature: options.temperature,
|
|
22770
|
+
topP: options.topP,
|
|
22771
|
+
presencePenalty: options.presencePenalty,
|
|
22772
|
+
frequencyPenalty: options.frequencyPenalty
|
|
22773
|
+
}, reasoning),
|
|
22774
|
+
...deepseekReasoningParams(model, reasoning),
|
|
22775
|
+
stop: deepseekStopSequences(options.stop),
|
|
22776
|
+
stream: useStreaming,
|
|
22777
|
+
max_tokens: options.maxTokens,
|
|
22778
|
+
...useStreaming && { stream_options: { include_usage: true } }
|
|
22779
|
+
});
|
|
22780
|
+
if (options.tools?.length) {
|
|
22781
|
+
parameters.tools = this.formatTools(options.tools);
|
|
22782
|
+
if (options.tool_choice !== void 0) parameters.tool_choice = options.tool_choice;
|
|
22783
|
+
}
|
|
22784
|
+
if (options.responseFormat?.type === "json_schema") parameters.response_format = { type: "json_object" };
|
|
22785
|
+
else if (options.responseFormat?.type === "text") parameters.response_format = { type: "text" };
|
|
22786
|
+
const cacheStrategy = options.cacheStrategy;
|
|
22787
|
+
const response = await this._api.chat.completions.create(parameters, { signal: options.abortSignal });
|
|
22788
|
+
let inputTokens = 0;
|
|
22789
|
+
let outputTokens = 0;
|
|
22790
|
+
if (!(response instanceof Stream)) {
|
|
22791
|
+
const streamedText = [];
|
|
22792
|
+
if (!response.choices || response.choices.length === 0) throw new Error("No choices returned from the DeepSeek API");
|
|
22793
|
+
const turnCacheReadTokens = cachedTokensFromUsage(response.usage);
|
|
22794
|
+
for (const c of response.choices) {
|
|
22795
|
+
if (!c.message) continue;
|
|
22796
|
+
const reasoningContent = c.message.reasoning_content;
|
|
22797
|
+
if (c.message.tool_calls && c.message.tool_calls.length > 0) {
|
|
22798
|
+
for (const toolCall of c.message.tool_calls) {
|
|
22799
|
+
if (toolCall.type !== "function") continue;
|
|
22800
|
+
if (toolCall.function.arguments) toolsUsed.push({
|
|
22801
|
+
name: toolCall.function.name,
|
|
22802
|
+
arguments: toolCall.function.arguments,
|
|
22803
|
+
id: toolCall.id
|
|
22804
|
+
});
|
|
22805
|
+
}
|
|
22806
|
+
if (options.executeTools !== false) {
|
|
22807
|
+
const resolvedTools = [];
|
|
22808
|
+
for (const toolCall of c.message.tool_calls) {
|
|
22809
|
+
if (toolCall.type !== "function" || !toolCall.function.arguments) continue;
|
|
22810
|
+
const toolFn = options.tools?.find((t) => t.toolSchema.name === toolCall.function.name)?.toolFn;
|
|
22811
|
+
if (!toolFn) continue;
|
|
22812
|
+
try {
|
|
22813
|
+
const parsedParams = JSON.parse(toolCall.function.arguments);
|
|
22814
|
+
resolvedTools.push({
|
|
22815
|
+
id: toolCall.id,
|
|
22816
|
+
name: toolCall.function.name,
|
|
22817
|
+
parameters: toolCall.function.arguments,
|
|
22818
|
+
parsedParams,
|
|
22819
|
+
toolFn
|
|
22820
|
+
});
|
|
22821
|
+
} catch {
|
|
22822
|
+
this.logger.warn(`JSON parse error for ${toolCall.function.name} arguments`);
|
|
22823
|
+
const entry = toolsUsed.find((t) => t.name === toolCall.function.name && t.id === toolCall.id);
|
|
22824
|
+
if (entry) entry.arguments = "{}";
|
|
22825
|
+
recordToolResult(toolsUsed, {
|
|
22826
|
+
id: toolCall.id,
|
|
22827
|
+
name: toolCall.function.name
|
|
22828
|
+
}, "Error: Tool arguments were malformed and could not be parsed.", false);
|
|
22829
|
+
}
|
|
22830
|
+
}
|
|
22831
|
+
const parallelEnabled = options.parallelToolExecution !== false;
|
|
22832
|
+
this.logger.debug("[Tool Execution] Executing tools (DeepSeek non-streaming)", {
|
|
22833
|
+
mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
|
|
22834
|
+
toolNames: resolvedTools.map((t) => t.name)
|
|
22835
|
+
});
|
|
22836
|
+
const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
|
|
22837
|
+
return {
|
|
22838
|
+
id,
|
|
22839
|
+
name,
|
|
22840
|
+
parameters: toolParams,
|
|
22841
|
+
result: await toolFn(parsedParams)
|
|
22842
|
+
};
|
|
22843
|
+
}), {
|
|
22844
|
+
parallel: parallelEnabled,
|
|
22845
|
+
maxConcurrency: options.maxParallelTools
|
|
22846
|
+
})).map((outcome, i) => outcome.ok ? {
|
|
22847
|
+
ok: true,
|
|
22848
|
+
...outcome.result
|
|
22849
|
+
} : {
|
|
22850
|
+
ok: false,
|
|
22851
|
+
id: resolvedTools[i].id,
|
|
22852
|
+
name: resolvedTools[i].name,
|
|
22853
|
+
parameters: resolvedTools[i].parameters,
|
|
22854
|
+
error: outcome.error
|
|
22855
|
+
});
|
|
22856
|
+
let turnReasoning = reasoningContent;
|
|
22857
|
+
for (const outcome of outcomes) {
|
|
22858
|
+
if (outcome.ok) {
|
|
22859
|
+
const resultStr = outcome.result.toString();
|
|
22860
|
+
recordToolResult(toolsUsed, {
|
|
22861
|
+
id: outcome.id,
|
|
22862
|
+
name: outcome.name
|
|
22863
|
+
}, resultStr, true);
|
|
22864
|
+
this.pushToolMessages(messages, {
|
|
22865
|
+
id: outcome.id,
|
|
22866
|
+
name: outcome.name,
|
|
22867
|
+
parameters: outcome.parameters
|
|
22868
|
+
}, resultStr, turnReasoning ? [turnReasoning] : void 0);
|
|
22869
|
+
} else {
|
|
22870
|
+
if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
|
|
22871
|
+
const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
|
|
22872
|
+
const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
|
|
22873
|
+
recordToolResult(toolsUsed, {
|
|
22874
|
+
id: outcome.id,
|
|
22875
|
+
name: outcome.name
|
|
22876
|
+
}, observation, false);
|
|
22877
|
+
this.pushToolMessages(messages, {
|
|
22878
|
+
id: outcome.id,
|
|
22879
|
+
name: outcome.name,
|
|
22880
|
+
parameters: outcome.parameters
|
|
22881
|
+
}, observation, turnReasoning ? [turnReasoning] : void 0);
|
|
22882
|
+
}
|
|
22883
|
+
turnReasoning = void 0;
|
|
22884
|
+
}
|
|
22885
|
+
await this.complete(model, messages, {
|
|
22886
|
+
...options,
|
|
22887
|
+
_internal: {
|
|
22888
|
+
...options._internal,
|
|
22889
|
+
toolCallCount: toolCallCount + 1,
|
|
22890
|
+
accumInputTokens: accumInputTokens + (response.usage?.prompt_tokens || 0),
|
|
22891
|
+
accumOutputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
|
|
22892
|
+
accumCacheReadTokens: accumCacheReadTokens + turnCacheReadTokens
|
|
22893
|
+
}
|
|
22894
|
+
}, callback, toolsUsed);
|
|
22895
|
+
return;
|
|
22896
|
+
} else {
|
|
22897
|
+
this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
|
|
22898
|
+
await callback([null], {
|
|
22899
|
+
...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), accumCacheReadTokens + turnCacheReadTokens),
|
|
22900
|
+
outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
|
|
22901
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
|
|
22902
|
+
});
|
|
22903
|
+
return;
|
|
22904
|
+
}
|
|
22905
|
+
} else {
|
|
22906
|
+
const content = c.message.content || "";
|
|
22907
|
+
streamedText[c.index] = reasoningContent ? `<think>${reasoningContent}</think>${content}` : content;
|
|
22908
|
+
}
|
|
22909
|
+
}
|
|
22910
|
+
if (streamedText.every((text) => !text) && toolsUsed.length === 0) {
|
|
22911
|
+
const finish = response.choices[0]?.finish_reason;
|
|
22912
|
+
throw new Error(finish === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${finish ?? "unknown"}).`);
|
|
22913
|
+
}
|
|
22914
|
+
let cacheStats;
|
|
22915
|
+
if (cacheStrategy?.enableCaching && response.usage) {
|
|
22916
|
+
cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats(response, model);
|
|
22917
|
+
if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: false });
|
|
22918
|
+
}
|
|
22919
|
+
const finishReason = normalizeOpenAIFinishReason(response.choices[0]?.finish_reason);
|
|
22920
|
+
const totalCacheReadTokens = accumCacheReadTokens + turnCacheReadTokens;
|
|
22921
|
+
await callback(streamedText, {
|
|
22922
|
+
...splitCacheInclusiveInput(accumInputTokens + (response.usage?.prompt_tokens || 0), totalCacheReadTokens),
|
|
22923
|
+
outputTokens: accumOutputTokens + (response.usage?.completion_tokens || 0),
|
|
22924
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
|
|
22925
|
+
cacheStats,
|
|
22926
|
+
...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
|
|
22927
|
+
...finishReason ? { stopReason: finishReason } : {}
|
|
22928
|
+
});
|
|
22929
|
+
return;
|
|
22930
|
+
}
|
|
22931
|
+
const func = [];
|
|
22932
|
+
let isInThinkingBlock = false;
|
|
22933
|
+
let streamedReasoning = "";
|
|
22934
|
+
let cachedTokensFromStream = 0;
|
|
22935
|
+
let streamFinishReason;
|
|
22936
|
+
let sawAnyText = false;
|
|
22937
|
+
for await (const chunk of response) {
|
|
22938
|
+
const streamedText = [];
|
|
22939
|
+
if (chunk.usage) {
|
|
22940
|
+
inputTokens = Math.max(inputTokens, chunk.usage?.prompt_tokens || 0);
|
|
22941
|
+
outputTokens += chunk.usage?.completion_tokens || 0;
|
|
22942
|
+
const chunkCached = cachedTokensFromUsage(chunk.usage);
|
|
22943
|
+
if (chunkCached > 0) cachedTokensFromStream = chunkCached;
|
|
22944
|
+
}
|
|
22945
|
+
chunk?.choices.forEach((c) => {
|
|
22946
|
+
if (c.finish_reason) streamFinishReason = c.finish_reason;
|
|
22947
|
+
const deltaReasoning = c.delta.reasoning_content;
|
|
22948
|
+
if (deltaReasoning) {
|
|
22949
|
+
streamedReasoning += deltaReasoning;
|
|
22950
|
+
if (!isInThinkingBlock) {
|
|
22951
|
+
isInThinkingBlock = true;
|
|
22952
|
+
streamedText[c.index] = "<think>" + deltaReasoning;
|
|
22953
|
+
} else streamedText[c.index] = deltaReasoning;
|
|
22954
|
+
if (!c.delta.content) return;
|
|
22955
|
+
}
|
|
22956
|
+
if (isInThinkingBlock && c.delta.content) {
|
|
22957
|
+
isInThinkingBlock = false;
|
|
22958
|
+
streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
|
|
22959
|
+
return;
|
|
22960
|
+
}
|
|
22961
|
+
c.delta.tool_calls?.map((tool) => {
|
|
22962
|
+
func[tool.index] ||= {};
|
|
22963
|
+
func[tool.index].name ||= tool.function?.name;
|
|
22964
|
+
func[tool.index].id ||= tool.id;
|
|
22965
|
+
func[tool.index].parameters ??= "";
|
|
22966
|
+
func[tool.index].parameters += tool.function?.arguments || "";
|
|
22967
|
+
});
|
|
22968
|
+
if (func.length > 0) return;
|
|
22969
|
+
streamedText[c.index] = c.delta.content || "";
|
|
22970
|
+
});
|
|
22971
|
+
if (streamedText.some((t) => t)) sawAnyText = true;
|
|
22972
|
+
const normalizedFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
|
|
22973
|
+
await callback(streamedText, {
|
|
22974
|
+
...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
|
|
22975
|
+
outputTokens: accumOutputTokens + outputTokens,
|
|
22976
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
|
|
22977
|
+
...normalizedFinishReason ? { stopReason: normalizedFinishReason } : {}
|
|
22978
|
+
});
|
|
22979
|
+
}
|
|
22980
|
+
if (isInThinkingBlock) {
|
|
22981
|
+
await callback(["</think>"], {
|
|
22982
|
+
...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
|
|
22983
|
+
outputTokens: accumOutputTokens + outputTokens,
|
|
22984
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0
|
|
22985
|
+
});
|
|
22986
|
+
isInThinkingBlock = false;
|
|
22987
|
+
}
|
|
22988
|
+
if (!sawAnyText && func.length === 0 && toolsUsed.length === 0) throw new Error(streamFinishReason === "length" ? `DeepSeek returned no content for ${model}: the output budget was exhausted before any answer was produced (finish_reason: length). Raise maxTokens or lower the reasoning effort.` : `DeepSeek returned no content for ${model} (finish_reason: ${streamFinishReason ?? "unknown"}).`);
|
|
22989
|
+
let cacheStats;
|
|
22990
|
+
if (cacheStrategy?.enableCaching && inputTokens > 0) {
|
|
22991
|
+
cacheStats = getCachingAdapter(ModelBackend.DeepSeek).extractCacheStats({ usage: {
|
|
22992
|
+
prompt_tokens: inputTokens,
|
|
22993
|
+
completion_tokens: outputTokens,
|
|
22994
|
+
prompt_cache_hit_tokens: cachedTokensFromStream
|
|
22995
|
+
} }, model);
|
|
22996
|
+
if (cacheStats) logCacheStats(this.logger, cacheStats, { streaming: true });
|
|
22997
|
+
}
|
|
22998
|
+
if ((cacheStats || bestEffortFormat) && func.length === 0) {
|
|
22999
|
+
const terminalFinishReason = normalizeOpenAIFinishReason(streamFinishReason);
|
|
23000
|
+
await callback([""], {
|
|
23001
|
+
...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
|
|
23002
|
+
outputTokens: accumOutputTokens + outputTokens,
|
|
23003
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
|
|
23004
|
+
...cacheStats ? { cacheStats } : {},
|
|
23005
|
+
...bestEffortFormat ? { responseFormatMode: "best-effort" } : {},
|
|
23006
|
+
...terminalFinishReason ? { stopReason: terminalFinishReason } : {}
|
|
23007
|
+
});
|
|
23008
|
+
}
|
|
23009
|
+
if (func.length > 0) {
|
|
23010
|
+
for await (const tool of func) {
|
|
23011
|
+
const { name, parameters: toolParams, id } = tool;
|
|
23012
|
+
if (name) toolsUsed.push({
|
|
23013
|
+
name,
|
|
23014
|
+
arguments: toolParams || "{}",
|
|
23015
|
+
id
|
|
23016
|
+
});
|
|
23017
|
+
}
|
|
23018
|
+
if (options.executeTools !== false) {
|
|
23019
|
+
const resolvedTools = [];
|
|
23020
|
+
for (const tool of func) {
|
|
23021
|
+
const { id, name } = tool;
|
|
23022
|
+
if (!id || !name) continue;
|
|
23023
|
+
const toolParams = tool.parameters || "{}";
|
|
23024
|
+
const toolFn = options.tools?.find((t) => t.toolSchema.name === name)?.toolFn;
|
|
23025
|
+
if (!toolFn) continue;
|
|
23026
|
+
try {
|
|
23027
|
+
const parsedParams = JSON.parse(toolParams);
|
|
23028
|
+
resolvedTools.push({
|
|
23029
|
+
id,
|
|
23030
|
+
name,
|
|
23031
|
+
parameters: toolParams,
|
|
23032
|
+
parsedParams,
|
|
23033
|
+
toolFn
|
|
23034
|
+
});
|
|
23035
|
+
} catch {
|
|
23036
|
+
this.logger.warn(`JSON parse error for ${name} arguments (streaming)`);
|
|
23037
|
+
const entry = toolsUsed.find((t) => t.name === name && t.id === id);
|
|
23038
|
+
if (entry) entry.arguments = "{}";
|
|
23039
|
+
recordToolResult(toolsUsed, {
|
|
23040
|
+
id,
|
|
23041
|
+
name
|
|
23042
|
+
}, "Error: Tool arguments were malformed and could not be parsed.", false);
|
|
23043
|
+
}
|
|
23044
|
+
}
|
|
23045
|
+
const parallelEnabled = options.parallelToolExecution !== false;
|
|
23046
|
+
this.logger.debug("[Tool Execution] Executing tools (DeepSeek streaming)", {
|
|
23047
|
+
mode: parallelEnabled && resolvedTools.length > 1 ? "parallel" : "sequential",
|
|
23048
|
+
toolNames: resolvedTools.map((t) => t.name)
|
|
23049
|
+
});
|
|
23050
|
+
const outcomes = (await executeToolsBatch(resolvedTools.map(({ id, name, parameters: toolParams, parsedParams, toolFn }) => async () => {
|
|
23051
|
+
return {
|
|
23052
|
+
id,
|
|
23053
|
+
name,
|
|
23054
|
+
parameters: toolParams,
|
|
23055
|
+
result: await toolFn(parsedParams)
|
|
23056
|
+
};
|
|
23057
|
+
}), {
|
|
23058
|
+
parallel: parallelEnabled,
|
|
23059
|
+
maxConcurrency: options.maxParallelTools
|
|
23060
|
+
})).map((outcome, i) => outcome.ok ? {
|
|
23061
|
+
ok: true,
|
|
23062
|
+
...outcome.result
|
|
23063
|
+
} : {
|
|
23064
|
+
ok: false,
|
|
23065
|
+
id: resolvedTools[i].id,
|
|
23066
|
+
name: resolvedTools[i].name,
|
|
23067
|
+
parameters: resolvedTools[i].parameters,
|
|
23068
|
+
error: outcome.error
|
|
23069
|
+
});
|
|
23070
|
+
let turnReasoning = streamedReasoning || void 0;
|
|
23071
|
+
for (const outcome of outcomes) {
|
|
23072
|
+
if (outcome.ok) {
|
|
23073
|
+
const resultStr = outcome.result.toString();
|
|
23074
|
+
recordToolResult(toolsUsed, {
|
|
23075
|
+
id: outcome.id,
|
|
23076
|
+
name: outcome.name
|
|
23077
|
+
}, resultStr, true);
|
|
23078
|
+
this.pushToolMessages(messages, {
|
|
23079
|
+
id: outcome.id,
|
|
23080
|
+
name: outcome.name,
|
|
23081
|
+
parameters: outcome.parameters
|
|
23082
|
+
}, resultStr, turnReasoning ? [turnReasoning] : void 0);
|
|
23083
|
+
} else {
|
|
23084
|
+
if (outcome.error instanceof PermissionDeniedError) throw outcome.error;
|
|
23085
|
+
const errorMessage = outcome.error instanceof Error ? outcome.error.message : "Unknown error";
|
|
23086
|
+
const observation = `Error processing ${outcome.name} tool: ${errorMessage}`;
|
|
23087
|
+
recordToolResult(toolsUsed, {
|
|
23088
|
+
id: outcome.id,
|
|
23089
|
+
name: outcome.name
|
|
23090
|
+
}, observation, false);
|
|
23091
|
+
this.pushToolMessages(messages, {
|
|
23092
|
+
id: outcome.id,
|
|
23093
|
+
name: outcome.name,
|
|
23094
|
+
parameters: outcome.parameters
|
|
23095
|
+
}, observation, turnReasoning ? [turnReasoning] : void 0);
|
|
23096
|
+
}
|
|
23097
|
+
turnReasoning = void 0;
|
|
23098
|
+
}
|
|
23099
|
+
await this.complete(model, messages, {
|
|
23100
|
+
...options,
|
|
23101
|
+
_internal: {
|
|
23102
|
+
...options._internal,
|
|
23103
|
+
toolCallCount: toolCallCount + 1,
|
|
23104
|
+
accumInputTokens: accumInputTokens + inputTokens,
|
|
23105
|
+
accumOutputTokens: accumOutputTokens + outputTokens,
|
|
23106
|
+
accumCacheReadTokens: accumCacheReadTokens + cachedTokensFromStream
|
|
23107
|
+
}
|
|
23108
|
+
}, callback, toolsUsed);
|
|
23109
|
+
} else {
|
|
23110
|
+
this.logger.debug(`[Tool Execution] executeTools=false, passing tool calls to callback`);
|
|
23111
|
+
await callback([null], {
|
|
23112
|
+
...splitCacheInclusiveInput(accumInputTokens + inputTokens, accumCacheReadTokens + cachedTokensFromStream),
|
|
23113
|
+
outputTokens: accumOutputTokens + outputTokens,
|
|
23114
|
+
toolsUsed: toolsUsed.length > 0 ? toolsUsed : void 0,
|
|
23115
|
+
...cacheStats ? { cacheStats } : {}
|
|
23116
|
+
});
|
|
23117
|
+
}
|
|
23118
|
+
}
|
|
23119
|
+
}
|
|
23120
|
+
formatMessages(messages) {
|
|
23121
|
+
return convertMessagesToOpenAIFormat(messages, { preserveReasoningContent: true });
|
|
23122
|
+
}
|
|
23123
|
+
formatTools(tools = []) {
|
|
23124
|
+
return tools.map((tool) => ({
|
|
23125
|
+
type: "function",
|
|
23126
|
+
function: tool.toolSchema
|
|
23127
|
+
}));
|
|
23128
|
+
}
|
|
23129
|
+
/**
|
|
23130
|
+
* `thinkingBlocks` carries the turn's `reasoning_content` as a single string
|
|
23131
|
+
* entry. DeepSeek inverts the usual rule: when a request carries `tools`, the
|
|
23132
|
+
* prior turn's monologue MUST be replayed on the assistant tool-call message or
|
|
23133
|
+
* reasoning continuity breaks across the loop. formatMessages opts into the
|
|
23134
|
+
* converter's `preserveReasoningContent` for exactly this path; every other
|
|
23135
|
+
* target strips it, because this array is shared with the fallback hop.
|
|
23136
|
+
*/
|
|
23137
|
+
pushToolMessages(messages, tool, result, thinkingBlocks) {
|
|
23138
|
+
const reasoningContent = typeof thinkingBlocks?.[0] === "string" ? thinkingBlocks[0] : void 0;
|
|
23139
|
+
messages.push({
|
|
23140
|
+
content: null,
|
|
23141
|
+
role: "assistant",
|
|
23142
|
+
...reasoningContent ? { reasoning_content: reasoningContent } : {},
|
|
23143
|
+
tool_calls: [{
|
|
23144
|
+
id: tool.id,
|
|
23145
|
+
type: "function",
|
|
23146
|
+
function: {
|
|
23147
|
+
name: tool.name,
|
|
23148
|
+
arguments: tool.parameters
|
|
23149
|
+
}
|
|
23150
|
+
}]
|
|
23151
|
+
});
|
|
23152
|
+
messages.push({
|
|
23153
|
+
role: "tool",
|
|
23154
|
+
content: JSON.stringify({ result }),
|
|
23155
|
+
tool_call_id: tool.id
|
|
23156
|
+
});
|
|
23157
|
+
}
|
|
23158
|
+
replaceLastToolResultObservation(messages, toolCallId, newObservation) {
|
|
23159
|
+
replaceLastToolResultObservationOpenAI(messages, toolCallId, newObservation);
|
|
23160
|
+
}
|
|
23161
|
+
getLatestToolCallId(messages, toolName) {
|
|
23162
|
+
return getLatestToolCallIdOpenAI(messages, toolName);
|
|
23163
|
+
}
|
|
23164
|
+
};
|
|
23165
|
+
/**
|
|
23166
|
+
* Request shaping for Moonshot's Kimi models. Kept separate from kimiBackend's
|
|
23167
|
+
* transport so every "which parameter does this id accept" rule is one pure
|
|
23168
|
+
* function with a test, rather than a conditional buried in a 400-line complete().
|
|
23169
|
+
*
|
|
23170
|
+
* Moonshot is OpenAI-compatible in envelope only. The reasoning controls, the
|
|
23171
|
+
* sampling pins, and the max-tokens parameter all differ per model, and sending
|
|
23172
|
+
* the wrong one is a 400 rather than a silently ignored field.
|
|
23173
|
+
* @see https://platform.kimi.ai/docs/api/chat
|
|
23174
|
+
*/
|
|
23175
|
+
/** Kimi's own effort vocabulary, which is not OpenAI's and not B4M's. */
|
|
23176
|
+
const KIMI_EFFORT_LEVELS = [
|
|
23177
|
+
"low",
|
|
23178
|
+
"high",
|
|
23179
|
+
"max"
|
|
23180
|
+
];
|
|
23181
|
+
/**
|
|
23182
|
+
* Takes `reasoning_effort`. K3 only, and K3 always reasons - there is no way to
|
|
23183
|
+
* turn thinking off, so the parameter selects depth, never whether.
|
|
23184
|
+
*/
|
|
23185
|
+
const EFFORT_MODELS = /* @__PURE__ */ new Set([ChatModels.KIMI_K3]);
|
|
23186
|
+
/** Takes the `thinking` object instead of `reasoning_effort`. */
|
|
23187
|
+
const THINKING_MODELS = /* @__PURE__ */ new Set([
|
|
23188
|
+
ChatModels.KIMI_K2_7_CODE,
|
|
23189
|
+
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
23190
|
+
ChatModels.KIMI_K2_6,
|
|
23191
|
+
ChatModels.KIMI_K2_5
|
|
23192
|
+
]);
|
|
23193
|
+
/**
|
|
23194
|
+
* `thinking.type` accepts only 'enabled' on the K2.7 code models - 'disabled' is
|
|
23195
|
+
* rejected. So a caller asking for no thinking gets thinking anyway; the
|
|
23196
|
+
* alternative is a 400, and the parameter is omitted rather than fought.
|
|
23197
|
+
*/
|
|
23198
|
+
const THINKING_ALWAYS_ON = /* @__PURE__ */ new Set([ChatModels.KIMI_K2_7_CODE, ChatModels.KIMI_K2_7_CODE_HIGHSPEED]);
|
|
23199
|
+
/**
|
|
23200
|
+
* Rejects `tool_choice: 'required'` (auto and none are fine, as is the explicit
|
|
23201
|
+
* function form). Downgraded to 'auto' rather than dropped: a caller that asked
|
|
23202
|
+
* for a forced tool still wants tools offered.
|
|
23203
|
+
*/
|
|
23204
|
+
const NO_REQUIRED_TOOL_CHOICE = /* @__PURE__ */ new Set([
|
|
23205
|
+
ChatModels.KIMI_K2_7_CODE,
|
|
23206
|
+
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
23207
|
+
ChatModels.KIMI_K2_6
|
|
23208
|
+
]);
|
|
23209
|
+
/** Every Kimi id this build ships, direct-served. Bedrock-served Kimi is not here. */
|
|
23210
|
+
const KIMI_MODELS = /* @__PURE__ */ new Set([
|
|
23211
|
+
ChatModels.KIMI_K3,
|
|
23212
|
+
ChatModels.KIMI_K2_7_CODE,
|
|
23213
|
+
ChatModels.KIMI_K2_7_CODE_HIGHSPEED,
|
|
23214
|
+
ChatModels.KIMI_K2_6,
|
|
23215
|
+
ChatModels.KIMI_K2_5
|
|
23216
|
+
]);
|
|
23217
|
+
/**
|
|
23218
|
+
* B4M's six-level effort onto Kimi's three. 'none' and 'minimal' map to 'low'
|
|
23219
|
+
* rather than to omission because K3 cannot be asked not to think - claiming
|
|
23220
|
+
* otherwise by dropping the parameter would silently bill max-effort reasoning
|
|
23221
|
+
* (Moonshot's default is 'max').
|
|
23222
|
+
*/
|
|
23223
|
+
function toKimiEffort(effort) {
|
|
23224
|
+
if (!effort) return void 0;
|
|
23225
|
+
switch (effort) {
|
|
23226
|
+
case "none":
|
|
23227
|
+
case "minimal":
|
|
23228
|
+
case "low": return "low";
|
|
23229
|
+
case "medium":
|
|
23230
|
+
case "high": return "high";
|
|
23231
|
+
case "xhigh": return "max";
|
|
23232
|
+
default: return;
|
|
23233
|
+
}
|
|
23234
|
+
}
|
|
23235
|
+
/**
|
|
23236
|
+
* The reasoning parameters for one model, or an empty object when it takes none.
|
|
23237
|
+
* Mutually exclusive by construction: no Kimi model accepts both spellings, and
|
|
23238
|
+
* sending both is a 400.
|
|
23239
|
+
*/
|
|
23240
|
+
function kimiReasoningParams(model, input) {
|
|
23241
|
+
if (EFFORT_MODELS.has(model)) {
|
|
23242
|
+
const effort = toKimiEffort(input.reasoningEffort);
|
|
23243
|
+
return effort ? { reasoning_effort: effort } : {};
|
|
23244
|
+
}
|
|
23245
|
+
if (THINKING_MODELS.has(model)) {
|
|
23246
|
+
if (THINKING_ALWAYS_ON.has(model)) return { thinking: { type: "enabled" } };
|
|
23247
|
+
if (input.thinking?.enabled === void 0) return {};
|
|
23248
|
+
return { thinking: { type: input.thinking.enabled ? "enabled" : "disabled" } };
|
|
23249
|
+
}
|
|
23250
|
+
return {};
|
|
23251
|
+
}
|
|
23252
|
+
/**
|
|
23253
|
+
* Sampling parameters for one model. Moonshot pins temperature (1.0) and top_p
|
|
23254
|
+
* (0.95) on every current Kimi and documents them as unmodifiable, so they are
|
|
23255
|
+
* omitted rather than sent-and-ignored; NO_TEMPERATURE_MODELS is the shared set
|
|
23256
|
+
* the catalog's temperatureMode also lands on.
|
|
23257
|
+
*
|
|
23258
|
+
* The penalties and `n` ride the same gate. Moonshot documents the whole sampling
|
|
23259
|
+
* group as fixed on these ids, B4M sends penalties on essentially every turn, and
|
|
23260
|
+
* an unmodifiable parameter here is a 400 rather than a silently ignored field -
|
|
23261
|
+
* so the conservative reading is the safe one. Only the moonshot-v1 family, which
|
|
23262
|
+
* this build does not ship, accepts any of them.
|
|
23263
|
+
*/
|
|
23264
|
+
function kimiSamplingParams(model, input) {
|
|
23265
|
+
if (NO_TEMPERATURE_MODELS.has(model)) return {};
|
|
23266
|
+
const params = {};
|
|
23267
|
+
if (input.temperature !== void 0) params.temperature = input.temperature;
|
|
23268
|
+
if (input.topP !== void 0) params.top_p = input.topP;
|
|
23269
|
+
if (input.presencePenalty !== void 0) params.presence_penalty = input.presencePenalty;
|
|
23270
|
+
if (input.frequencyPenalty !== void 0) params.frequency_penalty = input.frequencyPenalty;
|
|
23271
|
+
if (input.n !== void 0) params.n = input.n;
|
|
23272
|
+
return params;
|
|
23273
|
+
}
|
|
23274
|
+
/** `tool_choice`, downgraded to 'auto' on the ids that reject 'required'. */
|
|
23275
|
+
function kimiToolChoice(model, choice) {
|
|
23276
|
+
if (choice === void 0) return void 0;
|
|
23277
|
+
if (choice === "required" && NO_REQUIRED_TOOL_CHOICE.has(model)) return "auto";
|
|
23278
|
+
return choice;
|
|
22498
23279
|
}
|
|
22499
23280
|
/**
|
|
22500
23281
|
* Moonshot AI's Kimi models, served from their OpenAI-compatible endpoint.
|
|
@@ -22634,6 +23415,8 @@ var KimiBackend = class {
|
|
|
22634
23415
|
supportsImageVariation: false,
|
|
22635
23416
|
releaseDate: "2026-01-01",
|
|
22636
23417
|
trainingCutoff: "2025-01-01",
|
|
23418
|
+
deprecationDate: "2026-08-31",
|
|
23419
|
+
replacedBy: ChatModels.KIMI_K2_6,
|
|
22637
23420
|
description: "The previous-generation Kimi, still the cheapest of the family. Superseded by K2.6 on quality at a modest price increase."
|
|
22638
23421
|
}
|
|
22639
23422
|
];
|
|
@@ -22852,16 +23635,17 @@ var KimiBackend = class {
|
|
|
22852
23635
|
}
|
|
22853
23636
|
chunk?.choices.forEach((c) => {
|
|
22854
23637
|
if (c.finish_reason) streamFinishReason = c.finish_reason;
|
|
22855
|
-
|
|
23638
|
+
const deltaReasoning = c.delta.reasoning_content;
|
|
23639
|
+
if (deltaReasoning) {
|
|
22856
23640
|
if (!isInThinkingBlock) {
|
|
22857
23641
|
isInThinkingBlock = true;
|
|
22858
|
-
streamedText[c.index] = "<think>" +
|
|
22859
|
-
} else streamedText[c.index] =
|
|
22860
|
-
return;
|
|
23642
|
+
streamedText[c.index] = "<think>" + deltaReasoning;
|
|
23643
|
+
} else streamedText[c.index] = deltaReasoning;
|
|
23644
|
+
if (!c.delta.content) return;
|
|
22861
23645
|
}
|
|
22862
|
-
if (isInThinkingBlock && c.delta.content
|
|
23646
|
+
if (isInThinkingBlock && c.delta.content) {
|
|
22863
23647
|
isInThinkingBlock = false;
|
|
22864
|
-
streamedText[c.index] = "</think>" +
|
|
23648
|
+
streamedText[c.index] = (streamedText[c.index] ?? "") + "</think>" + c.delta.content;
|
|
22865
23649
|
return;
|
|
22866
23650
|
}
|
|
22867
23651
|
c.delta.tool_calls?.map((tool) => {
|
|
@@ -23772,7 +24556,7 @@ var OpenAIBackend = class {
|
|
|
23772
24556
|
supportsTools: true,
|
|
23773
24557
|
supportsImageVariation: false,
|
|
23774
24558
|
logoFile: "OpenAI_Logo.svg",
|
|
23775
|
-
rank:
|
|
24559
|
+
rank: 4,
|
|
23776
24560
|
trainingCutoff: "2024-06-01",
|
|
23777
24561
|
description: "Reliable for general-purpose text generation and analysis with a standard context window, suitable for a wide range of applications."
|
|
23778
24562
|
},
|
|
@@ -23793,7 +24577,7 @@ var OpenAIBackend = class {
|
|
|
23793
24577
|
supportsTools: true,
|
|
23794
24578
|
supportsImageVariation: false,
|
|
23795
24579
|
logoFile: "OpenAI_Logo.svg",
|
|
23796
|
-
rank:
|
|
24580
|
+
rank: 4,
|
|
23797
24581
|
trainingCutoff: "2024-06-01",
|
|
23798
24582
|
description: "OpenAI's balanced GPT-4.1 model offering optimal price-performance ratio. Ideal for tasks requiring intelligence and cost efficiency."
|
|
23799
24583
|
},
|
|
@@ -23814,7 +24598,7 @@ var OpenAIBackend = class {
|
|
|
23814
24598
|
supportsTools: true,
|
|
23815
24599
|
supportsImageVariation: false,
|
|
23816
24600
|
logoFile: "OpenAI_Logo.svg",
|
|
23817
|
-
rank:
|
|
24601
|
+
rank: 4,
|
|
23818
24602
|
trainingCutoff: "2024-06-01",
|
|
23819
24603
|
deprecationDate: "2026-10-23",
|
|
23820
24604
|
description: "Designed for high-volume, low-cost processing with rapid response times, ideal for budget-conscious applications."
|
|
@@ -23878,7 +24662,7 @@ var OpenAIBackend = class {
|
|
|
23878
24662
|
supportsTools: true,
|
|
23879
24663
|
supportsImageVariation: false,
|
|
23880
24664
|
logoFile: "OpenAI_Logo.svg",
|
|
23881
|
-
rank:
|
|
24665
|
+
rank: 3,
|
|
23882
24666
|
trainingCutoff: "2024-06-01",
|
|
23883
24667
|
deprecationDate: "2026-12-11",
|
|
23884
24668
|
description: "OpenAI's O3 reasoning model with broad capabilities and up-to-date training data. Superseded by O4 Mini for most use cases.",
|
|
@@ -24050,7 +24834,7 @@ var OpenAIBackend = class {
|
|
|
24050
24834
|
supportsImageVariation: false,
|
|
24051
24835
|
supportsTools: true,
|
|
24052
24836
|
logoFile: "OpenAI_Logo.svg",
|
|
24053
|
-
rank:
|
|
24837
|
+
rank: 2,
|
|
24054
24838
|
trainingCutoff: "2026-01-01",
|
|
24055
24839
|
releaseDate: "2026-06-23",
|
|
24056
24840
|
description: "GPT-5.6 Luna - the fast, cost-efficient GPT-5.6 variant. Great for high-volume workloads that still need solid reasoning, vision, and tool use."
|
|
@@ -24116,7 +24900,7 @@ var OpenAIBackend = class {
|
|
|
24116
24900
|
supportsImageVariation: false,
|
|
24117
24901
|
supportsTools: true,
|
|
24118
24902
|
logoFile: "OpenAI_Logo.svg",
|
|
24119
|
-
rank:
|
|
24903
|
+
rank: 2,
|
|
24120
24904
|
trainingCutoff: "2025-08-31",
|
|
24121
24905
|
releaseDate: "2026-03-17",
|
|
24122
24906
|
description: "Compact GPT-5.4 variant balancing strong performance with lower cost. Great for everyday tasks needing solid reasoning and vision."
|
|
@@ -24138,7 +24922,7 @@ var OpenAIBackend = class {
|
|
|
24138
24922
|
supportsImageVariation: false,
|
|
24139
24923
|
supportsTools: true,
|
|
24140
24924
|
logoFile: "OpenAI_Logo.svg",
|
|
24141
|
-
rank:
|
|
24925
|
+
rank: 2,
|
|
24142
24926
|
trainingCutoff: "2025-08-31",
|
|
24143
24927
|
releaseDate: "2026-03-17",
|
|
24144
24928
|
description: "Ultra-lightweight GPT-5.4 model optimized for speed and cost efficiency. Ideal for high-volume workloads and quick interactions."
|
|
@@ -26011,6 +26795,10 @@ function backendForAdapterFamily(family, ctx) {
|
|
|
26011
26795
|
const key = keyOrThrow(apiKeyTable.kimi, "Moonshot");
|
|
26012
26796
|
return key ? new KimiBackend(key, logger) : null;
|
|
26013
26797
|
}
|
|
26798
|
+
case "deepseek": {
|
|
26799
|
+
const key = keyOrThrow(apiKeyTable.deepseek, "DeepSeek");
|
|
26800
|
+
return key ? new DeepSeekBackend(key, logger) : null;
|
|
26801
|
+
}
|
|
26014
26802
|
case "bfl": return new BFLBackend(keyOrThrow(apiKeyTable.bfl, "BFL") ?? "demo-key");
|
|
26015
26803
|
case "local-image": {
|
|
26016
26804
|
const baseUrl = keyOrThrow(apiKeyTable["local-image"], "Local image");
|
|
@@ -26051,6 +26839,7 @@ function buildApiKeyTable(keys) {
|
|
|
26051
26839
|
[ModelBackend.Ollama]: keys.ollama || void 0,
|
|
26052
26840
|
[ModelBackend.XAI]: keys.xai || void 0,
|
|
26053
26841
|
[ModelBackend.Kimi]: keys.kimi || void 0,
|
|
26842
|
+
[ModelBackend.DeepSeek]: keys.deepseek || void 0,
|
|
26054
26843
|
[ModelBackend.VoyageAI]: keys.voyageai || void 0,
|
|
26055
26844
|
[ModelBackend.LocalImage]: keys.imageGen || void 0,
|
|
26056
26845
|
[ModelBackend.Bedrock]: void 0,
|
|
@@ -26077,6 +26866,7 @@ const KEYED_LISTING_BACKENDS = [
|
|
|
26077
26866
|
ModelBackend.BFL,
|
|
26078
26867
|
ModelBackend.XAI,
|
|
26079
26868
|
ModelBackend.Kimi,
|
|
26869
|
+
ModelBackend.DeepSeek,
|
|
26080
26870
|
ModelBackend.LocalImage
|
|
26081
26871
|
];
|
|
26082
26872
|
/**
|
|
@@ -26134,6 +26924,7 @@ const DISPATCHABLE_ADAPTER_FAMILIES = [
|
|
|
26134
26924
|
"gemini",
|
|
26135
26925
|
"xai",
|
|
26136
26926
|
"kimi",
|
|
26927
|
+
"deepseek",
|
|
26137
26928
|
"ollama",
|
|
26138
26929
|
"bfl",
|
|
26139
26930
|
"local-image",
|
|
@@ -26211,6 +27002,15 @@ function mergeCatalogWithDrops(seedModels, rows, ctx) {
|
|
|
26211
27002
|
for (const [modelId, bucket] of rowsByModel) {
|
|
26212
27003
|
if (seeded.has(modelId)) continue;
|
|
26213
27004
|
const { draft } = mergeRows(bucket, null);
|
|
27005
|
+
const status = draft.lifecycle?.status;
|
|
27006
|
+
const lifecycleReason = inactiveLifecycleReason(status);
|
|
27007
|
+
if (lifecycleReason) {
|
|
27008
|
+
dropped.push({
|
|
27009
|
+
modelId,
|
|
27010
|
+
reason: lifecycleReason
|
|
27011
|
+
});
|
|
27012
|
+
continue;
|
|
27013
|
+
}
|
|
26214
27014
|
const parsed = asRenderableRecord(draft);
|
|
26215
27015
|
if ("reason" in parsed) {
|
|
26216
27016
|
dropped.push({
|
|
@@ -26327,10 +27127,19 @@ function asRenderableRecord(draft) {
|
|
|
26327
27127
|
if (typeof draft.type !== "string" || !isRenderableModelType(draft.type)) return { reason: `unsupported model type "${String(draft.type)}"` };
|
|
26328
27128
|
return { record: draft };
|
|
26329
27129
|
}
|
|
27130
|
+
/**
|
|
27131
|
+
* Why a lifecycle status is not invocable, or null when it is "active". Shared
|
|
27132
|
+
* between invocabilityBlocker and the catalog-only tier's pre-parse check, so
|
|
27133
|
+
* both agree on the exact wording.
|
|
27134
|
+
*/
|
|
27135
|
+
function inactiveLifecycleReason(status) {
|
|
27136
|
+
if (status !== "active") return `lifecycle status "${status ?? "unset"}" is not invocable`;
|
|
27137
|
+
return null;
|
|
27138
|
+
}
|
|
26330
27139
|
/** Why a catalog-only record is metadata-only, or null when it is invocable. */
|
|
26331
27140
|
function invocabilityBlocker(record) {
|
|
26332
|
-
const
|
|
26333
|
-
if (
|
|
27141
|
+
const lifecycleReason = inactiveLifecycleReason(record.lifecycle?.status);
|
|
27142
|
+
if (lifecycleReason) return lifecycleReason;
|
|
26334
27143
|
if (!record.adapterFamily) return "no adapterFamily";
|
|
26335
27144
|
if (!DISPATCHABLE_ADAPTER_FAMILIES.includes(record.adapterFamily)) return `adapterFamily "${record.adapterFamily}" is not dispatchable by this build`;
|
|
26336
27145
|
if (!record.dispatchProfile) return "no dispatchProfile";
|
|
@@ -26591,6 +27400,7 @@ const DEPRECATED_MODEL_MAP = {
|
|
|
26591
27400
|
"grok-2-vision-1212": "grok-4.5",
|
|
26592
27401
|
"grok-beta": "grok-4.5",
|
|
26593
27402
|
"grok-vision-beta": "grok-4.5",
|
|
27403
|
+
"kimi-k2.5": "kimi-k2.6",
|
|
26594
27404
|
"grok-3-mini-fast": "grok-3-mini"
|
|
26595
27405
|
};
|
|
26596
27406
|
/**
|
|
@@ -26715,6 +27525,58 @@ var UndifferentiatedBedrockBackend = class extends BaseBedrockBackend {
|
|
|
26715
27525
|
}
|
|
26716
27526
|
};
|
|
26717
27527
|
/**
|
|
27528
|
+
* The prices this build ships in code, keyed by model id.
|
|
27529
|
+
*
|
|
27530
|
+
* Same provenance as packages/database's modelPrices.seed.json - the adapter
|
|
27531
|
+
* `getModelInfo()` literals - reachable without a database, which is what the
|
|
27532
|
+
* price planner needs: a model's FIRST discovery-written row has no row in force
|
|
27533
|
+
* to carry the rates no feed publishes from, and a tier that reaches
|
|
27534
|
+
* getTextModelCost without `cache_read` settles cached reads at
|
|
27535
|
+
* input * CACHE_READ_MULTIPLIER. On DeepSeek Flash that default is 0.03/1M
|
|
27536
|
+
* against a real 0.006/1M. MUST STAY IN SYNC with collectStaticTextModels in
|
|
27537
|
+
* packages/database/src/seeds/generateModelPriceSeed.ts: both lists are "every
|
|
27538
|
+
* backend whose getModelInfo() is a static table", and Ollama is absent from
|
|
27539
|
+
* both because its listing is a live server call.
|
|
27540
|
+
*/
|
|
27541
|
+
const STATIC_PRICE_BACKENDS = () => [
|
|
27542
|
+
new OpenAIBackend("price-literal"),
|
|
27543
|
+
new AnthropicBackend("price-literal"),
|
|
27544
|
+
new UndifferentiatedBedrockBackend(),
|
|
27545
|
+
new GeminiBackend("price-literal"),
|
|
27546
|
+
new XAIBackend("price-literal"),
|
|
27547
|
+
new KimiBackend("price-literal"),
|
|
27548
|
+
new DeepSeekBackend("price-literal"),
|
|
27549
|
+
new AWSBackend()
|
|
27550
|
+
];
|
|
27551
|
+
let cached;
|
|
27552
|
+
/**
|
|
27553
|
+
* The lowest-threshold tier of each priced text model's adapter literal.
|
|
27554
|
+
*
|
|
27555
|
+
* Lowest tier on purpose: this is a last-resort carry for rates no feed
|
|
27556
|
+
* publishes (cache and audio), and those do not vary by context bracket in any
|
|
27557
|
+
* literal we ship, while the threshold keys of a discovered ladder need not
|
|
27558
|
+
* match the literal's. Memoized - the tables are static, and the planner runs
|
|
27559
|
+
* once per convergence pass.
|
|
27560
|
+
*/
|
|
27561
|
+
async function adapterPriceTiers() {
|
|
27562
|
+
cached ??= collect();
|
|
27563
|
+
return cached;
|
|
27564
|
+
}
|
|
27565
|
+
async function collect() {
|
|
27566
|
+
const tables = await Promise.all(STATIC_PRICE_BACKENDS().map((backend) => backend.getModelInfo()));
|
|
27567
|
+
const tiers = /* @__PURE__ */ new Map();
|
|
27568
|
+
for (const model of tables.flat()) {
|
|
27569
|
+
if (model.type !== "text" || model.freeToRun) continue;
|
|
27570
|
+
const tier = lowestTier(model);
|
|
27571
|
+
if (tier) tiers.set(String(model.id), tier);
|
|
27572
|
+
}
|
|
27573
|
+
return tiers;
|
|
27574
|
+
}
|
|
27575
|
+
function lowestTier(model) {
|
|
27576
|
+
const thresholds = Object.keys(model.pricing).map(Number).filter((threshold) => Number.isFinite(threshold)).sort((a, b) => a - b);
|
|
27577
|
+
return thresholds.length > 0 ? model.pricing[thresholds[0]] : void 0;
|
|
27578
|
+
}
|
|
27579
|
+
/**
|
|
26718
27580
|
* The dispatch group for a family whose request builder shapes its payload from
|
|
26719
27581
|
* the provider's own contract and reads nothing out of the profile (Bedrock,
|
|
26720
27582
|
* Gemini, xAI, Ollama, and the image/speech backends). Promotion still requires
|
|
@@ -26755,6 +27617,15 @@ const KIMI_PROFILE = {
|
|
|
26755
27617
|
maxTokensParam: "max_completion_tokens",
|
|
26756
27618
|
toolTransport: "chat"
|
|
26757
27619
|
};
|
|
27620
|
+
/**
|
|
27621
|
+
* DeepSeek direct. Its own constant rather than PROVIDER_NATIVE_PROFILE because
|
|
27622
|
+
* tools ride Chat Completions rather than a provider-native field, which is what
|
|
27623
|
+
* deepseekBackend sends; the token parameter is still `max_tokens`.
|
|
27624
|
+
*/
|
|
27625
|
+
const DEEPSEEK_PROFILE = {
|
|
27626
|
+
maxTokensParam: "max_tokens",
|
|
27627
|
+
toolTransport: "chat"
|
|
27628
|
+
};
|
|
26758
27629
|
/** Backends whose family is the backend, with a request shape this build fixes. */
|
|
26759
27630
|
const FAMILY_BY_BACKEND = {
|
|
26760
27631
|
[ModelBackend.Anthropic]: "anthropic-messages",
|
|
@@ -26796,6 +27667,10 @@ function resolveDispatchForRecord(record) {
|
|
|
26796
27667
|
adapterFamily: "kimi",
|
|
26797
27668
|
dispatchProfile: KIMI_PROFILE
|
|
26798
27669
|
};
|
|
27670
|
+
if (record.backend === ModelBackend.DeepSeek) return {
|
|
27671
|
+
adapterFamily: "deepseek",
|
|
27672
|
+
dispatchProfile: DEEPSEEK_PROFILE
|
|
27673
|
+
};
|
|
26799
27674
|
const adapterFamily = FAMILY_BY_BACKEND[record.backend];
|
|
26800
27675
|
return adapterFamily ? {
|
|
26801
27676
|
adapterFamily,
|
|
@@ -26915,7 +27790,11 @@ var AnthropicBatchService = class AnthropicBatchService {
|
|
|
26915
27790
|
reply: msg.content.filter((b) => b.type === "text").map((b) => b.text).join(""),
|
|
26916
27791
|
tokenUsage: {
|
|
26917
27792
|
inputTokens: msg.usage?.input_tokens ?? 0,
|
|
26918
|
-
outputTokens: msg.usage?.output_tokens ?? 0
|
|
27793
|
+
outputTokens: msg.usage?.output_tokens ?? 0,
|
|
27794
|
+
cacheReadInputTokens: msg.usage?.cache_read_input_tokens ?? void 0,
|
|
27795
|
+
cacheCreationInputTokens: msg.usage?.cache_creation_input_tokens ?? void 0,
|
|
27796
|
+
cacheWrite5mInputTokens: msg.usage?.cache_creation?.ephemeral_5m_input_tokens ?? void 0,
|
|
27797
|
+
cacheWrite1hInputTokens: msg.usage?.cache_creation?.ephemeral_1h_input_tokens ?? void 0
|
|
26919
27798
|
}
|
|
26920
27799
|
};
|
|
26921
27800
|
}
|
|
@@ -27155,6 +28034,10 @@ function getLlmByModel(apiKeyTable, options) {
|
|
|
27155
28034
|
if (apiKeyTable.kimi === "expired") throw new Error("Moonshot API key is expired");
|
|
27156
28035
|
backend = apiKeyTable.kimi ? new KimiBackend(apiKeyTable.kimi, logger) : null;
|
|
27157
28036
|
break;
|
|
28037
|
+
case "deepseek":
|
|
28038
|
+
if (apiKeyTable.deepseek === "expired") throw new Error("DeepSeek API key is expired");
|
|
28039
|
+
backend = apiKeyTable.deepseek ? new DeepSeekBackend(apiKeyTable.deepseek, logger) : null;
|
|
28040
|
+
break;
|
|
27158
28041
|
case "aws":
|
|
27159
28042
|
backend = new AWSBackend();
|
|
27160
28043
|
break;
|
|
@@ -27249,6 +28132,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
|
|
|
27249
28132
|
const bflKey = resolveListingKey(ModelBackend.BFL, gateCtx);
|
|
27250
28133
|
const xaiKey = resolveListingKey(ModelBackend.XAI, gateCtx);
|
|
27251
28134
|
const kimiKey = resolveListingKey(ModelBackend.Kimi, gateCtx);
|
|
28135
|
+
const deepseekKey = resolveListingKey(ModelBackend.DeepSeek, gateCtx);
|
|
27252
28136
|
const localImageBaseUrl = resolveListingKey(ModelBackend.LocalImage, gateCtx);
|
|
27253
28137
|
const backends = {
|
|
27254
28138
|
[ModelBackend.OpenAI]: openaiKey ? new OpenAIBackend(openaiKey) : null,
|
|
@@ -27259,6 +28143,7 @@ const getAvailableModels = async (apiKeys, options = {}) => {
|
|
|
27259
28143
|
[ModelBackend.BFL]: bflKey ? new BFLBackend(bflKey) : null,
|
|
27260
28144
|
[ModelBackend.XAI]: xaiKey ? new XAIBackend(xaiKey) : null,
|
|
27261
28145
|
[ModelBackend.Kimi]: kimiKey ? new KimiBackend(kimiKey) : null,
|
|
28146
|
+
[ModelBackend.DeepSeek]: deepseekKey ? new DeepSeekBackend(deepseekKey) : null,
|
|
27262
28147
|
[ModelBackend.AWS]: isBackendUsable(ModelBackend.AWS, gateCtx) ? new AWSBackend() : null,
|
|
27263
28148
|
[ModelBackend.LocalImage]: localImageBaseUrl ? new LocalImageBackend(localImageBaseUrl, Logger.globalInstance) : null
|
|
27264
28149
|
};
|
|
@@ -28781,7 +29666,7 @@ const MODEL_ALIASES = {
|
|
|
28781
29666
|
"grok-3-mini-fast": ChatModels.GROK_3_MINI_FAST,
|
|
28782
29667
|
"grok-2": ChatModels.GROK_2,
|
|
28783
29668
|
"grok-2-vision": ChatModels.GROK_2_VISION,
|
|
28784
|
-
deepseek: ChatModels.
|
|
29669
|
+
deepseek: ChatModels.DEEPSEEK_FLASH,
|
|
28785
29670
|
"deepseek-r1": ChatModels.DEEPSEEK_R1,
|
|
28786
29671
|
llama: ChatModels.LLAMA3_LOCAL,
|
|
28787
29672
|
llama3: ChatModels.LLAMA3_LOCAL,
|
|
@@ -29630,8 +30515,11 @@ function effectiveContextWindow(modelInfo) {
|
|
|
29630
30515
|
* budget below - and they must not drift apart.
|
|
29631
30516
|
*
|
|
29632
30517
|
* The static catalog tables are held to the positive-budget property by
|
|
29633
|
-
* modelCatalogInputBudget.test.ts
|
|
29634
|
-
*
|
|
30518
|
+
* modelCatalogInputBudget.test.ts. A discovered claim is guarded in two places, one per direction:
|
|
30519
|
+
* modelDiscoveryService/catalogWrite refuses a TEXT row whose output cap starves its own window,
|
|
30520
|
+
* and the docs parsers refuse a window or an output cap past MAX_PLAUSIBLE_TOKENS
|
|
30521
|
+
* (modelDiscoveryService/sources/openaiDocs.ts) - an overstated window is not a non-positive
|
|
30522
|
+
* budget, so catalogWrite would never see it, and no aggregator may correct a provider's figure.
|
|
29635
30523
|
*
|
|
29636
30524
|
* The buffer figure is imported rather than redeclared here: common owns it, and two copies of the
|
|
29637
30525
|
* same number is the drift that made it a shared export in the first place.
|
|
@@ -31331,7 +32219,8 @@ async function processFabFilesServer(embeddingFactory, fabFiles, userPrompt, att
|
|
|
31331
32219
|
switch (modelInfo?.backend) {
|
|
31332
32220
|
case ModelBackend.OpenAI:
|
|
31333
32221
|
case ModelBackend.XAI:
|
|
31334
|
-
case ModelBackend.Kimi:
|
|
32222
|
+
case ModelBackend.Kimi:
|
|
32223
|
+
case ModelBackend.DeepSeek: {
|
|
31335
32224
|
const openaiImageBuffer = await storage.download(file.filePath);
|
|
31336
32225
|
const { mime: openaiMimeType } = await getFileType(openaiImageBuffer, file.fileName, file.mimeType);
|
|
31337
32226
|
const openaiBase64 = openaiImageBuffer.toString("base64");
|
|
@@ -36743,6 +37632,7 @@ __reExport(/* @__PURE__ */ __exportAll({
|
|
|
36743
37632
|
registrableDomain: () => registrableDomain,
|
|
36744
37633
|
reservationOutputTokens: () => reservationOutputTokens,
|
|
36745
37634
|
resolveEmbeddingConfig: () => resolveEmbeddingConfig,
|
|
37635
|
+
resolveEmbeddingWithKeylessFallback: () => resolveEmbeddingWithKeylessFallback,
|
|
36746
37636
|
resolveSupportedMimeType: () => resolveSupportedMimeType,
|
|
36747
37637
|
safeInputWindow: () => safeInputWindow,
|
|
36748
37638
|
scopedOverrideKey: () => scopedOverrideKey,
|