@oh-my-pi/pi-catalog 18.2.0 → 18.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/dist/types/compat/behavior.d.ts +11 -0
- package/dist/types/compat/types.d.ts +21 -0
- package/dist/types/discovery/antigravity.d.ts +10 -1
- package/dist/types/model-thinking.d.ts +7 -0
- package/dist/types/provider-models/openai-compat.d.ts +2 -0
- package/dist/types/types.d.ts +26 -1
- package/package.json +4 -4
- package/src/compat/axes.ts +14 -0
- package/src/compat/behavior.ts +22 -2
- package/src/compat/cascade.ts +3 -1
- package/src/compat/context-window.ts +11 -1
- package/src/compat/resolve.ts +42 -16
- package/src/compat/rules/README.md +3 -1
- package/src/compat/rules/classes/deepseek.kdl +9 -1
- package/src/compat/rules/classes/kimi.kdl +6 -0
- package/src/compat/rules/providers/alibaba-token-plan.kdl +16 -8
- package/src/compat/rules/providers/amazon-bedrock.kdl +30 -0
- package/src/compat/rules/providers/azure.kdl +6 -0
- package/src/compat/rules/providers/cerebras.kdl +10 -0
- package/src/compat/rules/providers/commandcode.kdl +20 -4
- package/src/compat/rules/providers/cursor.kdl +32 -0
- package/src/compat/rules/providers/deepseek.kdl +5 -5
- package/src/compat/rules/providers/google-vertex.kdl +15 -0
- package/src/compat/rules/providers/meta.kdl +3 -0
- package/src/compat/rules/providers/muse-code.kdl +3 -0
- package/src/compat/rules/providers/openrouter.kdl +6 -0
- package/src/compat/rules/runtime/behavior.kdl +17 -0
- package/src/compat/rules/taxonomy/deepseek.kdl +5 -0
- package/src/compat/rules.json +1 -1
- package/src/compat/types.ts +23 -0
- package/src/discovery/antigravity.ts +80 -43
- package/src/identity/bundled.ts +4 -3
- package/src/model-cache.ts +115 -37
- package/src/model-thinking.ts +10 -7
- package/src/models.json +1 -1
- package/src/provider-models/bundled-references.ts +4 -3
- package/src/provider-models/cache-provider-id.ts +14 -8
- package/src/provider-models/ollama.ts +11 -31
- package/src/provider-models/openai-compat.ts +94 -24
- package/src/types.ts +28 -0
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { isBareIdReferenceProvider } from "../compat/behavior";
|
|
1
2
|
import { isZeroCostXaiOAuthReference } from "../identity/reference";
|
|
2
3
|
import { getBundledModels, getBundledProviders } from "../models";
|
|
3
4
|
import type { Api, Model, ModelSpec } from "../types";
|
|
@@ -45,9 +46,9 @@ function getGlobalReferences(): Map<string, Model<Api>> {
|
|
|
45
46
|
for (const provider of getBundledProviders()) {
|
|
46
47
|
for (const model of getBundledModels(provider as Parameters<typeof getBundledModels>[0])) {
|
|
47
48
|
const candidate = model as Model<Api>;
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
if (candidate.provider
|
|
49
|
+
// Gateway-specific metadata must remain provider-local when proxy
|
|
50
|
+
// discovery resolves references by bare model id.
|
|
51
|
+
if (!isBareIdReferenceProvider(candidate.provider) || isZeroCostXaiOAuthReference(candidate)) {
|
|
51
52
|
continue;
|
|
52
53
|
}
|
|
53
54
|
const existing = references.get(candidate.id);
|
|
@@ -89,15 +89,21 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
|
|
89
89
|
}
|
|
90
90
|
case "litellm": {
|
|
91
91
|
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
|
|
92
|
-
// rich-
|
|
93
|
-
//
|
|
94
|
-
//
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
// (issue #9938).
|
|
99
|
-
return `litellm:rich-
|
|
92
|
+
// rich-v11 invalidates rows that inherited ClinePass gateway metadata
|
|
93
|
+
// through generic models.dev bare-id enrichment (issue #10932). rich-v10
|
|
94
|
+
// filtered known non-conversational LiteLLM modes, unioned compat across
|
|
95
|
+
// the management endpoints, and keyed the deployment's `supports_vision`
|
|
96
|
+
// declaration into it; earlier versions invalidated rows whose
|
|
97
|
+
// `compatConfig` retained a colliding bundled model's provider-specific
|
|
98
|
+
// transport (e.g. Fireworks `wireModelIdMode`) (issue #9938).
|
|
99
|
+
return `litellm:rich-v11:${Bun.hash(baseUrl).toString(36)}`;
|
|
100
100
|
}
|
|
101
|
+
case "gmi-cloud":
|
|
102
|
+
case "siliconflow":
|
|
103
|
+
case "siliconflow-cn":
|
|
104
|
+
// models-v1 moves rows enriched before cross-provider reference
|
|
105
|
+
// isolation out of the legacy bare-provider namespaces (#10932).
|
|
106
|
+
return `${providerId}:models-v1`;
|
|
101
107
|
case "opencode-go":
|
|
102
108
|
case "opencode-zen": {
|
|
103
109
|
// v3: gateway-first rows cached before stencil enrichment carry null
|
|
@@ -1,9 +1,6 @@
|
|
|
1
1
|
import { fetchWithRetry } from "@oh-my-pi/pi-utils";
|
|
2
|
-
import { compareRevision, parseRevision } from "../compat/revision";
|
|
3
|
-
import { classifyModel } from "../compat/taxonomy";
|
|
4
|
-
import { Effort } from "../effort";
|
|
5
2
|
import type { ModelManagerOptions } from "../model-manager";
|
|
6
|
-
import type { FetchImpl, ModelSpec
|
|
3
|
+
import type { FetchImpl, ModelSpec } from "../types";
|
|
7
4
|
import { discoveryFetch } from "../utils";
|
|
8
5
|
import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references";
|
|
9
6
|
|
|
@@ -54,11 +51,6 @@ export function isOllamaCloudOutputCapped(id: string): boolean {
|
|
|
54
51
|
return OLLAMA_CLOUD_OUTPUT_CAPPED_BASE_IDS[baseId] === true;
|
|
55
52
|
}
|
|
56
53
|
|
|
57
|
-
const OLLAMA_CLOUD_GLM_52_THINKING: ThinkingConfig = {
|
|
58
|
-
mode: "effort",
|
|
59
|
-
efforts: [Effort.High, Effort.Max],
|
|
60
|
-
};
|
|
61
|
-
|
|
62
54
|
function trimTrailingSlash(value: string): string {
|
|
63
55
|
return value.endsWith("/") ? value.slice(0, -1) : value;
|
|
64
56
|
}
|
|
@@ -93,27 +85,6 @@ function getContextWindow(modelInfo: Record<string, unknown> | undefined): numbe
|
|
|
93
85
|
}
|
|
94
86
|
}
|
|
95
87
|
|
|
96
|
-
function getThinkingConfig(modelId: string, capabilities: string[] | undefined): ThinkingConfig | undefined {
|
|
97
|
-
if (!capabilities?.includes("thinking")) {
|
|
98
|
-
return undefined;
|
|
99
|
-
}
|
|
100
|
-
const identity = classifyModel("ollama-cloud", modelId, { lenient: true });
|
|
101
|
-
const revision = identity.revision === undefined ? undefined : parseRevision(identity.revision);
|
|
102
|
-
const floor = parseRevision(identity.family === "flash" ? "5.3" : "5.2");
|
|
103
|
-
const isGlmEffortModel =
|
|
104
|
-
identity.class === "glm" &&
|
|
105
|
-
(identity.family === undefined ||
|
|
106
|
-
identity.family === "air" ||
|
|
107
|
-
identity.family === "turbo" ||
|
|
108
|
-
identity.family === "flash") &&
|
|
109
|
-
revision !== undefined &&
|
|
110
|
-
floor !== undefined &&
|
|
111
|
-
compareRevision(revision, floor) >= 0;
|
|
112
|
-
if (isGlmEffortModel) {
|
|
113
|
-
return OLLAMA_CLOUD_GLM_52_THINKING;
|
|
114
|
-
}
|
|
115
|
-
return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
|
|
116
|
-
}
|
|
117
88
|
async function fetchShowMetadata(
|
|
118
89
|
baseUrl: string,
|
|
119
90
|
apiKey: string,
|
|
@@ -182,7 +153,16 @@ export function ollamaCloudModelManagerOptions(
|
|
|
182
153
|
// reference limit, falling back to the historical safe cap otherwise.
|
|
183
154
|
const contextWindow = discoveredContextWindow ?? 128000;
|
|
184
155
|
const reasoning = capabilities ? capabilities.includes("thinking") : (reference?.reasoning ?? false);
|
|
185
|
-
|
|
156
|
+
// `/api/show` reports only a boolean `thinking` capability, never a
|
|
157
|
+
// tier vocabulary, so the effort ladder is left to the compat rules
|
|
158
|
+
// (which own every other host's ladder too). Synthesizing one here
|
|
159
|
+
// shadowed the rules: discovery rows carry explicit `thinking`, and
|
|
160
|
+
// `resolveThinkingPolicy` treats explicit metadata as authoritative
|
|
161
|
+
// over the KDL, so a fabricated `minimal..high` ladder overrode the
|
|
162
|
+
// DeepSeek V4 `low/high/max` contract and silently clamped `max`
|
|
163
|
+
// down to `high` (#8334 regression). Models whose ladder the rules
|
|
164
|
+
// do not know still fall back to the generic four-tier default.
|
|
165
|
+
const thinking = capabilities ? undefined : reference?.thinking;
|
|
186
166
|
const input = capabilities
|
|
187
167
|
? capabilities.includes("vision")
|
|
188
168
|
? (["text", "image"] as Array<"text" | "image">)
|
|
@@ -4,6 +4,8 @@ import { toClinePassPublicModelId } from "../cline-pass-model-id";
|
|
|
4
4
|
import {
|
|
5
5
|
apiRouteExactModelIds,
|
|
6
6
|
apiRouteFor,
|
|
7
|
+
isBareIdReferenceProvider,
|
|
8
|
+
isExcludedDiscoveryMode,
|
|
7
9
|
isExcludedModel,
|
|
8
10
|
isLikelyOpenAIResponsesId,
|
|
9
11
|
modelLimitsFor,
|
|
@@ -579,6 +581,7 @@ type OpenAICompatibleModelManagerBuilderOptions<TApi extends Api> = {
|
|
|
579
581
|
dynamicModelsAuthoritative?: true;
|
|
580
582
|
requireApiKey?: true;
|
|
581
583
|
dropCachedModelIdsOnStaticMismatch?: readonly string[];
|
|
584
|
+
cacheProviderId?: string;
|
|
582
585
|
filterModel?: (
|
|
583
586
|
entry: OpenAICompatibleModelRecord,
|
|
584
587
|
model: ModelSpec<TApi>,
|
|
@@ -600,6 +603,7 @@ function createOpenAICompatibleModelManagerOptions<TApi extends Api>(
|
|
|
600
603
|
const filterModel = options.filterModel;
|
|
601
604
|
return {
|
|
602
605
|
providerId: options.providerId,
|
|
606
|
+
...(options.cacheProviderId && { cacheProviderId: options.cacheProviderId }),
|
|
603
607
|
...(options.dynamicModelsAuthoritative && { dynamicModelsAuthoritative: true }),
|
|
604
608
|
...(options.dropCachedModelIdsOnStaticMismatch && {
|
|
605
609
|
dropCachedModelIdsOnStaticMismatch: options.dropCachedModelIdsOnStaticMismatch,
|
|
@@ -982,6 +986,7 @@ export function gmiCloudModelManagerOptions(
|
|
|
982
986
|
api: "openai-completions",
|
|
983
987
|
providerId: "gmi-cloud",
|
|
984
988
|
defaultBaseUrl: GMI_CLOUD_BASE_URL,
|
|
989
|
+
cacheProviderId: resolveModelCacheProviderId("gmi-cloud"),
|
|
985
990
|
config,
|
|
986
991
|
requireApiKey: true,
|
|
987
992
|
mapModel: mapGmiCloudModel,
|
|
@@ -1201,6 +1206,13 @@ function mapDeepinfraModel(
|
|
|
1201
1206
|
return null;
|
|
1202
1207
|
}
|
|
1203
1208
|
const pricing = isRecord(metadata.pricing) ? metadata.pricing : {};
|
|
1209
|
+
// `metadata.discount` is a promotional fraction in [0, 1): DeepInfra bills
|
|
1210
|
+
// `pricing * (1 - discount)` (verified against the site — GLM-5.2 lists
|
|
1211
|
+
// input 0.75 with discount 0.35 and charges 0.4875), while `pricing.*`
|
|
1212
|
+
// stays at list price. Fold it into the rate card so cost reporting matches
|
|
1213
|
+
// what the user is actually billed. Values outside (0, 1) are ignored.
|
|
1214
|
+
const discount = toNumber(metadata.discount);
|
|
1215
|
+
const discountMultiplier = discount !== undefined && discount > 0 && discount < 1 ? 1 - discount : 1;
|
|
1204
1216
|
// `reasoning_effort` marks models whose effort dial is advertised. The
|
|
1205
1217
|
// parameter itself is validated and accepted platform-wide on DeepInfra
|
|
1206
1218
|
// (verified: 200 on effort-tagged, reasoning-only, and plain-chat models;
|
|
@@ -1243,9 +1255,9 @@ function mapDeepinfraModel(
|
|
|
1243
1255
|
...(thinking ? { thinking } : {}),
|
|
1244
1256
|
input: tags.includes("vision") || tags.includes("vlm") ? ["text", "image"] : ["text"],
|
|
1245
1257
|
cost: {
|
|
1246
|
-
input: toPositiveNumber(pricing.input_tokens, 0),
|
|
1247
|
-
output: toPositiveNumber(pricing.output_tokens, 0),
|
|
1248
|
-
cacheRead: toPositiveNumber(pricing.cache_read_tokens, 0),
|
|
1258
|
+
input: toPositiveNumber(pricing.input_tokens, 0) * discountMultiplier,
|
|
1259
|
+
output: toPositiveNumber(pricing.output_tokens, 0) * discountMultiplier,
|
|
1260
|
+
cacheRead: toPositiveNumber(pricing.cache_read_tokens, 0) * discountMultiplier,
|
|
1249
1261
|
cacheWrite: 0,
|
|
1250
1262
|
},
|
|
1251
1263
|
contextWindow,
|
|
@@ -1711,6 +1723,7 @@ function createSiliconFlowModelManagerOptions(
|
|
|
1711
1723
|
const baseUrl = config?.baseUrl ?? defaultBaseUrl;
|
|
1712
1724
|
return {
|
|
1713
1725
|
providerId,
|
|
1726
|
+
cacheProviderId: resolveModelCacheProviderId(providerId),
|
|
1714
1727
|
dynamicModelsAuthoritative: true,
|
|
1715
1728
|
...(apiKey && {
|
|
1716
1729
|
fetchDynamicModels: async () => {
|
|
@@ -2134,6 +2147,9 @@ function createModelsDevReferenceMap<TApi extends Api>(
|
|
|
2134
2147
|
const references = new Map<string, ModelSpec<TApi>>();
|
|
2135
2148
|
for (const model of models) {
|
|
2136
2149
|
const candidate = model as ModelSpec<TApi>;
|
|
2150
|
+
if (!isBareIdReferenceProvider(candidate.provider)) {
|
|
2151
|
+
continue;
|
|
2152
|
+
}
|
|
2137
2153
|
const existing = references.get(candidate.id);
|
|
2138
2154
|
if (!existing) {
|
|
2139
2155
|
references.set(candidate.id, candidate);
|
|
@@ -3866,10 +3882,17 @@ export interface BasetenModelManagerConfig {
|
|
|
3866
3882
|
fetch?: FetchImpl;
|
|
3867
3883
|
}
|
|
3868
3884
|
|
|
3869
|
-
// A previous version of OMP shipped
|
|
3870
|
-
// since fixed that. This const lets us bust
|
|
3871
|
-
// version of OMP pick up the reasoning levels
|
|
3872
|
-
|
|
3885
|
+
// A previous version of OMP shipped these models without reasoning levels.
|
|
3886
|
+
// We've since fixed that (V4-generation whitelist). This const lets us bust
|
|
3887
|
+
// the cache so that users on that version of OMP pick up the reasoning levels
|
|
3888
|
+
// immediately.
|
|
3889
|
+
const BASETEN_CACHE_MIGRATION_MODEL_IDS = [
|
|
3890
|
+
"zai-org/GLM-5.3",
|
|
3891
|
+
"zai-org/GLM-5.3-Flash",
|
|
3892
|
+
"deepseek-ai/DeepSeek-V4-Flash-0731",
|
|
3893
|
+
"deepseek-ai/DeepSeek-V4.1-Flash",
|
|
3894
|
+
"deepseek-ai/DeepSeek-V4-Pro-0813",
|
|
3895
|
+
] as const;
|
|
3873
3896
|
|
|
3874
3897
|
export function basetenModelManagerOptions(
|
|
3875
3898
|
config?: BasetenModelManagerConfig,
|
|
@@ -3900,7 +3923,7 @@ export function basetenModelManagerOptions(
|
|
|
3900
3923
|
(identity.class === "kimi" && identity.family === "k3") ||
|
|
3901
3924
|
isGlmReasoningIdentity("baseten", defaults.id, "5.2") ||
|
|
3902
3925
|
defaults.id === "openai/gpt-oss-120b" ||
|
|
3903
|
-
defaults.id
|
|
3926
|
+
isDeepseekV4Generation("baseten", defaults.id);
|
|
3904
3927
|
const reasoning =
|
|
3905
3928
|
isSupportedBasetenReasoningModel &&
|
|
3906
3929
|
(features.includes("reasoning") || features.includes("reasoning_effort"));
|
|
@@ -4678,7 +4701,11 @@ type LiteLLMRichEndpointFailure = {
|
|
|
4678
4701
|
error?: unknown;
|
|
4679
4702
|
};
|
|
4680
4703
|
type LiteLLMRichEndpointResult<TApi extends Api> =
|
|
4681
|
-
| {
|
|
4704
|
+
| {
|
|
4705
|
+
models: LiteLLMRichEndpointModel<TApi>[];
|
|
4706
|
+
excludedModelIds: ReadonlySet<string>;
|
|
4707
|
+
incompleteVisionMetadata: boolean;
|
|
4708
|
+
}
|
|
4682
4709
|
| { failure: LiteLLMRichEndpointFailure };
|
|
4683
4710
|
|
|
4684
4711
|
const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const;
|
|
@@ -4707,6 +4734,11 @@ function warnLiteLLMMetadataFallback(managementBaseUrl: string, failure: LiteLLM
|
|
|
4707
4734
|
});
|
|
4708
4735
|
}
|
|
4709
4736
|
|
|
4737
|
+
/** Exclude only known non-conversational modes; unknown and non-string modes remain selectable for aliases. */
|
|
4738
|
+
export function isSelectableLiteLLMModelMode(mode: unknown): boolean {
|
|
4739
|
+
return typeof mode !== "string" || !isExcludedDiscoveryMode("litellm", mode);
|
|
4740
|
+
}
|
|
4741
|
+
|
|
4710
4742
|
export function normalizeLiteLLMManagementBaseUrl(baseUrl: string): string {
|
|
4711
4743
|
const trimmed = baseUrl.trim().replace(/\/+$/g, "");
|
|
4712
4744
|
if (!trimmed) {
|
|
@@ -4747,7 +4779,10 @@ function mapLiteLLMOpenAICompatibleModel(
|
|
|
4747
4779
|
entry: OpenAICompatibleModelRecord,
|
|
4748
4780
|
defaults: ModelSpec<Api>,
|
|
4749
4781
|
reference: ModelSpec<Api> | undefined,
|
|
4750
|
-
): ModelSpec<Api> {
|
|
4782
|
+
): ModelSpec<Api> | null {
|
|
4783
|
+
if (!isSelectableLiteLLMModelMode(entry.mode)) {
|
|
4784
|
+
return null;
|
|
4785
|
+
}
|
|
4751
4786
|
const model = mapWithBundledReference(entry, defaults, reference);
|
|
4752
4787
|
return {
|
|
4753
4788
|
...model,
|
|
@@ -4943,7 +4978,10 @@ function mapLiteLLMRichEntry<TApi extends Api>(
|
|
|
4943
4978
|
options: FetchLiteLLMRichModelsOptions<TApi>,
|
|
4944
4979
|
runtimeBaseUrl: string,
|
|
4945
4980
|
): ModelSpec<TApi> | null {
|
|
4946
|
-
if (
|
|
4981
|
+
if (
|
|
4982
|
+
!isSelectableLiteLLMModelMode(getLiteLLMMetadataValue(entry, "mode")) ||
|
|
4983
|
+
isLiteLLMUnusableSentinelPlaceholder(entry)
|
|
4984
|
+
) {
|
|
4947
4985
|
return null;
|
|
4948
4986
|
}
|
|
4949
4987
|
const id = getLiteLLMRichModelId(entry);
|
|
@@ -5151,7 +5189,22 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
|
|
|
5151
5189
|
return null;
|
|
5152
5190
|
}
|
|
5153
5191
|
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
|
|
5192
|
+
const excludedModelIds = new Set<string>();
|
|
5154
5193
|
for (const entry of entries) {
|
|
5194
|
+
if (isLiteLLMUnusableSentinelPlaceholder(entry)) {
|
|
5195
|
+
continue;
|
|
5196
|
+
}
|
|
5197
|
+
const modelId = getLiteLLMRichModelId(entry);
|
|
5198
|
+
if (!isSelectableLiteLLMModelMode(getLiteLLMMetadataValue(entry, "mode"))) {
|
|
5199
|
+
if (modelId) {
|
|
5200
|
+
excludedModelIds.add(modelId);
|
|
5201
|
+
deduped.delete(modelId);
|
|
5202
|
+
}
|
|
5203
|
+
continue;
|
|
5204
|
+
}
|
|
5205
|
+
if (modelId && excludedModelIds.has(modelId)) {
|
|
5206
|
+
continue;
|
|
5207
|
+
}
|
|
5155
5208
|
const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl);
|
|
5156
5209
|
if (model) {
|
|
5157
5210
|
const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision");
|
|
@@ -5177,12 +5230,13 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
|
|
|
5177
5230
|
deduped.set(model.id, existing ? mergeLiteLLMRichEndpointModels(existing, next) : next);
|
|
5178
5231
|
}
|
|
5179
5232
|
}
|
|
5180
|
-
if (deduped.size === 0) {
|
|
5233
|
+
if (deduped.size === 0 && excludedModelIds.size === 0) {
|
|
5181
5234
|
return null;
|
|
5182
5235
|
}
|
|
5183
5236
|
const models = Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id));
|
|
5184
5237
|
return {
|
|
5185
5238
|
models,
|
|
5239
|
+
excludedModelIds,
|
|
5186
5240
|
incompleteVisionMetadata: models.some(entry => entry.supportsVision !== true && entry.supportsVision !== false),
|
|
5187
5241
|
};
|
|
5188
5242
|
}
|
|
@@ -5197,6 +5251,7 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
|
|
|
5197
5251
|
}
|
|
5198
5252
|
const fetchModels = async (signal?: AbortSignal): Promise<ModelSpec<TApi>[] | null> => {
|
|
5199
5253
|
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
|
|
5254
|
+
const excludedModelIds = new Set<string>();
|
|
5200
5255
|
let metadataFailure: LiteLLMRichEndpointFailure | undefined;
|
|
5201
5256
|
for (const endpoint of LITELLM_RICH_ENDPOINTS) {
|
|
5202
5257
|
const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
|
|
@@ -5218,8 +5273,15 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
|
|
|
5218
5273
|
}
|
|
5219
5274
|
continue;
|
|
5220
5275
|
}
|
|
5276
|
+
for (const modelId of result.excludedModelIds) {
|
|
5277
|
+
excludedModelIds.add(modelId);
|
|
5278
|
+
deduped.delete(modelId);
|
|
5279
|
+
}
|
|
5221
5280
|
const hadPriorModels = deduped.size > 0;
|
|
5222
5281
|
for (const next of result.models) {
|
|
5282
|
+
if (excludedModelIds.has(next.model.id)) {
|
|
5283
|
+
continue;
|
|
5284
|
+
}
|
|
5223
5285
|
const existing = deduped.get(next.model.id);
|
|
5224
5286
|
if (!existing) {
|
|
5225
5287
|
if (!hadPriorModels) {
|
|
@@ -5229,6 +5291,9 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
|
|
|
5229
5291
|
}
|
|
5230
5292
|
deduped.set(next.model.id, mergeLiteLLMRichEndpointModels(existing, next));
|
|
5231
5293
|
}
|
|
5294
|
+
if (deduped.size === 0) {
|
|
5295
|
+
continue;
|
|
5296
|
+
}
|
|
5232
5297
|
let needsMoreMetadata = false;
|
|
5233
5298
|
for (const entry of deduped.values()) {
|
|
5234
5299
|
if (
|
|
@@ -5249,6 +5314,9 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
|
|
|
5249
5314
|
}
|
|
5250
5315
|
}
|
|
5251
5316
|
if (deduped.size === 0) {
|
|
5317
|
+
if (excludedModelIds.size > 0) {
|
|
5318
|
+
return [];
|
|
5319
|
+
}
|
|
5252
5320
|
if (metadataFailure) {
|
|
5253
5321
|
warnLiteLLMMetadataFallback(managementBaseUrl, metadataFailure);
|
|
5254
5322
|
}
|
|
@@ -5275,17 +5343,18 @@ export function litellmModelManagerOptions(config?: LiteLLMModelManagerConfig):
|
|
|
5275
5343
|
const baseUrl = config?.baseUrl ?? getDefaultModelDiscoveryBaseUrl("litellm")!;
|
|
5276
5344
|
return {
|
|
5277
5345
|
providerId: "litellm",
|
|
5278
|
-
// rich-
|
|
5279
|
-
//
|
|
5280
|
-
//
|
|
5281
|
-
//
|
|
5282
|
-
//
|
|
5283
|
-
//
|
|
5284
|
-
//
|
|
5285
|
-
// past incomplete vision/API
|
|
5286
|
-
// pricing, stripped reseller usage
|
|
5287
|
-
// and mapped rich pricing. Bump the
|
|
5288
|
-
//
|
|
5346
|
+
// rich-v11 invalidates rows that inherited ClinePass gateway metadata
|
|
5347
|
+
// through generic models.dev bare-id enrichment (issue #10932). rich-v10
|
|
5348
|
+
// filtered known non-conversational LiteLLM modes, keyed the deployment's
|
|
5349
|
+
// `supports_vision` declaration into cached compat, and unioned compat
|
|
5350
|
+
// across management endpoints instead of letting a later endpoint retract
|
|
5351
|
+
// what an earlier one reported (issue #11982). Earlier versions fixed
|
|
5352
|
+
// provider-specific transport leakage, added bundled reference fallback,
|
|
5353
|
+
// moved OpenAI models to Responses, continued past incomplete vision/API
|
|
5354
|
+
// metadata and endpoints omitting cache pricing, stripped reseller usage
|
|
5355
|
+
// suffixes, filtered placeholder rows, and mapped rich pricing. Bump the
|
|
5356
|
+
// version whenever these mappers change, or warm authoritative caches keep
|
|
5357
|
+
// serving pre-change rows for the full TTL.
|
|
5289
5358
|
cacheProviderId: resolveModelCacheProviderId("litellm", { baseUrl }),
|
|
5290
5359
|
// litellm is a local-only proxy and is never bundled in models.json (that
|
|
5291
5360
|
// would leak the machine's localhost catalog). Prefer the proxy's richer
|
|
@@ -5304,7 +5373,7 @@ export function litellmModelManagerOptions(config?: LiteLLMModelManagerConfig):
|
|
|
5304
5373
|
resolveApi: resolveLiteLLMApi,
|
|
5305
5374
|
timeoutMs: 10_000,
|
|
5306
5375
|
});
|
|
5307
|
-
if (richModels
|
|
5376
|
+
if (richModels !== null) {
|
|
5308
5377
|
return richModels;
|
|
5309
5378
|
}
|
|
5310
5379
|
return fetchOpenAICompatibleModels<Api>({
|
|
@@ -6308,6 +6377,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
|
|
|
6308
6377
|
return {
|
|
6309
6378
|
...model,
|
|
6310
6379
|
id,
|
|
6380
|
+
name: id,
|
|
6311
6381
|
thinking: model.reasoning ? buildClinePassThinking(raw, model) : undefined,
|
|
6312
6382
|
};
|
|
6313
6383
|
},
|
package/src/types.ts
CHANGED
|
@@ -327,6 +327,12 @@ export interface OpenAICompat {
|
|
|
327
327
|
* Default: auto-detected (DeepSeek reasoning models).
|
|
328
328
|
*/
|
|
329
329
|
disableReasoningOnToolChoice?: boolean;
|
|
330
|
+
/**
|
|
331
|
+
* Disable reasoning whenever the request advertises function tools.
|
|
332
|
+
* Use for model surfaces that reject every tools-plus-reasoning combination.
|
|
333
|
+
* Default: false.
|
|
334
|
+
*/
|
|
335
|
+
disableReasoningWithTools?: boolean;
|
|
330
336
|
/** OpenRouter-specific routing preferences. Only used when baseUrl points to OpenRouter. */
|
|
331
337
|
openRouterRouting?: OpenRouterRouting;
|
|
332
338
|
/** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */
|
|
@@ -421,12 +427,24 @@ export interface OpenAICompat {
|
|
|
421
427
|
strictResponsesPairing?: boolean;
|
|
422
428
|
/** Whether the Responses API accepts the `detail: "original"` image hint. Default: auto-detected (false for GitHub Copilot, which rejects it with a 400). */
|
|
423
429
|
supportsImageDetailOriginal?: boolean;
|
|
430
|
+
/**
|
|
431
|
+
* Whether the Responses endpoint accepts `configuration_update` input items
|
|
432
|
+
* that change `reasoning.effort` mid-conversation while the request-level
|
|
433
|
+
* effort stays pinned for prompt caching (GPT-6 Astra). Default:
|
|
434
|
+
* rule-detected (`true` for `gpt-6-astra` on any host, `false` otherwise).
|
|
435
|
+
* Set `false` for custom `openai-responses` / `openai-codex-responses`
|
|
436
|
+
* endpoints that reject the item type with HTTP 400; effort changes are then
|
|
437
|
+
* sent as the top-level `reasoning.effort`.
|
|
438
|
+
*/
|
|
439
|
+
supportsConfigurationUpdate?: boolean;
|
|
424
440
|
/** Whether streamed reasoning deltas for the same field may repeat the full cumulative text snapshot. Default: false. */
|
|
425
441
|
reasoningDeltasMayBeCumulative?: boolean;
|
|
426
442
|
/** Strip leaked DeepSeek chat-template special tokens from visible content deltas. Default: auto-detected. */
|
|
427
443
|
stripDeepseekSpecialTokens?: boolean;
|
|
428
444
|
/** Heal leaked chat-template/tool-call/thinking markup from visible content deltas. Default: auto-detected. */
|
|
429
445
|
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
|
|
446
|
+
/** Whether this wire may revise already-streamed text (`stream-revision` axis). Unassigned: append-only. */
|
|
447
|
+
streamRevision?: "none" | "possible";
|
|
430
448
|
/** Treat an empty length-finished stream as a context-window error. Default: auto-detected. */
|
|
431
449
|
emptyLengthFinishIsContextError?: boolean;
|
|
432
450
|
/** Normalize tool call ids to OpenAI's 40-character limit. Default: auto-detected. */
|
|
@@ -599,6 +617,8 @@ export interface AnthropicCompat {
|
|
|
599
617
|
export interface BedrockCompat {
|
|
600
618
|
/** Whether this endpoint accepts no checkpoints, automatic caching, or explicit cachePoint blocks. */
|
|
601
619
|
promptCacheMode?: "none" | "automatic" | "explicit";
|
|
620
|
+
/** Whether this wire may revise already-streamed text (`stream-revision` axis). Unassigned: append-only. */
|
|
621
|
+
streamRevision?: "none" | "possible";
|
|
602
622
|
/** Whether explicit cachePoint blocks accept `ttl: "1h"`; omitted TTL means Bedrock's 5-minute default. */
|
|
603
623
|
supportsLongPromptCacheRetention?: boolean;
|
|
604
624
|
/**
|
|
@@ -623,6 +643,8 @@ export interface BedrockCompat {
|
|
|
623
643
|
/** Fully-resolved Bedrock Converse prompt-cache capabilities, materialized once by `buildModel`. */
|
|
624
644
|
export interface ResolvedBedrockCompat {
|
|
625
645
|
promptCacheMode: NonNullable<BedrockCompat["promptCacheMode"]>;
|
|
646
|
+
/** See {@link BedrockCompat.streamRevision}. */
|
|
647
|
+
streamRevision?: BedrockCompat["streamRevision"];
|
|
626
648
|
supportsLongPromptCacheRetention: boolean;
|
|
627
649
|
promptCacheMinimumTokens: number;
|
|
628
650
|
promptCacheMaximumCheckpoints: number;
|
|
@@ -688,6 +710,7 @@ export interface ResolvedOpenAISharedCompat {
|
|
|
688
710
|
filterReasoningHistory: boolean;
|
|
689
711
|
disableReasoningOnForcedToolChoice: boolean;
|
|
690
712
|
disableReasoningOnToolChoice: boolean;
|
|
713
|
+
disableReasoningWithTools?: boolean;
|
|
691
714
|
supportsToolChoice: boolean;
|
|
692
715
|
supportsForcedToolChoice: boolean;
|
|
693
716
|
supportsNamedToolChoice: boolean;
|
|
@@ -705,6 +728,8 @@ export interface ResolvedOpenAISharedCompat {
|
|
|
705
728
|
requiresAssistantContentForToolCalls: boolean;
|
|
706
729
|
stripDeepseekSpecialTokens: boolean;
|
|
707
730
|
streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
|
|
731
|
+
/** See {@link OpenAICompat.streamRevision}. */
|
|
732
|
+
streamRevision?: OpenAICompat["streamRevision"];
|
|
708
733
|
/** See {@link OpenAICompat.streamFirstEventTimeoutMs}. */
|
|
709
734
|
streamFirstEventTimeoutMs?: number;
|
|
710
735
|
reasoningDeltasMayBeCumulative: boolean;
|
|
@@ -765,6 +790,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
|
|
765
790
|
| "filterReasoningHistory"
|
|
766
791
|
| "disableReasoningOnForcedToolChoice"
|
|
767
792
|
| "disableReasoningOnToolChoice"
|
|
793
|
+
| "disableReasoningWithTools"
|
|
768
794
|
| "supportsToolChoice"
|
|
769
795
|
| "supportsForcedToolChoice"
|
|
770
796
|
| "supportsNamedToolChoice"
|
|
@@ -801,10 +827,12 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
|
|
801
827
|
| "toolSchemaFlavor"
|
|
802
828
|
| "streamFirstEventTimeoutMs"
|
|
803
829
|
| "streamIdleTimeoutMs"
|
|
830
|
+
| "streamRevision"
|
|
804
831
|
| "cacheControlFormat"
|
|
805
832
|
| "thinkingKeep"
|
|
806
833
|
| "strictResponsesPairing"
|
|
807
834
|
| "supportsImageDetailOriginal"
|
|
835
|
+
| "supportsConfigurationUpdate"
|
|
808
836
|
| "stripImageInput"
|
|
809
837
|
| "thinkingLoopGuard"
|
|
810
838
|
| "whenThinking"
|