@oh-my-pi/pi-catalog 18.2.0 → 18.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +27 -0
  2. package/dist/types/compat/behavior.d.ts +11 -0
  3. package/dist/types/compat/types.d.ts +21 -0
  4. package/dist/types/discovery/antigravity.d.ts +10 -1
  5. package/dist/types/model-thinking.d.ts +7 -0
  6. package/dist/types/provider-models/openai-compat.d.ts +2 -0
  7. package/dist/types/types.d.ts +26 -1
  8. package/package.json +4 -4
  9. package/src/compat/axes.ts +14 -0
  10. package/src/compat/behavior.ts +22 -2
  11. package/src/compat/cascade.ts +3 -1
  12. package/src/compat/context-window.ts +11 -1
  13. package/src/compat/resolve.ts +42 -16
  14. package/src/compat/rules/README.md +3 -1
  15. package/src/compat/rules/classes/deepseek.kdl +9 -1
  16. package/src/compat/rules/classes/kimi.kdl +6 -0
  17. package/src/compat/rules/providers/alibaba-token-plan.kdl +16 -8
  18. package/src/compat/rules/providers/amazon-bedrock.kdl +30 -0
  19. package/src/compat/rules/providers/azure.kdl +6 -0
  20. package/src/compat/rules/providers/cerebras.kdl +10 -0
  21. package/src/compat/rules/providers/commandcode.kdl +20 -4
  22. package/src/compat/rules/providers/cursor.kdl +32 -0
  23. package/src/compat/rules/providers/deepseek.kdl +5 -5
  24. package/src/compat/rules/providers/google-vertex.kdl +15 -0
  25. package/src/compat/rules/providers/meta.kdl +3 -0
  26. package/src/compat/rules/providers/muse-code.kdl +3 -0
  27. package/src/compat/rules/providers/openrouter.kdl +6 -0
  28. package/src/compat/rules/runtime/behavior.kdl +17 -0
  29. package/src/compat/rules/taxonomy/deepseek.kdl +5 -0
  30. package/src/compat/rules.json +1 -1
  31. package/src/compat/types.ts +23 -0
  32. package/src/discovery/antigravity.ts +80 -43
  33. package/src/identity/bundled.ts +4 -3
  34. package/src/model-cache.ts +115 -37
  35. package/src/model-thinking.ts +10 -7
  36. package/src/models.json +1 -1
  37. package/src/provider-models/bundled-references.ts +4 -3
  38. package/src/provider-models/cache-provider-id.ts +14 -8
  39. package/src/provider-models/ollama.ts +11 -31
  40. package/src/provider-models/openai-compat.ts +94 -24
  41. package/src/types.ts +28 -0
@@ -1,3 +1,4 @@
1
+ import { isBareIdReferenceProvider } from "../compat/behavior";
1
2
  import { isZeroCostXaiOAuthReference } from "../identity/reference";
2
3
  import { getBundledModels, getBundledProviders } from "../models";
3
4
  import type { Api, Model, ModelSpec } from "../types";
@@ -45,9 +46,9 @@ function getGlobalReferences(): Map<string, Model<Api>> {
45
46
  for (const provider of getBundledProviders()) {
46
47
  for (const model of getBundledModels(provider as Parameters<typeof getBundledModels>[0])) {
47
48
  const candidate = model as Model<Api>;
48
- // ClinePass limits, pricing, and reasoning controls are gateway-specific;
49
- // matching them by bare id would contaminate unrelated proxy models.
50
- if (candidate.provider === "cline-pass" || isZeroCostXaiOAuthReference(candidate)) {
49
+ // Gateway-specific metadata must remain provider-local when proxy
50
+ // discovery resolves references by bare model id.
51
+ if (!isBareIdReferenceProvider(candidate.provider) || isZeroCostXaiOAuthReference(candidate)) {
51
52
  continue;
52
53
  }
53
54
  const existing = references.get(candidate.id);
@@ -89,15 +89,21 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
89
89
  }
90
90
  case "litellm": {
91
91
  const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
92
- // rich-v9 unions compat across the management endpoints and keys the
93
- // deployment's `supports_vision` declaration into it, so a warm
94
- // rich-v8 row would keep retracting axes an earlier endpoint reported
95
- // (issue #11982). rich-v8 invalidated rows whose `compatConfig`
96
- // retained a colliding bundled model's provider-specific transport
97
- // (e.g. Fireworks `wireModelIdMode`) before that leak was fixed
98
- // (issue #9938).
99
- return `litellm:rich-v9:${Bun.hash(baseUrl).toString(36)}`;
92
+ // rich-v11 invalidates rows that inherited ClinePass gateway metadata
93
+ // through generic models.dev bare-id enrichment (issue #10932). rich-v10
94
+ // filtered known non-conversational LiteLLM modes, unioned compat across
95
+ // the management endpoints, and keyed the deployment's `supports_vision`
96
+ // declaration into it; earlier versions invalidated rows whose
97
+ // `compatConfig` retained a colliding bundled model's provider-specific
98
+ // transport (e.g. Fireworks `wireModelIdMode`) (issue #9938).
99
+ return `litellm:rich-v11:${Bun.hash(baseUrl).toString(36)}`;
100
100
  }
101
+ case "gmi-cloud":
102
+ case "siliconflow":
103
+ case "siliconflow-cn":
104
+ // models-v1 moves rows enriched before cross-provider reference
105
+ // isolation out of the legacy bare-provider namespaces (#10932).
106
+ return `${providerId}:models-v1`;
101
107
  case "opencode-go":
102
108
  case "opencode-zen": {
103
109
  // v3: gateway-first rows cached before stencil enrichment carry null
@@ -1,9 +1,6 @@
1
1
  import { fetchWithRetry } from "@oh-my-pi/pi-utils";
2
- import { compareRevision, parseRevision } from "../compat/revision";
3
- import { classifyModel } from "../compat/taxonomy";
4
- import { Effort } from "../effort";
5
2
  import type { ModelManagerOptions } from "../model-manager";
6
- import type { FetchImpl, ModelSpec, ThinkingConfig } from "../types";
3
+ import type { FetchImpl, ModelSpec } from "../types";
7
4
  import { discoveryFetch } from "../utils";
8
5
  import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references";
9
6
 
@@ -54,11 +51,6 @@ export function isOllamaCloudOutputCapped(id: string): boolean {
54
51
  return OLLAMA_CLOUD_OUTPUT_CAPPED_BASE_IDS[baseId] === true;
55
52
  }
56
53
 
57
- const OLLAMA_CLOUD_GLM_52_THINKING: ThinkingConfig = {
58
- mode: "effort",
59
- efforts: [Effort.High, Effort.Max],
60
- };
61
-
62
54
  function trimTrailingSlash(value: string): string {
63
55
  return value.endsWith("/") ? value.slice(0, -1) : value;
64
56
  }
@@ -93,27 +85,6 @@ function getContextWindow(modelInfo: Record<string, unknown> | undefined): numbe
93
85
  }
94
86
  }
95
87
 
96
- function getThinkingConfig(modelId: string, capabilities: string[] | undefined): ThinkingConfig | undefined {
97
- if (!capabilities?.includes("thinking")) {
98
- return undefined;
99
- }
100
- const identity = classifyModel("ollama-cloud", modelId, { lenient: true });
101
- const revision = identity.revision === undefined ? undefined : parseRevision(identity.revision);
102
- const floor = parseRevision(identity.family === "flash" ? "5.3" : "5.2");
103
- const isGlmEffortModel =
104
- identity.class === "glm" &&
105
- (identity.family === undefined ||
106
- identity.family === "air" ||
107
- identity.family === "turbo" ||
108
- identity.family === "flash") &&
109
- revision !== undefined &&
110
- floor !== undefined &&
111
- compareRevision(revision, floor) >= 0;
112
- if (isGlmEffortModel) {
113
- return OLLAMA_CLOUD_GLM_52_THINKING;
114
- }
115
- return { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] };
116
- }
117
88
  async function fetchShowMetadata(
118
89
  baseUrl: string,
119
90
  apiKey: string,
@@ -182,7 +153,16 @@ export function ollamaCloudModelManagerOptions(
182
153
  // reference limit, falling back to the historical safe cap otherwise.
183
154
  const contextWindow = discoveredContextWindow ?? 128000;
184
155
  const reasoning = capabilities ? capabilities.includes("thinking") : (reference?.reasoning ?? false);
185
- const thinking = capabilities ? getThinkingConfig(id, capabilities) : reference?.thinking;
156
+ // `/api/show` reports only a boolean `thinking` capability, never a
157
+ // tier vocabulary, so the effort ladder is left to the compat rules
158
+ // (which own every other host's ladder too). Synthesizing one here
159
+ // shadowed the rules: discovery rows carry explicit `thinking`, and
160
+ // `resolveThinkingPolicy` treats explicit metadata as authoritative
161
+ // over the KDL, so a fabricated `minimal..high` ladder overrode the
162
+ // DeepSeek V4 `low/high/max` contract and silently clamped `max`
163
+ // down to `high` (#8334 regression). Models whose ladder the rules
164
+ // do not know still fall back to the generic four-tier default.
165
+ const thinking = capabilities ? undefined : reference?.thinking;
186
166
  const input = capabilities
187
167
  ? capabilities.includes("vision")
188
168
  ? (["text", "image"] as Array<"text" | "image">)
@@ -4,6 +4,8 @@ import { toClinePassPublicModelId } from "../cline-pass-model-id";
4
4
  import {
5
5
  apiRouteExactModelIds,
6
6
  apiRouteFor,
7
+ isBareIdReferenceProvider,
8
+ isExcludedDiscoveryMode,
7
9
  isExcludedModel,
8
10
  isLikelyOpenAIResponsesId,
9
11
  modelLimitsFor,
@@ -579,6 +581,7 @@ type OpenAICompatibleModelManagerBuilderOptions<TApi extends Api> = {
579
581
  dynamicModelsAuthoritative?: true;
580
582
  requireApiKey?: true;
581
583
  dropCachedModelIdsOnStaticMismatch?: readonly string[];
584
+ cacheProviderId?: string;
582
585
  filterModel?: (
583
586
  entry: OpenAICompatibleModelRecord,
584
587
  model: ModelSpec<TApi>,
@@ -600,6 +603,7 @@ function createOpenAICompatibleModelManagerOptions<TApi extends Api>(
600
603
  const filterModel = options.filterModel;
601
604
  return {
602
605
  providerId: options.providerId,
606
+ ...(options.cacheProviderId && { cacheProviderId: options.cacheProviderId }),
603
607
  ...(options.dynamicModelsAuthoritative && { dynamicModelsAuthoritative: true }),
604
608
  ...(options.dropCachedModelIdsOnStaticMismatch && {
605
609
  dropCachedModelIdsOnStaticMismatch: options.dropCachedModelIdsOnStaticMismatch,
@@ -982,6 +986,7 @@ export function gmiCloudModelManagerOptions(
982
986
  api: "openai-completions",
983
987
  providerId: "gmi-cloud",
984
988
  defaultBaseUrl: GMI_CLOUD_BASE_URL,
989
+ cacheProviderId: resolveModelCacheProviderId("gmi-cloud"),
985
990
  config,
986
991
  requireApiKey: true,
987
992
  mapModel: mapGmiCloudModel,
@@ -1201,6 +1206,13 @@ function mapDeepinfraModel(
1201
1206
  return null;
1202
1207
  }
1203
1208
  const pricing = isRecord(metadata.pricing) ? metadata.pricing : {};
1209
+ // `metadata.discount` is a promotional fraction in [0, 1): DeepInfra bills
1210
+ // `pricing * (1 - discount)` (verified against the site — GLM-5.2 lists
1211
+ // input 0.75 with discount 0.35 and charges 0.4875), while `pricing.*`
1212
+ // stays at list price. Fold it into the rate card so cost reporting matches
1213
+ // what the user is actually billed. Values outside (0, 1) are ignored.
1214
+ const discount = toNumber(metadata.discount);
1215
+ const discountMultiplier = discount !== undefined && discount > 0 && discount < 1 ? 1 - discount : 1;
1204
1216
  // `reasoning_effort` marks models whose effort dial is advertised. The
1205
1217
  // parameter itself is validated and accepted platform-wide on DeepInfra
1206
1218
  // (verified: 200 on effort-tagged, reasoning-only, and plain-chat models;
@@ -1243,9 +1255,9 @@ function mapDeepinfraModel(
1243
1255
  ...(thinking ? { thinking } : {}),
1244
1256
  input: tags.includes("vision") || tags.includes("vlm") ? ["text", "image"] : ["text"],
1245
1257
  cost: {
1246
- input: toPositiveNumber(pricing.input_tokens, 0),
1247
- output: toPositiveNumber(pricing.output_tokens, 0),
1248
- cacheRead: toPositiveNumber(pricing.cache_read_tokens, 0),
1258
+ input: toPositiveNumber(pricing.input_tokens, 0) * discountMultiplier,
1259
+ output: toPositiveNumber(pricing.output_tokens, 0) * discountMultiplier,
1260
+ cacheRead: toPositiveNumber(pricing.cache_read_tokens, 0) * discountMultiplier,
1249
1261
  cacheWrite: 0,
1250
1262
  },
1251
1263
  contextWindow,
@@ -1711,6 +1723,7 @@ function createSiliconFlowModelManagerOptions(
1711
1723
  const baseUrl = config?.baseUrl ?? defaultBaseUrl;
1712
1724
  return {
1713
1725
  providerId,
1726
+ cacheProviderId: resolveModelCacheProviderId(providerId),
1714
1727
  dynamicModelsAuthoritative: true,
1715
1728
  ...(apiKey && {
1716
1729
  fetchDynamicModels: async () => {
@@ -2134,6 +2147,9 @@ function createModelsDevReferenceMap<TApi extends Api>(
2134
2147
  const references = new Map<string, ModelSpec<TApi>>();
2135
2148
  for (const model of models) {
2136
2149
  const candidate = model as ModelSpec<TApi>;
2150
+ if (!isBareIdReferenceProvider(candidate.provider)) {
2151
+ continue;
2152
+ }
2137
2153
  const existing = references.get(candidate.id);
2138
2154
  if (!existing) {
2139
2155
  references.set(candidate.id, candidate);
@@ -3866,10 +3882,17 @@ export interface BasetenModelManagerConfig {
3866
3882
  fetch?: FetchImpl;
3867
3883
  }
3868
3884
 
3869
- // A previous version of OMP shipped this model without reasoning levels. We've
3870
- // since fixed that. This const lets us bust the cache so that users on that
3871
- // version of OMP pick up the reasoning levels immediately.
3872
- const BASETEN_CACHE_MIGRATION_MODEL_IDS = ["zai-org/GLM-5.3", "zai-org/GLM-5.3-Flash"] as const;
3885
+ // A previous version of OMP shipped these models without reasoning levels.
3886
+ // We've since fixed that (V4-generation whitelist). This const lets us bust
3887
+ // the cache so that users on that version of OMP pick up the reasoning levels
3888
+ // immediately.
3889
+ const BASETEN_CACHE_MIGRATION_MODEL_IDS = [
3890
+ "zai-org/GLM-5.3",
3891
+ "zai-org/GLM-5.3-Flash",
3892
+ "deepseek-ai/DeepSeek-V4-Flash-0731",
3893
+ "deepseek-ai/DeepSeek-V4.1-Flash",
3894
+ "deepseek-ai/DeepSeek-V4-Pro-0813",
3895
+ ] as const;
3873
3896
 
3874
3897
  export function basetenModelManagerOptions(
3875
3898
  config?: BasetenModelManagerConfig,
@@ -3900,7 +3923,7 @@ export function basetenModelManagerOptions(
3900
3923
  (identity.class === "kimi" && identity.family === "k3") ||
3901
3924
  isGlmReasoningIdentity("baseten", defaults.id, "5.2") ||
3902
3925
  defaults.id === "openai/gpt-oss-120b" ||
3903
- defaults.id === "deepseek-ai/DeepSeek-V4-Pro";
3926
+ isDeepseekV4Generation("baseten", defaults.id);
3904
3927
  const reasoning =
3905
3928
  isSupportedBasetenReasoningModel &&
3906
3929
  (features.includes("reasoning") || features.includes("reasoning_effort"));
@@ -4678,7 +4701,11 @@ type LiteLLMRichEndpointFailure = {
4678
4701
  error?: unknown;
4679
4702
  };
4680
4703
  type LiteLLMRichEndpointResult<TApi extends Api> =
4681
- | { models: LiteLLMRichEndpointModel<TApi>[]; incompleteVisionMetadata: boolean }
4704
+ | {
4705
+ models: LiteLLMRichEndpointModel<TApi>[];
4706
+ excludedModelIds: ReadonlySet<string>;
4707
+ incompleteVisionMetadata: boolean;
4708
+ }
4682
4709
  | { failure: LiteLLMRichEndpointFailure };
4683
4710
 
4684
4711
  const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const;
@@ -4707,6 +4734,11 @@ function warnLiteLLMMetadataFallback(managementBaseUrl: string, failure: LiteLLM
4707
4734
  });
4708
4735
  }
4709
4736
 
4737
+ /** Exclude only known non-conversational modes; unknown and non-string modes remain selectable for aliases. */
4738
+ export function isSelectableLiteLLMModelMode(mode: unknown): boolean {
4739
+ return typeof mode !== "string" || !isExcludedDiscoveryMode("litellm", mode);
4740
+ }
4741
+
4710
4742
  export function normalizeLiteLLMManagementBaseUrl(baseUrl: string): string {
4711
4743
  const trimmed = baseUrl.trim().replace(/\/+$/g, "");
4712
4744
  if (!trimmed) {
@@ -4747,7 +4779,10 @@ function mapLiteLLMOpenAICompatibleModel(
4747
4779
  entry: OpenAICompatibleModelRecord,
4748
4780
  defaults: ModelSpec<Api>,
4749
4781
  reference: ModelSpec<Api> | undefined,
4750
- ): ModelSpec<Api> {
4782
+ ): ModelSpec<Api> | null {
4783
+ if (!isSelectableLiteLLMModelMode(entry.mode)) {
4784
+ return null;
4785
+ }
4751
4786
  const model = mapWithBundledReference(entry, defaults, reference);
4752
4787
  return {
4753
4788
  ...model,
@@ -4943,7 +4978,10 @@ function mapLiteLLMRichEntry<TApi extends Api>(
4943
4978
  options: FetchLiteLLMRichModelsOptions<TApi>,
4944
4979
  runtimeBaseUrl: string,
4945
4980
  ): ModelSpec<TApi> | null {
4946
- if (isLiteLLMUnusableSentinelPlaceholder(entry)) {
4981
+ if (
4982
+ !isSelectableLiteLLMModelMode(getLiteLLMMetadataValue(entry, "mode")) ||
4983
+ isLiteLLMUnusableSentinelPlaceholder(entry)
4984
+ ) {
4947
4985
  return null;
4948
4986
  }
4949
4987
  const id = getLiteLLMRichModelId(entry);
@@ -5151,7 +5189,22 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
5151
5189
  return null;
5152
5190
  }
5153
5191
  const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
5192
+ const excludedModelIds = new Set<string>();
5154
5193
  for (const entry of entries) {
5194
+ if (isLiteLLMUnusableSentinelPlaceholder(entry)) {
5195
+ continue;
5196
+ }
5197
+ const modelId = getLiteLLMRichModelId(entry);
5198
+ if (!isSelectableLiteLLMModelMode(getLiteLLMMetadataValue(entry, "mode"))) {
5199
+ if (modelId) {
5200
+ excludedModelIds.add(modelId);
5201
+ deduped.delete(modelId);
5202
+ }
5203
+ continue;
5204
+ }
5205
+ if (modelId && excludedModelIds.has(modelId)) {
5206
+ continue;
5207
+ }
5155
5208
  const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl);
5156
5209
  if (model) {
5157
5210
  const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision");
@@ -5177,12 +5230,13 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
5177
5230
  deduped.set(model.id, existing ? mergeLiteLLMRichEndpointModels(existing, next) : next);
5178
5231
  }
5179
5232
  }
5180
- if (deduped.size === 0) {
5233
+ if (deduped.size === 0 && excludedModelIds.size === 0) {
5181
5234
  return null;
5182
5235
  }
5183
5236
  const models = Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id));
5184
5237
  return {
5185
5238
  models,
5239
+ excludedModelIds,
5186
5240
  incompleteVisionMetadata: models.some(entry => entry.supportsVision !== true && entry.supportsVision !== false),
5187
5241
  };
5188
5242
  }
@@ -5197,6 +5251,7 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
5197
5251
  }
5198
5252
  const fetchModels = async (signal?: AbortSignal): Promise<ModelSpec<TApi>[] | null> => {
5199
5253
  const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
5254
+ const excludedModelIds = new Set<string>();
5200
5255
  let metadataFailure: LiteLLMRichEndpointFailure | undefined;
5201
5256
  for (const endpoint of LITELLM_RICH_ENDPOINTS) {
5202
5257
  const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
@@ -5218,8 +5273,15 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
5218
5273
  }
5219
5274
  continue;
5220
5275
  }
5276
+ for (const modelId of result.excludedModelIds) {
5277
+ excludedModelIds.add(modelId);
5278
+ deduped.delete(modelId);
5279
+ }
5221
5280
  const hadPriorModels = deduped.size > 0;
5222
5281
  for (const next of result.models) {
5282
+ if (excludedModelIds.has(next.model.id)) {
5283
+ continue;
5284
+ }
5223
5285
  const existing = deduped.get(next.model.id);
5224
5286
  if (!existing) {
5225
5287
  if (!hadPriorModels) {
@@ -5229,6 +5291,9 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
5229
5291
  }
5230
5292
  deduped.set(next.model.id, mergeLiteLLMRichEndpointModels(existing, next));
5231
5293
  }
5294
+ if (deduped.size === 0) {
5295
+ continue;
5296
+ }
5232
5297
  let needsMoreMetadata = false;
5233
5298
  for (const entry of deduped.values()) {
5234
5299
  if (
@@ -5249,6 +5314,9 @@ async function fetchLiteLLMRichModelsInternal<TApi extends Api>(
5249
5314
  }
5250
5315
  }
5251
5316
  if (deduped.size === 0) {
5317
+ if (excludedModelIds.size > 0) {
5318
+ return [];
5319
+ }
5252
5320
  if (metadataFailure) {
5253
5321
  warnLiteLLMMetadataFallback(managementBaseUrl, metadataFailure);
5254
5322
  }
@@ -5275,17 +5343,18 @@ export function litellmModelManagerOptions(config?: LiteLLMModelManagerConfig):
5275
5343
  const baseUrl = config?.baseUrl ?? getDefaultModelDiscoveryBaseUrl("litellm")!;
5276
5344
  return {
5277
5345
  providerId: "litellm",
5278
- // rich-v9 keys the deployment's `supports_vision` declaration into the
5279
- // cached compat and unions compat across management endpoints instead of
5280
- // letting a later one retract what an earlier one reported (issue
5281
- // #11982). rich-v8 invalidated rows whose `compatConfig` retained a
5282
- // colliding bundled model's provider-specific transport (e.g. Fireworks
5283
- // `wireModelIdMode`) before that leak was fixed. Earlier versions added
5284
- // bundled reference fallback, moved OpenAI models to Responses, continued
5285
- // past incomplete vision/API metadata and endpoints omitting cache
5286
- // pricing, stripped reseller usage suffixes, filtered placeholder rows,
5287
- // and mapped rich pricing. Bump the version whenever these mappers change,
5288
- // or warm authoritative caches keep serving pre-change rows for the full TTL.
5346
+ // rich-v11 invalidates rows that inherited ClinePass gateway metadata
5347
+ // through generic models.dev bare-id enrichment (issue #10932). rich-v10
5348
+ // filtered known non-conversational LiteLLM modes, keyed the deployment's
5349
+ // `supports_vision` declaration into cached compat, and unioned compat
5350
+ // across management endpoints instead of letting a later endpoint retract
5351
+ // what an earlier one reported (issue #11982). Earlier versions fixed
5352
+ // provider-specific transport leakage, added bundled reference fallback,
5353
+ // moved OpenAI models to Responses, continued past incomplete vision/API
5354
+ // metadata and endpoints omitting cache pricing, stripped reseller usage
5355
+ // suffixes, filtered placeholder rows, and mapped rich pricing. Bump the
5356
+ // version whenever these mappers change, or warm authoritative caches keep
5357
+ // serving pre-change rows for the full TTL.
5289
5358
  cacheProviderId: resolveModelCacheProviderId("litellm", { baseUrl }),
5290
5359
  // litellm is a local-only proxy and is never bundled in models.json (that
5291
5360
  // would leak the machine's localhost catalog). Prefer the proxy's richer
@@ -5304,7 +5373,7 @@ export function litellmModelManagerOptions(config?: LiteLLMModelManagerConfig):
5304
5373
  resolveApi: resolveLiteLLMApi,
5305
5374
  timeoutMs: 10_000,
5306
5375
  });
5307
- if (richModels && richModels.length > 0) {
5376
+ if (richModels !== null) {
5308
5377
  return richModels;
5309
5378
  }
5310
5379
  return fetchOpenAICompatibleModels<Api>({
@@ -6308,6 +6377,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
6308
6377
  return {
6309
6378
  ...model,
6310
6379
  id,
6380
+ name: id,
6311
6381
  thinking: model.reasoning ? buildClinePassThinking(raw, model) : undefined,
6312
6382
  };
6313
6383
  },
package/src/types.ts CHANGED
@@ -327,6 +327,12 @@ export interface OpenAICompat {
327
327
  * Default: auto-detected (DeepSeek reasoning models).
328
328
  */
329
329
  disableReasoningOnToolChoice?: boolean;
330
+ /**
331
+ * Disable reasoning whenever the request advertises function tools.
332
+ * Use for model surfaces that reject every tools-plus-reasoning combination.
333
+ * Default: false.
334
+ */
335
+ disableReasoningWithTools?: boolean;
330
336
  /** OpenRouter-specific routing preferences. Only used when baseUrl points to OpenRouter. */
331
337
  openRouterRouting?: OpenRouterRouting;
332
338
  /** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */
@@ -421,12 +427,24 @@ export interface OpenAICompat {
421
427
  strictResponsesPairing?: boolean;
422
428
  /** Whether the Responses API accepts the `detail: "original"` image hint. Default: auto-detected (false for GitHub Copilot, which rejects it with a 400). */
423
429
  supportsImageDetailOriginal?: boolean;
430
+ /**
431
+ * Whether the Responses endpoint accepts `configuration_update` input items
432
+ * that change `reasoning.effort` mid-conversation while the request-level
433
+ * effort stays pinned for prompt caching (GPT-6 Astra). Default:
434
+ * rule-detected (`true` for `gpt-6-astra` on any host, `false` otherwise).
435
+ * Set `false` for custom `openai-responses` / `openai-codex-responses`
436
+ * endpoints that reject the item type with HTTP 400; effort changes are then
437
+ * sent as the top-level `reasoning.effort`.
438
+ */
439
+ supportsConfigurationUpdate?: boolean;
424
440
  /** Whether streamed reasoning deltas for the same field may repeat the full cumulative text snapshot. Default: false. */
425
441
  reasoningDeltasMayBeCumulative?: boolean;
426
442
  /** Strip leaked DeepSeek chat-template special tokens from visible content deltas. Default: auto-detected. */
427
443
  stripDeepseekSpecialTokens?: boolean;
428
444
  /** Heal leaked chat-template/tool-call/thinking markup from visible content deltas. Default: auto-detected. */
429
445
  streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
446
+ /** Whether this wire may revise already-streamed text (`stream-revision` axis). Unassigned: append-only. */
447
+ streamRevision?: "none" | "possible";
430
448
  /** Treat an empty length-finished stream as a context-window error. Default: auto-detected. */
431
449
  emptyLengthFinishIsContextError?: boolean;
432
450
  /** Normalize tool call ids to OpenAI's 40-character limit. Default: auto-detected. */
@@ -599,6 +617,8 @@ export interface AnthropicCompat {
599
617
  export interface BedrockCompat {
600
618
  /** Whether this endpoint accepts no checkpoints, automatic caching, or explicit cachePoint blocks. */
601
619
  promptCacheMode?: "none" | "automatic" | "explicit";
620
+ /** Whether this wire may revise already-streamed text (`stream-revision` axis). Unassigned: append-only. */
621
+ streamRevision?: "none" | "possible";
602
622
  /** Whether explicit cachePoint blocks accept `ttl: "1h"`; omitted TTL means Bedrock's 5-minute default. */
603
623
  supportsLongPromptCacheRetention?: boolean;
604
624
  /**
@@ -623,6 +643,8 @@ export interface BedrockCompat {
623
643
  /** Fully-resolved Bedrock Converse prompt-cache capabilities, materialized once by `buildModel`. */
624
644
  export interface ResolvedBedrockCompat {
625
645
  promptCacheMode: NonNullable<BedrockCompat["promptCacheMode"]>;
646
+ /** See {@link BedrockCompat.streamRevision}. */
647
+ streamRevision?: BedrockCompat["streamRevision"];
626
648
  supportsLongPromptCacheRetention: boolean;
627
649
  promptCacheMinimumTokens: number;
628
650
  promptCacheMaximumCheckpoints: number;
@@ -688,6 +710,7 @@ export interface ResolvedOpenAISharedCompat {
688
710
  filterReasoningHistory: boolean;
689
711
  disableReasoningOnForcedToolChoice: boolean;
690
712
  disableReasoningOnToolChoice: boolean;
713
+ disableReasoningWithTools?: boolean;
691
714
  supportsToolChoice: boolean;
692
715
  supportsForcedToolChoice: boolean;
693
716
  supportsNamedToolChoice: boolean;
@@ -705,6 +728,8 @@ export interface ResolvedOpenAISharedCompat {
705
728
  requiresAssistantContentForToolCalls: boolean;
706
729
  stripDeepseekSpecialTokens: boolean;
707
730
  streamMarkupHealingPattern?: OpenAIStreamMarkupHealingPattern;
731
+ /** See {@link OpenAICompat.streamRevision}. */
732
+ streamRevision?: OpenAICompat["streamRevision"];
708
733
  /** See {@link OpenAICompat.streamFirstEventTimeoutMs}. */
709
734
  streamFirstEventTimeoutMs?: number;
710
735
  reasoningDeltasMayBeCumulative: boolean;
@@ -765,6 +790,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
765
790
  | "filterReasoningHistory"
766
791
  | "disableReasoningOnForcedToolChoice"
767
792
  | "disableReasoningOnToolChoice"
793
+ | "disableReasoningWithTools"
768
794
  | "supportsToolChoice"
769
795
  | "supportsForcedToolChoice"
770
796
  | "supportsNamedToolChoice"
@@ -801,10 +827,12 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
801
827
  | "toolSchemaFlavor"
802
828
  | "streamFirstEventTimeoutMs"
803
829
  | "streamIdleTimeoutMs"
830
+ | "streamRevision"
804
831
  | "cacheControlFormat"
805
832
  | "thinkingKeep"
806
833
  | "strictResponsesPairing"
807
834
  | "supportsImageDetailOriginal"
835
+ | "supportsConfigurationUpdate"
808
836
  | "stripImageInput"
809
837
  | "thinkingLoopGuard"
810
838
  | "whenThinking"