@oh-my-pi/pi-catalog 17.3.7 → 17.3.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/dist/types/discovery/cursor-gen/agent_pb.d.ts +88 -3
- package/dist/types/identity/family.d.ts +13 -1
- package/dist/types/model-thinking.d.ts +6 -2
- package/dist/types/provider-models/descriptors.d.ts +1 -0
- package/dist/types/types.d.ts +12 -1
- package/dist/types/variant-collapse.d.ts +2 -0
- package/package.json +4 -4
- package/src/compat/openai.ts +27 -1
- package/src/discovery/codex.ts +118 -24
- package/src/discovery/cursor-gen/agent_pb.ts +614 -519
- package/src/discovery/cursor.ts +14 -1
- package/src/identity/family.ts +21 -1
- package/src/model-cache.ts +67 -6
- package/src/model-thinking.ts +38 -6
- package/src/models.json +3 -7
- package/src/provider-models/cache-provider-id.ts +3 -1
- package/src/provider-models/descriptors.ts +1 -0
- package/src/provider-models/openai-compat.ts +83 -3
- package/src/types.ts +12 -0
- package/src/variant-collapse.ts +67 -34
package/src/discovery/cursor.ts
CHANGED
|
@@ -28,6 +28,15 @@ const CURSOR_MAX_MODE_1M_ID_PATTERN = /claude|gemini/;
|
|
|
28
28
|
/** Kimi's official bare K3 id (`k3`, `kimi/k3`); `k3-256k` is the 256k SKU and stays out. */
|
|
29
29
|
const CURSOR_KIMI_K3_BARE_ID_PATTERN = /(^|\/)k3$/i;
|
|
30
30
|
|
|
31
|
+
/**
|
|
32
|
+
* Versioned Cursor Grok ids (`cursor-grok-4.5`, `cursor-grok-4.6-high`) are
|
|
33
|
+
* reasoning models whose effort is carried in the per-tier sibling id.
|
|
34
|
+
* `GetUsableModels` ships no `thinkingDetails` and the bundled references read
|
|
35
|
+
* `reasoning: false`, so classification falls back to the id. The non-reasoning
|
|
36
|
+
* `grok-code-*` coding models lack the version digit and stay out.
|
|
37
|
+
*/
|
|
38
|
+
const CURSOR_GROK_REASONING_ID_PATTERN = /^cursor-grok-\d/i;
|
|
39
|
+
|
|
31
40
|
/**
|
|
32
41
|
* Model-id families whose native catalogs (anthropic, openai/openai-codex,
|
|
33
42
|
* google) are multimodal. Cursor-only or text-only families (`composer-*`,
|
|
@@ -297,7 +306,11 @@ function normalizeCursorModel(
|
|
|
297
306
|
|
|
298
307
|
const name = pickModelDisplayName(details, id);
|
|
299
308
|
const reference = references.get(id);
|
|
300
|
-
const reasoning =
|
|
309
|
+
const reasoning =
|
|
310
|
+
isKimiK3ModelId(id) ||
|
|
311
|
+
CURSOR_GROK_REASONING_ID_PATTERN.test(id) ||
|
|
312
|
+
Boolean(details.thinkingDetails) ||
|
|
313
|
+
reference?.reasoning === true;
|
|
301
314
|
|
|
302
315
|
if (reference) {
|
|
303
316
|
return {
|
package/src/identity/family.ts
CHANGED
|
@@ -73,6 +73,25 @@ export const isQwenModelId = memo((modelId: string): boolean => {
|
|
|
73
73
|
return modelId.toLowerCase().includes("qwen");
|
|
74
74
|
});
|
|
75
75
|
|
|
76
|
+
/**
|
|
77
|
+
* Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF
|
|
78
|
+
* names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking
|
|
79
|
+
* depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`,
|
|
80
|
+
* template default `xhigh`; thinking itself cannot be disabled). Compared
|
|
81
|
+
* component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are
|
|
82
|
+
* excluded — Dashscope drives them through OpenAI-style `reasoning_effort`
|
|
83
|
+
* with curated compat. The trailing guard rejects parameter-count lookalikes
|
|
84
|
+
* (`qwen-3.8b`) without breaking `qwen3.8-27b`.
|
|
85
|
+
*/
|
|
86
|
+
export const isQwen38PlusTemplateEffortModelId = memo((modelId: string): boolean => {
|
|
87
|
+
const match = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId);
|
|
88
|
+
if (!match) return false;
|
|
89
|
+
const major = Number.parseInt(match[1], 10);
|
|
90
|
+
const minor = Number.parseInt(match[2], 10);
|
|
91
|
+
if (major < 3 || (major === 3 && minor < 8)) return false;
|
|
92
|
+
return !/^-max(?:$|[-.:])/i.test(modelId.slice(match.index + match[0].length));
|
|
93
|
+
});
|
|
94
|
+
|
|
76
95
|
/** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */
|
|
77
96
|
export const isGemmaModelId = memo((modelId: string): boolean => {
|
|
78
97
|
return /(^|\/)gemma[-.]?\d/i.test(modelId);
|
|
@@ -121,7 +140,8 @@ const GROK_EFFORT_CAPABLE_PREFIXES = [
|
|
|
121
140
|
/**
|
|
122
141
|
* Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
|
|
123
142
|
* (e.g. `grok-build`, `grok-4.20-0309-reasoning`) think natively but reject the
|
|
124
|
-
* param, so callers must omit reasoning effort for them.
|
|
143
|
+
* param, so callers must omit reasoning effort for them. `grok-4.6` accepts
|
|
144
|
+
* `low`/`medium`/`high`/`xhigh` and 400s on `max`.
|
|
125
145
|
*/
|
|
126
146
|
export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
|
|
127
147
|
const bare = bareModelId(modelId).trim().toLowerCase();
|
package/src/model-cache.ts
CHANGED
|
@@ -3,7 +3,8 @@
|
|
|
3
3
|
* Replaces per-provider JSON files with a single cache.db.
|
|
4
4
|
*/
|
|
5
5
|
import { Database } from "bun:sqlite";
|
|
6
|
-
import {
|
|
6
|
+
import { renameSync } from "node:fs";
|
|
7
|
+
import { getModelDbPath, isEnoent, isSqliteCorruptionError, logger } from "@oh-my-pi/pi-utils";
|
|
7
8
|
import type { Api, Model, ModelSpec } from "./types";
|
|
8
9
|
|
|
9
10
|
// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
|
|
@@ -91,13 +92,14 @@ function openDb(resolvedPath: string): Database {
|
|
|
91
92
|
return db;
|
|
92
93
|
}
|
|
93
94
|
|
|
94
|
-
function getSharedDb(): Database {
|
|
95
|
-
const resolvedPath = getModelDbPath();
|
|
95
|
+
function getSharedDb(resolvedPath: string): Database {
|
|
96
96
|
if (sharedDb && sharedDbPath === resolvedPath) {
|
|
97
97
|
return sharedDb;
|
|
98
98
|
}
|
|
99
99
|
if (sharedDb) {
|
|
100
100
|
sharedDb.close();
|
|
101
|
+
sharedDb = null;
|
|
102
|
+
sharedDbPath = null;
|
|
101
103
|
}
|
|
102
104
|
const db = openDb(resolvedPath);
|
|
103
105
|
sharedDb = db;
|
|
@@ -105,9 +107,9 @@ function getSharedDb(): Database {
|
|
|
105
107
|
return db;
|
|
106
108
|
}
|
|
107
109
|
|
|
108
|
-
function
|
|
109
|
-
if (
|
|
110
|
-
const db = openDb(
|
|
110
|
+
function runModelCacheDb<T>(resolvedPath: string, shared: boolean, useDb: (db: Database) => T): T {
|
|
111
|
+
if (shared) return useDb(getSharedDb(resolvedPath));
|
|
112
|
+
const db = openDb(resolvedPath);
|
|
111
113
|
try {
|
|
112
114
|
return useDb(db);
|
|
113
115
|
} finally {
|
|
@@ -115,6 +117,65 @@ function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) =
|
|
|
115
117
|
}
|
|
116
118
|
}
|
|
117
119
|
|
|
120
|
+
// Paths already reported corrupt this process: the first unrecoverable failure
|
|
121
|
+
// is logged at `error`, later heals at `debug`, so a dying disk cannot spam.
|
|
122
|
+
const reportedCorruptPaths = new Set<string>();
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Move a physically corrupt `models.db` (plus its `-wal`/`-shm` sidecars) aside
|
|
126
|
+
* so {@link openDb} can recreate a fresh cache at the original path. Renames are
|
|
127
|
+
* best-effort: a vanished sidecar (already healed by a peer process) is fine,
|
|
128
|
+
* and any other rename failure is left for {@link openDb} to surface.
|
|
129
|
+
*/
|
|
130
|
+
function quarantineCorruptModelCache(resolvedPath: string): void {
|
|
131
|
+
const stamp = Date.now();
|
|
132
|
+
for (const suffix of ["", "-wal", "-shm"]) {
|
|
133
|
+
try {
|
|
134
|
+
renameSync(`${resolvedPath}${suffix}`, `${resolvedPath}.corrupt-${stamp}${suffix}`);
|
|
135
|
+
} catch (err) {
|
|
136
|
+
if (!isEnoent(err)) {
|
|
137
|
+
logger.debug("model cache: could not quarantine corrupt file", { path: `${resolvedPath}${suffix}` });
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Recover from unrecoverable `models.db` corruption: drop the cached handle,
|
|
145
|
+
* quarantine the broken files, and let the next open recreate the cache. A
|
|
146
|
+
* corrupt cache would otherwise be re-queried on every read/write forever,
|
|
147
|
+
* permanently masking a successful live catalog (issue #8867). Only
|
|
148
|
+
* {@link isSqliteCorruptionError} codes reach here; BUSY/permission errors keep
|
|
149
|
+
* their existing best-effort paths.
|
|
150
|
+
*/
|
|
151
|
+
function healCorruptModelCache(resolvedPath: string, shared: boolean, err: unknown): void {
|
|
152
|
+
if (shared && sharedDb) {
|
|
153
|
+
sharedDb.close();
|
|
154
|
+
sharedDb = null;
|
|
155
|
+
sharedDbPath = null;
|
|
156
|
+
}
|
|
157
|
+
quarantineCorruptModelCache(resolvedPath);
|
|
158
|
+
const code = err && typeof err === "object" && "code" in err ? err.code : undefined;
|
|
159
|
+
if (reportedCorruptPaths.has(resolvedPath)) {
|
|
160
|
+
logger.debug("model cache: re-healed corrupt database", { path: resolvedPath, code });
|
|
161
|
+
} else {
|
|
162
|
+
reportedCorruptPaths.add(resolvedPath);
|
|
163
|
+
logger.error("model cache corrupt; quarantined and recreated a fresh cache", { path: resolvedPath, code });
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
|
|
168
|
+
const resolvedPath = dbPath ?? getModelDbPath();
|
|
169
|
+
const shared = dbPath === undefined;
|
|
170
|
+
try {
|
|
171
|
+
return runModelCacheDb(resolvedPath, shared, useDb);
|
|
172
|
+
} catch (err) {
|
|
173
|
+
if (!isSqliteCorruptionError(err)) throw err;
|
|
174
|
+
healCorruptModelCache(resolvedPath, shared, err);
|
|
175
|
+
return runModelCacheDb(resolvedPath, shared, useDb);
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
118
179
|
function migrateCacheSchema(db: Database): void {
|
|
119
180
|
const stmt = db.prepare("PRAGMA table_info(model_cache)");
|
|
120
181
|
try {
|
package/src/model-thinking.ts
CHANGED
|
@@ -71,6 +71,11 @@ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Hi
|
|
|
71
71
|
const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
|
|
72
72
|
/** OpenRouter's DeepSeek route accepts only `high`. */
|
|
73
73
|
const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High];
|
|
74
|
+
/**
|
|
75
|
+
* Qwen 3.8+ open-weight chat template: prompt-steered `reasoning_effort`
|
|
76
|
+
* kwarg with exactly three wire tiers (template default is `xhigh`).
|
|
77
|
+
*/
|
|
78
|
+
const QWEN38_TEMPLATE_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.XHigh];
|
|
74
79
|
/**
|
|
75
80
|
* Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models
|
|
76
81
|
* with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the
|
|
@@ -179,7 +184,9 @@ function fillThinkingWireDefaults<TApi extends Api>(
|
|
|
179
184
|
thinking.supportsDisplay === undefined &&
|
|
180
185
|
(spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
|
|
181
186
|
supportsAdaptiveThinkingDisplay(spec.id);
|
|
182
|
-
const needsRequiresEffort =
|
|
187
|
+
const needsRequiresEffort =
|
|
188
|
+
thinking.requiresEffort === undefined &&
|
|
189
|
+
(impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat));
|
|
183
190
|
const needsDefaultLevel =
|
|
184
191
|
thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
|
|
185
192
|
if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
|
|
@@ -232,7 +239,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
|
|
|
232
239
|
) {
|
|
233
240
|
config.supportsDisplay = true;
|
|
234
241
|
}
|
|
235
|
-
if (impliesMandatoryReasoning(parsed, spec.id)) {
|
|
242
|
+
if (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)) {
|
|
236
243
|
config.requiresEffort = true;
|
|
237
244
|
}
|
|
238
245
|
return config;
|
|
@@ -376,6 +383,13 @@ function getModelDefinedEfforts<TApi extends Api>(
|
|
|
376
383
|
if (spec.provider === "ollama") {
|
|
377
384
|
return OLLAMA_REASONING_EFFORTS;
|
|
378
385
|
}
|
|
386
|
+
// Qwen 3.8+ served through a local llama.cpp-style backend: the chat
|
|
387
|
+
// template's prompt-steered `reasoning_effort` kwarg accepts exactly
|
|
388
|
+
// low/medium/xhigh (and thinking cannot be turned off — the official 3.8
|
|
389
|
+
// template raises on `enable_thinking: false`, hence requiresEffort).
|
|
390
|
+
if (isOpenAICompatReasoningApi(spec.api) && isQwenTemplateReasoningEffortCompat(compat)) {
|
|
391
|
+
return QWEN38_TEMPLATE_REASONING_EFFORTS;
|
|
392
|
+
}
|
|
379
393
|
if (
|
|
380
394
|
(isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
|
|
381
395
|
isDeepseekReasoningModel(spec)
|
|
@@ -483,6 +497,12 @@ function isOpenRouterThinkingFormat(compat: CompatOf<Api>): boolean {
|
|
|
483
497
|
function isZaiThinkingFormat(compat: CompatOf<Api>): boolean {
|
|
484
498
|
return compat !== undefined && "thinkingFormat" in compat && compat.thinkingFormat === "zai";
|
|
485
499
|
}
|
|
500
|
+
/** Resolved-compat gate for the Qwen 3.8+ local template `reasoning_effort` dialect. */
|
|
501
|
+
function isQwenTemplateReasoningEffortCompat(compat: CompatOf<Api>): boolean {
|
|
502
|
+
return (
|
|
503
|
+
compat !== undefined && "qwenTemplateReasoningEffort" in compat && compat.qwenTemplateReasoningEffort === true
|
|
504
|
+
);
|
|
505
|
+
}
|
|
486
506
|
|
|
487
507
|
function inferDetectedEffortMap<TApi extends Api>(
|
|
488
508
|
spec: ModelSpec<TApi>,
|
|
@@ -796,11 +816,23 @@ export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>,
|
|
|
796
816
|
return effort;
|
|
797
817
|
}
|
|
798
818
|
|
|
799
|
-
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values.
|
|
800
|
-
|
|
819
|
+
/** Maps a normalized thinking effort to Google's `thinkingLevel` enum values.
|
|
820
|
+
* When a collapsed family routes `minimal` onto the same wire id as `low`
|
|
821
|
+
* (Antigravity Gemini 3.6/3.7 Flash), emit `LOW` — Cloud Code Assist rejects
|
|
822
|
+
* `MINIMAL` on those `-low` SKUs.
|
|
823
|
+
*/
|
|
824
|
+
export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
|
|
825
|
+
effort: Effort,
|
|
826
|
+
model?: ApiModel<TApi>,
|
|
827
|
+
): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
|
|
828
|
+
if (effort === Effort.Minimal) {
|
|
829
|
+
const routing = model?.thinking?.effortRouting;
|
|
830
|
+
if (routing?.[Effort.Minimal] && routing[Effort.Minimal] === routing[Effort.Low]) {
|
|
831
|
+
return "LOW";
|
|
832
|
+
}
|
|
833
|
+
return "MINIMAL";
|
|
834
|
+
}
|
|
801
835
|
switch (effort) {
|
|
802
|
-
case Effort.Minimal:
|
|
803
|
-
return "MINIMAL";
|
|
804
836
|
case Effort.Low:
|
|
805
837
|
return "LOW";
|
|
806
838
|
case Effort.Medium:
|
package/src/models.json
CHANGED
|
@@ -24448,7 +24448,7 @@
|
|
|
24448
24448
|
"grok-4.6": {
|
|
24449
24449
|
"id": "grok-4.6",
|
|
24450
24450
|
"name": "Grok 4.6",
|
|
24451
|
-
"api": "openai-
|
|
24451
|
+
"api": "openai-responses",
|
|
24452
24452
|
"provider": "github-copilot",
|
|
24453
24453
|
"baseUrl": "https://api.githubcopilot.com",
|
|
24454
24454
|
"reasoning": true,
|
|
@@ -24468,18 +24468,14 @@
|
|
|
24468
24468
|
"User-Agent": "opencode/1.3.15",
|
|
24469
24469
|
"X-GitHub-Api-Version": "2026-06-01"
|
|
24470
24470
|
},
|
|
24471
|
-
"compat": {
|
|
24472
|
-
"supportsStore": false,
|
|
24473
|
-
"supportsDeveloperRole": false,
|
|
24474
|
-
"supportsReasoningEffort": false
|
|
24475
|
-
},
|
|
24476
24471
|
"thinking": {
|
|
24477
24472
|
"mode": "effort",
|
|
24478
24473
|
"efforts": [
|
|
24479
24474
|
"minimal",
|
|
24480
24475
|
"low",
|
|
24481
24476
|
"medium",
|
|
24482
|
-
"high"
|
|
24477
|
+
"high",
|
|
24478
|
+
"xhigh"
|
|
24483
24479
|
]
|
|
24484
24480
|
}
|
|
24485
24481
|
},
|
|
@@ -73,8 +73,10 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
|
|
|
73
73
|
case "openrouter":
|
|
74
74
|
return "openrouter:pseudo-api";
|
|
75
75
|
case "vllm": {
|
|
76
|
+
// v2: qwen3.8 rows cached before the reasoning/template-effort upgrade
|
|
77
|
+
// carry `reasoning: false` and must be refetched.
|
|
76
78
|
const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
|
|
77
|
-
return `vllm:${Bun.hash(baseUrl).toString(36)}`;
|
|
79
|
+
return `vllm:models-v2:${Bun.hash(baseUrl).toString(36)}`;
|
|
78
80
|
}
|
|
79
81
|
default:
|
|
80
82
|
return providerId;
|
|
@@ -474,6 +474,7 @@ export const CATALOG_PROVIDERS = [
|
|
|
474
474
|
defaultModel: "openai/gpt-oss-120b",
|
|
475
475
|
envVars: ["COREWEAVE_API_KEY", "WANDB_API_KEY"],
|
|
476
476
|
createModelManagerOptions: (config: ModelManagerConfig) => coreWeaveModelManagerOptions(config),
|
|
477
|
+
dynamicModelsAuthoritative: true,
|
|
477
478
|
catalogDiscovery: { label: "CoreWeave Serverless Inference" },
|
|
478
479
|
},
|
|
479
480
|
{
|
|
@@ -16,6 +16,7 @@ import {
|
|
|
16
16
|
isGrokReasoningEffortCapable,
|
|
17
17
|
isKimiK3ModelId,
|
|
18
18
|
isKimiModelId,
|
|
19
|
+
isQwen38PlusTemplateEffortModelId,
|
|
19
20
|
isReasoningGlmModelId,
|
|
20
21
|
} from "../identity/family";
|
|
21
22
|
import { resolveModelReference } from "../identity/reference";
|
|
@@ -1073,10 +1074,61 @@ export interface GmiCloudModelManagerConfig {
|
|
|
1073
1074
|
fetch?: FetchImpl;
|
|
1074
1075
|
}
|
|
1075
1076
|
|
|
1077
|
+
/**
|
|
1078
|
+
* Map a discovered GMI Cloud model to a full spec.
|
|
1079
|
+
*
|
|
1080
|
+
* GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults carry
|
|
1081
|
+
* no limits, reasoning, or thinking metadata. When a gmi-cloud bundled
|
|
1082
|
+
* reference exists (the seeded default) it supplies GMI's published tariff and
|
|
1083
|
+
* limits directly. Every other id is an open-weight model GMI resells under its
|
|
1084
|
+
* canonical id (`deepseek-ai/…`, `moonshotai/…`, `zai-org/…`, `Qwen/…`), so its
|
|
1085
|
+
* intrinsic capabilities — context window, output limit, reasoning, thinking
|
|
1086
|
+
* ladder — are recovered from any bundled upstream entry via the canonical
|
|
1087
|
+
* reference index. Pricing is deliberately never borrowed across providers:
|
|
1088
|
+
* GMI's per-model tariff is unknown for these ids, so cost stays zeroed rather
|
|
1089
|
+
* than inheriting another provider's rate.
|
|
1090
|
+
*/
|
|
1091
|
+
function mapGmiCloudModel(
|
|
1092
|
+
entry: OpenAICompatibleModelRecord,
|
|
1093
|
+
defaults: ModelSpec<"openai-completions">,
|
|
1094
|
+
reference: ModelSpec<"openai-completions"> | undefined,
|
|
1095
|
+
): ModelSpec<"openai-completions"> {
|
|
1096
|
+
if (reference) {
|
|
1097
|
+
return mapWithBundledReference(entry, defaults, reference);
|
|
1098
|
+
}
|
|
1099
|
+
const canonical = resolveModelReference(defaults.id, getBundledModelReferenceIndex()) as
|
|
1100
|
+
| ModelSpec<"openai-completions">
|
|
1101
|
+
| undefined;
|
|
1102
|
+
if (!canonical) {
|
|
1103
|
+
return { ...defaults, name: toModelName(entry.name, defaults.name) };
|
|
1104
|
+
}
|
|
1105
|
+
const contextWindow = canonical.contextWindow ?? defaults.contextWindow;
|
|
1106
|
+
const maxTokens =
|
|
1107
|
+
canonical.maxTokens != null && contextWindow != null
|
|
1108
|
+
? Math.min(canonical.maxTokens, contextWindow)
|
|
1109
|
+
: (canonical.maxTokens ?? defaults.maxTokens);
|
|
1110
|
+
return {
|
|
1111
|
+
...defaults,
|
|
1112
|
+
name: toModelName(entry.name, canonical.name ?? defaults.name),
|
|
1113
|
+
reasoning: canonical.reasoning,
|
|
1114
|
+
input: canonical.input,
|
|
1115
|
+
...(canonical.thinking && { thinking: canonical.thinking }),
|
|
1116
|
+
contextWindow,
|
|
1117
|
+
maxTokens,
|
|
1118
|
+
};
|
|
1119
|
+
}
|
|
1120
|
+
|
|
1076
1121
|
export function gmiCloudModelManagerOptions(
|
|
1077
1122
|
config?: GmiCloudModelManagerConfig,
|
|
1078
1123
|
): ModelManagerOptions<"openai-completions"> {
|
|
1079
|
-
return
|
|
1124
|
+
return createOpenAICompatibleModelManagerOptions({
|
|
1125
|
+
api: "openai-completions",
|
|
1126
|
+
providerId: "gmi-cloud",
|
|
1127
|
+
defaultBaseUrl: GMI_CLOUD_BASE_URL,
|
|
1128
|
+
config,
|
|
1129
|
+
requireApiKey: true,
|
|
1130
|
+
mapModel: mapGmiCloudModel,
|
|
1131
|
+
});
|
|
1080
1132
|
}
|
|
1081
1133
|
|
|
1082
1134
|
// ---------------------------------------------------------------------------
|
|
@@ -2980,6 +3032,10 @@ export const ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS: Readonly<Record<string,
|
|
|
2980
3032
|
contextWindow: 1_000_000,
|
|
2981
3033
|
maxTokens: 384_000,
|
|
2982
3034
|
},
|
|
3035
|
+
"deepseek-v4-pro-0813": {
|
|
3036
|
+
contextWindow: 1_000_000,
|
|
3037
|
+
maxTokens: 384_000,
|
|
3038
|
+
},
|
|
2983
3039
|
"deepseek-v3.2": {
|
|
2984
3040
|
contextWindow: 131_072,
|
|
2985
3041
|
maxTokens: 65_536,
|
|
@@ -4994,6 +5050,11 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
|
|
|
4994
5050
|
return {
|
|
4995
5051
|
...model,
|
|
4996
5052
|
contextWindow: toPositiveNumber(entry.max_model_len, model.contextWindow),
|
|
5053
|
+
// vLLM's /v1/models reports no reasoning capability. Qwen 3.8+
|
|
5054
|
+
// open weights always think (the template cannot disable it), so
|
|
5055
|
+
// light up the effort dial; buildModel derives the template
|
|
5056
|
+
// ladder from the id + local-backend compat.
|
|
5057
|
+
reasoning: model.reasoning || isQwen38PlusTemplateEffortModelId(model.id),
|
|
4997
5058
|
};
|
|
4998
5059
|
},
|
|
4999
5060
|
fetch: config?.fetch,
|
|
@@ -5071,8 +5132,18 @@ export interface GithubCopilotModelManagerConfig {
|
|
|
5071
5132
|
|
|
5072
5133
|
const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus|fable|mythos)-\d/;
|
|
5073
5134
|
const isCopilotResponsesModelId = (modelId: string): boolean =>
|
|
5074
|
-
modelId === "grok-4.5" ||
|
|
5075
|
-
|
|
5135
|
+
modelId === "grok-4.5" ||
|
|
5136
|
+
modelId === "grok-4.6" ||
|
|
5137
|
+
modelId.startsWith("gpt-5") ||
|
|
5138
|
+
modelId.startsWith("oswe") ||
|
|
5139
|
+
modelId.startsWith("mai-");
|
|
5140
|
+
const COPILOT_CACHE_INVALIDATED_MODEL_IDS = [
|
|
5141
|
+
"grok-4.5",
|
|
5142
|
+
"grok-4.5-1m",
|
|
5143
|
+
"grok-4.6",
|
|
5144
|
+
"grok-4.6-1m",
|
|
5145
|
+
"mai-code-1-flash-picker",
|
|
5146
|
+
];
|
|
5076
5147
|
|
|
5077
5148
|
function inferCopilotApi(modelId: string): Api {
|
|
5078
5149
|
if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
|
|
@@ -5705,8 +5776,17 @@ const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution("https://opencod
|
|
|
5705
5776
|
// /zen/go/v1/chat/completions route does not work for this model while
|
|
5706
5777
|
// /zen/go/v1/responses does (user-verified against the live gateway,
|
|
5707
5778
|
// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
|
|
5779
|
+
//
|
|
5780
|
+
// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case: the
|
|
5781
|
+
// Go gateway's /zen/go/v1/models discovery drops the `provider.npm` hint, so
|
|
5782
|
+
// without an override they fall through to openai-completions even though the
|
|
5783
|
+
// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
|
|
5784
|
+
// https://opencode.ai/docs/go/#endpoints). The completions parser then closes
|
|
5785
|
+
// the stream with no finish_reason on every tool-call turn (#8957).
|
|
5708
5786
|
const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen/go", {
|
|
5709
5787
|
"deepseek-v4-flash": "openai-responses",
|
|
5788
|
+
"muse-spark-1.2": "openai-responses",
|
|
5789
|
+
"muse-spark-1.2-contributor": "openai-responses",
|
|
5710
5790
|
"minimax-m2.7": "openai-completions",
|
|
5711
5791
|
"minimax-m3": "openai-completions",
|
|
5712
5792
|
"minimax-m3-free": "openai-completions",
|
package/src/types.ts
CHANGED
|
@@ -259,6 +259,16 @@ export interface OpenAICompat {
|
|
|
259
259
|
* Non-Qwen templates ignore the flag, so the auto-detection is safe.
|
|
260
260
|
*/
|
|
261
261
|
qwenPreserveThinking?: boolean;
|
|
262
|
+
/**
|
|
263
|
+
* Route the requested thinking effort onto the Qwen 3.8+ chat template's
|
|
264
|
+
* `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default
|
|
265
|
+
* `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects
|
|
266
|
+
* (plus the top-level field on the `qwen` dialect, which newer llama.cpp
|
|
267
|
+
* builds map natively). Without it the qwen dialects only toggle
|
|
268
|
+
* `enable_thinking` and the template always thinks at its `xhigh` default.
|
|
269
|
+
* Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend).
|
|
270
|
+
*/
|
|
271
|
+
qwenTemplateReasoningEffort?: boolean;
|
|
262
272
|
/** Whether assistant tool-call messages must include non-empty content. Default: false. */
|
|
263
273
|
requiresAssistantContentForToolCalls?: boolean;
|
|
264
274
|
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
|
|
@@ -604,6 +614,7 @@ export interface ResolvedOpenAISharedCompat {
|
|
|
604
614
|
allowsSyntheticReasoningContentForToolCalls: boolean;
|
|
605
615
|
replayReasoningContent: boolean;
|
|
606
616
|
qwenPreserveThinking: boolean;
|
|
617
|
+
qwenTemplateReasoningEffort: boolean;
|
|
607
618
|
requiresThinkingAsText: boolean;
|
|
608
619
|
requiresMistralToolIds: boolean;
|
|
609
620
|
requiresToolResultName: boolean;
|
|
@@ -669,6 +680,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
|
|
|
669
680
|
| "allowsSyntheticReasoningContentForToolCalls"
|
|
670
681
|
| "replayReasoningContent"
|
|
671
682
|
| "qwenPreserveThinking"
|
|
683
|
+
| "qwenTemplateReasoningEffort"
|
|
672
684
|
| "requiresThinkingAsText"
|
|
673
685
|
| "requiresMistralToolIds"
|
|
674
686
|
| "requiresToolResultName"
|