@oh-my-pi/pi-catalog 17.3.7 → 17.3.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,6 +28,15 @@ const CURSOR_MAX_MODE_1M_ID_PATTERN = /claude|gemini/;
28
28
  /** Kimi's official bare K3 id (`k3`, `kimi/k3`); `k3-256k` is the 256k SKU and stays out. */
29
29
  const CURSOR_KIMI_K3_BARE_ID_PATTERN = /(^|\/)k3$/i;
30
30
 
31
+ /**
32
+ * Versioned Cursor Grok ids (`cursor-grok-4.5`, `cursor-grok-4.6-high`) are
33
+ * reasoning models whose effort is carried in the per-tier sibling id.
34
+ * `GetUsableModels` ships no `thinkingDetails` and the bundled references read
35
+ * `reasoning: false`, so classification falls back to the id. The non-reasoning
36
+ * `grok-code-*` coding models lack the version digit and stay out.
37
+ */
38
+ const CURSOR_GROK_REASONING_ID_PATTERN = /^cursor-grok-\d/i;
39
+
31
40
  /**
32
41
  * Model-id families whose native catalogs (anthropic, openai/openai-codex,
33
42
  * google) are multimodal. Cursor-only or text-only families (`composer-*`,
@@ -297,7 +306,11 @@ function normalizeCursorModel(
297
306
 
298
307
  const name = pickModelDisplayName(details, id);
299
308
  const reference = references.get(id);
300
- const reasoning = isKimiK3ModelId(id) || Boolean(details.thinkingDetails) || reference?.reasoning === true;
309
+ const reasoning =
310
+ isKimiK3ModelId(id) ||
311
+ CURSOR_GROK_REASONING_ID_PATTERN.test(id) ||
312
+ Boolean(details.thinkingDetails) ||
313
+ reference?.reasoning === true;
301
314
 
302
315
  if (reference) {
303
316
  return {
@@ -73,6 +73,25 @@ export const isQwenModelId = memo((modelId: string): boolean => {
73
73
  return modelId.toLowerCase().includes("qwen");
74
74
  });
75
75
 
76
+ /**
77
+ * Open-weight Qwen 3.8+ releases (`qwen3.8-27b`, `qwen3.8-2.4t-a95b`, GGUF
78
+ * names like `Qwen3.8-27B-UD-Q6_K_XL`) whose chat template steers thinking
79
+ * depth through a `reasoning_effort` template kwarg (`low`/`medium`/`xhigh`,
80
+ * template default `xhigh`; thinking itself cannot be disabled). Compared
81
+ * component-wise so `qwen3.10` sorts after `qwen3.8`. API-only `-max` SKUs are
82
+ * excluded — Dashscope drives them through OpenAI-style `reasoning_effort`
83
+ * with curated compat. The trailing guard rejects parameter-count lookalikes
84
+ * (`qwen-3.8b`) without breaking `qwen3.8-27b`.
85
+ */
86
+ export const isQwen38PlusTemplateEffortModelId = memo((modelId: string): boolean => {
87
+ const match = /qwen[-_ ]?(\d+)\.(\d+)(?![\dbB])/i.exec(modelId);
88
+ if (!match) return false;
89
+ const major = Number.parseInt(match[1], 10);
90
+ const minor = Number.parseInt(match[2], 10);
91
+ if (major < 3 || (major === 3 && minor < 8)) return false;
92
+ return !/^-max(?:$|[-.:])/i.test(modelId.slice(match.index + match[0].length));
93
+ });
94
+
76
95
  /** Gemma open-weights family (`gemma-3-27b-it`, `google/gemma-4-E2B-it`, `gemma2-9b`). */
77
96
  export const isGemmaModelId = memo((modelId: string): boolean => {
78
97
  return /(^|\/)gemma[-.]?\d/i.test(modelId);
@@ -121,7 +140,8 @@ const GROK_EFFORT_CAPABLE_PREFIXES = [
121
140
  /**
122
141
  * Grok SKUs that expose the wire `reasoning.effort` dial. Other Grok reasoners
123
142
  * (e.g. `grok-build`, `grok-4.20-0309-reasoning`) think natively but reject the
124
- * param, so callers must omit reasoning effort for them.
143
+ * param, so callers must omit reasoning effort for them. `grok-4.6` accepts
144
+ * `low`/`medium`/`high`/`xhigh` and 400s on `max`.
125
145
  */
126
146
  export const isGrokReasoningEffortCapable = memo((modelId: string): boolean => {
127
147
  const bare = bareModelId(modelId).trim().toLowerCase();
@@ -3,7 +3,8 @@
3
3
  * Replaces per-provider JSON files with a single cache.db.
4
4
  */
5
5
  import { Database } from "bun:sqlite";
6
- import { getModelDbPath } from "@oh-my-pi/pi-utils";
6
+ import { renameSync } from "node:fs";
7
+ import { getModelDbPath, isEnoent, isSqliteCorruptionError, logger } from "@oh-my-pi/pi-utils";
7
8
  import type { Api, Model, ModelSpec } from "./types";
8
9
 
9
10
  // Rows persist ModelSpec JSON (sparse `compat`, never the resolved record);
@@ -91,13 +92,14 @@ function openDb(resolvedPath: string): Database {
91
92
  return db;
92
93
  }
93
94
 
94
- function getSharedDb(): Database {
95
- const resolvedPath = getModelDbPath();
95
+ function getSharedDb(resolvedPath: string): Database {
96
96
  if (sharedDb && sharedDbPath === resolvedPath) {
97
97
  return sharedDb;
98
98
  }
99
99
  if (sharedDb) {
100
100
  sharedDb.close();
101
+ sharedDb = null;
102
+ sharedDbPath = null;
101
103
  }
102
104
  const db = openDb(resolvedPath);
103
105
  sharedDb = db;
@@ -105,9 +107,9 @@ function getSharedDb(): Database {
105
107
  return db;
106
108
  }
107
109
 
108
- function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
109
- if (!dbPath) return useDb(getSharedDb());
110
- const db = openDb(dbPath);
110
+ function runModelCacheDb<T>(resolvedPath: string, shared: boolean, useDb: (db: Database) => T): T {
111
+ if (shared) return useDb(getSharedDb(resolvedPath));
112
+ const db = openDb(resolvedPath);
111
113
  try {
112
114
  return useDb(db);
113
115
  } finally {
@@ -115,6 +117,65 @@ function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) =
115
117
  }
116
118
  }
117
119
 
120
+ // Paths already reported corrupt this process: the first unrecoverable failure
121
+ // is logged at `error`, later heals at `debug`, so a dying disk cannot spam.
122
+ const reportedCorruptPaths = new Set<string>();
123
+
124
+ /**
125
+ * Move a physically corrupt `models.db` (plus its `-wal`/`-shm` sidecars) aside
126
+ * so {@link openDb} can recreate a fresh cache at the original path. Renames are
127
+ * best-effort: a vanished sidecar (already healed by a peer process) is fine,
128
+ * and any other rename failure is left for {@link openDb} to surface.
129
+ */
130
+ function quarantineCorruptModelCache(resolvedPath: string): void {
131
+ const stamp = Date.now();
132
+ for (const suffix of ["", "-wal", "-shm"]) {
133
+ try {
134
+ renameSync(`${resolvedPath}${suffix}`, `${resolvedPath}.corrupt-${stamp}${suffix}`);
135
+ } catch (err) {
136
+ if (!isEnoent(err)) {
137
+ logger.debug("model cache: could not quarantine corrupt file", { path: `${resolvedPath}${suffix}` });
138
+ }
139
+ }
140
+ }
141
+ }
142
+
143
+ /**
144
+ * Recover from unrecoverable `models.db` corruption: drop the cached handle,
145
+ * quarantine the broken files, and let the next open recreate the cache. A
146
+ * corrupt cache would otherwise be re-queried on every read/write forever,
147
+ * permanently masking a successful live catalog (issue #8867). Only
148
+ * {@link isSqliteCorruptionError} codes reach here; BUSY/permission errors keep
149
+ * their existing best-effort paths.
150
+ */
151
+ function healCorruptModelCache(resolvedPath: string, shared: boolean, err: unknown): void {
152
+ if (shared && sharedDb) {
153
+ sharedDb.close();
154
+ sharedDb = null;
155
+ sharedDbPath = null;
156
+ }
157
+ quarantineCorruptModelCache(resolvedPath);
158
+ const code = err && typeof err === "object" && "code" in err ? err.code : undefined;
159
+ if (reportedCorruptPaths.has(resolvedPath)) {
160
+ logger.debug("model cache: re-healed corrupt database", { path: resolvedPath, code });
161
+ } else {
162
+ reportedCorruptPaths.add(resolvedPath);
163
+ logger.error("model cache corrupt; quarantined and recreated a fresh cache", { path: resolvedPath, code });
164
+ }
165
+ }
166
+
167
+ function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
168
+ const resolvedPath = dbPath ?? getModelDbPath();
169
+ const shared = dbPath === undefined;
170
+ try {
171
+ return runModelCacheDb(resolvedPath, shared, useDb);
172
+ } catch (err) {
173
+ if (!isSqliteCorruptionError(err)) throw err;
174
+ healCorruptModelCache(resolvedPath, shared, err);
175
+ return runModelCacheDb(resolvedPath, shared, useDb);
176
+ }
177
+ }
178
+
118
179
  function migrateCacheSchema(db: Database): void {
119
180
  const stmt = db.prepare("PRAGMA table_info(model_cache)");
120
181
  try {
@@ -71,6 +71,11 @@ const LOW_HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Hi
71
71
  const HIGH_MAX_REASONING_EFFORTS: readonly Effort[] = [Effort.High, Effort.Max];
72
72
  /** OpenRouter's DeepSeek route accepts only `high`. */
73
73
  const HIGH_ONLY_REASONING_EFFORTS: readonly Effort[] = [Effort.High];
74
+ /**
75
+ * Qwen 3.8+ open-weight chat template: prompt-steered `reasoning_effort`
76
+ * kwarg with exactly three wire tiers (template default is `xhigh`).
77
+ */
78
+ const QWEN38_TEMPLATE_REASONING_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.XHigh];
74
79
  /**
75
80
  * Five wire tiers with a `low` floor: GPT-5.6+, Anthropic adaptive models
76
81
  * with the real xhigh tier (Opus 4.7+, Sonnet 5+, Fable/Mythos 5), and the
@@ -179,7 +184,9 @@ function fillThinkingWireDefaults<TApi extends Api>(
179
184
  thinking.supportsDisplay === undefined &&
180
185
  (spec.api === "anthropic-messages" || spec.api === "bedrock-converse-stream") &&
181
186
  supportsAdaptiveThinkingDisplay(spec.id);
182
- const needsRequiresEffort = thinking.requiresEffort === undefined && impliesMandatoryReasoning(parsed, spec.id);
187
+ const needsRequiresEffort =
188
+ thinking.requiresEffort === undefined &&
189
+ (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat));
183
190
  const needsDefaultLevel =
184
191
  thinking.defaultLevel === undefined && (isKimiK3ModelId(spec.id) || isGlm53ReasoningEffortModelId(spec.id));
185
192
  if (!effortsChanged && !shouldReplaceEffortMap && !needsDisplay && !needsRequiresEffort && !needsDefaultLevel) {
@@ -232,7 +239,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
232
239
  ) {
233
240
  config.supportsDisplay = true;
234
241
  }
235
- if (impliesMandatoryReasoning(parsed, spec.id)) {
242
+ if (impliesMandatoryReasoning(parsed, spec.id) || isQwenTemplateReasoningEffortCompat(compat)) {
236
243
  config.requiresEffort = true;
237
244
  }
238
245
  return config;
@@ -376,6 +383,13 @@ function getModelDefinedEfforts<TApi extends Api>(
376
383
  if (spec.provider === "ollama") {
377
384
  return OLLAMA_REASONING_EFFORTS;
378
385
  }
386
+ // Qwen 3.8+ served through a local llama.cpp-style backend: the chat
387
+ // template's prompt-steered `reasoning_effort` kwarg accepts exactly
388
+ // low/medium/xhigh (and thinking cannot be turned off — the official 3.8
389
+ // template raises on `enable_thinking: false`, hence requiresEffort).
390
+ if (isOpenAICompatReasoningApi(spec.api) && isQwenTemplateReasoningEffortCompat(compat)) {
391
+ return QWEN38_TEMPLATE_REASONING_EFFORTS;
392
+ }
379
393
  if (
380
394
  (isOpenAICompatReasoningApi(spec.api) || (spec.api === "ollama-chat" && spec.provider === "ollama-cloud")) &&
381
395
  isDeepseekReasoningModel(spec)
@@ -483,6 +497,12 @@ function isOpenRouterThinkingFormat(compat: CompatOf<Api>): boolean {
483
497
  function isZaiThinkingFormat(compat: CompatOf<Api>): boolean {
484
498
  return compat !== undefined && "thinkingFormat" in compat && compat.thinkingFormat === "zai";
485
499
  }
500
+ /** Resolved-compat gate for the Qwen 3.8+ local template `reasoning_effort` dialect. */
501
+ function isQwenTemplateReasoningEffortCompat(compat: CompatOf<Api>): boolean {
502
+ return (
503
+ compat !== undefined && "qwenTemplateReasoningEffort" in compat && compat.qwenTemplateReasoningEffort === true
504
+ );
505
+ }
486
506
 
487
507
  function inferDetectedEffortMap<TApi extends Api>(
488
508
  spec: ModelSpec<TApi>,
@@ -796,11 +816,23 @@ export function requireSupportedEffort<TApi extends Api>(model: ApiModel<TApi>,
796
816
  return effort;
797
817
  }
798
818
 
799
- /** Maps a normalized thinking effort to Google's `thinkingLevel` enum values. */
800
- export function mapEffortToGoogleThinkingLevel(effort: Effort): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
819
+ /** Maps a normalized thinking effort to Google's `thinkingLevel` enum values.
820
+ * When a collapsed family routes `minimal` onto the same wire id as `low`
821
+ * (Antigravity Gemini 3.6/3.7 Flash), emit `LOW` — Cloud Code Assist rejects
822
+ * `MINIMAL` on those `-low` SKUs.
823
+ */
824
+ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
825
+ effort: Effort,
826
+ model?: ApiModel<TApi>,
827
+ ): "MINIMAL" | "LOW" | "MEDIUM" | "HIGH" {
828
+ if (effort === Effort.Minimal) {
829
+ const routing = model?.thinking?.effortRouting;
830
+ if (routing?.[Effort.Minimal] && routing[Effort.Minimal] === routing[Effort.Low]) {
831
+ return "LOW";
832
+ }
833
+ return "MINIMAL";
834
+ }
801
835
  switch (effort) {
802
- case Effort.Minimal:
803
- return "MINIMAL";
804
836
  case Effort.Low:
805
837
  return "LOW";
806
838
  case Effort.Medium:
package/src/models.json CHANGED
@@ -24448,7 +24448,7 @@
24448
24448
  "grok-4.6": {
24449
24449
  "id": "grok-4.6",
24450
24450
  "name": "Grok 4.6",
24451
- "api": "openai-completions",
24451
+ "api": "openai-responses",
24452
24452
  "provider": "github-copilot",
24453
24453
  "baseUrl": "https://api.githubcopilot.com",
24454
24454
  "reasoning": true,
@@ -24468,18 +24468,14 @@
24468
24468
  "User-Agent": "opencode/1.3.15",
24469
24469
  "X-GitHub-Api-Version": "2026-06-01"
24470
24470
  },
24471
- "compat": {
24472
- "supportsStore": false,
24473
- "supportsDeveloperRole": false,
24474
- "supportsReasoningEffort": false
24475
- },
24476
24471
  "thinking": {
24477
24472
  "mode": "effort",
24478
24473
  "efforts": [
24479
24474
  "minimal",
24480
24475
  "low",
24481
24476
  "medium",
24482
- "high"
24477
+ "high",
24478
+ "xhigh"
24483
24479
  ]
24484
24480
  }
24485
24481
  },
@@ -73,8 +73,10 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
73
73
  case "openrouter":
74
74
  return "openrouter:pseudo-api";
75
75
  case "vllm": {
76
+ // v2: qwen3.8 rows cached before the reasoning/template-effort upgrade
77
+ // carry `reasoning: false` and must be refetched.
76
78
  const baseUrl = options.baseUrl ?? getDefaultModelDiscoveryBaseUrl(providerId)!;
77
- return `vllm:${Bun.hash(baseUrl).toString(36)}`;
79
+ return `vllm:models-v2:${Bun.hash(baseUrl).toString(36)}`;
78
80
  }
79
81
  default:
80
82
  return providerId;
@@ -474,6 +474,7 @@ export const CATALOG_PROVIDERS = [
474
474
  defaultModel: "openai/gpt-oss-120b",
475
475
  envVars: ["COREWEAVE_API_KEY", "WANDB_API_KEY"],
476
476
  createModelManagerOptions: (config: ModelManagerConfig) => coreWeaveModelManagerOptions(config),
477
+ dynamicModelsAuthoritative: true,
477
478
  catalogDiscovery: { label: "CoreWeave Serverless Inference" },
478
479
  },
479
480
  {
@@ -16,6 +16,7 @@ import {
16
16
  isGrokReasoningEffortCapable,
17
17
  isKimiK3ModelId,
18
18
  isKimiModelId,
19
+ isQwen38PlusTemplateEffortModelId,
19
20
  isReasoningGlmModelId,
20
21
  } from "../identity/family";
21
22
  import { resolveModelReference } from "../identity/reference";
@@ -1073,10 +1074,61 @@ export interface GmiCloudModelManagerConfig {
1073
1074
  fetch?: FetchImpl;
1074
1075
  }
1075
1076
 
1077
+ /**
1078
+ * Map a discovered GMI Cloud model to a full spec.
1079
+ *
1080
+ * GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults carry
1081
+ * no limits, reasoning, or thinking metadata. When a gmi-cloud bundled
1082
+ * reference exists (the seeded default) it supplies GMI's published tariff and
1083
+ * limits directly. Every other id is an open-weight model GMI resells under its
1084
+ * canonical id (`deepseek-ai/…`, `moonshotai/…`, `zai-org/…`, `Qwen/…`), so its
1085
+ * intrinsic capabilities — context window, output limit, reasoning, thinking
1086
+ * ladder — are recovered from any bundled upstream entry via the canonical
1087
+ * reference index. Pricing is deliberately never borrowed across providers:
1088
+ * GMI's per-model tariff is unknown for these ids, so cost stays zeroed rather
1089
+ * than inheriting another provider's rate.
1090
+ */
1091
+ function mapGmiCloudModel(
1092
+ entry: OpenAICompatibleModelRecord,
1093
+ defaults: ModelSpec<"openai-completions">,
1094
+ reference: ModelSpec<"openai-completions"> | undefined,
1095
+ ): ModelSpec<"openai-completions"> {
1096
+ if (reference) {
1097
+ return mapWithBundledReference(entry, defaults, reference);
1098
+ }
1099
+ const canonical = resolveModelReference(defaults.id, getBundledModelReferenceIndex()) as
1100
+ | ModelSpec<"openai-completions">
1101
+ | undefined;
1102
+ if (!canonical) {
1103
+ return { ...defaults, name: toModelName(entry.name, defaults.name) };
1104
+ }
1105
+ const contextWindow = canonical.contextWindow ?? defaults.contextWindow;
1106
+ const maxTokens =
1107
+ canonical.maxTokens != null && contextWindow != null
1108
+ ? Math.min(canonical.maxTokens, contextWindow)
1109
+ : (canonical.maxTokens ?? defaults.maxTokens);
1110
+ return {
1111
+ ...defaults,
1112
+ name: toModelName(entry.name, canonical.name ?? defaults.name),
1113
+ reasoning: canonical.reasoning,
1114
+ input: canonical.input,
1115
+ ...(canonical.thinking && { thinking: canonical.thinking }),
1116
+ contextWindow,
1117
+ maxTokens,
1118
+ };
1119
+ }
1120
+
1076
1121
  export function gmiCloudModelManagerOptions(
1077
1122
  config?: GmiCloudModelManagerConfig,
1078
1123
  ): ModelManagerOptions<"openai-completions"> {
1079
- return createSimpleOpenAICompletionsOptions("gmi-cloud", GMI_CLOUD_BASE_URL, config);
1124
+ return createOpenAICompatibleModelManagerOptions({
1125
+ api: "openai-completions",
1126
+ providerId: "gmi-cloud",
1127
+ defaultBaseUrl: GMI_CLOUD_BASE_URL,
1128
+ config,
1129
+ requireApiKey: true,
1130
+ mapModel: mapGmiCloudModel,
1131
+ });
1080
1132
  }
1081
1133
 
1082
1134
  // ---------------------------------------------------------------------------
@@ -2980,6 +3032,10 @@ export const ALIBABA_TOKEN_PLAN_DISCOVERED_MODEL_LIMITS: Readonly<Record<string,
2980
3032
  contextWindow: 1_000_000,
2981
3033
  maxTokens: 384_000,
2982
3034
  },
3035
+ "deepseek-v4-pro-0813": {
3036
+ contextWindow: 1_000_000,
3037
+ maxTokens: 384_000,
3038
+ },
2983
3039
  "deepseek-v3.2": {
2984
3040
  contextWindow: 131_072,
2985
3041
  maxTokens: 65_536,
@@ -4994,6 +5050,11 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
4994
5050
  return {
4995
5051
  ...model,
4996
5052
  contextWindow: toPositiveNumber(entry.max_model_len, model.contextWindow),
5053
+ // vLLM's /v1/models reports no reasoning capability. Qwen 3.8+
5054
+ // open weights always think (the template cannot disable it), so
5055
+ // light up the effort dial; buildModel derives the template
5056
+ // ladder from the id + local-backend compat.
5057
+ reasoning: model.reasoning || isQwen38PlusTemplateEffortModelId(model.id),
4997
5058
  };
4998
5059
  },
4999
5060
  fetch: config?.fetch,
@@ -5071,8 +5132,18 @@ export interface GithubCopilotModelManagerConfig {
5071
5132
 
5072
5133
  const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus|fable|mythos)-\d/;
5073
5134
  const isCopilotResponsesModelId = (modelId: string): boolean =>
5074
- modelId === "grok-4.5" || modelId.startsWith("gpt-5") || modelId.startsWith("oswe") || modelId.startsWith("mai-");
5075
- const COPILOT_CACHE_INVALIDATED_MODEL_IDS = ["grok-4.5", "grok-4.5-1m", "mai-code-1-flash-picker"];
5135
+ modelId === "grok-4.5" ||
5136
+ modelId === "grok-4.6" ||
5137
+ modelId.startsWith("gpt-5") ||
5138
+ modelId.startsWith("oswe") ||
5139
+ modelId.startsWith("mai-");
5140
+ const COPILOT_CACHE_INVALIDATED_MODEL_IDS = [
5141
+ "grok-4.5",
5142
+ "grok-4.5-1m",
5143
+ "grok-4.6",
5144
+ "grok-4.6-1m",
5145
+ "mai-code-1-flash-picker",
5146
+ ];
5076
5147
 
5077
5148
  function inferCopilotApi(modelId: string): Api {
5078
5149
  if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
@@ -5705,8 +5776,17 @@ const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution("https://opencod
5705
5776
  // /zen/go/v1/chat/completions route does not work for this model while
5706
5777
  // /zen/go/v1/responses does (user-verified against the live gateway,
5707
5778
  // 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
5779
+ //
5780
+ // muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case: the
5781
+ // Go gateway's /zen/go/v1/models discovery drops the `provider.npm` hint, so
5782
+ // without an override they fall through to openai-completions even though the
5783
+ // gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
5784
+ // https://opencode.ai/docs/go/#endpoints). The completions parser then closes
5785
+ // the stream with no finish_reason on every tool-call turn (#8957).
5708
5786
  const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen/go", {
5709
5787
  "deepseek-v4-flash": "openai-responses",
5788
+ "muse-spark-1.2": "openai-responses",
5789
+ "muse-spark-1.2-contributor": "openai-responses",
5710
5790
  "minimax-m2.7": "openai-completions",
5711
5791
  "minimax-m3": "openai-completions",
5712
5792
  "minimax-m3-free": "openai-completions",
package/src/types.ts CHANGED
@@ -259,6 +259,16 @@ export interface OpenAICompat {
259
259
  * Non-Qwen templates ignore the flag, so the auto-detection is safe.
260
260
  */
261
261
  qwenPreserveThinking?: boolean;
262
+ /**
263
+ * Route the requested thinking effort onto the Qwen 3.8+ chat template's
264
+ * `reasoning_effort` kwarg (`low`/`medium`/`xhigh`; template default
265
+ * `xhigh`). Emitted inside `chat_template_kwargs` for both Qwen dialects
266
+ * (plus the top-level field on the `qwen` dialect, which newer llama.cpp
267
+ * builds map natively). Without it the qwen dialects only toggle
268
+ * `enable_thinking` and the template always thinks at its `xhigh` default.
269
+ * Default: auto-detected (Qwen 3.8+ id on a local llama.cpp-style backend).
270
+ */
271
+ qwenTemplateReasoningEffort?: boolean;
262
272
  /** Whether assistant tool-call messages must include non-empty content. Default: false. */
263
273
  requiresAssistantContentForToolCalls?: boolean;
264
274
  /** Whether the provider supports the `tool_choice` parameter. Default: true. */
@@ -604,6 +614,7 @@ export interface ResolvedOpenAISharedCompat {
604
614
  allowsSyntheticReasoningContentForToolCalls: boolean;
605
615
  replayReasoningContent: boolean;
606
616
  qwenPreserveThinking: boolean;
617
+ qwenTemplateReasoningEffort: boolean;
607
618
  requiresThinkingAsText: boolean;
608
619
  requiresMistralToolIds: boolean;
609
620
  requiresToolResultName: boolean;
@@ -669,6 +680,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
669
680
  | "allowsSyntheticReasoningContentForToolCalls"
670
681
  | "replayReasoningContent"
671
682
  | "qwenPreserveThinking"
683
+ | "qwenTemplateReasoningEffort"
672
684
  | "requiresThinkingAsText"
673
685
  | "requiresMistralToolIds"
674
686
  | "requiresToolResultName"