pi-ollama-cloud 0.11.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,7 +2,18 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
- ## [Unreleased]
5
+ ## [0.12.1] - 2026-09-14
6
+
7
+ - Omit empty `openRouterRouting` / `vercelGatewayRouting` from `buildCompat` (set to `undefined`, not `{}`): pi-ai reads the raw `model.compat` and treats `{}` as truthy, sending a stray `provider: {}` on every Ollama chat completion. Fixes #60. Thanks @0xbentang (#61).
8
+
9
+ ## [0.12.0] - 2026-09-11
10
+
11
+ - Source per-model thinking levels from models.dev instead of hardcoded maps. `scripts/generate-reasoning.ts` fetches the `ollama-cloud` provider's `reasoning_options` into `reasoning.generated.ts`, and `thinking-levels.ts` maps each model's effort values onto Pi's levels (toggle-only models become a binary on/off map; models with no models.dev entry fall back to `DEFAULT`). The `off` switch is handled by a small override table for models verified not to honor `reasoning_effort:"none"` (`gpt-oss:20b`, `gpt-oss:120b`, `minimax-m2.7`). Removed the now-stale per-family maps and the `docs/think-experiment.md` doc.
12
+ - `generate-models` now also refreshes `reasoning.generated.ts` (runs `generate-pricing`, `generate-reasoning`, then `generate-models`).
13
+ - Fix `generate-pricing` mis-dropping models whose pricing-page cached-input cell is `-` (no cache rate): those rows now match and their `cacheRead` equals `input`. This restored pricing for `mistral-large-3:675b`, `nemotron-3-nano:30b`, and `qwen3.5:397b`, which the earlier regex had left at zero cost.
14
+ - Refresh the model catalog: added `deepseek-v4.1-flash` (probed max output 393216).
15
+
16
+ ## [0.11.0] - 2026-09-07
6
17
 
7
18
  - Fix `/ollama-cloud-usage` and the usage status bar failing with "unexpected response shape" after the undocumented `/api/usage` endpoint flipped between a single `limits.monthly` bucket and `limits.session` plus `limits.weekly` (the shape has flip-flopped repeatedly as of 2026-09). Any bucket present (`monthly`, `session`, `weekly`) is accepted alone or in combination, and whichever are present are displayed as `5h`/`7d`/`30d` segments. Thanks @johanngyger (#56).
8
19
  - Cache `ollama_web_search` results (24h) and `ollama_web_fetch` pages (24h success / 15 min failure) on disk under the pi agent home, so repeated queries and page reads cost 0 API calls. Expired entries are pruned on write, the cache is capped at 500 entries per kind (oldest evicted beyond the cap; `PI_OLLAMA_SEARCH_MAX_ENTRIES`), a partially corrupted cache file is validated per entry and degrades to "no cache" instead of crashing tool calls, and the file is written with `0600` permissions since it stores page content and URLs that can embed credentials. Tune with `PI_OLLAMA_SEARCH_TTL_HOURS`, `PI_OLLAMA_SEARCH_FAIL_TTL_MINUTES`, `PI_OLLAMA_SEARCH_MAX_ENTRIES`, and `PI_OLLAMA_SEARCH_CACHE_PATH`.
package/README.md CHANGED
@@ -7,7 +7,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
7
7
  ## Features
8
8
 
9
9
  - **Dynamic model discovery** - Fetches the full model list from `ollama.com/v1/models`, then fetches per-model details via `/api/show` to determine capabilities, context length, and tool support.
10
- - **Curated thinking levels** - Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible `reasoning_effort` values via `thinking-levels.ts`, with per-model exceptions based on API testing.
10
+ - **Data-driven thinking levels** - Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible `reasoning_effort` values via `thinking-levels.ts`, sourced from models.dev per-model reasoning options with a small override table for the models where `none` doesn't disable thinking.
11
11
  - **Baked-in model list** - A generated fallback list (`models.generated.ts`) ships with the extension so models are available on first launch without any network calls. It is only a fallback: pi refreshes the live catalog at runtime, so shipping a new release for catalog freshness is no longer needed.
12
12
  - **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
13
13
  - **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
@@ -132,7 +132,7 @@ Model metadata is derived from the `/api/show` response:
132
132
  | Field | Source |
133
133
  |---|---|
134
134
  | `reasoning` | `capabilities` includes `"thinking"` |
135
- | `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 5 maps (DEFAULT, GPT_OSS, QWEN3, GLM_52, NO_OFF) based on API testing |
135
+ | `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) + [`reasoning.generated.ts`](reasoning.generated.ts) (models.dev reasoning options), with an `off` override table for models that ignore `none` |
136
136
  | `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
137
137
  | `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
138
138
  | `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
@@ -146,17 +146,11 @@ Cache pricing is informational only: the `/pricing` page lists a "Cached input"
146
146
 
147
147
  ### Thinking level mapping
148
148
 
149
- Pi's thinking levels are mapped to Ollama Cloud's OpenAI-compatible `reasoning_effort` parameter in [`thinking-levels.ts`](thinking-levels.ts). The API accepts `none`, `low`, `medium`, `high`, and `max`. Effects of `max` over `high` vary by model and prompt difficulty - see [`docs/think-experiment.md`](docs/think-experiment.md) for details.
149
+ Pi's thinking levels are mapped to Ollama Cloud's OpenAI-compatible `reasoning_effort` parameter in [`thinking-levels.ts`](thinking-levels.ts). The API accepts `none`, `low`, `medium`, `high`, `xhigh`, and `max`. Effects of `max` over `high` vary by model and prompt difficulty.
150
150
 
151
- | Map | Models | Levels exposed | Notes |
152
- |---|---|---|---|
153
- | `DEFAULT` | Most thinking models | off, low, medium, high, xhigh | `minimal` hidden (duplicate of low) |
154
- | `GPT_OSS` | `gpt-oss*` | low, medium, high | Can't disable thinking, no off or xhigh |
155
- | `QWEN3` | `qwen3*` (except `qwen3-vl*`) | off, medium | Binary-only (think/nothink), no gradation |
156
- | `GLM_52` | `glm-5.2` | off, high, xhigh | GLM supports disabled thinking; Ollama's model page confirms `high` and `max` reasoning efforts |
157
- | `NO_OFF` | `qwen3-vl*`, `kimi-k2-thinking`, `minimax*` | low, medium, high, xhigh | "none" doesn't disable thinking on these models |
151
+ Per-model support is sourced from models.dev: [`scripts/generate-reasoning.ts`](scripts/generate-reasoning.ts) fetches the `ollama-cloud` provider's `reasoning_options` into `reasoning.generated.ts`, and `resolve()` maps each model's effort values onto Pi's levels. Models with `effort` values expose those grades; `toggle`-only models expose a single on/off level. Models with no models.dev entry fall back to `DEFAULT`.
158
152
 
159
- See [docs/think-experiment.md](docs/think-experiment.md) for the testing methodology and results.
153
+ Because the API reports only a boolean `thinking` capability and models.dev does not reliably encode the `none` behavior, the `off` switch is handled via a small override table in `thinking-levels.ts`: it defaults to enabled, and is hidden only for models verified (by live probing) not to honor `reasoning_effort:"none"` - currently `gpt-oss:20b`, `gpt-oss:120b`, and `minimax-m2.7`. The per-model metadata gaps behind the models.dev sourcing are tracked upstream in [ollama/ollama#18385](https://github.com/ollama/ollama/issues/18385).
160
154
 
161
155
  ## Tools
162
156
 
@@ -1,10 +1,11 @@
1
1
  // Auto-generated by scripts/generate-limits.ts
2
2
  // Do not edit manually.
3
- // Probed models: 19 (0 failed)
3
+ // Entries: 20
4
4
 
5
5
  export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
6
6
  "deepseek-v4-flash:0731": 65536,
7
7
  "deepseek-v4-pro:0813": 65536,
8
+ "deepseek-v4.1-flash": 393216,
8
9
  "gemma4:31b": 262144,
9
10
  "glm-5.1": 131072,
10
11
  "glm-5.2": 131072,
@@ -1,7 +1,7 @@
1
1
  // Auto-generated by scripts/generate-models.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-09-03T10:12:02.243Z
4
- // Model count: 19
3
+ // Generated: 2026-09-14T10:27:20.856Z
4
+ // Model count: 20
5
5
 
6
6
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
7
7
 
@@ -11,7 +11,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
11
11
  name: "deepseek-v4-flash:0731",
12
12
  compat: {
13
13
  maxTokensField: "max_tokens",
14
- openRouterRouting: {},
15
14
  requiresAssistantAfterToolResult: false,
16
15
  requiresReasoningContentOnAssistantMessages: false,
17
16
  requiresThinkingAsText: false,
@@ -24,7 +23,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
24
23
  supportsStrictMode: false,
25
24
  supportsUsageInStreaming: true,
26
25
  thinkingFormat: "openai",
27
- vercelGatewayRouting: {},
28
26
  zaiToolStream: false,
29
27
  },
30
28
  contextWindow: 1048576,
@@ -39,8 +37,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
39
37
  reasoning: true,
40
38
  thinkingLevelMap: {
41
39
  high: "high",
42
- low: "low",
43
- medium: "medium",
40
+ low: null,
41
+ medium: null,
44
42
  minimal: null,
45
43
  off: "none",
46
44
  xhigh: "max",
@@ -51,7 +49,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
51
49
  name: "deepseek-v4-pro:0813",
52
50
  compat: {
53
51
  maxTokensField: "max_tokens",
54
- openRouterRouting: {},
55
52
  requiresAssistantAfterToolResult: false,
56
53
  requiresReasoningContentOnAssistantMessages: false,
57
54
  requiresThinkingAsText: false,
@@ -64,7 +61,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
64
61
  supportsStrictMode: false,
65
62
  supportsUsageInStreaming: true,
66
63
  thinkingFormat: "openai",
67
- vercelGatewayRouting: {},
68
64
  zaiToolStream: false,
69
65
  },
70
66
  contextWindow: 1048576,
@@ -77,10 +73,48 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
77
73
  input: ["text"],
78
74
  maxTokens: 65536,
79
75
  reasoning: true,
76
+ thinkingLevelMap: {
77
+ high: "high",
78
+ low: null,
79
+ medium: null,
80
+ minimal: null,
81
+ off: "none",
82
+ xhigh: "max",
83
+ },
84
+ },
85
+ {
86
+ id: "deepseek-v4.1-flash",
87
+ name: "deepseek-v4.1-flash",
88
+ compat: {
89
+ maxTokensField: "max_tokens",
90
+ requiresAssistantAfterToolResult: false,
91
+ requiresReasoningContentOnAssistantMessages: false,
92
+ requiresThinkingAsText: false,
93
+ requiresToolResultName: false,
94
+ sendSessionAffinityHeaders: false,
95
+ supportsDeveloperRole: false,
96
+ supportsLongCacheRetention: false,
97
+ supportsReasoningEffort: true,
98
+ supportsStore: false,
99
+ supportsStrictMode: false,
100
+ supportsUsageInStreaming: true,
101
+ thinkingFormat: "openai",
102
+ zaiToolStream: false,
103
+ },
104
+ contextWindow: 1048576,
105
+ cost: {
106
+ cacheRead: 0.006,
107
+ cacheWrite: 0,
108
+ input: 0.3,
109
+ output: 1.2,
110
+ },
111
+ input: ["text", "image"],
112
+ maxTokens: 393216,
113
+ reasoning: true,
80
114
  thinkingLevelMap: {
81
115
  high: "high",
82
116
  low: "low",
83
- medium: "medium",
117
+ medium: null,
84
118
  minimal: null,
85
119
  off: "none",
86
120
  xhigh: "max",
@@ -91,7 +125,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
91
125
  name: "gemma4:31b",
92
126
  compat: {
93
127
  maxTokensField: "max_tokens",
94
- openRouterRouting: {},
95
128
  requiresAssistantAfterToolResult: false,
96
129
  requiresReasoningContentOnAssistantMessages: false,
97
130
  requiresThinkingAsText: false,
@@ -104,7 +137,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
104
137
  supportsStrictMode: false,
105
138
  supportsUsageInStreaming: true,
106
139
  thinkingFormat: "openai",
107
- vercelGatewayRouting: {},
108
140
  zaiToolStream: false,
109
141
  },
110
142
  contextWindow: 262144,
@@ -118,12 +150,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
118
150
  maxTokens: 262144,
119
151
  reasoning: true,
120
152
  thinkingLevelMap: {
121
- high: "high",
122
- low: "low",
153
+ high: null,
154
+ low: null,
123
155
  medium: "medium",
124
156
  minimal: null,
125
157
  off: "none",
126
- xhigh: "max",
158
+ xhigh: null,
127
159
  },
128
160
  },
129
161
  {
@@ -131,7 +163,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
131
163
  name: "glm-5.1",
132
164
  compat: {
133
165
  maxTokensField: "max_tokens",
134
- openRouterRouting: {},
135
166
  requiresAssistantAfterToolResult: false,
136
167
  requiresReasoningContentOnAssistantMessages: false,
137
168
  requiresThinkingAsText: false,
@@ -144,7 +175,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
144
175
  supportsStrictMode: false,
145
176
  supportsUsageInStreaming: true,
146
177
  thinkingFormat: "openai",
147
- vercelGatewayRouting: {},
148
178
  zaiToolStream: false,
149
179
  },
150
180
  contextWindow: 202752,
@@ -158,12 +188,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
158
188
  maxTokens: 131072,
159
189
  reasoning: true,
160
190
  thinkingLevelMap: {
161
- high: "high",
162
- low: "low",
191
+ high: null,
192
+ low: null,
163
193
  medium: "medium",
164
194
  minimal: null,
165
195
  off: "none",
166
- xhigh: "max",
196
+ xhigh: null,
167
197
  },
168
198
  },
169
199
  {
@@ -171,7 +201,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
171
201
  name: "glm-5.2",
172
202
  compat: {
173
203
  maxTokensField: "max_tokens",
174
- openRouterRouting: {},
175
204
  requiresAssistantAfterToolResult: false,
176
205
  requiresReasoningContentOnAssistantMessages: false,
177
206
  requiresThinkingAsText: false,
@@ -184,7 +213,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
184
213
  supportsStrictMode: false,
185
214
  supportsUsageInStreaming: true,
186
215
  thinkingFormat: "openai",
187
- vercelGatewayRouting: {},
188
216
  zaiToolStream: false,
189
217
  },
190
218
  contextWindow: 1048576,
@@ -211,7 +239,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
211
239
  name: "glm-5.3",
212
240
  compat: {
213
241
  maxTokensField: "max_tokens",
214
- openRouterRouting: {},
215
242
  requiresAssistantAfterToolResult: false,
216
243
  requiresReasoningContentOnAssistantMessages: false,
217
244
  requiresThinkingAsText: false,
@@ -224,7 +251,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
224
251
  supportsStrictMode: false,
225
252
  supportsUsageInStreaming: true,
226
253
  thinkingFormat: "openai",
227
- vercelGatewayRouting: {},
228
254
  zaiToolStream: false,
229
255
  },
230
256
  contextWindow: 1048576,
@@ -240,7 +266,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
240
266
  thinkingLevelMap: {
241
267
  high: "high",
242
268
  low: "low",
243
- medium: "medium",
269
+ medium: null,
244
270
  minimal: null,
245
271
  off: "none",
246
272
  xhigh: "max",
@@ -251,7 +277,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
251
277
  name: "glm-5.3-flash",
252
278
  compat: {
253
279
  maxTokensField: "max_tokens",
254
- openRouterRouting: {},
255
280
  requiresAssistantAfterToolResult: false,
256
281
  requiresReasoningContentOnAssistantMessages: false,
257
282
  requiresThinkingAsText: false,
@@ -264,7 +289,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
264
289
  supportsStrictMode: false,
265
290
  supportsUsageInStreaming: true,
266
291
  thinkingFormat: "openai",
267
- vercelGatewayRouting: {},
268
292
  zaiToolStream: false,
269
293
  },
270
294
  contextWindow: 1048576,
@@ -280,7 +304,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
280
304
  thinkingLevelMap: {
281
305
  high: "high",
282
306
  low: "low",
283
- medium: "medium",
307
+ medium: null,
284
308
  minimal: null,
285
309
  off: "none",
286
310
  xhigh: "max",
@@ -291,7 +315,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
291
315
  name: "gpt-oss:120b",
292
316
  compat: {
293
317
  maxTokensField: "max_tokens",
294
- openRouterRouting: {},
295
318
  requiresAssistantAfterToolResult: false,
296
319
  requiresReasoningContentOnAssistantMessages: false,
297
320
  requiresThinkingAsText: false,
@@ -304,7 +327,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
304
327
  supportsStrictMode: false,
305
328
  supportsUsageInStreaming: true,
306
329
  thinkingFormat: "openai",
307
- vercelGatewayRouting: {},
308
330
  zaiToolStream: false,
309
331
  },
310
332
  contextWindow: 131072,
@@ -331,7 +353,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
331
353
  name: "gpt-oss:20b",
332
354
  compat: {
333
355
  maxTokensField: "max_tokens",
334
- openRouterRouting: {},
335
356
  requiresAssistantAfterToolResult: false,
336
357
  requiresReasoningContentOnAssistantMessages: false,
337
358
  requiresThinkingAsText: false,
@@ -344,7 +365,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
344
365
  supportsStrictMode: false,
345
366
  supportsUsageInStreaming: true,
346
367
  thinkingFormat: "openai",
347
- vercelGatewayRouting: {},
348
368
  zaiToolStream: false,
349
369
  },
350
370
  contextWindow: 131072,
@@ -371,7 +391,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
371
391
  name: "kimi-k2.6",
372
392
  compat: {
373
393
  maxTokensField: "max_tokens",
374
- openRouterRouting: {},
375
394
  requiresAssistantAfterToolResult: false,
376
395
  requiresReasoningContentOnAssistantMessages: false,
377
396
  requiresThinkingAsText: false,
@@ -384,7 +403,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
384
403
  supportsStrictMode: false,
385
404
  supportsUsageInStreaming: true,
386
405
  thinkingFormat: "openai",
387
- vercelGatewayRouting: {},
388
406
  zaiToolStream: false,
389
407
  },
390
408
  contextWindow: 262144,
@@ -398,12 +416,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
398
416
  maxTokens: 262144,
399
417
  reasoning: true,
400
418
  thinkingLevelMap: {
401
- high: "high",
402
- low: "low",
419
+ high: null,
420
+ low: null,
403
421
  medium: "medium",
404
422
  minimal: null,
405
423
  off: "none",
406
- xhigh: "max",
424
+ xhigh: null,
407
425
  },
408
426
  },
409
427
  {
@@ -411,7 +429,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
411
429
  name: "kimi-k2.7-code",
412
430
  compat: {
413
431
  maxTokensField: "max_tokens",
414
- openRouterRouting: {},
415
432
  requiresAssistantAfterToolResult: false,
416
433
  requiresReasoningContentOnAssistantMessages: false,
417
434
  requiresThinkingAsText: false,
@@ -424,7 +441,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
424
441
  supportsStrictMode: false,
425
442
  supportsUsageInStreaming: true,
426
443
  thinkingFormat: "openai",
427
- vercelGatewayRouting: {},
428
444
  zaiToolStream: false,
429
445
  },
430
446
  contextWindow: 262144,
@@ -438,12 +454,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
438
454
  maxTokens: 262144,
439
455
  reasoning: true,
440
456
  thinkingLevelMap: {
441
- high: "high",
442
- low: "low",
457
+ high: null,
458
+ low: null,
443
459
  medium: "medium",
444
460
  minimal: null,
445
461
  off: "none",
446
- xhigh: "max",
462
+ xhigh: null,
447
463
  },
448
464
  },
449
465
  {
@@ -451,7 +467,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
451
467
  name: "kimi-k3",
452
468
  compat: {
453
469
  maxTokensField: "max_tokens",
454
- openRouterRouting: {},
455
470
  requiresAssistantAfterToolResult: false,
456
471
  requiresReasoningContentOnAssistantMessages: false,
457
472
  requiresThinkingAsText: false,
@@ -464,7 +479,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
464
479
  supportsStrictMode: false,
465
480
  supportsUsageInStreaming: true,
466
481
  thinkingFormat: "openai",
467
- vercelGatewayRouting: {},
468
482
  zaiToolStream: false,
469
483
  },
470
484
  contextWindow: 1048576,
@@ -480,7 +494,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
480
494
  thinkingLevelMap: {
481
495
  high: "high",
482
496
  low: "low",
483
- medium: "medium",
497
+ medium: null,
484
498
  minimal: null,
485
499
  off: "none",
486
500
  xhigh: "max",
@@ -491,7 +505,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
491
505
  name: "minimax-m2.7",
492
506
  compat: {
493
507
  maxTokensField: "max_tokens",
494
- openRouterRouting: {},
495
508
  requiresAssistantAfterToolResult: false,
496
509
  requiresReasoningContentOnAssistantMessages: false,
497
510
  requiresThinkingAsText: false,
@@ -504,7 +517,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
504
517
  supportsStrictMode: false,
505
518
  supportsUsageInStreaming: true,
506
519
  thinkingFormat: "openai",
507
- vercelGatewayRouting: {},
508
520
  zaiToolStream: false,
509
521
  },
510
522
  contextWindow: 196608,
@@ -518,12 +530,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
518
530
  maxTokens: 131072,
519
531
  reasoning: true,
520
532
  thinkingLevelMap: {
521
- high: "high",
522
- low: "low",
533
+ high: null,
534
+ low: null,
523
535
  medium: "medium",
524
536
  minimal: null,
525
537
  off: null,
526
- xhigh: "max",
538
+ xhigh: null,
527
539
  },
528
540
  },
529
541
  {
@@ -531,7 +543,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
531
543
  name: "minimax-m3",
532
544
  compat: {
533
545
  maxTokensField: "max_tokens",
534
- openRouterRouting: {},
535
546
  requiresAssistantAfterToolResult: false,
536
547
  requiresReasoningContentOnAssistantMessages: false,
537
548
  requiresThinkingAsText: false,
@@ -544,7 +555,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
544
555
  supportsStrictMode: false,
545
556
  supportsUsageInStreaming: true,
546
557
  thinkingFormat: "openai",
547
- vercelGatewayRouting: {},
548
558
  zaiToolStream: false,
549
559
  },
550
560
  contextWindow: 512000,
@@ -562,7 +572,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
562
572
  low: "low",
563
573
  medium: "medium",
564
574
  minimal: null,
565
- off: null,
575
+ off: "none",
566
576
  xhigh: "max",
567
577
  },
568
578
  },
@@ -571,7 +581,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
571
581
  name: "mistral-large-3:675b",
572
582
  compat: {
573
583
  maxTokensField: "max_tokens",
574
- openRouterRouting: {},
575
584
  requiresAssistantAfterToolResult: false,
576
585
  requiresReasoningContentOnAssistantMessages: false,
577
586
  requiresThinkingAsText: false,
@@ -584,7 +593,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
584
593
  supportsStrictMode: false,
585
594
  supportsUsageInStreaming: true,
586
595
  thinkingFormat: "openai",
587
- vercelGatewayRouting: {},
588
596
  zaiToolStream: false,
589
597
  },
590
598
  contextWindow: 262144,
@@ -603,7 +611,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
603
611
  name: "nemotron-3-nano:30b",
604
612
  compat: {
605
613
  maxTokensField: "max_tokens",
606
- openRouterRouting: {},
607
614
  requiresAssistantAfterToolResult: false,
608
615
  requiresReasoningContentOnAssistantMessages: false,
609
616
  requiresThinkingAsText: false,
@@ -616,7 +623,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
616
623
  supportsStrictMode: false,
617
624
  supportsUsageInStreaming: true,
618
625
  thinkingFormat: "openai",
619
- vercelGatewayRouting: {},
620
626
  zaiToolStream: false,
621
627
  },
622
628
  contextWindow: 262144,
@@ -630,12 +636,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
630
636
  maxTokens: 131072,
631
637
  reasoning: true,
632
638
  thinkingLevelMap: {
633
- high: "high",
634
- low: "low",
639
+ high: null,
640
+ low: null,
635
641
  medium: "medium",
636
642
  minimal: null,
637
643
  off: "none",
638
- xhigh: "max",
644
+ xhigh: null,
639
645
  },
640
646
  },
641
647
  {
@@ -643,7 +649,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
643
649
  name: "nemotron-3-super",
644
650
  compat: {
645
651
  maxTokensField: "max_tokens",
646
- openRouterRouting: {},
647
652
  requiresAssistantAfterToolResult: false,
648
653
  requiresReasoningContentOnAssistantMessages: false,
649
654
  requiresThinkingAsText: false,
@@ -656,7 +661,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
656
661
  supportsStrictMode: false,
657
662
  supportsUsageInStreaming: true,
658
663
  thinkingFormat: "openai",
659
- vercelGatewayRouting: {},
660
664
  zaiToolStream: false,
661
665
  },
662
666
  contextWindow: 262144,
@@ -670,12 +674,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
670
674
  maxTokens: 65536,
671
675
  reasoning: true,
672
676
  thinkingLevelMap: {
673
- high: "high",
674
- low: "low",
677
+ high: null,
678
+ low: null,
675
679
  medium: "medium",
676
680
  minimal: null,
677
681
  off: "none",
678
- xhigh: "max",
682
+ xhigh: null,
679
683
  },
680
684
  },
681
685
  {
@@ -683,7 +687,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
683
687
  name: "nemotron-3-ultra",
684
688
  compat: {
685
689
  maxTokensField: "max_tokens",
686
- openRouterRouting: {},
687
690
  requiresAssistantAfterToolResult: false,
688
691
  requiresReasoningContentOnAssistantMessages: false,
689
692
  requiresThinkingAsText: false,
@@ -696,7 +699,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
696
699
  supportsStrictMode: false,
697
700
  supportsUsageInStreaming: true,
698
701
  thinkingFormat: "openai",
699
- vercelGatewayRouting: {},
700
702
  zaiToolStream: false,
701
703
  },
702
704
  contextWindow: 262144,
@@ -710,12 +712,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
710
712
  maxTokens: 65536,
711
713
  reasoning: true,
712
714
  thinkingLevelMap: {
713
- high: "high",
714
- low: "low",
715
+ high: null,
716
+ low: null,
715
717
  medium: "medium",
716
718
  minimal: null,
717
719
  off: "none",
718
- xhigh: "max",
720
+ xhigh: null,
719
721
  },
720
722
  },
721
723
  {
@@ -723,7 +725,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
723
725
  name: "qwen3.5:397b",
724
726
  compat: {
725
727
  maxTokensField: "max_tokens",
726
- openRouterRouting: {},
727
728
  requiresAssistantAfterToolResult: false,
728
729
  requiresReasoningContentOnAssistantMessages: false,
729
730
  requiresThinkingAsText: false,
@@ -736,7 +737,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
736
737
  supportsStrictMode: false,
737
738
  supportsUsageInStreaming: true,
738
739
  thinkingFormat: "openai",
739
- vercelGatewayRouting: {},
740
740
  zaiToolStream: false,
741
741
  },
742
742
  contextWindow: 262144,
package/models.ts CHANGED
@@ -93,7 +93,7 @@ function buildCompat(): ProviderModelConfig["compat"] {
93
93
  return {
94
94
  // Ollama uses "system" role, not "developer" (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsDeveloperRole).
95
95
  supportsDeveloperRole: false,
96
- // reasoning_effort works (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsReasoningEffort, tested in think-experiment.md).
96
+ // reasoning_effort works (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsReasoningEffort).
97
97
  supportsReasoningEffort: true,
98
98
  // "store" is not a supported field (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsStore).
99
99
  supportsStore: false,
@@ -109,22 +109,32 @@ function buildCompat(): ProviderModelConfig["compat"] {
109
109
  requiresThinkingAsText: false,
110
110
  // DeepSeek-specific, not needed for Ollama (pi: types.ts#requiresReasoningContentOnAssistantMessages).
111
111
  requiresReasoningContentOnAssistantMessages: false,
112
- // reasoning_effort format works (pi: types.ts#thinkingFormat, tested in think-experiment.md).
112
+ // reasoning_effort format works (pi: types.ts#thinkingFormat).
113
113
  thinkingFormat: "openai",
114
114
  // Ollama does not support tool_choice, so strict mode is unavailable (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsStrictMode).
115
115
  supportsStrictMode: false,
116
- // Anthropic cache_control not relevant; Ollama has implicit KV cache only (pi: types.ts#cacheControlFormat).
117
- // Explicitly undefined: JSON.stringify drops undefined values, keeping
118
- // models.generated.ts structurally consistent with assembleModels() runtime output.
119
116
  // Session affinity headers not relevant for Ollama (pi: types.ts#sendSessionAffinityHeaders).
120
117
  sendSessionAffinityHeaders: false,
121
118
  // No explicit cache-retention API (pi: types.ts#supportsLongCacheRetention).
122
119
  supportsLongCacheRetention: false,
123
120
  // Not z.ai (pi: types.ts#zaiToolStream).
124
121
  zaiToolStream: false,
122
+ // Anthropic cache_control not relevant; Ollama has implicit KV cache only (pi: types.ts#cacheControlFormat).
123
+ // Explicitly undefined: JSON.stringify drops undefined values, keeping
124
+ // models.generated.ts structurally consistent with assembleModels() runtime output.
125
125
  cacheControlFormat: undefined,
126
- openRouterRouting: {},
127
- vercelGatewayRouting: {},
126
+ // OpenRouter / Vercel AI Gateway routing prefs. pi-ai's OpenAI transport
127
+ // truthiness-checks the raw model.compat (not the resolved getCompat() value)
128
+ // and forwards it verbatim: `if (model.compat?.openRouterRouting)
129
+ // params.provider = ...` in packages/ai/src/api/openai-completions.ts. `{}` is
130
+ // truthy, so baking it in sent a stray `provider: {}` on every Ollama request.
131
+ // Express "none" as undefined, never `{}`; undefined is falsy at the wire site
132
+ // and JSON.stringify drops it from models.generated.ts. The boolean flags above
133
+ // stay explicit `false` on purpose: omitting a boolean makes getCompat() fall
134
+ // back to detectCompat's generic-OpenAI defaults, which are wrong for Ollama
135
+ // (e.g. maxTokensField -> "max_completion_tokens", supportsStrictMode -> true).
136
+ openRouterRouting: undefined,
137
+ vercelGatewayRouting: undefined,
128
138
  };
129
139
  }
130
140
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-ollama-cloud",
3
- "version": "0.11.0",
3
+ "version": "0.12.1",
4
4
  "type": "module",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -13,6 +13,7 @@
13
13
  "models.ts",
14
14
  "models.generated.ts",
15
15
  "pricing.generated.ts",
16
+ "reasoning.generated.ts",
16
17
  "thinking-levels.ts",
17
18
  "usage.ts",
18
19
  "utils.ts",
@@ -32,8 +33,9 @@
32
33
  "format": "biome format --write .",
33
34
  "test": "vitest run",
34
35
  "smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
35
- "generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts",
36
- "generate-limits": "tsx scripts/generate-limits.ts && biome format --write limits.generated.ts"
36
+ "generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-reasoning.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts reasoning.generated.ts",
37
+ "generate-limits": "tsx scripts/generate-limits.ts",
38
+ "generate-reasoning": "tsx scripts/generate-reasoning.ts && biome format --write reasoning.generated.ts"
37
39
  },
38
40
  "pi": {
39
41
  "extensions": [
@@ -51,6 +53,6 @@
51
53
  "@types/node": "^26.1.2",
52
54
  "@typescript/native-preview": "7.0.0-dev.20260707.2",
53
55
  "tsx": "^4.19.0",
54
- "vitest": "^4.1.6"
56
+ "vitest": "^4.1.11"
55
57
  }
56
58
  }
@@ -1,7 +1,7 @@
1
1
  // Auto-generated by scripts/generate-pricing.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-09-03T10:12:00.135Z
4
- // Model count: 19
3
+ // Generated: 2026-09-14T10:27:19.243Z
4
+ // Model count: 20
5
5
 
6
6
  export interface ModelPrice {
7
7
  input: number;
@@ -13,6 +13,7 @@ export interface ModelPrice {
13
13
  export const MODEL_PRICING: Record<string, ModelPrice> = {
14
14
  "deepseek-v4-flash:0731": { input: 0.44, output: 1.32, cacheRead: 0.014, cacheWrite: 0 },
15
15
  "deepseek-v4-pro:0813": { input: 1.32, output: 3.96, cacheRead: 0.044, cacheWrite: 0 },
16
+ "deepseek-v4.1-flash": { input: 0.3, output: 1.2, cacheRead: 0.006, cacheWrite: 0 },
16
17
  "gemma4:31b": { input: 0.14, output: 0.4, cacheRead: 0.05, cacheWrite: 0 },
17
18
  "glm-5.1": { input: 1, output: 3.2, cacheRead: 0.2, cacheWrite: 0 },
18
19
  "glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
@@ -0,0 +1,36 @@
1
+ // Auto-generated by scripts/generate-reasoning.ts
2
+ // Do not edit manually.
3
+ // Model count: 23
4
+
5
+ export type ModelsDevReasoningOption =
6
+ | { type: "toggle" }
7
+ | {
8
+ type: "effort";
9
+ values: Array<"none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | "ultra" | "default" | null>;
10
+ };
11
+
12
+ export const MODEL_REASONING_OPTIONS: Record<string, ModelsDevReasoningOption[]> = {
13
+ "deepseek-v4-flash": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
14
+ "deepseek-v4-flash:0731": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
15
+ "deepseek-v4-pro": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
16
+ "deepseek-v4-pro:0813": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
17
+ "deepseek-v4.1-flash": [{ type: "toggle" }, { type: "effort", values: ["low", "high", "max"] }],
18
+ "gemma4:31b": [{ type: "toggle" }],
19
+ "glm-5.1": [{ type: "toggle" }],
20
+ "glm-5.2": [{ type: "effort", values: ["high", "max"] }],
21
+ "glm-5.3": [{ type: "effort", values: ["low", "high", "max"] }],
22
+ "glm-5.3-flash": [{ type: "effort", values: ["low", "high", "max"] }],
23
+ "gpt-oss:120b": [{ type: "effort", values: ["low", "medium", "high"] }],
24
+ "gpt-oss:20b": [{ type: "effort", values: ["low", "medium", "high"] }],
25
+ "kimi-k2.5": [{ type: "toggle" }],
26
+ "kimi-k2.6": [{ type: "toggle" }],
27
+ "kimi-k2.7-code": [{ type: "toggle" }],
28
+ "kimi-k3": [{ type: "toggle" }, { type: "effort", values: ["low", "high", "max"] }],
29
+ "minimax-m2.5": [],
30
+ "minimax-m2.7": [{ type: "toggle" }],
31
+ "minimax-m3": [{ type: "toggle" }, { type: "effort", values: ["low", "medium", "high", "max"] }],
32
+ "nemotron-3-nano:30b": [{ type: "toggle" }],
33
+ "nemotron-3-super": [{ type: "toggle" }],
34
+ "nemotron-3-ultra": [{ type: "toggle" }],
35
+ "qwen3.5:397b": [{ type: "toggle" }],
36
+ };
@@ -2,26 +2,36 @@
2
2
  * Thinking level mapping for Ollama Cloud models.
3
3
  *
4
4
  * Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible
5
- * `reasoning_effort` values. The API accepts "none", "low", "medium",
6
- * "high", and "max". On simple prompts, "max" can be a no-op over
7
- * "high", but on harder prompts it can increase thinking substantially
8
- * (e.g. deepseek-v4-pro: ~32k tokens on high vs ~55k on max).
5
+ * `reasoning_effort` values. The API accepts "minimal", "none", "low",
6
+ * "medium", "high", "xhigh", "ultra", and "max". On simple prompts, "max" can
7
+ * be a no-op over "high", but on harder prompts it can increase thinking
8
+ * substantially.
9
9
  *
10
- * A `null` value means the level is hidden in Pi's UI.
10
+ * Every value EFFORT_TO_LEVEL can send was verified against the live chat
11
+ * completions API (2026-09-11: "minimal", "xhigh", and "ultra" probed across
12
+ * gpt-oss, deepseek-v4, glm, minimax, and qwen thinking models, all accepted
13
+ * with graded reasoning), so a future models.dev row that lists them passes
14
+ * through a value the endpoint demonstrably accepts.
15
+ *
16
+ * The per-model level support comes from models.dev: scripts/generate-reasoning.ts
17
+ * fetches the `ollama-cloud` provider's `reasoning_options` into
18
+ * reasoning.generated.ts (the same data source pi uses for its built-in
19
+ * providers), and resolve() maps a model's effort values onto Pi's levels.
20
+ * We fall back to models.dev because the Cloud API does not yet expose
21
+ * per-model supported levels (tracked upstream: https://github.com/ollama/ollama/issues/18385).
11
22
  *
12
- * Model-specific behavior discovered through testing (see docs/think-experiment.md):
13
- * - Most models: all levels work, "none" disables thinking
14
- * - GPT-OSS: no off mode, only low/medium/high
15
- * - Qwen 3.x (non-VL): binary-only (think/nothink) - off works
16
- * - Qwen 3 VL: "none" doesn't disable thinking - off is hidden
17
- * - GLM 5.2: off/high/max are exposed; low/medium are hidden
18
- * - Kimi K2 Thinking: "none" doesn't disable thinking - off is hidden
19
- * - MiniMax M2.x: "none" doesn't disable thinking - off is hidden
23
+ * The API exposes only a boolean `thinking` capability plus a global effort
24
+ * vocabulary, and models.dev does not reliably encode the `none` behavior, so
25
+ * the `off` switch is handled separately: it defaults to "none" (a live probe
26
+ * of the current catalog confirmed every model except the OFF_NULL overrides
27
+ * below honors it), and models verified not to honor `none` pin it to null
28
+ * (hidden) via OFF_NULL.
20
29
  *
21
- * Reference: https://docs.ollama.com/api/openai-compatibility
30
+ * A `null` value means the level is hidden in Pi's UI.
22
31
  */
23
32
 
24
33
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
34
+ import { MODEL_REASONING_OPTIONS, type ModelsDevReasoningOption } from "./reasoning.generated.ts";
25
35
 
26
36
  export type ThinkingLevelMap = NonNullable<ProviderModelConfig["thinkingLevelMap"]>;
27
37
 
@@ -35,63 +45,101 @@ export const DEFAULT: ThinkingLevelMap = {
35
45
  xhigh: "max",
36
46
  };
37
47
 
38
- /** GPT-OSS: can't disable thinking, only low/medium/high.
39
- * https://ollama.com/library/gpt-oss */
40
- export const GPT_OSS: ThinkingLevelMap = {
41
- off: null,
42
- minimal: null,
43
- low: "low",
44
- medium: "medium",
45
- high: "high",
46
- xhigh: null,
47
- };
48
+ /**
49
+ * Models where a live probe of `reasoning_effort:"none"` still produced
50
+ * reasoning (i.e. thinking cannot be disabled), so the `off` level is hidden.
51
+ * Confirmed against the current catalog by scripts/../test probing; the API
52
+ * and models.dev do not expose this behavior.
53
+ */
54
+ /**
55
+ * Models where a live probe of `reasoning_effort:"none"` still produced
56
+ * reasoning (i.e. thinking cannot be disabled), so the `off` level is hidden.
57
+ * Confirmed against the current catalog by scripts/../test probing; the API
58
+ * and models.dev do not expose this behavior.
59
+ *
60
+ * Exact ids hide only the named variant; family prefixes cover future model
61
+ * revisions in the same family. gpt-oss is a family-wide prefix because both
62
+ * probed variants leak and the behavior is documented for the family.
63
+ * minimax is NOT matched family-wide here: minimax-m3 was probed to honor
64
+ * `none`, so only the verified-leaking minimax-m2.7 is pinned by exact id.
65
+ */
66
+ const OFF_NULL_EXACT = new Set(["minimax-m2.7"]);
67
+ const OFF_NULL_FAMILIES = ["gpt-oss"];
48
68
 
49
- /** Qwen 3.x: binary-only (think/nothink), no gradation.
50
- * https://docs.ollama.com/capabilities/thinking */
51
- export const QWEN3: ThinkingLevelMap = {
52
- off: "none",
53
- minimal: null,
54
- low: null,
55
- medium: "medium",
56
- high: null,
57
- xhigh: null,
58
- };
69
+ function hidesOff(id: string): boolean {
70
+ return OFF_NULL_EXACT.has(id) || OFF_NULL_FAMILIES.some((prefix) => id.startsWith(prefix));
71
+ }
59
72
 
60
- /** GLM 5.2: Ollama's model page confirms support for "high" and "max" reasoning efforts.
61
- * https://ollama.com/library/glm-5.2 */
62
- export const GLM_52: ThinkingLevelMap = {
63
- off: "none",
64
- minimal: null,
65
- low: null,
66
- medium: null,
67
- high: "high",
68
- xhigh: "max",
73
+ /**
74
+ * Map a models.dev `effort` value onto the Pi level key and the reasoning_effort
75
+ * string to send for it. Ollama's top effort value is "max"; Pi exposes it via
76
+ * the extra-high level, so "max" (and "xhigh"/"ultra") map to the xhigh key.
77
+ */
78
+ const EFFORT_TO_LEVEL: Record<string, { key: "minimal" | "low" | "medium" | "high" | "xhigh"; value: string }> = {
79
+ minimal: { key: "minimal", value: "minimal" },
80
+ low: { key: "low", value: "low" },
81
+ medium: { key: "medium", value: "medium" },
82
+ high: { key: "high", value: "high" },
83
+ xhigh: { key: "xhigh", value: "xhigh" },
84
+ max: { key: "xhigh", value: "max" },
85
+ ultra: { key: "xhigh", value: "ultra" },
69
86
  };
87
+ /**
88
+ * Build a ThinkingLevelMap from models.dev reasoning_options.
89
+ * Levels come from `effort` values (mapped via EFFORT_TO_LEVEL); `off` defaults
90
+ * to "none" (probe-derived, see file header) and is hidden only via the
91
+ * OFF_NULL exact/family sets (see hidesOff).
92
+ * A toggle-only model is binary (on/off) and exposes a single "medium" level.
93
+ */
94
+ function buildMap(options: readonly ModelsDevReasoningOption[], id: string): ThinkingLevelMap {
95
+ const map: ThinkingLevelMap = {
96
+ off: hidesOff(id) ? null : "none",
97
+ minimal: null,
98
+ low: null,
99
+ medium: null,
100
+ high: null,
101
+ xhigh: null,
102
+ };
70
103
 
71
- /** "none" doesn't disable thinking - off is hidden.
72
- * Used by kimi and minimax families. */
73
- export const NO_OFF: ThinkingLevelMap = {
74
- off: null,
75
- minimal: null,
76
- low: "low",
77
- medium: "medium",
78
- high: "high",
79
- xhigh: "max",
80
- };
104
+ if (options.length > 0 && options.every((option) => option.type === "toggle")) {
105
+ // Binary on/off model: no graded effort, expose a single level.
106
+ return { ...map, medium: "medium" };
107
+ }
108
+
109
+ const efforts = options.flatMap((option) => (option.type === "effort" ? (option.values ?? []) : []));
110
+ for (const effort of efforts) {
111
+ const target = effort !== null && effort !== "default" ? EFFORT_TO_LEVEL[effort] : undefined;
112
+ if (target) map[target.key] = target.value;
113
+ }
114
+ return map;
115
+ }
116
+
117
+ /**
118
+ * Read MODEL_REASONING_OPTIONS[id] without tripping over inherited keys (e.g.
119
+ * "constructor"), which would otherwise resolve to the Object constructor and
120
+ * crash buildMap.
121
+ */
122
+ function ownOptions(id: string): ModelsDevReasoningOption[] | undefined {
123
+ return Object.hasOwn(MODEL_REASONING_OPTIONS, id) ? MODEL_REASONING_OPTIONS[id] : undefined;
124
+ }
81
125
 
82
126
  /**
83
127
  * Resolve the thinking level map for a model.
84
- * Matches by model ID prefix (case-sensitive, checks first chars).
128
+ * Looks up the model id (exact, then `:tag` family) in the generated models.dev
129
+ * table, falling back to DEFAULT for models with no entry. The matched key is
130
+ * the one passed to buildMap so the OFF_NULL set (keyed on bare family names)
131
+ * applies to tagged ids that resolve through a family match.
85
132
  */
86
133
  export function resolve(id: string, capabilities: string[]): ThinkingLevelMap | undefined {
87
134
  if (!capabilities.includes("thinking")) return undefined;
88
135
 
89
- if (id.startsWith("gpt-oss")) return GPT_OSS;
90
- if (id === "glm-5.2") return GLM_52;
91
- if (id.startsWith("qwen3-vl")) return NO_OFF;
92
- if (id.startsWith("qwen3")) return QWEN3;
93
- if (id === "kimi-k2-thinking") return NO_OFF;
94
- if (id.startsWith("minimax")) return NO_OFF;
95
-
96
- return DEFAULT;
136
+ const colon = id.lastIndexOf(":");
137
+ const exact = ownOptions(id);
138
+ const matchedKey = exact !== undefined ? id : colon > 0 ? id.slice(0, colon) : "";
139
+ const options = exact ?? ownOptions(matchedKey);
140
+ // An empty array (e.g. minimax-m2.5) carries no verified options; fall back
141
+ // to DEFAULT rather than a degenerate map whose only selectable level can
142
+ // be a leaking off.
143
+ if (options === undefined || options.length === 0) return DEFAULT;
144
+ return buildMap(options, matchedKey);
97
145
  }