pi-ollama-cloud 0.11.0 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -1
- package/README.md +5 -11
- package/limits.generated.ts +2 -1
- package/models.generated.ts +71 -71
- package/models.ts +17 -7
- package/package.json +6 -4
- package/pricing.generated.ts +3 -2
- package/reasoning.generated.ts +36 -0
- package/thinking-levels.ts +110 -62
package/CHANGELOG.md
CHANGED
|
@@ -2,7 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
-
## [
|
|
5
|
+
## [0.12.1] - 2026-09-14
|
|
6
|
+
|
|
7
|
+
- Omit empty `openRouterRouting` / `vercelGatewayRouting` from `buildCompat` (set to `undefined`, not `{}`): pi-ai reads the raw `model.compat` and treats `{}` as truthy, sending a stray `provider: {}` on every Ollama chat completion. Fixes #60. Thanks @0xbentang (#61).
|
|
8
|
+
|
|
9
|
+
## [0.12.0] - 2026-09-11
|
|
10
|
+
|
|
11
|
+
- Source per-model thinking levels from models.dev instead of hardcoded maps. `scripts/generate-reasoning.ts` fetches the `ollama-cloud` provider's `reasoning_options` into `reasoning.generated.ts`, and `thinking-levels.ts` maps each model's effort values onto Pi's levels (toggle-only models become a binary on/off map; models with no models.dev entry fall back to `DEFAULT`). The `off` switch is handled by a small override table for models verified not to honor `reasoning_effort:"none"` (`gpt-oss:20b`, `gpt-oss:120b`, `minimax-m2.7`). Removed the now-stale per-family maps and the `docs/think-experiment.md` doc.
|
|
12
|
+
- `generate-models` now also refreshes `reasoning.generated.ts` (runs `generate-pricing`, `generate-reasoning`, then `generate-models`).
|
|
13
|
+
- Fix `generate-pricing` mis-dropping models whose pricing-page cached-input cell is `-` (no cache rate): those rows now match and their `cacheRead` equals `input`. This restored pricing for `mistral-large-3:675b`, `nemotron-3-nano:30b`, and `qwen3.5:397b`, which the earlier regex had left at zero cost.
|
|
14
|
+
- Refresh the model catalog: added `deepseek-v4.1-flash` (probed max output 393216).
|
|
15
|
+
|
|
16
|
+
## [0.11.0] - 2026-09-07
|
|
6
17
|
|
|
7
18
|
- Fix `/ollama-cloud-usage` and the usage status bar failing with "unexpected response shape" after the undocumented `/api/usage` endpoint flipped between a single `limits.monthly` bucket and `limits.session` plus `limits.weekly` (the shape has flip-flopped repeatedly as of 2026-09). Any bucket present (`monthly`, `session`, `weekly`) is accepted alone or in combination, and whichever are present are displayed as `5h`/`7d`/`30d` segments. Thanks @johanngyger (#56).
|
|
8
19
|
- Cache `ollama_web_search` results (24h) and `ollama_web_fetch` pages (24h success / 15 min failure) on disk under the pi agent home, so repeated queries and page reads cost 0 API calls. Expired entries are pruned on write, the cache is capped at 500 entries per kind (oldest evicted beyond the cap; `PI_OLLAMA_SEARCH_MAX_ENTRIES`), a partially corrupted cache file is validated per entry and degrades to "no cache" instead of crashing tool calls, and the file is written with `0600` permissions since it stores page content and URLs that can embed credentials. Tune with `PI_OLLAMA_SEARCH_TTL_HOURS`, `PI_OLLAMA_SEARCH_FAIL_TTL_MINUTES`, `PI_OLLAMA_SEARCH_MAX_ENTRIES`, and `PI_OLLAMA_SEARCH_CACHE_PATH`.
|
package/README.md
CHANGED
|
@@ -7,7 +7,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
|
|
|
7
7
|
## Features
|
|
8
8
|
|
|
9
9
|
- **Dynamic model discovery** - Fetches the full model list from `ollama.com/v1/models`, then fetches per-model details via `/api/show` to determine capabilities, context length, and tool support.
|
|
10
|
-
- **
|
|
10
|
+
- **Data-driven thinking levels** - Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible `reasoning_effort` values via `thinking-levels.ts`, sourced from models.dev per-model reasoning options with a small override table for the models where `none` doesn't disable thinking.
|
|
11
11
|
- **Baked-in model list** - A generated fallback list (`models.generated.ts`) ships with the extension so models are available on first launch without any network calls. It is only a fallback: pi refreshes the live catalog at runtime, so shipping a new release for catalog freshness is no longer needed.
|
|
12
12
|
- **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
|
|
13
13
|
- **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
|
|
@@ -132,7 +132,7 @@ Model metadata is derived from the `/api/show` response:
|
|
|
132
132
|
| Field | Source |
|
|
133
133
|
|---|---|
|
|
134
134
|
| `reasoning` | `capabilities` includes `"thinking"` |
|
|
135
|
-
| `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts)
|
|
135
|
+
| `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) + [`reasoning.generated.ts`](reasoning.generated.ts) (models.dev reasoning options), with an `off` override table for models that ignore `none` |
|
|
136
136
|
| `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
|
|
137
137
|
| `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
|
|
138
138
|
| `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
|
|
@@ -146,17 +146,11 @@ Cache pricing is informational only: the `/pricing` page lists a "Cached input"
|
|
|
146
146
|
|
|
147
147
|
### Thinking level mapping
|
|
148
148
|
|
|
149
|
-
Pi's thinking levels are mapped to Ollama Cloud's OpenAI-compatible `reasoning_effort` parameter in [`thinking-levels.ts`](thinking-levels.ts). The API accepts `none`, `low`, `medium`, `high`, and `max`. Effects of `max` over `high` vary by model and prompt difficulty
|
|
149
|
+
Pi's thinking levels are mapped to Ollama Cloud's OpenAI-compatible `reasoning_effort` parameter in [`thinking-levels.ts`](thinking-levels.ts). The API accepts `none`, `low`, `medium`, `high`, `xhigh`, and `max`. Effects of `max` over `high` vary by model and prompt difficulty.
|
|
150
150
|
|
|
151
|
-
|
|
152
|
-
|---|---|---|---|
|
|
153
|
-
| `DEFAULT` | Most thinking models | off, low, medium, high, xhigh | `minimal` hidden (duplicate of low) |
|
|
154
|
-
| `GPT_OSS` | `gpt-oss*` | low, medium, high | Can't disable thinking, no off or xhigh |
|
|
155
|
-
| `QWEN3` | `qwen3*` (except `qwen3-vl*`) | off, medium | Binary-only (think/nothink), no gradation |
|
|
156
|
-
| `GLM_52` | `glm-5.2` | off, high, xhigh | GLM supports disabled thinking; Ollama's model page confirms `high` and `max` reasoning efforts |
|
|
157
|
-
| `NO_OFF` | `qwen3-vl*`, `kimi-k2-thinking`, `minimax*` | low, medium, high, xhigh | "none" doesn't disable thinking on these models |
|
|
151
|
+
Per-model support is sourced from models.dev: [`scripts/generate-reasoning.ts`](scripts/generate-reasoning.ts) fetches the `ollama-cloud` provider's `reasoning_options` into `reasoning.generated.ts`, and `resolve()` maps each model's effort values onto Pi's levels. Models with `effort` values expose those grades; `toggle`-only models expose a single on/off level. Models with no models.dev entry fall back to `DEFAULT`.
|
|
158
152
|
|
|
159
|
-
|
|
153
|
+
Because the API reports only a boolean `thinking` capability and models.dev does not reliably encode the `none` behavior, the `off` switch is handled via a small override table in `thinking-levels.ts`: it defaults to enabled, and is hidden only for models verified (by live probing) not to honor `reasoning_effort:"none"` - currently `gpt-oss:20b`, `gpt-oss:120b`, and `minimax-m2.7`. The per-model metadata gaps behind the models.dev sourcing are tracked upstream in [ollama/ollama#18385](https://github.com/ollama/ollama/issues/18385).
|
|
160
154
|
|
|
161
155
|
## Tools
|
|
162
156
|
|
package/limits.generated.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-limits.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
//
|
|
3
|
+
// Entries: 20
|
|
4
4
|
|
|
5
5
|
export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
6
6
|
"deepseek-v4-flash:0731": 65536,
|
|
7
7
|
"deepseek-v4-pro:0813": 65536,
|
|
8
|
+
"deepseek-v4.1-flash": 393216,
|
|
8
9
|
"gemma4:31b": 262144,
|
|
9
10
|
"glm-5.1": 131072,
|
|
10
11
|
"glm-5.2": 131072,
|
package/models.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-models.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-09-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-14T10:27:20.856Z
|
|
4
|
+
// Model count: 20
|
|
5
5
|
|
|
6
6
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
7
7
|
|
|
@@ -11,7 +11,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
11
11
|
name: "deepseek-v4-flash:0731",
|
|
12
12
|
compat: {
|
|
13
13
|
maxTokensField: "max_tokens",
|
|
14
|
-
openRouterRouting: {},
|
|
15
14
|
requiresAssistantAfterToolResult: false,
|
|
16
15
|
requiresReasoningContentOnAssistantMessages: false,
|
|
17
16
|
requiresThinkingAsText: false,
|
|
@@ -24,7 +23,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
24
23
|
supportsStrictMode: false,
|
|
25
24
|
supportsUsageInStreaming: true,
|
|
26
25
|
thinkingFormat: "openai",
|
|
27
|
-
vercelGatewayRouting: {},
|
|
28
26
|
zaiToolStream: false,
|
|
29
27
|
},
|
|
30
28
|
contextWindow: 1048576,
|
|
@@ -39,8 +37,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
39
37
|
reasoning: true,
|
|
40
38
|
thinkingLevelMap: {
|
|
41
39
|
high: "high",
|
|
42
|
-
low:
|
|
43
|
-
medium:
|
|
40
|
+
low: null,
|
|
41
|
+
medium: null,
|
|
44
42
|
minimal: null,
|
|
45
43
|
off: "none",
|
|
46
44
|
xhigh: "max",
|
|
@@ -51,7 +49,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
51
49
|
name: "deepseek-v4-pro:0813",
|
|
52
50
|
compat: {
|
|
53
51
|
maxTokensField: "max_tokens",
|
|
54
|
-
openRouterRouting: {},
|
|
55
52
|
requiresAssistantAfterToolResult: false,
|
|
56
53
|
requiresReasoningContentOnAssistantMessages: false,
|
|
57
54
|
requiresThinkingAsText: false,
|
|
@@ -64,7 +61,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
64
61
|
supportsStrictMode: false,
|
|
65
62
|
supportsUsageInStreaming: true,
|
|
66
63
|
thinkingFormat: "openai",
|
|
67
|
-
vercelGatewayRouting: {},
|
|
68
64
|
zaiToolStream: false,
|
|
69
65
|
},
|
|
70
66
|
contextWindow: 1048576,
|
|
@@ -77,10 +73,48 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
77
73
|
input: ["text"],
|
|
78
74
|
maxTokens: 65536,
|
|
79
75
|
reasoning: true,
|
|
76
|
+
thinkingLevelMap: {
|
|
77
|
+
high: "high",
|
|
78
|
+
low: null,
|
|
79
|
+
medium: null,
|
|
80
|
+
minimal: null,
|
|
81
|
+
off: "none",
|
|
82
|
+
xhigh: "max",
|
|
83
|
+
},
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
id: "deepseek-v4.1-flash",
|
|
87
|
+
name: "deepseek-v4.1-flash",
|
|
88
|
+
compat: {
|
|
89
|
+
maxTokensField: "max_tokens",
|
|
90
|
+
requiresAssistantAfterToolResult: false,
|
|
91
|
+
requiresReasoningContentOnAssistantMessages: false,
|
|
92
|
+
requiresThinkingAsText: false,
|
|
93
|
+
requiresToolResultName: false,
|
|
94
|
+
sendSessionAffinityHeaders: false,
|
|
95
|
+
supportsDeveloperRole: false,
|
|
96
|
+
supportsLongCacheRetention: false,
|
|
97
|
+
supportsReasoningEffort: true,
|
|
98
|
+
supportsStore: false,
|
|
99
|
+
supportsStrictMode: false,
|
|
100
|
+
supportsUsageInStreaming: true,
|
|
101
|
+
thinkingFormat: "openai",
|
|
102
|
+
zaiToolStream: false,
|
|
103
|
+
},
|
|
104
|
+
contextWindow: 1048576,
|
|
105
|
+
cost: {
|
|
106
|
+
cacheRead: 0.006,
|
|
107
|
+
cacheWrite: 0,
|
|
108
|
+
input: 0.3,
|
|
109
|
+
output: 1.2,
|
|
110
|
+
},
|
|
111
|
+
input: ["text", "image"],
|
|
112
|
+
maxTokens: 393216,
|
|
113
|
+
reasoning: true,
|
|
80
114
|
thinkingLevelMap: {
|
|
81
115
|
high: "high",
|
|
82
116
|
low: "low",
|
|
83
|
-
medium:
|
|
117
|
+
medium: null,
|
|
84
118
|
minimal: null,
|
|
85
119
|
off: "none",
|
|
86
120
|
xhigh: "max",
|
|
@@ -91,7 +125,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
91
125
|
name: "gemma4:31b",
|
|
92
126
|
compat: {
|
|
93
127
|
maxTokensField: "max_tokens",
|
|
94
|
-
openRouterRouting: {},
|
|
95
128
|
requiresAssistantAfterToolResult: false,
|
|
96
129
|
requiresReasoningContentOnAssistantMessages: false,
|
|
97
130
|
requiresThinkingAsText: false,
|
|
@@ -104,7 +137,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
104
137
|
supportsStrictMode: false,
|
|
105
138
|
supportsUsageInStreaming: true,
|
|
106
139
|
thinkingFormat: "openai",
|
|
107
|
-
vercelGatewayRouting: {},
|
|
108
140
|
zaiToolStream: false,
|
|
109
141
|
},
|
|
110
142
|
contextWindow: 262144,
|
|
@@ -118,12 +150,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
118
150
|
maxTokens: 262144,
|
|
119
151
|
reasoning: true,
|
|
120
152
|
thinkingLevelMap: {
|
|
121
|
-
high:
|
|
122
|
-
low:
|
|
153
|
+
high: null,
|
|
154
|
+
low: null,
|
|
123
155
|
medium: "medium",
|
|
124
156
|
minimal: null,
|
|
125
157
|
off: "none",
|
|
126
|
-
xhigh:
|
|
158
|
+
xhigh: null,
|
|
127
159
|
},
|
|
128
160
|
},
|
|
129
161
|
{
|
|
@@ -131,7 +163,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
131
163
|
name: "glm-5.1",
|
|
132
164
|
compat: {
|
|
133
165
|
maxTokensField: "max_tokens",
|
|
134
|
-
openRouterRouting: {},
|
|
135
166
|
requiresAssistantAfterToolResult: false,
|
|
136
167
|
requiresReasoningContentOnAssistantMessages: false,
|
|
137
168
|
requiresThinkingAsText: false,
|
|
@@ -144,7 +175,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
144
175
|
supportsStrictMode: false,
|
|
145
176
|
supportsUsageInStreaming: true,
|
|
146
177
|
thinkingFormat: "openai",
|
|
147
|
-
vercelGatewayRouting: {},
|
|
148
178
|
zaiToolStream: false,
|
|
149
179
|
},
|
|
150
180
|
contextWindow: 202752,
|
|
@@ -158,12 +188,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
158
188
|
maxTokens: 131072,
|
|
159
189
|
reasoning: true,
|
|
160
190
|
thinkingLevelMap: {
|
|
161
|
-
high:
|
|
162
|
-
low:
|
|
191
|
+
high: null,
|
|
192
|
+
low: null,
|
|
163
193
|
medium: "medium",
|
|
164
194
|
minimal: null,
|
|
165
195
|
off: "none",
|
|
166
|
-
xhigh:
|
|
196
|
+
xhigh: null,
|
|
167
197
|
},
|
|
168
198
|
},
|
|
169
199
|
{
|
|
@@ -171,7 +201,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
171
201
|
name: "glm-5.2",
|
|
172
202
|
compat: {
|
|
173
203
|
maxTokensField: "max_tokens",
|
|
174
|
-
openRouterRouting: {},
|
|
175
204
|
requiresAssistantAfterToolResult: false,
|
|
176
205
|
requiresReasoningContentOnAssistantMessages: false,
|
|
177
206
|
requiresThinkingAsText: false,
|
|
@@ -184,7 +213,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
184
213
|
supportsStrictMode: false,
|
|
185
214
|
supportsUsageInStreaming: true,
|
|
186
215
|
thinkingFormat: "openai",
|
|
187
|
-
vercelGatewayRouting: {},
|
|
188
216
|
zaiToolStream: false,
|
|
189
217
|
},
|
|
190
218
|
contextWindow: 1048576,
|
|
@@ -211,7 +239,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
211
239
|
name: "glm-5.3",
|
|
212
240
|
compat: {
|
|
213
241
|
maxTokensField: "max_tokens",
|
|
214
|
-
openRouterRouting: {},
|
|
215
242
|
requiresAssistantAfterToolResult: false,
|
|
216
243
|
requiresReasoningContentOnAssistantMessages: false,
|
|
217
244
|
requiresThinkingAsText: false,
|
|
@@ -224,7 +251,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
224
251
|
supportsStrictMode: false,
|
|
225
252
|
supportsUsageInStreaming: true,
|
|
226
253
|
thinkingFormat: "openai",
|
|
227
|
-
vercelGatewayRouting: {},
|
|
228
254
|
zaiToolStream: false,
|
|
229
255
|
},
|
|
230
256
|
contextWindow: 1048576,
|
|
@@ -240,7 +266,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
240
266
|
thinkingLevelMap: {
|
|
241
267
|
high: "high",
|
|
242
268
|
low: "low",
|
|
243
|
-
medium:
|
|
269
|
+
medium: null,
|
|
244
270
|
minimal: null,
|
|
245
271
|
off: "none",
|
|
246
272
|
xhigh: "max",
|
|
@@ -251,7 +277,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
251
277
|
name: "glm-5.3-flash",
|
|
252
278
|
compat: {
|
|
253
279
|
maxTokensField: "max_tokens",
|
|
254
|
-
openRouterRouting: {},
|
|
255
280
|
requiresAssistantAfterToolResult: false,
|
|
256
281
|
requiresReasoningContentOnAssistantMessages: false,
|
|
257
282
|
requiresThinkingAsText: false,
|
|
@@ -264,7 +289,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
264
289
|
supportsStrictMode: false,
|
|
265
290
|
supportsUsageInStreaming: true,
|
|
266
291
|
thinkingFormat: "openai",
|
|
267
|
-
vercelGatewayRouting: {},
|
|
268
292
|
zaiToolStream: false,
|
|
269
293
|
},
|
|
270
294
|
contextWindow: 1048576,
|
|
@@ -280,7 +304,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
280
304
|
thinkingLevelMap: {
|
|
281
305
|
high: "high",
|
|
282
306
|
low: "low",
|
|
283
|
-
medium:
|
|
307
|
+
medium: null,
|
|
284
308
|
minimal: null,
|
|
285
309
|
off: "none",
|
|
286
310
|
xhigh: "max",
|
|
@@ -291,7 +315,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
291
315
|
name: "gpt-oss:120b",
|
|
292
316
|
compat: {
|
|
293
317
|
maxTokensField: "max_tokens",
|
|
294
|
-
openRouterRouting: {},
|
|
295
318
|
requiresAssistantAfterToolResult: false,
|
|
296
319
|
requiresReasoningContentOnAssistantMessages: false,
|
|
297
320
|
requiresThinkingAsText: false,
|
|
@@ -304,7 +327,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
304
327
|
supportsStrictMode: false,
|
|
305
328
|
supportsUsageInStreaming: true,
|
|
306
329
|
thinkingFormat: "openai",
|
|
307
|
-
vercelGatewayRouting: {},
|
|
308
330
|
zaiToolStream: false,
|
|
309
331
|
},
|
|
310
332
|
contextWindow: 131072,
|
|
@@ -331,7 +353,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
331
353
|
name: "gpt-oss:20b",
|
|
332
354
|
compat: {
|
|
333
355
|
maxTokensField: "max_tokens",
|
|
334
|
-
openRouterRouting: {},
|
|
335
356
|
requiresAssistantAfterToolResult: false,
|
|
336
357
|
requiresReasoningContentOnAssistantMessages: false,
|
|
337
358
|
requiresThinkingAsText: false,
|
|
@@ -344,7 +365,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
344
365
|
supportsStrictMode: false,
|
|
345
366
|
supportsUsageInStreaming: true,
|
|
346
367
|
thinkingFormat: "openai",
|
|
347
|
-
vercelGatewayRouting: {},
|
|
348
368
|
zaiToolStream: false,
|
|
349
369
|
},
|
|
350
370
|
contextWindow: 131072,
|
|
@@ -371,7 +391,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
371
391
|
name: "kimi-k2.6",
|
|
372
392
|
compat: {
|
|
373
393
|
maxTokensField: "max_tokens",
|
|
374
|
-
openRouterRouting: {},
|
|
375
394
|
requiresAssistantAfterToolResult: false,
|
|
376
395
|
requiresReasoningContentOnAssistantMessages: false,
|
|
377
396
|
requiresThinkingAsText: false,
|
|
@@ -384,7 +403,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
384
403
|
supportsStrictMode: false,
|
|
385
404
|
supportsUsageInStreaming: true,
|
|
386
405
|
thinkingFormat: "openai",
|
|
387
|
-
vercelGatewayRouting: {},
|
|
388
406
|
zaiToolStream: false,
|
|
389
407
|
},
|
|
390
408
|
contextWindow: 262144,
|
|
@@ -398,12 +416,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
398
416
|
maxTokens: 262144,
|
|
399
417
|
reasoning: true,
|
|
400
418
|
thinkingLevelMap: {
|
|
401
|
-
high:
|
|
402
|
-
low:
|
|
419
|
+
high: null,
|
|
420
|
+
low: null,
|
|
403
421
|
medium: "medium",
|
|
404
422
|
minimal: null,
|
|
405
423
|
off: "none",
|
|
406
|
-
xhigh:
|
|
424
|
+
xhigh: null,
|
|
407
425
|
},
|
|
408
426
|
},
|
|
409
427
|
{
|
|
@@ -411,7 +429,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
411
429
|
name: "kimi-k2.7-code",
|
|
412
430
|
compat: {
|
|
413
431
|
maxTokensField: "max_tokens",
|
|
414
|
-
openRouterRouting: {},
|
|
415
432
|
requiresAssistantAfterToolResult: false,
|
|
416
433
|
requiresReasoningContentOnAssistantMessages: false,
|
|
417
434
|
requiresThinkingAsText: false,
|
|
@@ -424,7 +441,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
424
441
|
supportsStrictMode: false,
|
|
425
442
|
supportsUsageInStreaming: true,
|
|
426
443
|
thinkingFormat: "openai",
|
|
427
|
-
vercelGatewayRouting: {},
|
|
428
444
|
zaiToolStream: false,
|
|
429
445
|
},
|
|
430
446
|
contextWindow: 262144,
|
|
@@ -438,12 +454,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
438
454
|
maxTokens: 262144,
|
|
439
455
|
reasoning: true,
|
|
440
456
|
thinkingLevelMap: {
|
|
441
|
-
high:
|
|
442
|
-
low:
|
|
457
|
+
high: null,
|
|
458
|
+
low: null,
|
|
443
459
|
medium: "medium",
|
|
444
460
|
minimal: null,
|
|
445
461
|
off: "none",
|
|
446
|
-
xhigh:
|
|
462
|
+
xhigh: null,
|
|
447
463
|
},
|
|
448
464
|
},
|
|
449
465
|
{
|
|
@@ -451,7 +467,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
451
467
|
name: "kimi-k3",
|
|
452
468
|
compat: {
|
|
453
469
|
maxTokensField: "max_tokens",
|
|
454
|
-
openRouterRouting: {},
|
|
455
470
|
requiresAssistantAfterToolResult: false,
|
|
456
471
|
requiresReasoningContentOnAssistantMessages: false,
|
|
457
472
|
requiresThinkingAsText: false,
|
|
@@ -464,7 +479,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
464
479
|
supportsStrictMode: false,
|
|
465
480
|
supportsUsageInStreaming: true,
|
|
466
481
|
thinkingFormat: "openai",
|
|
467
|
-
vercelGatewayRouting: {},
|
|
468
482
|
zaiToolStream: false,
|
|
469
483
|
},
|
|
470
484
|
contextWindow: 1048576,
|
|
@@ -480,7 +494,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
480
494
|
thinkingLevelMap: {
|
|
481
495
|
high: "high",
|
|
482
496
|
low: "low",
|
|
483
|
-
medium:
|
|
497
|
+
medium: null,
|
|
484
498
|
minimal: null,
|
|
485
499
|
off: "none",
|
|
486
500
|
xhigh: "max",
|
|
@@ -491,7 +505,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
491
505
|
name: "minimax-m2.7",
|
|
492
506
|
compat: {
|
|
493
507
|
maxTokensField: "max_tokens",
|
|
494
|
-
openRouterRouting: {},
|
|
495
508
|
requiresAssistantAfterToolResult: false,
|
|
496
509
|
requiresReasoningContentOnAssistantMessages: false,
|
|
497
510
|
requiresThinkingAsText: false,
|
|
@@ -504,7 +517,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
504
517
|
supportsStrictMode: false,
|
|
505
518
|
supportsUsageInStreaming: true,
|
|
506
519
|
thinkingFormat: "openai",
|
|
507
|
-
vercelGatewayRouting: {},
|
|
508
520
|
zaiToolStream: false,
|
|
509
521
|
},
|
|
510
522
|
contextWindow: 196608,
|
|
@@ -518,12 +530,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
518
530
|
maxTokens: 131072,
|
|
519
531
|
reasoning: true,
|
|
520
532
|
thinkingLevelMap: {
|
|
521
|
-
high:
|
|
522
|
-
low:
|
|
533
|
+
high: null,
|
|
534
|
+
low: null,
|
|
523
535
|
medium: "medium",
|
|
524
536
|
minimal: null,
|
|
525
537
|
off: null,
|
|
526
|
-
xhigh:
|
|
538
|
+
xhigh: null,
|
|
527
539
|
},
|
|
528
540
|
},
|
|
529
541
|
{
|
|
@@ -531,7 +543,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
531
543
|
name: "minimax-m3",
|
|
532
544
|
compat: {
|
|
533
545
|
maxTokensField: "max_tokens",
|
|
534
|
-
openRouterRouting: {},
|
|
535
546
|
requiresAssistantAfterToolResult: false,
|
|
536
547
|
requiresReasoningContentOnAssistantMessages: false,
|
|
537
548
|
requiresThinkingAsText: false,
|
|
@@ -544,7 +555,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
544
555
|
supportsStrictMode: false,
|
|
545
556
|
supportsUsageInStreaming: true,
|
|
546
557
|
thinkingFormat: "openai",
|
|
547
|
-
vercelGatewayRouting: {},
|
|
548
558
|
zaiToolStream: false,
|
|
549
559
|
},
|
|
550
560
|
contextWindow: 512000,
|
|
@@ -562,7 +572,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
562
572
|
low: "low",
|
|
563
573
|
medium: "medium",
|
|
564
574
|
minimal: null,
|
|
565
|
-
off:
|
|
575
|
+
off: "none",
|
|
566
576
|
xhigh: "max",
|
|
567
577
|
},
|
|
568
578
|
},
|
|
@@ -571,7 +581,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
571
581
|
name: "mistral-large-3:675b",
|
|
572
582
|
compat: {
|
|
573
583
|
maxTokensField: "max_tokens",
|
|
574
|
-
openRouterRouting: {},
|
|
575
584
|
requiresAssistantAfterToolResult: false,
|
|
576
585
|
requiresReasoningContentOnAssistantMessages: false,
|
|
577
586
|
requiresThinkingAsText: false,
|
|
@@ -584,7 +593,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
584
593
|
supportsStrictMode: false,
|
|
585
594
|
supportsUsageInStreaming: true,
|
|
586
595
|
thinkingFormat: "openai",
|
|
587
|
-
vercelGatewayRouting: {},
|
|
588
596
|
zaiToolStream: false,
|
|
589
597
|
},
|
|
590
598
|
contextWindow: 262144,
|
|
@@ -603,7 +611,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
603
611
|
name: "nemotron-3-nano:30b",
|
|
604
612
|
compat: {
|
|
605
613
|
maxTokensField: "max_tokens",
|
|
606
|
-
openRouterRouting: {},
|
|
607
614
|
requiresAssistantAfterToolResult: false,
|
|
608
615
|
requiresReasoningContentOnAssistantMessages: false,
|
|
609
616
|
requiresThinkingAsText: false,
|
|
@@ -616,7 +623,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
616
623
|
supportsStrictMode: false,
|
|
617
624
|
supportsUsageInStreaming: true,
|
|
618
625
|
thinkingFormat: "openai",
|
|
619
|
-
vercelGatewayRouting: {},
|
|
620
626
|
zaiToolStream: false,
|
|
621
627
|
},
|
|
622
628
|
contextWindow: 262144,
|
|
@@ -630,12 +636,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
630
636
|
maxTokens: 131072,
|
|
631
637
|
reasoning: true,
|
|
632
638
|
thinkingLevelMap: {
|
|
633
|
-
high:
|
|
634
|
-
low:
|
|
639
|
+
high: null,
|
|
640
|
+
low: null,
|
|
635
641
|
medium: "medium",
|
|
636
642
|
minimal: null,
|
|
637
643
|
off: "none",
|
|
638
|
-
xhigh:
|
|
644
|
+
xhigh: null,
|
|
639
645
|
},
|
|
640
646
|
},
|
|
641
647
|
{
|
|
@@ -643,7 +649,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
643
649
|
name: "nemotron-3-super",
|
|
644
650
|
compat: {
|
|
645
651
|
maxTokensField: "max_tokens",
|
|
646
|
-
openRouterRouting: {},
|
|
647
652
|
requiresAssistantAfterToolResult: false,
|
|
648
653
|
requiresReasoningContentOnAssistantMessages: false,
|
|
649
654
|
requiresThinkingAsText: false,
|
|
@@ -656,7 +661,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
656
661
|
supportsStrictMode: false,
|
|
657
662
|
supportsUsageInStreaming: true,
|
|
658
663
|
thinkingFormat: "openai",
|
|
659
|
-
vercelGatewayRouting: {},
|
|
660
664
|
zaiToolStream: false,
|
|
661
665
|
},
|
|
662
666
|
contextWindow: 262144,
|
|
@@ -670,12 +674,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
670
674
|
maxTokens: 65536,
|
|
671
675
|
reasoning: true,
|
|
672
676
|
thinkingLevelMap: {
|
|
673
|
-
high:
|
|
674
|
-
low:
|
|
677
|
+
high: null,
|
|
678
|
+
low: null,
|
|
675
679
|
medium: "medium",
|
|
676
680
|
minimal: null,
|
|
677
681
|
off: "none",
|
|
678
|
-
xhigh:
|
|
682
|
+
xhigh: null,
|
|
679
683
|
},
|
|
680
684
|
},
|
|
681
685
|
{
|
|
@@ -683,7 +687,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
683
687
|
name: "nemotron-3-ultra",
|
|
684
688
|
compat: {
|
|
685
689
|
maxTokensField: "max_tokens",
|
|
686
|
-
openRouterRouting: {},
|
|
687
690
|
requiresAssistantAfterToolResult: false,
|
|
688
691
|
requiresReasoningContentOnAssistantMessages: false,
|
|
689
692
|
requiresThinkingAsText: false,
|
|
@@ -696,7 +699,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
696
699
|
supportsStrictMode: false,
|
|
697
700
|
supportsUsageInStreaming: true,
|
|
698
701
|
thinkingFormat: "openai",
|
|
699
|
-
vercelGatewayRouting: {},
|
|
700
702
|
zaiToolStream: false,
|
|
701
703
|
},
|
|
702
704
|
contextWindow: 262144,
|
|
@@ -710,12 +712,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
710
712
|
maxTokens: 65536,
|
|
711
713
|
reasoning: true,
|
|
712
714
|
thinkingLevelMap: {
|
|
713
|
-
high:
|
|
714
|
-
low:
|
|
715
|
+
high: null,
|
|
716
|
+
low: null,
|
|
715
717
|
medium: "medium",
|
|
716
718
|
minimal: null,
|
|
717
719
|
off: "none",
|
|
718
|
-
xhigh:
|
|
720
|
+
xhigh: null,
|
|
719
721
|
},
|
|
720
722
|
},
|
|
721
723
|
{
|
|
@@ -723,7 +725,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
723
725
|
name: "qwen3.5:397b",
|
|
724
726
|
compat: {
|
|
725
727
|
maxTokensField: "max_tokens",
|
|
726
|
-
openRouterRouting: {},
|
|
727
728
|
requiresAssistantAfterToolResult: false,
|
|
728
729
|
requiresReasoningContentOnAssistantMessages: false,
|
|
729
730
|
requiresThinkingAsText: false,
|
|
@@ -736,7 +737,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
736
737
|
supportsStrictMode: false,
|
|
737
738
|
supportsUsageInStreaming: true,
|
|
738
739
|
thinkingFormat: "openai",
|
|
739
|
-
vercelGatewayRouting: {},
|
|
740
740
|
zaiToolStream: false,
|
|
741
741
|
},
|
|
742
742
|
contextWindow: 262144,
|
package/models.ts
CHANGED
|
@@ -93,7 +93,7 @@ function buildCompat(): ProviderModelConfig["compat"] {
|
|
|
93
93
|
return {
|
|
94
94
|
// Ollama uses "system" role, not "developer" (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsDeveloperRole).
|
|
95
95
|
supportsDeveloperRole: false,
|
|
96
|
-
// reasoning_effort works (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsReasoningEffort
|
|
96
|
+
// reasoning_effort works (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsReasoningEffort).
|
|
97
97
|
supportsReasoningEffort: true,
|
|
98
98
|
// "store" is not a supported field (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsStore).
|
|
99
99
|
supportsStore: false,
|
|
@@ -109,22 +109,32 @@ function buildCompat(): ProviderModelConfig["compat"] {
|
|
|
109
109
|
requiresThinkingAsText: false,
|
|
110
110
|
// DeepSeek-specific, not needed for Ollama (pi: types.ts#requiresReasoningContentOnAssistantMessages).
|
|
111
111
|
requiresReasoningContentOnAssistantMessages: false,
|
|
112
|
-
// reasoning_effort format works (pi: types.ts#thinkingFormat
|
|
112
|
+
// reasoning_effort format works (pi: types.ts#thinkingFormat).
|
|
113
113
|
thinkingFormat: "openai",
|
|
114
114
|
// Ollama does not support tool_choice, so strict mode is unavailable (ollama: docs.ollama.com/api/openai-compatibility, pi: types.ts#supportsStrictMode).
|
|
115
115
|
supportsStrictMode: false,
|
|
116
|
-
// Anthropic cache_control not relevant; Ollama has implicit KV cache only (pi: types.ts#cacheControlFormat).
|
|
117
|
-
// Explicitly undefined: JSON.stringify drops undefined values, keeping
|
|
118
|
-
// models.generated.ts structurally consistent with assembleModels() runtime output.
|
|
119
116
|
// Session affinity headers not relevant for Ollama (pi: types.ts#sendSessionAffinityHeaders).
|
|
120
117
|
sendSessionAffinityHeaders: false,
|
|
121
118
|
// No explicit cache-retention API (pi: types.ts#supportsLongCacheRetention).
|
|
122
119
|
supportsLongCacheRetention: false,
|
|
123
120
|
// Not z.ai (pi: types.ts#zaiToolStream).
|
|
124
121
|
zaiToolStream: false,
|
|
122
|
+
// Anthropic cache_control not relevant; Ollama has implicit KV cache only (pi: types.ts#cacheControlFormat).
|
|
123
|
+
// Explicitly undefined: JSON.stringify drops undefined values, keeping
|
|
124
|
+
// models.generated.ts structurally consistent with assembleModels() runtime output.
|
|
125
125
|
cacheControlFormat: undefined,
|
|
126
|
-
|
|
127
|
-
|
|
126
|
+
// OpenRouter / Vercel AI Gateway routing prefs. pi-ai's OpenAI transport
|
|
127
|
+
// truthiness-checks the raw model.compat (not the resolved getCompat() value)
|
|
128
|
+
// and forwards it verbatim: `if (model.compat?.openRouterRouting)
|
|
129
|
+
// params.provider = ...` in packages/ai/src/api/openai-completions.ts. `{}` is
|
|
130
|
+
// truthy, so baking it in sent a stray `provider: {}` on every Ollama request.
|
|
131
|
+
// Express "none" as undefined, never `{}`; undefined is falsy at the wire site
|
|
132
|
+
// and JSON.stringify drops it from models.generated.ts. The boolean flags above
|
|
133
|
+
// stay explicit `false` on purpose: omitting a boolean makes getCompat() fall
|
|
134
|
+
// back to detectCompat's generic-OpenAI defaults, which are wrong for Ollama
|
|
135
|
+
// (e.g. maxTokensField -> "max_completion_tokens", supportsStrictMode -> true).
|
|
136
|
+
openRouterRouting: undefined,
|
|
137
|
+
vercelGatewayRouting: undefined,
|
|
128
138
|
};
|
|
129
139
|
}
|
|
130
140
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-ollama-cloud",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.12.1",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package"
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
"models.ts",
|
|
14
14
|
"models.generated.ts",
|
|
15
15
|
"pricing.generated.ts",
|
|
16
|
+
"reasoning.generated.ts",
|
|
16
17
|
"thinking-levels.ts",
|
|
17
18
|
"usage.ts",
|
|
18
19
|
"utils.ts",
|
|
@@ -32,8 +33,9 @@
|
|
|
32
33
|
"format": "biome format --write .",
|
|
33
34
|
"test": "vitest run",
|
|
34
35
|
"smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
|
|
35
|
-
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts",
|
|
36
|
-
"generate-limits": "tsx scripts/generate-limits.ts
|
|
36
|
+
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-reasoning.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts reasoning.generated.ts",
|
|
37
|
+
"generate-limits": "tsx scripts/generate-limits.ts",
|
|
38
|
+
"generate-reasoning": "tsx scripts/generate-reasoning.ts && biome format --write reasoning.generated.ts"
|
|
37
39
|
},
|
|
38
40
|
"pi": {
|
|
39
41
|
"extensions": [
|
|
@@ -51,6 +53,6 @@
|
|
|
51
53
|
"@types/node": "^26.1.2",
|
|
52
54
|
"@typescript/native-preview": "7.0.0-dev.20260707.2",
|
|
53
55
|
"tsx": "^4.19.0",
|
|
54
|
-
"vitest": "^4.1.
|
|
56
|
+
"vitest": "^4.1.11"
|
|
55
57
|
}
|
|
56
58
|
}
|
package/pricing.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-pricing.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-09-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-14T10:27:19.243Z
|
|
4
|
+
// Model count: 20
|
|
5
5
|
|
|
6
6
|
export interface ModelPrice {
|
|
7
7
|
input: number;
|
|
@@ -13,6 +13,7 @@ export interface ModelPrice {
|
|
|
13
13
|
export const MODEL_PRICING: Record<string, ModelPrice> = {
|
|
14
14
|
"deepseek-v4-flash:0731": { input: 0.44, output: 1.32, cacheRead: 0.014, cacheWrite: 0 },
|
|
15
15
|
"deepseek-v4-pro:0813": { input: 1.32, output: 3.96, cacheRead: 0.044, cacheWrite: 0 },
|
|
16
|
+
"deepseek-v4.1-flash": { input: 0.3, output: 1.2, cacheRead: 0.006, cacheWrite: 0 },
|
|
16
17
|
"gemma4:31b": { input: 0.14, output: 0.4, cacheRead: 0.05, cacheWrite: 0 },
|
|
17
18
|
"glm-5.1": { input: 1, output: 3.2, cacheRead: 0.2, cacheWrite: 0 },
|
|
18
19
|
"glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
// Auto-generated by scripts/generate-reasoning.ts
|
|
2
|
+
// Do not edit manually.
|
|
3
|
+
// Model count: 23
|
|
4
|
+
|
|
5
|
+
export type ModelsDevReasoningOption =
|
|
6
|
+
| { type: "toggle" }
|
|
7
|
+
| {
|
|
8
|
+
type: "effort";
|
|
9
|
+
values: Array<"none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max" | "ultra" | "default" | null>;
|
|
10
|
+
};
|
|
11
|
+
|
|
12
|
+
export const MODEL_REASONING_OPTIONS: Record<string, ModelsDevReasoningOption[]> = {
|
|
13
|
+
"deepseek-v4-flash": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
|
|
14
|
+
"deepseek-v4-flash:0731": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
|
|
15
|
+
"deepseek-v4-pro": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
|
|
16
|
+
"deepseek-v4-pro:0813": [{ type: "toggle" }, { type: "effort", values: ["high", "max"] }],
|
|
17
|
+
"deepseek-v4.1-flash": [{ type: "toggle" }, { type: "effort", values: ["low", "high", "max"] }],
|
|
18
|
+
"gemma4:31b": [{ type: "toggle" }],
|
|
19
|
+
"glm-5.1": [{ type: "toggle" }],
|
|
20
|
+
"glm-5.2": [{ type: "effort", values: ["high", "max"] }],
|
|
21
|
+
"glm-5.3": [{ type: "effort", values: ["low", "high", "max"] }],
|
|
22
|
+
"glm-5.3-flash": [{ type: "effort", values: ["low", "high", "max"] }],
|
|
23
|
+
"gpt-oss:120b": [{ type: "effort", values: ["low", "medium", "high"] }],
|
|
24
|
+
"gpt-oss:20b": [{ type: "effort", values: ["low", "medium", "high"] }],
|
|
25
|
+
"kimi-k2.5": [{ type: "toggle" }],
|
|
26
|
+
"kimi-k2.6": [{ type: "toggle" }],
|
|
27
|
+
"kimi-k2.7-code": [{ type: "toggle" }],
|
|
28
|
+
"kimi-k3": [{ type: "toggle" }, { type: "effort", values: ["low", "high", "max"] }],
|
|
29
|
+
"minimax-m2.5": [],
|
|
30
|
+
"minimax-m2.7": [{ type: "toggle" }],
|
|
31
|
+
"minimax-m3": [{ type: "toggle" }, { type: "effort", values: ["low", "medium", "high", "max"] }],
|
|
32
|
+
"nemotron-3-nano:30b": [{ type: "toggle" }],
|
|
33
|
+
"nemotron-3-super": [{ type: "toggle" }],
|
|
34
|
+
"nemotron-3-ultra": [{ type: "toggle" }],
|
|
35
|
+
"qwen3.5:397b": [{ type: "toggle" }],
|
|
36
|
+
};
|
package/thinking-levels.ts
CHANGED
|
@@ -2,26 +2,36 @@
|
|
|
2
2
|
* Thinking level mapping for Ollama Cloud models.
|
|
3
3
|
*
|
|
4
4
|
* Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible
|
|
5
|
-
* `reasoning_effort` values. The API accepts "
|
|
6
|
-
* "high", and "max". On simple prompts, "max" can
|
|
7
|
-
* "high", but on harder prompts it can increase thinking
|
|
8
|
-
*
|
|
5
|
+
* `reasoning_effort` values. The API accepts "minimal", "none", "low",
|
|
6
|
+
* "medium", "high", "xhigh", "ultra", and "max". On simple prompts, "max" can
|
|
7
|
+
* be a no-op over "high", but on harder prompts it can increase thinking
|
|
8
|
+
* substantially.
|
|
9
9
|
*
|
|
10
|
-
*
|
|
10
|
+
* Every value EFFORT_TO_LEVEL can send was verified against the live chat
|
|
11
|
+
* completions API (2026-09-11: "minimal", "xhigh", and "ultra" probed across
|
|
12
|
+
* gpt-oss, deepseek-v4, glm, minimax, and qwen thinking models, all accepted
|
|
13
|
+
* with graded reasoning), so a future models.dev row that lists them passes
|
|
14
|
+
* through a value the endpoint demonstrably accepts.
|
|
15
|
+
*
|
|
16
|
+
* The per-model level support comes from models.dev: scripts/generate-reasoning.ts
|
|
17
|
+
* fetches the `ollama-cloud` provider's `reasoning_options` into
|
|
18
|
+
* reasoning.generated.ts (the same data source pi uses for its built-in
|
|
19
|
+
* providers), and resolve() maps a model's effort values onto Pi's levels.
|
|
20
|
+
* We fall back to models.dev because the Cloud API does not yet expose
|
|
21
|
+
* per-model supported levels (tracked upstream: https://github.com/ollama/ollama/issues/18385).
|
|
11
22
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* - Kimi K2 Thinking: "none" doesn't disable thinking - off is hidden
|
|
19
|
-
* - MiniMax M2.x: "none" doesn't disable thinking - off is hidden
|
|
23
|
+
* The API exposes only a boolean `thinking` capability plus a global effort
|
|
24
|
+
* vocabulary, and models.dev does not reliably encode the `none` behavior, so
|
|
25
|
+
* the `off` switch is handled separately: it defaults to "none" (a live probe
|
|
26
|
+
* of the current catalog confirmed every model except the OFF_NULL overrides
|
|
27
|
+
* below honors it), and models verified not to honor `none` pin it to null
|
|
28
|
+
* (hidden) via OFF_NULL.
|
|
20
29
|
*
|
|
21
|
-
*
|
|
30
|
+
* A `null` value means the level is hidden in Pi's UI.
|
|
22
31
|
*/
|
|
23
32
|
|
|
24
33
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
34
|
+
import { MODEL_REASONING_OPTIONS, type ModelsDevReasoningOption } from "./reasoning.generated.ts";
|
|
25
35
|
|
|
26
36
|
export type ThinkingLevelMap = NonNullable<ProviderModelConfig["thinkingLevelMap"]>;
|
|
27
37
|
|
|
@@ -35,63 +45,101 @@ export const DEFAULT: ThinkingLevelMap = {
|
|
|
35
45
|
xhigh: "max",
|
|
36
46
|
};
|
|
37
47
|
|
|
38
|
-
/**
|
|
39
|
-
*
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
+
/**
|
|
49
|
+
* Models where a live probe of `reasoning_effort:"none"` still produced
|
|
50
|
+
* reasoning (i.e. thinking cannot be disabled), so the `off` level is hidden.
|
|
51
|
+
* Confirmed against the current catalog by scripts/../test probing; the API
|
|
52
|
+
* and models.dev do not expose this behavior.
|
|
53
|
+
*/
|
|
54
|
+
/**
|
|
55
|
+
* Models where a live probe of `reasoning_effort:"none"` still produced
|
|
56
|
+
* reasoning (i.e. thinking cannot be disabled), so the `off` level is hidden.
|
|
57
|
+
* Confirmed against the current catalog by scripts/../test probing; the API
|
|
58
|
+
* and models.dev do not expose this behavior.
|
|
59
|
+
*
|
|
60
|
+
* Exact ids hide only the named variant; family prefixes cover future model
|
|
61
|
+
* revisions in the same family. gpt-oss is a family-wide prefix because both
|
|
62
|
+
* probed variants leak and the behavior is documented for the family.
|
|
63
|
+
* minimax is NOT matched family-wide here: minimax-m3 was probed to honor
|
|
64
|
+
* `none`, so only the verified-leaking minimax-m2.7 is pinned by exact id.
|
|
65
|
+
*/
|
|
66
|
+
const OFF_NULL_EXACT = new Set(["minimax-m2.7"]);
|
|
67
|
+
const OFF_NULL_FAMILIES = ["gpt-oss"];
|
|
48
68
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
off: "none",
|
|
53
|
-
minimal: null,
|
|
54
|
-
low: null,
|
|
55
|
-
medium: "medium",
|
|
56
|
-
high: null,
|
|
57
|
-
xhigh: null,
|
|
58
|
-
};
|
|
69
|
+
function hidesOff(id: string): boolean {
|
|
70
|
+
return OFF_NULL_EXACT.has(id) || OFF_NULL_FAMILIES.some((prefix) => id.startsWith(prefix));
|
|
71
|
+
}
|
|
59
72
|
|
|
60
|
-
/**
|
|
61
|
-
*
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
73
|
+
/**
|
|
74
|
+
* Map a models.dev `effort` value onto the Pi level key and the reasoning_effort
|
|
75
|
+
* string to send for it. Ollama's top effort value is "max"; Pi exposes it via
|
|
76
|
+
* the extra-high level, so "max" (and "xhigh"/"ultra") map to the xhigh key.
|
|
77
|
+
*/
|
|
78
|
+
const EFFORT_TO_LEVEL: Record<string, { key: "minimal" | "low" | "medium" | "high" | "xhigh"; value: string }> = {
|
|
79
|
+
minimal: { key: "minimal", value: "minimal" },
|
|
80
|
+
low: { key: "low", value: "low" },
|
|
81
|
+
medium: { key: "medium", value: "medium" },
|
|
82
|
+
high: { key: "high", value: "high" },
|
|
83
|
+
xhigh: { key: "xhigh", value: "xhigh" },
|
|
84
|
+
max: { key: "xhigh", value: "max" },
|
|
85
|
+
ultra: { key: "xhigh", value: "ultra" },
|
|
69
86
|
};
|
|
87
|
+
/**
|
|
88
|
+
* Build a ThinkingLevelMap from models.dev reasoning_options.
|
|
89
|
+
* Levels come from `effort` values (mapped via EFFORT_TO_LEVEL); `off` defaults
|
|
90
|
+
* to "none" (probe-derived, see file header) and is hidden only via the
|
|
91
|
+
* OFF_NULL exact/family sets (see hidesOff).
|
|
92
|
+
* A toggle-only model is binary (on/off) and exposes a single "medium" level.
|
|
93
|
+
*/
|
|
94
|
+
function buildMap(options: readonly ModelsDevReasoningOption[], id: string): ThinkingLevelMap {
|
|
95
|
+
const map: ThinkingLevelMap = {
|
|
96
|
+
off: hidesOff(id) ? null : "none",
|
|
97
|
+
minimal: null,
|
|
98
|
+
low: null,
|
|
99
|
+
medium: null,
|
|
100
|
+
high: null,
|
|
101
|
+
xhigh: null,
|
|
102
|
+
};
|
|
70
103
|
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
}
|
|
104
|
+
if (options.length > 0 && options.every((option) => option.type === "toggle")) {
|
|
105
|
+
// Binary on/off model: no graded effort, expose a single level.
|
|
106
|
+
return { ...map, medium: "medium" };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const efforts = options.flatMap((option) => (option.type === "effort" ? (option.values ?? []) : []));
|
|
110
|
+
for (const effort of efforts) {
|
|
111
|
+
const target = effort !== null && effort !== "default" ? EFFORT_TO_LEVEL[effort] : undefined;
|
|
112
|
+
if (target) map[target.key] = target.value;
|
|
113
|
+
}
|
|
114
|
+
return map;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Read MODEL_REASONING_OPTIONS[id] without tripping over inherited keys (e.g.
|
|
119
|
+
* "constructor"), which would otherwise resolve to the Object constructor and
|
|
120
|
+
* crash buildMap.
|
|
121
|
+
*/
|
|
122
|
+
function ownOptions(id: string): ModelsDevReasoningOption[] | undefined {
|
|
123
|
+
return Object.hasOwn(MODEL_REASONING_OPTIONS, id) ? MODEL_REASONING_OPTIONS[id] : undefined;
|
|
124
|
+
}
|
|
81
125
|
|
|
82
126
|
/**
|
|
83
127
|
* Resolve the thinking level map for a model.
|
|
84
|
-
*
|
|
128
|
+
* Looks up the model id (exact, then `:tag` family) in the generated models.dev
|
|
129
|
+
* table, falling back to DEFAULT for models with no entry. The matched key is
|
|
130
|
+
* the one passed to buildMap so the OFF_NULL set (keyed on bare family names)
|
|
131
|
+
* applies to tagged ids that resolve through a family match.
|
|
85
132
|
*/
|
|
86
133
|
export function resolve(id: string, capabilities: string[]): ThinkingLevelMap | undefined {
|
|
87
134
|
if (!capabilities.includes("thinking")) return undefined;
|
|
88
135
|
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
return DEFAULT;
|
|
136
|
+
const colon = id.lastIndexOf(":");
|
|
137
|
+
const exact = ownOptions(id);
|
|
138
|
+
const matchedKey = exact !== undefined ? id : colon > 0 ? id.slice(0, colon) : "";
|
|
139
|
+
const options = exact ?? ownOptions(matchedKey);
|
|
140
|
+
// An empty array (e.g. minimax-m2.5) carries no verified options; fall back
|
|
141
|
+
// to DEFAULT rather than a degenerate map whose only selectable level can
|
|
142
|
+
// be a leaking off.
|
|
143
|
+
if (options === undefined || options.length === 0) return DEFAULT;
|
|
144
|
+
return buildMap(options, matchedKey);
|
|
97
145
|
}
|