pi-ollama-cloud 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,15 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
+ ## [Unreleased]
6
+
7
+ ## [0.10.0] - 2026-09-03
8
+
9
+ - **Breaking:** Adapt to the changed `/api/usage` response shape. The endpoint now returns a single `limits.monthly` bucket (replacing `limits.session` and `limits.weekly`) and adds an `activity.models` array. `UsageData`, `isUsageResponse`, `formatUsage`, and `formatUsageStatusColored` now read the monthly limit; the status bar shows a single `30d` segment instead of `5h`/`7d`.
10
+ - Refreshed the generated catalog from the live API: added `deepseek-v4-pro:0813`, `glm-5.3`, and `glm-5.3-flash`; removed `deepseek-v4-flash:preview` and `deepseek-v4-pro`, which are no longer listed.
11
+ - Source per-token pricing from the official model table on ollama.com/pricing instead of models.dev estimates, which no longer track Ollama's published rates (up to ~13x off per model). `scripts/generate-pricing.ts` now scrapes the pricing page (the table is server-rendered; no JSON endpoint exists) and matches catalog IDs to pricing rows by exact or `:tag`-family match, replacing the `OLLAMA_TO_MODELSDEV` mapping. Regenerated `pricing.generated.ts` with the official rates, including new models (`glm-5.3`, `glm-5.3-flash`, `deepseek-v4-pro:0813`). Fixes #51. Thanks @Hackbard (#52).
12
+ - Probe per-model max output tokens against the live API via `scripts/generate-limits.ts` into `limits.generated.ts`, replacing the fixed 32768 default for known models (unprobed models still fall back to 32768). The probe timeout is 60s so slow first-token models are not dropped from the table. Thanks @f440 (#49).
13
+
5
14
  ## [0.9.0] - 2026-08-11
6
15
 
7
16
  - Add `/ollama-cloud-usage` command to show Ollama Cloud session (5h) and weekly (7d) usage limits, per-model request counts, and the 4-week activity cost, fetched from the undocumented `/api/usage` endpoint with the already-resolved API key.
package/README.md CHANGED
@@ -12,7 +12,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
12
12
  - **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
13
13
  - **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
14
14
  - **`ollama_web_fetch` tool** - Fetch and extract text content from a web page URL using Ollama Cloud's `/api/web_fetch` endpoint. Returns page title, content, and links.
15
- - **Estimated cost tracking** - Models are registered with estimated per-token costs sourced from [models.dev](https://models.dev) (the same catalog pi uses), so Pi's `/cost` shows comparable usage. Ollama Cloud is subscription-billed (Free, Pro, Max), so these are equivalent pay-as-you-go estimates, not actual charges. See [ollama.com/pricing](https://ollama.com/pricing) for plan details.
15
+ - **Per-token cost tracking** - Models are registered with the official per-token prices from [ollama.com/pricing](https://ollama.com/pricing), so Pi's `/cost` shows comparable usage. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges.
16
16
 
17
17
  ## Prerequisites
18
18
 
@@ -135,8 +135,14 @@ Model metadata is derived from the `/api/show` response:
135
135
  | `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 5 maps (DEFAULT, GPT_OSS, QWEN3, GLM_52, NO_OFF) based on API testing |
136
136
  | `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
137
137
  | `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
138
- | `maxTokens` | Fixed at 32768 |
139
- | `cost` | Estimated per-1M-token prices from [models.dev](https://models.dev), generated by `scripts/generate-pricing.ts` into `pricing.generated.ts`. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go estimates, not actual charges. Unmapped models default to zero. Prices are pinned to the installed package version and only update on a new release, so newly added models register with zero cost until then. |
138
+ | `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
139
+ | `cost` | Official per-1M-token prices from the [ollama.com/pricing](https://ollama.com/pricing) model table, generated by `scripts/generate-pricing.ts` into `pricing.generated.ts`. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges. Catalog IDs with no matching pricing row default to zero. Prices are pinned to the installed package version and only update on a new release, so newly added models register with zero cost until then. |
140
+
141
+ The per-model max output token table (`limits.generated.ts`) is probed against the live API by `scripts/generate-limits.ts`, since `/api/show` does not expose the limit. It needs an API key: `OLLAMA_API_KEY=<key> npm run generate-limits`. Limits ship with the package, so regenerated values take effect on the next release.
142
+
143
+ The API itself returns no cost data: completion responses report only token counts (`prompt_tokens`/`completion_tokens`/`total_tokens`, including the final usage chunk when streaming), and `/api/show` exposes no pricing fields. The prices above come from the static `/pricing` page table and are only as fresh as the last regeneration.
144
+
145
+ Cache pricing is informational only: the `/pricing` page lists a "Cached input" column, but the completion API does not report cache token usage (there is no `prompt_tokens_details.cached_tokens` or equivalent in any response, verified against the live API in September 2026), so pi never sees cache hits and `/cost` estimates do not reflect them. `cacheWrite` is always zero because the pricing table has no cache-write column.
140
146
 
141
147
  ### Thinking level mapping
142
148
 
@@ -166,14 +172,14 @@ Both tools use the same Ollama Cloud API key configured for the provider. No loc
166
172
  | Command | Description |
167
173
  |---|---|
168
174
  | `/ollama-webtools [on\|off\|enable\|disable]` | Enable or disable the `ollama_web_search` and `ollama_web_fetch` tools. Toggles if no argument given. |
169
- | `/ollama-cloud-usage` | Show Ollama Cloud session (5h) and weekly (7d) usage limits, per-model request counts, and the 4-week activity cost. |
175
+ | `/ollama-cloud-usage` | Show Ollama Cloud monthly usage limits, per-model request counts, and the 4-week activity cost. |
170
176
  | `/ollama-usage-status [on\|off\|enable\|disable]` | Enable or disable the footer usage status bar. Toggles if no argument given. |
171
177
 
172
178
  ## Usage status bar
173
179
 
174
180
  While an `ollama-cloud` model is the active provider, the footer shows a compact
175
- live usage readout (`5h ▕███░░░░░░░▏ 34% 7d ▕████░░░░░░▏ 45%`) that refreshes
176
- every 5 minutes and after each agent turn (but no more often than every 5 minutes). Each segment is colored by how close
181
+ live usage readout (`30d ▕███░░░░░░░▏ 34%`) that refreshes
182
+ every 5 minutes and after each agent turn (but no more often than every 5 minutes). It is colored by how close
177
183
  it is to the cap: green below 60%, yellow at 60-79%, red at 80%+. It reads the
178
184
  same undocumented `/api/usage` endpoint as `/ollama-cloud-usage` and clears
179
185
  itself on transient errors or when you switch to a non-Ollama-Cloud provider.
@@ -227,6 +233,7 @@ npm run check # lint + format + type-check (auto-fix)
227
233
  npm run lint # lint only (no fixes)
228
234
  npm run typecheck # type-check only (tsgo --noEmit)
229
235
  npm run format # format only
236
+ OLLAMA_API_KEY=<key> npm run generate-limits # probe max output tokens (writes limits.generated.ts)
230
237
  ```
231
238
 
232
239
  The project uses [Biome](https://biomejs.dev/) for linting and formatting (2-space indent, line width 120) and [tsgo](https://github.com/microsoft/typescript-go) for type-checking.
@@ -295,7 +302,8 @@ git push --tags
295
302
  Because the model catalog refreshes automatically at runtime, a release is **not** needed to ship new models. Publish only when:
296
303
 
297
304
  - A model is retired and still listed by the API: add it to `RETIRED_MODEL_IDS` in `scripts/generate-models.ts` (check https://docs.ollama.com/cloud#retirements, then regenerate `models.generated.ts`).
298
- - Pricing changes: models.dev prices updated, or a new model needs an `OLLAMA_TO_MODELSDEV` mapping line (regenerate `pricing.generated.ts`).
305
+ - Pricing changes: Ollama updates the model pricing table, or a new model needs a pricing row (regenerate `pricing.generated.ts`).
306
+ - Max output token limits changed: run `OLLAMA_API_KEY=<key> npm run generate-limits` locally and commit.
299
307
 
300
308
  The tag version must match the version in `package.json` - `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
301
309
 
package/index.ts CHANGED
@@ -134,7 +134,7 @@ export default async function (pi: ExtensionAPI) {
134
134
  // --- Usage Command ---
135
135
 
136
136
  pi.registerCommand("ollama-cloud-usage", {
137
- description: "Show Ollama Cloud session and weekly usage limits.",
137
+ description: "Show Ollama Cloud monthly usage limits.",
138
138
  handler: async (_args, ctx) => {
139
139
  const apiKey = await getCloudApiKey(ctx);
140
140
  if (!apiKey) {
@@ -152,7 +152,7 @@ export default async function (pi: ExtensionAPI) {
152
152
 
153
153
  // --- Usage Status Bar ---
154
154
 
155
- // Footer status showing live session/weekly usage while ollama-cloud is the
155
+ // Footer status showing live monthly usage while ollama-cloud is the
156
156
  // active provider. Refreshes on a 5-minute timer; agent_end also triggers a
157
157
  // refresh but is throttled to the same cooldown so a turn never hammers the
158
158
  // undocumented /api/usage endpoint. The quota-bar concept is inspired by
@@ -0,0 +1,25 @@
1
+ // Auto-generated by scripts/generate-limits.ts
2
+ // Do not edit manually.
3
+ // Probed models: 19 (0 failed)
4
+
5
+ export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
6
+ "deepseek-v4-flash:0731": 65536,
7
+ "deepseek-v4-pro:0813": 65536,
8
+ "gemma4:31b": 262144,
9
+ "glm-5.1": 131072,
10
+ "glm-5.2": 131072,
11
+ "glm-5.3": 524288,
12
+ "glm-5.3-flash": 524288,
13
+ "gpt-oss:120b": 131072,
14
+ "gpt-oss:20b": 131072,
15
+ "kimi-k2.6": 262144,
16
+ "kimi-k2.7-code": 262144,
17
+ "kimi-k3": 524288,
18
+ "minimax-m2.7": 131072,
19
+ "minimax-m3": 131072,
20
+ "mistral-large-3:675b": 262144,
21
+ "nemotron-3-nano:30b": 131072,
22
+ "nemotron-3-super": 65536,
23
+ "nemotron-3-ultra": 65536,
24
+ "qwen3.5:397b": 65536,
25
+ };
@@ -1,7 +1,7 @@
1
1
  // Auto-generated by scripts/generate-models.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-08-08T04:19:46.959Z
4
- // Model count: 18
3
+ // Generated: 2026-09-03T10:12:02.243Z
4
+ // Model count: 19
5
5
 
6
6
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
7
7
 
@@ -29,13 +29,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
29
29
  },
30
30
  contextWindow: 1048576,
31
31
  cost: {
32
- cacheRead: 0.0028,
32
+ cacheRead: 0.014,
33
33
  cacheWrite: 0,
34
- input: 0.14,
35
- output: 0.28,
34
+ input: 0.44,
35
+ output: 1.32,
36
36
  },
37
37
  input: ["text"],
38
- maxTokens: 32768,
38
+ maxTokens: 65536,
39
39
  reasoning: true,
40
40
  thinkingLevelMap: {
41
41
  high: "high",
@@ -47,8 +47,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
47
47
  },
48
48
  },
49
49
  {
50
- id: "deepseek-v4-flash:preview",
51
- name: "deepseek-v4-flash:preview",
50
+ id: "deepseek-v4-pro:0813",
51
+ name: "deepseek-v4-pro:0813",
52
52
  compat: {
53
53
  maxTokensField: "max_tokens",
54
54
  openRouterRouting: {},
@@ -69,13 +69,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
69
69
  },
70
70
  contextWindow: 1048576,
71
71
  cost: {
72
- cacheRead: 0.0028,
72
+ cacheRead: 0.044,
73
73
  cacheWrite: 0,
74
- input: 0.14,
75
- output: 0.28,
74
+ input: 1.32,
75
+ output: 3.96,
76
76
  },
77
77
  input: ["text"],
78
- maxTokens: 32768,
78
+ maxTokens: 65536,
79
79
  reasoning: true,
80
80
  thinkingLevelMap: {
81
81
  high: "high",
@@ -87,8 +87,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
87
87
  },
88
88
  },
89
89
  {
90
- id: "deepseek-v4-pro",
91
- name: "deepseek-v4-pro",
90
+ id: "gemma4:31b",
91
+ name: "gemma4:31b",
92
92
  compat: {
93
93
  maxTokensField: "max_tokens",
94
94
  openRouterRouting: {},
@@ -107,15 +107,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
107
107
  vercelGatewayRouting: {},
108
108
  zaiToolStream: false,
109
109
  },
110
- contextWindow: 524288,
110
+ contextWindow: 262144,
111
111
  cost: {
112
- cacheRead: 0.003625,
112
+ cacheRead: 0.05,
113
113
  cacheWrite: 0,
114
- input: 0.435,
115
- output: 0.87,
114
+ input: 0.14,
115
+ output: 0.4,
116
116
  },
117
- input: ["text"],
118
- maxTokens: 32768,
117
+ input: ["text", "image"],
118
+ maxTokens: 262144,
119
119
  reasoning: true,
120
120
  thinkingLevelMap: {
121
121
  high: "high",
@@ -127,8 +127,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
127
127
  },
128
128
  },
129
129
  {
130
- id: "gemma4:31b",
131
- name: "gemma4:31b",
130
+ id: "glm-5.1",
131
+ name: "glm-5.1",
132
132
  compat: {
133
133
  maxTokensField: "max_tokens",
134
134
  openRouterRouting: {},
@@ -147,15 +147,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
147
147
  vercelGatewayRouting: {},
148
148
  zaiToolStream: false,
149
149
  },
150
- contextWindow: 262144,
150
+ contextWindow: 202752,
151
151
  cost: {
152
- cacheRead: 0.1,
152
+ cacheRead: 0.2,
153
153
  cacheWrite: 0,
154
- input: 0.1,
155
- output: 0.34,
154
+ input: 1,
155
+ output: 3.2,
156
156
  },
157
- input: ["text", "image"],
158
- maxTokens: 32768,
157
+ input: ["text"],
158
+ maxTokens: 131072,
159
159
  reasoning: true,
160
160
  thinkingLevelMap: {
161
161
  high: "high",
@@ -167,8 +167,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
167
167
  },
168
168
  },
169
169
  {
170
- id: "glm-5.1",
171
- name: "glm-5.1",
170
+ id: "glm-5.2",
171
+ name: "glm-5.2",
172
172
  compat: {
173
173
  maxTokensField: "max_tokens",
174
174
  openRouterRouting: {},
@@ -187,7 +187,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
187
187
  vercelGatewayRouting: {},
188
188
  zaiToolStream: false,
189
189
  },
190
- contextWindow: 202752,
190
+ contextWindow: 1048576,
191
191
  cost: {
192
192
  cacheRead: 0.26,
193
193
  cacheWrite: 0,
@@ -195,20 +195,20 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
195
195
  output: 4.4,
196
196
  },
197
197
  input: ["text"],
198
- maxTokens: 32768,
198
+ maxTokens: 131072,
199
199
  reasoning: true,
200
200
  thinkingLevelMap: {
201
201
  high: "high",
202
- low: "low",
203
- medium: "medium",
202
+ low: null,
203
+ medium: null,
204
204
  minimal: null,
205
205
  off: "none",
206
206
  xhigh: "max",
207
207
  },
208
208
  },
209
209
  {
210
- id: "glm-5.2",
211
- name: "glm-5.2",
210
+ id: "glm-5.3",
211
+ name: "glm-5.3",
212
212
  compat: {
213
213
  maxTokensField: "max_tokens",
214
214
  openRouterRouting: {},
@@ -227,7 +227,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
227
227
  vercelGatewayRouting: {},
228
228
  zaiToolStream: false,
229
229
  },
230
- contextWindow: 1000000,
230
+ contextWindow: 1048576,
231
231
  cost: {
232
232
  cacheRead: 0.26,
233
233
  cacheWrite: 0,
@@ -235,12 +235,52 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
235
235
  output: 4.4,
236
236
  },
237
237
  input: ["text"],
238
- maxTokens: 32768,
238
+ maxTokens: 524288,
239
239
  reasoning: true,
240
240
  thinkingLevelMap: {
241
241
  high: "high",
242
- low: null,
243
- medium: null,
242
+ low: "low",
243
+ medium: "medium",
244
+ minimal: null,
245
+ off: "none",
246
+ xhigh: "max",
247
+ },
248
+ },
249
+ {
250
+ id: "glm-5.3-flash",
251
+ name: "glm-5.3-flash",
252
+ compat: {
253
+ maxTokensField: "max_tokens",
254
+ openRouterRouting: {},
255
+ requiresAssistantAfterToolResult: false,
256
+ requiresReasoningContentOnAssistantMessages: false,
257
+ requiresThinkingAsText: false,
258
+ requiresToolResultName: false,
259
+ sendSessionAffinityHeaders: false,
260
+ supportsDeveloperRole: false,
261
+ supportsLongCacheRetention: false,
262
+ supportsReasoningEffort: true,
263
+ supportsStore: false,
264
+ supportsStrictMode: false,
265
+ supportsUsageInStreaming: true,
266
+ thinkingFormat: "openai",
267
+ vercelGatewayRouting: {},
268
+ zaiToolStream: false,
269
+ },
270
+ contextWindow: 1048576,
271
+ cost: {
272
+ cacheRead: 0.03,
273
+ cacheWrite: 0,
274
+ input: 0.15,
275
+ output: 0.5,
276
+ },
277
+ input: ["text", "image"],
278
+ maxTokens: 524288,
279
+ reasoning: true,
280
+ thinkingLevelMap: {
281
+ high: "high",
282
+ low: "low",
283
+ medium: "medium",
244
284
  minimal: null,
245
285
  off: "none",
246
286
  xhigh: "max",
@@ -269,13 +309,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
269
309
  },
270
310
  contextWindow: 131072,
271
311
  cost: {
272
- cacheRead: 0,
312
+ cacheRead: 0.014,
273
313
  cacheWrite: 0,
274
- input: 0.037,
275
- output: 0.17,
314
+ input: 0.15,
315
+ output: 0.6,
276
316
  },
277
317
  input: ["text"],
278
- maxTokens: 32768,
318
+ maxTokens: 131072,
279
319
  reasoning: true,
280
320
  thinkingLevelMap: {
281
321
  high: "high",
@@ -309,13 +349,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
309
349
  },
310
350
  contextWindow: 131072,
311
351
  cost: {
312
- cacheRead: 0.03,
352
+ cacheRead: 0.035,
313
353
  cacheWrite: 0,
314
- input: 0.03,
315
- output: 0.13,
354
+ input: 0.07,
355
+ output: 0.3,
316
356
  },
317
357
  input: ["text"],
318
- maxTokens: 32768,
358
+ maxTokens: 131072,
319
359
  reasoning: true,
320
360
  thinkingLevelMap: {
321
361
  high: "high",
@@ -355,7 +395,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
355
395
  output: 4,
356
396
  },
357
397
  input: ["text", "image"],
358
- maxTokens: 32768,
398
+ maxTokens: 262144,
359
399
  reasoning: true,
360
400
  thinkingLevelMap: {
361
401
  high: "high",
@@ -395,7 +435,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
395
435
  output: 4,
396
436
  },
397
437
  input: ["text", "image"],
398
- maxTokens: 32768,
438
+ maxTokens: 262144,
399
439
  reasoning: true,
400
440
  thinkingLevelMap: {
401
441
  high: "high",
@@ -435,7 +475,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
435
475
  output: 15,
436
476
  },
437
477
  input: ["text", "image"],
438
- maxTokens: 32768,
478
+ maxTokens: 524288,
439
479
  reasoning: true,
440
480
  thinkingLevelMap: {
441
481
  high: "high",
@@ -470,12 +510,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
470
510
  contextWindow: 196608,
471
511
  cost: {
472
512
  cacheRead: 0.06,
473
- cacheWrite: 0.375,
513
+ cacheWrite: 0,
474
514
  input: 0.3,
475
515
  output: 1.2,
476
516
  },
477
517
  input: ["text"],
478
- maxTokens: 32768,
518
+ maxTokens: 131072,
479
519
  reasoning: true,
480
520
  thinkingLevelMap: {
481
521
  high: "high",
@@ -507,15 +547,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
507
547
  vercelGatewayRouting: {},
508
548
  zaiToolStream: false,
509
549
  },
510
- contextWindow: 524288,
550
+ contextWindow: 512000,
511
551
  cost: {
512
- cacheRead: 0.06,
552
+ cacheRead: 0.12,
513
553
  cacheWrite: 0,
514
- input: 0.3,
515
- output: 1.2,
554
+ input: 0.6,
555
+ output: 2.4,
516
556
  },
517
557
  input: ["text", "image"],
518
- maxTokens: 32768,
558
+ maxTokens: 131072,
519
559
  reasoning: true,
520
560
  thinkingLevelMap: {
521
561
  high: "high",
@@ -549,13 +589,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
549
589
  },
550
590
  contextWindow: 262144,
551
591
  cost: {
552
- cacheRead: 0,
592
+ cacheRead: 0.5,
553
593
  cacheWrite: 0,
554
594
  input: 0.5,
555
595
  output: 1.5,
556
596
  },
557
597
  input: ["text", "image"],
558
- maxTokens: 32768,
598
+ maxTokens: 262144,
559
599
  reasoning: false,
560
600
  },
561
601
  {
@@ -581,13 +621,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
581
621
  },
582
622
  contextWindow: 262144,
583
623
  cost: {
584
- cacheRead: 0.03,
624
+ cacheRead: 0.06,
585
625
  cacheWrite: 0,
586
- input: 0.05,
587
- output: 0.2,
626
+ input: 0.06,
627
+ output: 0.24,
588
628
  },
589
629
  input: ["text"],
590
- maxTokens: 32768,
630
+ maxTokens: 131072,
591
631
  reasoning: true,
592
632
  thinkingLevelMap: {
593
633
  high: "high",
@@ -621,13 +661,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
621
661
  },
622
662
  contextWindow: 262144,
623
663
  cost: {
624
- cacheRead: 0,
664
+ cacheRead: 0.015,
625
665
  cacheWrite: 0,
626
- input: 0.2,
627
- output: 0.8,
666
+ input: 0.015,
667
+ output: 0.6,
628
668
  },
629
669
  input: ["text"],
630
- maxTokens: 32768,
670
+ maxTokens: 65536,
631
671
  reasoning: true,
632
672
  thinkingLevelMap: {
633
673
  high: "high",
@@ -661,13 +701,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
661
701
  },
662
702
  contextWindow: 262144,
663
703
  cost: {
664
- cacheRead: 0.15,
704
+ cacheRead: 0.1,
665
705
  cacheWrite: 0,
666
- input: 0.5,
667
- output: 2.5,
706
+ input: 0.1,
707
+ output: 3,
668
708
  },
669
709
  input: ["text"],
670
- maxTokens: 32768,
710
+ maxTokens: 65536,
671
711
  reasoning: true,
672
712
  thinkingLevelMap: {
673
713
  high: "high",
@@ -701,13 +741,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
701
741
  },
702
742
  contextWindow: 262144,
703
743
  cost: {
704
- cacheRead: 0,
744
+ cacheRead: 0.6,
705
745
  cacheWrite: 0,
706
746
  input: 0.6,
707
747
  output: 3.6,
708
748
  },
709
749
  input: ["text", "image"],
710
- maxTokens: 32768,
750
+ maxTokens: 65536,
711
751
  reasoning: true,
712
752
  thinkingLevelMap: {
713
753
  high: null,
package/models.ts CHANGED
@@ -1,22 +1,34 @@
1
1
  import type { RefreshModelsContext } from "@earendil-works/pi-ai";
2
2
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
3
+ import { MODEL_MAX_OUTPUT_TOKENS } from "./limits.generated.ts";
3
4
  import { GENERATED_MODELS } from "./models.generated.ts";
4
5
  import { MODEL_PRICING, type ModelPrice } from "./pricing.generated.ts";
5
6
  import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
6
7
  import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
7
8
 
8
9
  // --- Pricing ---
9
- // Estimated per-1M-token prices are generated from models.dev by
10
+ // Per-1M-token prices are generated from the ollama.com/pricing model table by
10
11
  // scripts/generate-pricing.ts (see pricing.generated.ts, do not edit by hand).
11
12
  // Ollama Cloud is subscription-billed; these are equivalent pay-as-you-go
12
- // estimates so /cost shows comparable usage, not actual charges.
13
+ // rates so /cost shows comparable usage, not actual charges.
13
14
 
14
- /** Resolve the estimated price for an Ollama Cloud model ID. Exact match only;
15
+ /** Resolve the price for an Ollama Cloud model ID. Exact match only;
15
16
  * unmapped models return zero. */
16
17
  function resolvePrice(id: string): ModelPrice {
17
18
  return MODEL_PRICING[id] ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
18
19
  }
19
20
 
21
+ // --- Max output tokens ---
22
+ // Per-model limits are probed against the live API by scripts/generate-limits.ts
23
+ // (see limits.generated.ts, do not edit by hand). /api/show does not expose the
24
+ // limit, so models without a probed value fall back to 32768.
25
+
26
+ /** Resolve the max output tokens for an Ollama Cloud model ID. Exact match only;
27
+ * unmapped models fall back to 32768. */
28
+ function resolveMaxTokens(id: string): number {
29
+ return MODEL_MAX_OUTPUT_TOKENS[id] ?? 32768;
30
+ }
31
+
20
32
  // --- Constants ---
21
33
  const FETCH_TIMEOUT_MS = 10000;
22
34
  // How long a stored catalog is considered fresh before the next network refresh
@@ -127,9 +139,7 @@ export function assembleModels(raw: Record<string, OllamaShowResponse>): Provide
127
139
  input: (data.capabilities?.includes("vision") ? ["text", "image"] : ["text"]) as ("text" | "image")[],
128
140
  cost: resolvePrice(id),
129
141
  contextWindow: getContextLength(data.model_info ?? {}),
130
- // No per-model limit exposed by the API (https://docs.ollama.com/api-reference/show-model-details,
131
- // https://github.com/ollama/ollama/issues/7222). 32768 matches most Ollama Cloud context windows.
132
- maxTokens: 32768,
142
+ maxTokens: resolveMaxTokens(id),
133
143
  compat: buildCompat(),
134
144
  }));
135
145
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-ollama-cloud",
3
- "version": "0.9.0",
3
+ "version": "0.10.0",
4
4
  "type": "module",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -8,6 +8,7 @@
8
8
  "files": [
9
9
  "index.ts",
10
10
  "config.ts",
11
+ "limits.generated.ts",
11
12
  "models.ts",
12
13
  "models.generated.ts",
13
14
  "pricing.generated.ts",
@@ -30,7 +31,8 @@
30
31
  "format": "biome format --write .",
31
32
  "test": "vitest run",
32
33
  "smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
33
- "generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts"
34
+ "generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts",
35
+ "generate-limits": "tsx scripts/generate-limits.ts && biome format --write limits.generated.ts"
34
36
  },
35
37
  "pi": {
36
38
  "extensions": [
@@ -1,7 +1,7 @@
1
1
  // Auto-generated by scripts/generate-pricing.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-08-08T04:19:45.594Z
4
- // Model count: 18
3
+ // Generated: 2026-09-03T10:12:00.135Z
4
+ // Model count: 19
5
5
 
6
6
  export interface ModelPrice {
7
7
  input: number;
@@ -11,22 +11,23 @@ export interface ModelPrice {
11
11
  }
12
12
 
13
13
  export const MODEL_PRICING: Record<string, ModelPrice> = {
14
- "deepseek-v4-flash:0731": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
15
- "deepseek-v4-flash:preview": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
16
- "deepseek-v4-pro": { input: 0.435, output: 0.87, cacheRead: 0.003625, cacheWrite: 0 },
17
- "gemma4:31b": { input: 0.1, output: 0.34, cacheRead: 0.1, cacheWrite: 0 },
18
- "glm-5.1": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
14
+ "deepseek-v4-flash:0731": { input: 0.44, output: 1.32, cacheRead: 0.014, cacheWrite: 0 },
15
+ "deepseek-v4-pro:0813": { input: 1.32, output: 3.96, cacheRead: 0.044, cacheWrite: 0 },
16
+ "gemma4:31b": { input: 0.14, output: 0.4, cacheRead: 0.05, cacheWrite: 0 },
17
+ "glm-5.1": { input: 1, output: 3.2, cacheRead: 0.2, cacheWrite: 0 },
19
18
  "glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
20
- "gpt-oss:120b": { input: 0.037, output: 0.17, cacheRead: 0, cacheWrite: 0 },
21
- "gpt-oss:20b": { input: 0.03, output: 0.13, cacheRead: 0.03, cacheWrite: 0 },
19
+ "glm-5.3": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
20
+ "glm-5.3-flash": { input: 0.15, output: 0.5, cacheRead: 0.03, cacheWrite: 0 },
21
+ "gpt-oss:120b": { input: 0.15, output: 0.6, cacheRead: 0.014, cacheWrite: 0 },
22
+ "gpt-oss:20b": { input: 0.07, output: 0.3, cacheRead: 0.035, cacheWrite: 0 },
22
23
  "kimi-k2.6": { input: 0.95, output: 4, cacheRead: 0.16, cacheWrite: 0 },
23
24
  "kimi-k2.7-code": { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 },
24
25
  "kimi-k3": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 },
25
- "minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0.375 },
26
- "minimax-m3": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
27
- "mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0, cacheWrite: 0 },
28
- "nemotron-3-nano:30b": { input: 0.05, output: 0.2, cacheRead: 0.03, cacheWrite: 0 },
29
- "nemotron-3-super": { input: 0.2, output: 0.8, cacheRead: 0, cacheWrite: 0 },
30
- "nemotron-3-ultra": { input: 0.5, output: 2.5, cacheRead: 0.15, cacheWrite: 0 },
31
- "qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0, cacheWrite: 0 },
26
+ "minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
27
+ "minimax-m3": { input: 0.6, output: 2.4, cacheRead: 0.12, cacheWrite: 0 },
28
+ "mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0.5, cacheWrite: 0 },
29
+ "nemotron-3-nano:30b": { input: 0.06, output: 0.24, cacheRead: 0.06, cacheWrite: 0 },
30
+ "nemotron-3-super": { input: 0.015, output: 0.6, cacheRead: 0.015, cacheWrite: 0 },
31
+ "nemotron-3-ultra": { input: 0.1, output: 3, cacheRead: 0.1, cacheWrite: 0 },
32
+ "qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0.6, cacheWrite: 0 },
32
33
  };
package/usage.ts CHANGED
@@ -38,12 +38,12 @@ export interface UsageActivity {
38
38
  starting_at?: string;
39
39
  ending_at?: string;
40
40
  };
41
+ models?: UsageModel[];
41
42
  }
42
43
 
43
44
  export interface UsageData {
44
45
  limits: {
45
- session: UsageLimit;
46
- weekly: UsageLimit;
46
+ monthly: UsageLimit;
47
47
  };
48
48
  activity?: UsageActivity;
49
49
  }
@@ -71,13 +71,11 @@ export function isUsageLimit(data: unknown): data is UsageLimit {
71
71
  );
72
72
  }
73
73
 
74
- /** Validate a parsed /api/usage response: must have session and weekly limits. */
74
+ /** Validate a parsed /api/usage response: must have a monthly limit. */
75
75
  export function isUsageResponse(data: unknown): data is UsageData {
76
76
  if (data == null || typeof data !== "object") return false;
77
77
  const d = data as UsageData;
78
- return (
79
- d.limits != null && typeof d.limits === "object" && isUsageLimit(d.limits.session) && isUsageLimit(d.limits.weekly)
80
- );
78
+ return d.limits != null && typeof d.limits === "object" && isUsageLimit(d.limits.monthly);
81
79
  }
82
80
 
83
81
  // --- Fetch ---
@@ -127,15 +125,9 @@ function usagePercent(usage: number): number {
127
125
  export function formatUsage(data: UsageData): string {
128
126
  const lines: string[] = ["Ollama Cloud usage:"];
129
127
 
130
- const sessionPct = usagePercent(data.limits.session.usage);
131
- lines.push(` Session (5h): ${sessionPct}%`);
132
- for (const m of data.limits.session.models) {
133
- lines.push(` - ${m.name}: ${m.request_count} request${m.request_count === 1 ? "" : "s"}`);
134
- }
135
-
136
- const weeklyPct = usagePercent(data.limits.weekly.usage);
137
- lines.push(` Weekly (7d): ${weeklyPct}%`);
138
- for (const m of data.limits.weekly.models) {
128
+ const monthlyPct = usagePercent(data.limits.monthly.usage);
129
+ lines.push(` Monthly (30d): ${monthlyPct}%`);
130
+ for (const m of data.limits.monthly.models) {
139
131
  lines.push(` - ${m.name}: ${m.request_count} request${m.request_count === 1 ? "" : "s"}`);
140
132
  }
141
133
 
@@ -160,11 +152,10 @@ function colorSegment(theme: Theme, label: string, pct: number): string {
160
152
 
161
153
  /**
162
154
  * Compact one-line usage for the footer status bar, colored by usage level.
163
- * The endpoint exposes reset periods (5h/7d) but not exact timestamps, so the
164
- * color reflects the usage fraction rather than pace.
155
+ * The endpoint exposes a monthly reset period but not an exact timestamp, so
156
+ * the color reflects the usage fraction rather than pace.
165
157
  */
166
158
  export function formatUsageStatusColored(theme: Theme, data: UsageData): string {
167
- const session = usagePercent(data.limits.session.usage);
168
- const weekly = usagePercent(data.limits.weekly.usage);
169
- return `${colorSegment(theme, "5h", session)} ${colorSegment(theme, "7d", weekly)}`;
159
+ const monthly = usagePercent(data.limits.monthly.usage);
160
+ return colorSegment(theme, "30d", monthly);
170
161
  }