pi-ollama-cloud 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/README.md +15 -7
- package/index.ts +2 -2
- package/limits.generated.ts +25 -0
- package/models.generated.ts +114 -74
- package/models.ts +16 -6
- package/package.json +4 -2
- package/pricing.generated.ts +17 -16
- package/usage.ts +11 -20
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,15 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [Unreleased]
|
|
6
|
+
|
|
7
|
+
## [0.10.0] - 2026-09-03
|
|
8
|
+
|
|
9
|
+
- **Breaking:** Adapt to the changed `/api/usage` response shape. The endpoint now returns a single `limits.monthly` bucket (replacing `limits.session` and `limits.weekly`) and adds an `activity.models` array. `UsageData`, `isUsageResponse`, `formatUsage`, and `formatUsageStatusColored` now read the monthly limit; the status bar shows a single `30d` segment instead of `5h`/`7d`.
|
|
10
|
+
- Refreshed the generated catalog from the live API: added `deepseek-v4-pro:0813`, `glm-5.3`, and `glm-5.3-flash`; removed `deepseek-v4-flash:preview` and `deepseek-v4-pro`, which are no longer listed.
|
|
11
|
+
- Source per-token pricing from the official model table on ollama.com/pricing instead of models.dev estimates, which no longer track Ollama's published rates (up to ~13x off per model). `scripts/generate-pricing.ts` now scrapes the pricing page (the table is server-rendered; no JSON endpoint exists) and matches catalog IDs to pricing rows by exact or `:tag`-family match, replacing the `OLLAMA_TO_MODELSDEV` mapping. Regenerated `pricing.generated.ts` with the official rates, including new models (`glm-5.3`, `glm-5.3-flash`, `deepseek-v4-pro:0813`). Fixes #51. Thanks @Hackbard (#52).
|
|
12
|
+
- Probe per-model max output tokens against the live API via `scripts/generate-limits.ts` into `limits.generated.ts`, replacing the fixed 32768 default for known models (unprobed models still fall back to 32768). The probe timeout is 60s so slow first-token models are not dropped from the table. Thanks @f440 (#49).
|
|
13
|
+
|
|
5
14
|
## [0.9.0] - 2026-08-11
|
|
6
15
|
|
|
7
16
|
- Add `/ollama-cloud-usage` command to show Ollama Cloud session (5h) and weekly (7d) usage limits, per-model request counts, and the 4-week activity cost, fetched from the undocumented `/api/usage` endpoint with the already-resolved API key.
|
package/README.md
CHANGED
|
@@ -12,7 +12,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
|
|
|
12
12
|
- **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
|
|
13
13
|
- **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
|
|
14
14
|
- **`ollama_web_fetch` tool** - Fetch and extract text content from a web page URL using Ollama Cloud's `/api/web_fetch` endpoint. Returns page title, content, and links.
|
|
15
|
-
- **
|
|
15
|
+
- **Per-token cost tracking** - Models are registered with the official per-token prices from [ollama.com/pricing](https://ollama.com/pricing), so Pi's `/cost` shows comparable usage. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges.
|
|
16
16
|
|
|
17
17
|
## Prerequisites
|
|
18
18
|
|
|
@@ -135,8 +135,14 @@ Model metadata is derived from the `/api/show` response:
|
|
|
135
135
|
| `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 5 maps (DEFAULT, GPT_OSS, QWEN3, GLM_52, NO_OFF) based on API testing |
|
|
136
136
|
| `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
|
|
137
137
|
| `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
|
|
138
|
-
| `maxTokens` |
|
|
139
|
-
| `cost` |
|
|
138
|
+
| `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
|
|
139
|
+
| `cost` | Official per-1M-token prices from the [ollama.com/pricing](https://ollama.com/pricing) model table, generated by `scripts/generate-pricing.ts` into `pricing.generated.ts`. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges. Catalog IDs with no matching pricing row default to zero. Prices are pinned to the installed package version and only update on a new release, so newly added models register with zero cost until then. |
|
|
140
|
+
|
|
141
|
+
The per-model max output token table (`limits.generated.ts`) is probed against the live API by `scripts/generate-limits.ts`, since `/api/show` does not expose the limit. It needs an API key: `OLLAMA_API_KEY=<key> npm run generate-limits`. Limits ship with the package, so regenerated values take effect on the next release.
|
|
142
|
+
|
|
143
|
+
The API itself returns no cost data: completion responses report only token counts (`prompt_tokens`/`completion_tokens`/`total_tokens`, including the final usage chunk when streaming), and `/api/show` exposes no pricing fields. The prices above come from the static `/pricing` page table and are only as fresh as the last regeneration.
|
|
144
|
+
|
|
145
|
+
Cache pricing is informational only: the `/pricing` page lists a "Cached input" column, but the completion API does not report cache token usage (there is no `prompt_tokens_details.cached_tokens` or equivalent in any response, verified against the live API in September 2026), so pi never sees cache hits and `/cost` estimates do not reflect them. `cacheWrite` is always zero because the pricing table has no cache-write column.
|
|
140
146
|
|
|
141
147
|
### Thinking level mapping
|
|
142
148
|
|
|
@@ -166,14 +172,14 @@ Both tools use the same Ollama Cloud API key configured for the provider. No loc
|
|
|
166
172
|
| Command | Description |
|
|
167
173
|
|---|---|
|
|
168
174
|
| `/ollama-webtools [on\|off\|enable\|disable]` | Enable or disable the `ollama_web_search` and `ollama_web_fetch` tools. Toggles if no argument given. |
|
|
169
|
-
| `/ollama-cloud-usage` | Show Ollama Cloud
|
|
175
|
+
| `/ollama-cloud-usage` | Show Ollama Cloud monthly usage limits, per-model request counts, and the 4-week activity cost. |
|
|
170
176
|
| `/ollama-usage-status [on\|off\|enable\|disable]` | Enable or disable the footer usage status bar. Toggles if no argument given. |
|
|
171
177
|
|
|
172
178
|
## Usage status bar
|
|
173
179
|
|
|
174
180
|
While an `ollama-cloud` model is the active provider, the footer shows a compact
|
|
175
|
-
live usage readout (`
|
|
176
|
-
every 5 minutes and after each agent turn (but no more often than every 5 minutes).
|
|
181
|
+
live usage readout (`30d ▕███░░░░░░░▏ 34%`) that refreshes
|
|
182
|
+
every 5 minutes and after each agent turn (but no more often than every 5 minutes). It is colored by how close
|
|
177
183
|
it is to the cap: green below 60%, yellow at 60-79%, red at 80%+. It reads the
|
|
178
184
|
same undocumented `/api/usage` endpoint as `/ollama-cloud-usage` and clears
|
|
179
185
|
itself on transient errors or when you switch to a non-Ollama-Cloud provider.
|
|
@@ -227,6 +233,7 @@ npm run check # lint + format + type-check (auto-fix)
|
|
|
227
233
|
npm run lint # lint only (no fixes)
|
|
228
234
|
npm run typecheck # type-check only (tsgo --noEmit)
|
|
229
235
|
npm run format # format only
|
|
236
|
+
OLLAMA_API_KEY=<key> npm run generate-limits # probe max output tokens (writes limits.generated.ts)
|
|
230
237
|
```
|
|
231
238
|
|
|
232
239
|
The project uses [Biome](https://biomejs.dev/) for linting and formatting (2-space indent, line width 120) and [tsgo](https://github.com/microsoft/typescript-go) for type-checking.
|
|
@@ -295,7 +302,8 @@ git push --tags
|
|
|
295
302
|
Because the model catalog refreshes automatically at runtime, a release is **not** needed to ship new models. Publish only when:
|
|
296
303
|
|
|
297
304
|
- A model is retired and still listed by the API: add it to `RETIRED_MODEL_IDS` in `scripts/generate-models.ts` (check https://docs.ollama.com/cloud#retirements, then regenerate `models.generated.ts`).
|
|
298
|
-
- Pricing changes:
|
|
305
|
+
- Pricing changes: Ollama updates the model pricing table, or a new model needs a pricing row (regenerate `pricing.generated.ts`).
|
|
306
|
+
- Max output token limits changed: run `OLLAMA_API_KEY=<key> npm run generate-limits` locally and commit.
|
|
299
307
|
|
|
300
308
|
The tag version must match the version in `package.json` - `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
|
|
301
309
|
|
package/index.ts
CHANGED
|
@@ -134,7 +134,7 @@ export default async function (pi: ExtensionAPI) {
|
|
|
134
134
|
// --- Usage Command ---
|
|
135
135
|
|
|
136
136
|
pi.registerCommand("ollama-cloud-usage", {
|
|
137
|
-
description: "Show Ollama Cloud
|
|
137
|
+
description: "Show Ollama Cloud monthly usage limits.",
|
|
138
138
|
handler: async (_args, ctx) => {
|
|
139
139
|
const apiKey = await getCloudApiKey(ctx);
|
|
140
140
|
if (!apiKey) {
|
|
@@ -152,7 +152,7 @@ export default async function (pi: ExtensionAPI) {
|
|
|
152
152
|
|
|
153
153
|
// --- Usage Status Bar ---
|
|
154
154
|
|
|
155
|
-
// Footer status showing live
|
|
155
|
+
// Footer status showing live monthly usage while ollama-cloud is the
|
|
156
156
|
// active provider. Refreshes on a 5-minute timer; agent_end also triggers a
|
|
157
157
|
// refresh but is throttled to the same cooldown so a turn never hammers the
|
|
158
158
|
// undocumented /api/usage endpoint. The quota-bar concept is inspired by
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
// Auto-generated by scripts/generate-limits.ts
|
|
2
|
+
// Do not edit manually.
|
|
3
|
+
// Probed models: 19 (0 failed)
|
|
4
|
+
|
|
5
|
+
export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
6
|
+
"deepseek-v4-flash:0731": 65536,
|
|
7
|
+
"deepseek-v4-pro:0813": 65536,
|
|
8
|
+
"gemma4:31b": 262144,
|
|
9
|
+
"glm-5.1": 131072,
|
|
10
|
+
"glm-5.2": 131072,
|
|
11
|
+
"glm-5.3": 524288,
|
|
12
|
+
"glm-5.3-flash": 524288,
|
|
13
|
+
"gpt-oss:120b": 131072,
|
|
14
|
+
"gpt-oss:20b": 131072,
|
|
15
|
+
"kimi-k2.6": 262144,
|
|
16
|
+
"kimi-k2.7-code": 262144,
|
|
17
|
+
"kimi-k3": 524288,
|
|
18
|
+
"minimax-m2.7": 131072,
|
|
19
|
+
"minimax-m3": 131072,
|
|
20
|
+
"mistral-large-3:675b": 262144,
|
|
21
|
+
"nemotron-3-nano:30b": 131072,
|
|
22
|
+
"nemotron-3-super": 65536,
|
|
23
|
+
"nemotron-3-ultra": 65536,
|
|
24
|
+
"qwen3.5:397b": 65536,
|
|
25
|
+
};
|
package/models.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-models.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-03T10:12:02.243Z
|
|
4
|
+
// Model count: 19
|
|
5
5
|
|
|
6
6
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
7
7
|
|
|
@@ -29,13 +29,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
29
29
|
},
|
|
30
30
|
contextWindow: 1048576,
|
|
31
31
|
cost: {
|
|
32
|
-
cacheRead: 0.
|
|
32
|
+
cacheRead: 0.014,
|
|
33
33
|
cacheWrite: 0,
|
|
34
|
-
input: 0.
|
|
35
|
-
output:
|
|
34
|
+
input: 0.44,
|
|
35
|
+
output: 1.32,
|
|
36
36
|
},
|
|
37
37
|
input: ["text"],
|
|
38
|
-
maxTokens:
|
|
38
|
+
maxTokens: 65536,
|
|
39
39
|
reasoning: true,
|
|
40
40
|
thinkingLevelMap: {
|
|
41
41
|
high: "high",
|
|
@@ -47,8 +47,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
47
47
|
},
|
|
48
48
|
},
|
|
49
49
|
{
|
|
50
|
-
id: "deepseek-v4-
|
|
51
|
-
name: "deepseek-v4-
|
|
50
|
+
id: "deepseek-v4-pro:0813",
|
|
51
|
+
name: "deepseek-v4-pro:0813",
|
|
52
52
|
compat: {
|
|
53
53
|
maxTokensField: "max_tokens",
|
|
54
54
|
openRouterRouting: {},
|
|
@@ -69,13 +69,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
69
69
|
},
|
|
70
70
|
contextWindow: 1048576,
|
|
71
71
|
cost: {
|
|
72
|
-
cacheRead: 0.
|
|
72
|
+
cacheRead: 0.044,
|
|
73
73
|
cacheWrite: 0,
|
|
74
|
-
input:
|
|
75
|
-
output:
|
|
74
|
+
input: 1.32,
|
|
75
|
+
output: 3.96,
|
|
76
76
|
},
|
|
77
77
|
input: ["text"],
|
|
78
|
-
maxTokens:
|
|
78
|
+
maxTokens: 65536,
|
|
79
79
|
reasoning: true,
|
|
80
80
|
thinkingLevelMap: {
|
|
81
81
|
high: "high",
|
|
@@ -87,8 +87,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
87
87
|
},
|
|
88
88
|
},
|
|
89
89
|
{
|
|
90
|
-
id: "
|
|
91
|
-
name: "
|
|
90
|
+
id: "gemma4:31b",
|
|
91
|
+
name: "gemma4:31b",
|
|
92
92
|
compat: {
|
|
93
93
|
maxTokensField: "max_tokens",
|
|
94
94
|
openRouterRouting: {},
|
|
@@ -107,15 +107,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
107
107
|
vercelGatewayRouting: {},
|
|
108
108
|
zaiToolStream: false,
|
|
109
109
|
},
|
|
110
|
-
contextWindow:
|
|
110
|
+
contextWindow: 262144,
|
|
111
111
|
cost: {
|
|
112
|
-
cacheRead: 0.
|
|
112
|
+
cacheRead: 0.05,
|
|
113
113
|
cacheWrite: 0,
|
|
114
|
-
input: 0.
|
|
115
|
-
output: 0.
|
|
114
|
+
input: 0.14,
|
|
115
|
+
output: 0.4,
|
|
116
116
|
},
|
|
117
|
-
input: ["text"],
|
|
118
|
-
maxTokens:
|
|
117
|
+
input: ["text", "image"],
|
|
118
|
+
maxTokens: 262144,
|
|
119
119
|
reasoning: true,
|
|
120
120
|
thinkingLevelMap: {
|
|
121
121
|
high: "high",
|
|
@@ -127,8 +127,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
127
127
|
},
|
|
128
128
|
},
|
|
129
129
|
{
|
|
130
|
-
id: "
|
|
131
|
-
name: "
|
|
130
|
+
id: "glm-5.1",
|
|
131
|
+
name: "glm-5.1",
|
|
132
132
|
compat: {
|
|
133
133
|
maxTokensField: "max_tokens",
|
|
134
134
|
openRouterRouting: {},
|
|
@@ -147,15 +147,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
147
147
|
vercelGatewayRouting: {},
|
|
148
148
|
zaiToolStream: false,
|
|
149
149
|
},
|
|
150
|
-
contextWindow:
|
|
150
|
+
contextWindow: 202752,
|
|
151
151
|
cost: {
|
|
152
|
-
cacheRead: 0.
|
|
152
|
+
cacheRead: 0.2,
|
|
153
153
|
cacheWrite: 0,
|
|
154
|
-
input:
|
|
155
|
-
output:
|
|
154
|
+
input: 1,
|
|
155
|
+
output: 3.2,
|
|
156
156
|
},
|
|
157
|
-
input: ["text"
|
|
158
|
-
maxTokens:
|
|
157
|
+
input: ["text"],
|
|
158
|
+
maxTokens: 131072,
|
|
159
159
|
reasoning: true,
|
|
160
160
|
thinkingLevelMap: {
|
|
161
161
|
high: "high",
|
|
@@ -167,8 +167,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
167
167
|
},
|
|
168
168
|
},
|
|
169
169
|
{
|
|
170
|
-
id: "glm-5.
|
|
171
|
-
name: "glm-5.
|
|
170
|
+
id: "glm-5.2",
|
|
171
|
+
name: "glm-5.2",
|
|
172
172
|
compat: {
|
|
173
173
|
maxTokensField: "max_tokens",
|
|
174
174
|
openRouterRouting: {},
|
|
@@ -187,7 +187,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
187
187
|
vercelGatewayRouting: {},
|
|
188
188
|
zaiToolStream: false,
|
|
189
189
|
},
|
|
190
|
-
contextWindow:
|
|
190
|
+
contextWindow: 1048576,
|
|
191
191
|
cost: {
|
|
192
192
|
cacheRead: 0.26,
|
|
193
193
|
cacheWrite: 0,
|
|
@@ -195,20 +195,20 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
195
195
|
output: 4.4,
|
|
196
196
|
},
|
|
197
197
|
input: ["text"],
|
|
198
|
-
maxTokens:
|
|
198
|
+
maxTokens: 131072,
|
|
199
199
|
reasoning: true,
|
|
200
200
|
thinkingLevelMap: {
|
|
201
201
|
high: "high",
|
|
202
|
-
low:
|
|
203
|
-
medium:
|
|
202
|
+
low: null,
|
|
203
|
+
medium: null,
|
|
204
204
|
minimal: null,
|
|
205
205
|
off: "none",
|
|
206
206
|
xhigh: "max",
|
|
207
207
|
},
|
|
208
208
|
},
|
|
209
209
|
{
|
|
210
|
-
id: "glm-5.
|
|
211
|
-
name: "glm-5.
|
|
210
|
+
id: "glm-5.3",
|
|
211
|
+
name: "glm-5.3",
|
|
212
212
|
compat: {
|
|
213
213
|
maxTokensField: "max_tokens",
|
|
214
214
|
openRouterRouting: {},
|
|
@@ -227,7 +227,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
227
227
|
vercelGatewayRouting: {},
|
|
228
228
|
zaiToolStream: false,
|
|
229
229
|
},
|
|
230
|
-
contextWindow:
|
|
230
|
+
contextWindow: 1048576,
|
|
231
231
|
cost: {
|
|
232
232
|
cacheRead: 0.26,
|
|
233
233
|
cacheWrite: 0,
|
|
@@ -235,12 +235,52 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
235
235
|
output: 4.4,
|
|
236
236
|
},
|
|
237
237
|
input: ["text"],
|
|
238
|
-
maxTokens:
|
|
238
|
+
maxTokens: 524288,
|
|
239
239
|
reasoning: true,
|
|
240
240
|
thinkingLevelMap: {
|
|
241
241
|
high: "high",
|
|
242
|
-
low:
|
|
243
|
-
medium:
|
|
242
|
+
low: "low",
|
|
243
|
+
medium: "medium",
|
|
244
|
+
minimal: null,
|
|
245
|
+
off: "none",
|
|
246
|
+
xhigh: "max",
|
|
247
|
+
},
|
|
248
|
+
},
|
|
249
|
+
{
|
|
250
|
+
id: "glm-5.3-flash",
|
|
251
|
+
name: "glm-5.3-flash",
|
|
252
|
+
compat: {
|
|
253
|
+
maxTokensField: "max_tokens",
|
|
254
|
+
openRouterRouting: {},
|
|
255
|
+
requiresAssistantAfterToolResult: false,
|
|
256
|
+
requiresReasoningContentOnAssistantMessages: false,
|
|
257
|
+
requiresThinkingAsText: false,
|
|
258
|
+
requiresToolResultName: false,
|
|
259
|
+
sendSessionAffinityHeaders: false,
|
|
260
|
+
supportsDeveloperRole: false,
|
|
261
|
+
supportsLongCacheRetention: false,
|
|
262
|
+
supportsReasoningEffort: true,
|
|
263
|
+
supportsStore: false,
|
|
264
|
+
supportsStrictMode: false,
|
|
265
|
+
supportsUsageInStreaming: true,
|
|
266
|
+
thinkingFormat: "openai",
|
|
267
|
+
vercelGatewayRouting: {},
|
|
268
|
+
zaiToolStream: false,
|
|
269
|
+
},
|
|
270
|
+
contextWindow: 1048576,
|
|
271
|
+
cost: {
|
|
272
|
+
cacheRead: 0.03,
|
|
273
|
+
cacheWrite: 0,
|
|
274
|
+
input: 0.15,
|
|
275
|
+
output: 0.5,
|
|
276
|
+
},
|
|
277
|
+
input: ["text", "image"],
|
|
278
|
+
maxTokens: 524288,
|
|
279
|
+
reasoning: true,
|
|
280
|
+
thinkingLevelMap: {
|
|
281
|
+
high: "high",
|
|
282
|
+
low: "low",
|
|
283
|
+
medium: "medium",
|
|
244
284
|
minimal: null,
|
|
245
285
|
off: "none",
|
|
246
286
|
xhigh: "max",
|
|
@@ -269,13 +309,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
269
309
|
},
|
|
270
310
|
contextWindow: 131072,
|
|
271
311
|
cost: {
|
|
272
|
-
cacheRead: 0,
|
|
312
|
+
cacheRead: 0.014,
|
|
273
313
|
cacheWrite: 0,
|
|
274
|
-
input: 0.
|
|
275
|
-
output: 0.
|
|
314
|
+
input: 0.15,
|
|
315
|
+
output: 0.6,
|
|
276
316
|
},
|
|
277
317
|
input: ["text"],
|
|
278
|
-
maxTokens:
|
|
318
|
+
maxTokens: 131072,
|
|
279
319
|
reasoning: true,
|
|
280
320
|
thinkingLevelMap: {
|
|
281
321
|
high: "high",
|
|
@@ -309,13 +349,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
309
349
|
},
|
|
310
350
|
contextWindow: 131072,
|
|
311
351
|
cost: {
|
|
312
|
-
cacheRead: 0.
|
|
352
|
+
cacheRead: 0.035,
|
|
313
353
|
cacheWrite: 0,
|
|
314
|
-
input: 0.
|
|
315
|
-
output: 0.
|
|
354
|
+
input: 0.07,
|
|
355
|
+
output: 0.3,
|
|
316
356
|
},
|
|
317
357
|
input: ["text"],
|
|
318
|
-
maxTokens:
|
|
358
|
+
maxTokens: 131072,
|
|
319
359
|
reasoning: true,
|
|
320
360
|
thinkingLevelMap: {
|
|
321
361
|
high: "high",
|
|
@@ -355,7 +395,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
355
395
|
output: 4,
|
|
356
396
|
},
|
|
357
397
|
input: ["text", "image"],
|
|
358
|
-
maxTokens:
|
|
398
|
+
maxTokens: 262144,
|
|
359
399
|
reasoning: true,
|
|
360
400
|
thinkingLevelMap: {
|
|
361
401
|
high: "high",
|
|
@@ -395,7 +435,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
395
435
|
output: 4,
|
|
396
436
|
},
|
|
397
437
|
input: ["text", "image"],
|
|
398
|
-
maxTokens:
|
|
438
|
+
maxTokens: 262144,
|
|
399
439
|
reasoning: true,
|
|
400
440
|
thinkingLevelMap: {
|
|
401
441
|
high: "high",
|
|
@@ -435,7 +475,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
435
475
|
output: 15,
|
|
436
476
|
},
|
|
437
477
|
input: ["text", "image"],
|
|
438
|
-
maxTokens:
|
|
478
|
+
maxTokens: 524288,
|
|
439
479
|
reasoning: true,
|
|
440
480
|
thinkingLevelMap: {
|
|
441
481
|
high: "high",
|
|
@@ -470,12 +510,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
470
510
|
contextWindow: 196608,
|
|
471
511
|
cost: {
|
|
472
512
|
cacheRead: 0.06,
|
|
473
|
-
cacheWrite: 0
|
|
513
|
+
cacheWrite: 0,
|
|
474
514
|
input: 0.3,
|
|
475
515
|
output: 1.2,
|
|
476
516
|
},
|
|
477
517
|
input: ["text"],
|
|
478
|
-
maxTokens:
|
|
518
|
+
maxTokens: 131072,
|
|
479
519
|
reasoning: true,
|
|
480
520
|
thinkingLevelMap: {
|
|
481
521
|
high: "high",
|
|
@@ -507,15 +547,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
507
547
|
vercelGatewayRouting: {},
|
|
508
548
|
zaiToolStream: false,
|
|
509
549
|
},
|
|
510
|
-
contextWindow:
|
|
550
|
+
contextWindow: 512000,
|
|
511
551
|
cost: {
|
|
512
|
-
cacheRead: 0.
|
|
552
|
+
cacheRead: 0.12,
|
|
513
553
|
cacheWrite: 0,
|
|
514
|
-
input: 0.
|
|
515
|
-
output:
|
|
554
|
+
input: 0.6,
|
|
555
|
+
output: 2.4,
|
|
516
556
|
},
|
|
517
557
|
input: ["text", "image"],
|
|
518
|
-
maxTokens:
|
|
558
|
+
maxTokens: 131072,
|
|
519
559
|
reasoning: true,
|
|
520
560
|
thinkingLevelMap: {
|
|
521
561
|
high: "high",
|
|
@@ -549,13 +589,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
549
589
|
},
|
|
550
590
|
contextWindow: 262144,
|
|
551
591
|
cost: {
|
|
552
|
-
cacheRead: 0,
|
|
592
|
+
cacheRead: 0.5,
|
|
553
593
|
cacheWrite: 0,
|
|
554
594
|
input: 0.5,
|
|
555
595
|
output: 1.5,
|
|
556
596
|
},
|
|
557
597
|
input: ["text", "image"],
|
|
558
|
-
maxTokens:
|
|
598
|
+
maxTokens: 262144,
|
|
559
599
|
reasoning: false,
|
|
560
600
|
},
|
|
561
601
|
{
|
|
@@ -581,13 +621,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
581
621
|
},
|
|
582
622
|
contextWindow: 262144,
|
|
583
623
|
cost: {
|
|
584
|
-
cacheRead: 0.
|
|
624
|
+
cacheRead: 0.06,
|
|
585
625
|
cacheWrite: 0,
|
|
586
|
-
input: 0.
|
|
587
|
-
output: 0.
|
|
626
|
+
input: 0.06,
|
|
627
|
+
output: 0.24,
|
|
588
628
|
},
|
|
589
629
|
input: ["text"],
|
|
590
|
-
maxTokens:
|
|
630
|
+
maxTokens: 131072,
|
|
591
631
|
reasoning: true,
|
|
592
632
|
thinkingLevelMap: {
|
|
593
633
|
high: "high",
|
|
@@ -621,13 +661,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
621
661
|
},
|
|
622
662
|
contextWindow: 262144,
|
|
623
663
|
cost: {
|
|
624
|
-
cacheRead: 0,
|
|
664
|
+
cacheRead: 0.015,
|
|
625
665
|
cacheWrite: 0,
|
|
626
|
-
input: 0.
|
|
627
|
-
output: 0.
|
|
666
|
+
input: 0.015,
|
|
667
|
+
output: 0.6,
|
|
628
668
|
},
|
|
629
669
|
input: ["text"],
|
|
630
|
-
maxTokens:
|
|
670
|
+
maxTokens: 65536,
|
|
631
671
|
reasoning: true,
|
|
632
672
|
thinkingLevelMap: {
|
|
633
673
|
high: "high",
|
|
@@ -661,13 +701,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
661
701
|
},
|
|
662
702
|
contextWindow: 262144,
|
|
663
703
|
cost: {
|
|
664
|
-
cacheRead: 0.
|
|
704
|
+
cacheRead: 0.1,
|
|
665
705
|
cacheWrite: 0,
|
|
666
|
-
input: 0.
|
|
667
|
-
output:
|
|
706
|
+
input: 0.1,
|
|
707
|
+
output: 3,
|
|
668
708
|
},
|
|
669
709
|
input: ["text"],
|
|
670
|
-
maxTokens:
|
|
710
|
+
maxTokens: 65536,
|
|
671
711
|
reasoning: true,
|
|
672
712
|
thinkingLevelMap: {
|
|
673
713
|
high: "high",
|
|
@@ -701,13 +741,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
701
741
|
},
|
|
702
742
|
contextWindow: 262144,
|
|
703
743
|
cost: {
|
|
704
|
-
cacheRead: 0,
|
|
744
|
+
cacheRead: 0.6,
|
|
705
745
|
cacheWrite: 0,
|
|
706
746
|
input: 0.6,
|
|
707
747
|
output: 3.6,
|
|
708
748
|
},
|
|
709
749
|
input: ["text", "image"],
|
|
710
|
-
maxTokens:
|
|
750
|
+
maxTokens: 65536,
|
|
711
751
|
reasoning: true,
|
|
712
752
|
thinkingLevelMap: {
|
|
713
753
|
high: null,
|
package/models.ts
CHANGED
|
@@ -1,22 +1,34 @@
|
|
|
1
1
|
import type { RefreshModelsContext } from "@earendil-works/pi-ai";
|
|
2
2
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { MODEL_MAX_OUTPUT_TOKENS } from "./limits.generated.ts";
|
|
3
4
|
import { GENERATED_MODELS } from "./models.generated.ts";
|
|
4
5
|
import { MODEL_PRICING, type ModelPrice } from "./pricing.generated.ts";
|
|
5
6
|
import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
|
|
6
7
|
import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
|
|
7
8
|
|
|
8
9
|
// --- Pricing ---
|
|
9
|
-
//
|
|
10
|
+
// Per-1M-token prices are generated from the ollama.com/pricing model table by
|
|
10
11
|
// scripts/generate-pricing.ts (see pricing.generated.ts, do not edit by hand).
|
|
11
12
|
// Ollama Cloud is subscription-billed; these are equivalent pay-as-you-go
|
|
12
|
-
//
|
|
13
|
+
// rates so /cost shows comparable usage, not actual charges.
|
|
13
14
|
|
|
14
|
-
/** Resolve the
|
|
15
|
+
/** Resolve the price for an Ollama Cloud model ID. Exact match only;
|
|
15
16
|
* unmapped models return zero. */
|
|
16
17
|
function resolvePrice(id: string): ModelPrice {
|
|
17
18
|
return MODEL_PRICING[id] ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
|
|
18
19
|
}
|
|
19
20
|
|
|
21
|
+
// --- Max output tokens ---
|
|
22
|
+
// Per-model limits are probed against the live API by scripts/generate-limits.ts
|
|
23
|
+
// (see limits.generated.ts, do not edit by hand). /api/show does not expose the
|
|
24
|
+
// limit, so models without a probed value fall back to 32768.
|
|
25
|
+
|
|
26
|
+
/** Resolve the max output tokens for an Ollama Cloud model ID. Exact match only;
|
|
27
|
+
* unmapped models fall back to 32768. */
|
|
28
|
+
function resolveMaxTokens(id: string): number {
|
|
29
|
+
return MODEL_MAX_OUTPUT_TOKENS[id] ?? 32768;
|
|
30
|
+
}
|
|
31
|
+
|
|
20
32
|
// --- Constants ---
|
|
21
33
|
const FETCH_TIMEOUT_MS = 10000;
|
|
22
34
|
// How long a stored catalog is considered fresh before the next network refresh
|
|
@@ -127,9 +139,7 @@ export function assembleModels(raw: Record<string, OllamaShowResponse>): Provide
|
|
|
127
139
|
input: (data.capabilities?.includes("vision") ? ["text", "image"] : ["text"]) as ("text" | "image")[],
|
|
128
140
|
cost: resolvePrice(id),
|
|
129
141
|
contextWindow: getContextLength(data.model_info ?? {}),
|
|
130
|
-
|
|
131
|
-
// https://github.com/ollama/ollama/issues/7222). 32768 matches most Ollama Cloud context windows.
|
|
132
|
-
maxTokens: 32768,
|
|
142
|
+
maxTokens: resolveMaxTokens(id),
|
|
133
143
|
compat: buildCompat(),
|
|
134
144
|
}));
|
|
135
145
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-ollama-cloud",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package"
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
"files": [
|
|
9
9
|
"index.ts",
|
|
10
10
|
"config.ts",
|
|
11
|
+
"limits.generated.ts",
|
|
11
12
|
"models.ts",
|
|
12
13
|
"models.generated.ts",
|
|
13
14
|
"pricing.generated.ts",
|
|
@@ -30,7 +31,8 @@
|
|
|
30
31
|
"format": "biome format --write .",
|
|
31
32
|
"test": "vitest run",
|
|
32
33
|
"smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
|
|
33
|
-
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts"
|
|
34
|
+
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts",
|
|
35
|
+
"generate-limits": "tsx scripts/generate-limits.ts && biome format --write limits.generated.ts"
|
|
34
36
|
},
|
|
35
37
|
"pi": {
|
|
36
38
|
"extensions": [
|
package/pricing.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-pricing.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-03T10:12:00.135Z
|
|
4
|
+
// Model count: 19
|
|
5
5
|
|
|
6
6
|
export interface ModelPrice {
|
|
7
7
|
input: number;
|
|
@@ -11,22 +11,23 @@ export interface ModelPrice {
|
|
|
11
11
|
}
|
|
12
12
|
|
|
13
13
|
export const MODEL_PRICING: Record<string, ModelPrice> = {
|
|
14
|
-
"deepseek-v4-flash:0731": { input: 0.
|
|
15
|
-
"deepseek-v4-
|
|
16
|
-
"
|
|
17
|
-
"
|
|
18
|
-
"glm-5.1": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
14
|
+
"deepseek-v4-flash:0731": { input: 0.44, output: 1.32, cacheRead: 0.014, cacheWrite: 0 },
|
|
15
|
+
"deepseek-v4-pro:0813": { input: 1.32, output: 3.96, cacheRead: 0.044, cacheWrite: 0 },
|
|
16
|
+
"gemma4:31b": { input: 0.14, output: 0.4, cacheRead: 0.05, cacheWrite: 0 },
|
|
17
|
+
"glm-5.1": { input: 1, output: 3.2, cacheRead: 0.2, cacheWrite: 0 },
|
|
19
18
|
"glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
20
|
-
"
|
|
21
|
-
"
|
|
19
|
+
"glm-5.3": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
20
|
+
"glm-5.3-flash": { input: 0.15, output: 0.5, cacheRead: 0.03, cacheWrite: 0 },
|
|
21
|
+
"gpt-oss:120b": { input: 0.15, output: 0.6, cacheRead: 0.014, cacheWrite: 0 },
|
|
22
|
+
"gpt-oss:20b": { input: 0.07, output: 0.3, cacheRead: 0.035, cacheWrite: 0 },
|
|
22
23
|
"kimi-k2.6": { input: 0.95, output: 4, cacheRead: 0.16, cacheWrite: 0 },
|
|
23
24
|
"kimi-k2.7-code": { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 },
|
|
24
25
|
"kimi-k3": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 },
|
|
25
|
-
"minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0
|
|
26
|
-
"minimax-m3": { input: 0.
|
|
27
|
-
"mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0, cacheWrite: 0 },
|
|
28
|
-
"nemotron-3-nano:30b": { input: 0.
|
|
29
|
-
"nemotron-3-super": { input: 0.
|
|
30
|
-
"nemotron-3-ultra": { input: 0.
|
|
31
|
-
"qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0, cacheWrite: 0 },
|
|
26
|
+
"minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
|
|
27
|
+
"minimax-m3": { input: 0.6, output: 2.4, cacheRead: 0.12, cacheWrite: 0 },
|
|
28
|
+
"mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0.5, cacheWrite: 0 },
|
|
29
|
+
"nemotron-3-nano:30b": { input: 0.06, output: 0.24, cacheRead: 0.06, cacheWrite: 0 },
|
|
30
|
+
"nemotron-3-super": { input: 0.015, output: 0.6, cacheRead: 0.015, cacheWrite: 0 },
|
|
31
|
+
"nemotron-3-ultra": { input: 0.1, output: 3, cacheRead: 0.1, cacheWrite: 0 },
|
|
32
|
+
"qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0.6, cacheWrite: 0 },
|
|
32
33
|
};
|
package/usage.ts
CHANGED
|
@@ -38,12 +38,12 @@ export interface UsageActivity {
|
|
|
38
38
|
starting_at?: string;
|
|
39
39
|
ending_at?: string;
|
|
40
40
|
};
|
|
41
|
+
models?: UsageModel[];
|
|
41
42
|
}
|
|
42
43
|
|
|
43
44
|
export interface UsageData {
|
|
44
45
|
limits: {
|
|
45
|
-
|
|
46
|
-
weekly: UsageLimit;
|
|
46
|
+
monthly: UsageLimit;
|
|
47
47
|
};
|
|
48
48
|
activity?: UsageActivity;
|
|
49
49
|
}
|
|
@@ -71,13 +71,11 @@ export function isUsageLimit(data: unknown): data is UsageLimit {
|
|
|
71
71
|
);
|
|
72
72
|
}
|
|
73
73
|
|
|
74
|
-
/** Validate a parsed /api/usage response: must have
|
|
74
|
+
/** Validate a parsed /api/usage response: must have a monthly limit. */
|
|
75
75
|
export function isUsageResponse(data: unknown): data is UsageData {
|
|
76
76
|
if (data == null || typeof data !== "object") return false;
|
|
77
77
|
const d = data as UsageData;
|
|
78
|
-
return (
|
|
79
|
-
d.limits != null && typeof d.limits === "object" && isUsageLimit(d.limits.session) && isUsageLimit(d.limits.weekly)
|
|
80
|
-
);
|
|
78
|
+
return d.limits != null && typeof d.limits === "object" && isUsageLimit(d.limits.monthly);
|
|
81
79
|
}
|
|
82
80
|
|
|
83
81
|
// --- Fetch ---
|
|
@@ -127,15 +125,9 @@ function usagePercent(usage: number): number {
|
|
|
127
125
|
export function formatUsage(data: UsageData): string {
|
|
128
126
|
const lines: string[] = ["Ollama Cloud usage:"];
|
|
129
127
|
|
|
130
|
-
const
|
|
131
|
-
lines.push(`
|
|
132
|
-
for (const m of data.limits.
|
|
133
|
-
lines.push(` - ${m.name}: ${m.request_count} request${m.request_count === 1 ? "" : "s"}`);
|
|
134
|
-
}
|
|
135
|
-
|
|
136
|
-
const weeklyPct = usagePercent(data.limits.weekly.usage);
|
|
137
|
-
lines.push(` Weekly (7d): ${weeklyPct}%`);
|
|
138
|
-
for (const m of data.limits.weekly.models) {
|
|
128
|
+
const monthlyPct = usagePercent(data.limits.monthly.usage);
|
|
129
|
+
lines.push(` Monthly (30d): ${monthlyPct}%`);
|
|
130
|
+
for (const m of data.limits.monthly.models) {
|
|
139
131
|
lines.push(` - ${m.name}: ${m.request_count} request${m.request_count === 1 ? "" : "s"}`);
|
|
140
132
|
}
|
|
141
133
|
|
|
@@ -160,11 +152,10 @@ function colorSegment(theme: Theme, label: string, pct: number): string {
|
|
|
160
152
|
|
|
161
153
|
/**
|
|
162
154
|
* Compact one-line usage for the footer status bar, colored by usage level.
|
|
163
|
-
* The endpoint exposes reset
|
|
164
|
-
* color reflects the usage fraction rather than pace.
|
|
155
|
+
* The endpoint exposes a monthly reset period but not an exact timestamp, so
|
|
156
|
+
* the color reflects the usage fraction rather than pace.
|
|
165
157
|
*/
|
|
166
158
|
export function formatUsageStatusColored(theme: Theme, data: UsageData): string {
|
|
167
|
-
const
|
|
168
|
-
|
|
169
|
-
return `${colorSegment(theme, "5h", session)} ${colorSegment(theme, "7d", weekly)}`;
|
|
159
|
+
const monthly = usagePercent(data.limits.monthly.usage);
|
|
160
|
+
return colorSegment(theme, "30d", monthly);
|
|
170
161
|
}
|