pi-ollama-cloud 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/README.md +68 -6
- package/config.ts +4 -0
- package/index.ts +149 -8
- package/limits.generated.ts +25 -0
- package/models.generated.ts +114 -74
- package/models.ts +16 -6
- package/package.json +5 -2
- package/pricing.generated.ts +17 -16
- package/usage.ts +161 -0
- package/utils.ts +36 -0
- package/web-tools.ts +3 -55
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [Unreleased]
|
|
6
|
+
|
|
7
|
+
## [0.10.0] - 2026-09-03
|
|
8
|
+
|
|
9
|
+
- **Breaking:** Adapt to the changed `/api/usage` response shape. The endpoint now returns a single `limits.monthly` bucket (replacing `limits.session` and `limits.weekly`) and adds an `activity.models` array. `UsageData`, `isUsageResponse`, `formatUsage`, and `formatUsageStatusColored` now read the monthly limit; the status bar shows a single `30d` segment instead of `5h`/`7d`.
|
|
10
|
+
- Refreshed the generated catalog from the live API: added `deepseek-v4-pro:0813`, `glm-5.3`, and `glm-5.3-flash`; removed `deepseek-v4-flash:preview` and `deepseek-v4-pro`, which are no longer listed.
|
|
11
|
+
- Source per-token pricing from the official model table on ollama.com/pricing instead of models.dev estimates, which no longer track Ollama's published rates (up to ~13x off per model). `scripts/generate-pricing.ts` now scrapes the pricing page (the table is server-rendered; no JSON endpoint exists) and matches catalog IDs to pricing rows by exact or `:tag`-family match, replacing the `OLLAMA_TO_MODELSDEV` mapping. Regenerated `pricing.generated.ts` with the official rates, including new models (`glm-5.3`, `glm-5.3-flash`, `deepseek-v4-pro:0813`). Fixes #51. Thanks @Hackbard (#52).
|
|
12
|
+
- Probe per-model max output tokens against the live API via `scripts/generate-limits.ts` into `limits.generated.ts`, replacing the fixed 32768 default for known models (unprobed models still fall back to 32768). The probe timeout is 60s so slow first-token models are not dropped from the table. Thanks @f440 (#49).
|
|
13
|
+
|
|
14
|
+
## [0.9.0] - 2026-08-11
|
|
15
|
+
|
|
16
|
+
- Add `/ollama-cloud-usage` command to show Ollama Cloud session (5h) and weekly (7d) usage limits, per-model request counts, and the 4-week activity cost, fetched from the undocumented `/api/usage` endpoint with the already-resolved API key.
|
|
17
|
+
- Add a footer usage status bar (`5h ▕███░░░░░░░▏ 34% 7d ▕████░░░░░░▏ 45%`) while an `ollama-cloud` model is active, refreshing every 5 minutes and after each agent turn (throttled so it never exceeds one /api/usage call per 5 minutes). Segments are colored by usage level (green <60%, yellow 60-79%, red 80%+). The quota-bar concept is inspired by `@entelligentsia/pi-ollama-cloud-usage-tracker`. Off by default; enable with `/ollama-usage-status on` or `"usageStatus": true` in `ollama-cloud.json`.
|
|
18
|
+
- Add `/ollama-usage-status [on|off|enable|disable]` to toggle the footer usage status bar at runtime (toggles without an argument).
|
|
19
|
+
- Document the exported usage API (`fetchUsage`, `formatUsageStatusColored`, `getCloudApiKey`, validators) so custom status bars can reuse it.
|
|
20
|
+
|
|
5
21
|
## [0.8.0] - 2026-08-09
|
|
6
22
|
|
|
7
23
|
- **Breaking:** Migrate model refresh to pi's native `refreshModels` mechanism. Remove the `/ollama-cloud-refresh` command and the manual `~/.pi/agent/cache/ollama-cloud-models.json` cache. The catalog now refreshes automatically on startup, on `/model` open, and via `pi update --models`, persisted through pi's own `FileModelsStore`. Users should delete the orphaned cache file after upgrade: `rm ~/.pi/agent/cache/ollama-cloud-models.json`.
|
package/README.md
CHANGED
|
@@ -12,7 +12,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
|
|
|
12
12
|
- **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
|
|
13
13
|
- **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
|
|
14
14
|
- **`ollama_web_fetch` tool** - Fetch and extract text content from a web page URL using Ollama Cloud's `/api/web_fetch` endpoint. Returns page title, content, and links.
|
|
15
|
-
- **
|
|
15
|
+
- **Per-token cost tracking** - Models are registered with the official per-token prices from [ollama.com/pricing](https://ollama.com/pricing), so Pi's `/cost` shows comparable usage. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges.
|
|
16
16
|
|
|
17
17
|
## Prerequisites
|
|
18
18
|
|
|
@@ -97,12 +97,14 @@ Extension settings can be set via JSON config files. Project-local settings over
|
|
|
97
97
|
| Setting | Type | Default | Description |
|
|
98
98
|
|---|---|---|---|
|
|
99
99
|
| `webTools` | boolean | `true` | Set to `false` to prevent `ollama_web_search` and `ollama_web_fetch` from being registered |
|
|
100
|
+
| `usageStatus` | boolean | `false` | Set to `true` to show the footer usage status bar (opt-in; enable at runtime with `/ollama-usage-status`) |
|
|
100
101
|
|
|
101
102
|
Example `ollama-cloud.json`:
|
|
102
103
|
|
|
103
104
|
```json
|
|
104
105
|
{
|
|
105
|
-
"webTools": false
|
|
106
|
+
"webTools": false,
|
|
107
|
+
"usageStatus": true
|
|
106
108
|
}
|
|
107
109
|
```
|
|
108
110
|
|
|
@@ -133,8 +135,14 @@ Model metadata is derived from the `/api/show` response:
|
|
|
133
135
|
| `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 5 maps (DEFAULT, GPT_OSS, QWEN3, GLM_52, NO_OFF) based on API testing |
|
|
134
136
|
| `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
|
|
135
137
|
| `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
|
|
136
|
-
| `maxTokens` |
|
|
137
|
-
| `cost` |
|
|
138
|
+
| `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
|
|
139
|
+
| `cost` | Official per-1M-token prices from the [ollama.com/pricing](https://ollama.com/pricing) model table, generated by `scripts/generate-pricing.ts` into `pricing.generated.ts`. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges. Catalog IDs with no matching pricing row default to zero. Prices are pinned to the installed package version and only update on a new release, so newly added models register with zero cost until then. |
|
|
140
|
+
|
|
141
|
+
The per-model max output token table (`limits.generated.ts`) is probed against the live API by `scripts/generate-limits.ts`, since `/api/show` does not expose the limit. It needs an API key: `OLLAMA_API_KEY=<key> npm run generate-limits`. Limits ship with the package, so regenerated values take effect on the next release.
|
|
142
|
+
|
|
143
|
+
The API itself returns no cost data: completion responses report only token counts (`prompt_tokens`/`completion_tokens`/`total_tokens`, including the final usage chunk when streaming), and `/api/show` exposes no pricing fields. The prices above come from the static `/pricing` page table and are only as fresh as the last regeneration.
|
|
144
|
+
|
|
145
|
+
Cache pricing is informational only: the `/pricing` page lists a "Cached input" column, but the completion API does not report cache token usage (there is no `prompt_tokens_details.cached_tokens` or equivalent in any response, verified against the live API in September 2026), so pi never sees cache hits and `/cost` estimates do not reflect them. `cacheWrite` is always zero because the pricing table has no cache-write column.
|
|
138
146
|
|
|
139
147
|
### Thinking level mapping
|
|
140
148
|
|
|
@@ -164,6 +172,58 @@ Both tools use the same Ollama Cloud API key configured for the provider. No loc
|
|
|
164
172
|
| Command | Description |
|
|
165
173
|
|---|---|
|
|
166
174
|
| `/ollama-webtools [on\|off\|enable\|disable]` | Enable or disable the `ollama_web_search` and `ollama_web_fetch` tools. Toggles if no argument given. |
|
|
175
|
+
| `/ollama-cloud-usage` | Show Ollama Cloud monthly usage limits, per-model request counts, and the 4-week activity cost. |
|
|
176
|
+
| `/ollama-usage-status [on\|off\|enable\|disable]` | Enable or disable the footer usage status bar. Toggles if no argument given. |
|
|
177
|
+
|
|
178
|
+
## Usage status bar
|
|
179
|
+
|
|
180
|
+
While an `ollama-cloud` model is the active provider, the footer shows a compact
|
|
181
|
+
live usage readout (`30d ▕███░░░░░░░▏ 34%`) that refreshes
|
|
182
|
+
every 5 minutes and after each agent turn (but no more often than every 5 minutes). It is colored by how close
|
|
183
|
+
it is to the cap: green below 60%, yellow at 60-79%, red at 80%+. It reads the
|
|
184
|
+
same undocumented `/api/usage` endpoint as `/ollama-cloud-usage` and clears
|
|
185
|
+
itself on transient errors or when you switch to a non-Ollama-Cloud provider.
|
|
186
|
+
|
|
187
|
+
It is off by default. Enable it at runtime with `/ollama-usage-status on`, or
|
|
188
|
+
enable it by default with `"usageStatus": true` in `ollama-cloud.json`. If the
|
|
189
|
+
bar never appears after enabling, run `/ollama-cloud-usage` to see the
|
|
190
|
+
underlying error (e.g. a misconfigured API key).
|
|
191
|
+
|
|
192
|
+
The quota-bar concept is inspired by
|
|
193
|
+
[`@entelligentsia/pi-ollama-cloud-usage-tracker`](https://github.com/Entelligentsia/pi-ollama-cloud-usage-tracker),
|
|
194
|
+
but this extension fetches usage from the `/api/usage` endpoint with the API key
|
|
195
|
+
it already resolves, rather than scraping the settings page with Chrome cookies.
|
|
196
|
+
|
|
197
|
+
## Usage API for custom status bars
|
|
198
|
+
|
|
199
|
+
The usage data plane is exported so you can plug it into your own footer or
|
|
200
|
+
status bar instead of (or alongside) the built-in one. The relevant modules ship
|
|
201
|
+
with the package and are importable directly:
|
|
202
|
+
|
|
203
|
+
```ts
|
|
204
|
+
import { fetchUsage, formatUsage, formatUsageStatusColored } from "pi-ollama-cloud/usage.ts";
|
|
205
|
+
import { getCloudApiKey } from "pi-ollama-cloud/utils.ts";
|
|
206
|
+
import type { UsageData } from "pi-ollama-cloud/usage.ts";
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
| Export | Description |
|
|
210
|
+
|---|---|
|
|
211
|
+
| `fetchUsage(apiKey, signal?)` | Fetch the raw `/api/usage` data, returning a typed `UsageData`. Throws a status-mapped error on 401/403/429/404/5xx. |
|
|
212
|
+
| `formatUsageStatusColored(theme, data)` | One-line status string with quota bars, colored by usage level. Takes a `Theme` (e.g. `ctx.ui.theme`). |
|
|
213
|
+
| `formatUsage(data)` | Multi-line human-readable output (percentages, per-model request counts, activity cost). |
|
|
214
|
+
| `getCloudApiKey(ctx)` | Resolve the Ollama Cloud API key the same way the extension does. |
|
|
215
|
+
| `isUsageResponse(data)` / `isUsageLimit(data)` | Validators for parsing the raw response yourself. |
|
|
216
|
+
|
|
217
|
+
Example custom status bar:
|
|
218
|
+
|
|
219
|
+
```ts
|
|
220
|
+
const apiKey = await getCloudApiKey(ctx);
|
|
221
|
+
const data = await fetchUsage(apiKey);
|
|
222
|
+
ctx.ui.setStatus("my-usage", formatUsageStatusColored(ctx.ui.theme, data));
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Note that the package ships raw TypeScript sources (no build step), so submodule
|
|
226
|
+
imports use the `.ts` extension, matching how the extension imports internally.
|
|
167
227
|
|
|
168
228
|
## Development
|
|
169
229
|
|
|
@@ -173,6 +233,7 @@ npm run check # lint + format + type-check (auto-fix)
|
|
|
173
233
|
npm run lint # lint only (no fixes)
|
|
174
234
|
npm run typecheck # type-check only (tsgo --noEmit)
|
|
175
235
|
npm run format # format only
|
|
236
|
+
OLLAMA_API_KEY=<key> npm run generate-limits # probe max output tokens (writes limits.generated.ts)
|
|
176
237
|
```
|
|
177
238
|
|
|
178
239
|
The project uses [Biome](https://biomejs.dev/) for linting and formatting (2-space indent, line width 120) and [tsgo](https://github.com/microsoft/typescript-go) for type-checking.
|
|
@@ -192,7 +253,7 @@ Live smoke against the real API (needs an `OLLAMA_API_KEY` or an `ollama-cloud`
|
|
|
192
253
|
```bash
|
|
193
254
|
# Run pi with the local extension, no install required. The --no-* flags isolate
|
|
194
255
|
# the run from other installed extensions, skills, prompt templates, themes,
|
|
195
|
-
# context files, and session storage so only the local checkout is exercised
|
|
256
|
+
# context files, and session storage so only the local checkout is exercised
|
|
196
257
|
pi --no-extensions --no-skills --no-prompt-templates --no-themes --no-context-files --no-session \
|
|
197
258
|
-e ./index.ts --model "ollama-cloud/gemma4:31b" --no-tools -p "Say hi in one word"
|
|
198
259
|
|
|
@@ -241,7 +302,8 @@ git push --tags
|
|
|
241
302
|
Because the model catalog refreshes automatically at runtime, a release is **not** needed to ship new models. Publish only when:
|
|
242
303
|
|
|
243
304
|
- A model is retired and still listed by the API: add it to `RETIRED_MODEL_IDS` in `scripts/generate-models.ts` (check https://docs.ollama.com/cloud#retirements, then regenerate `models.generated.ts`).
|
|
244
|
-
- Pricing changes:
|
|
305
|
+
- Pricing changes: Ollama updates the model pricing table, or a new model needs a pricing row (regenerate `pricing.generated.ts`).
|
|
306
|
+
- Max output token limits changed: run `OLLAMA_API_KEY=<key> npm run generate-limits` locally and commit.
|
|
245
307
|
|
|
246
308
|
The tag version must match the version in `package.json` - `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
|
|
247
309
|
|
package/config.ts
CHANGED
|
@@ -25,12 +25,15 @@ import { getAgentDir } from "@earendil-works/pi-coding-agent";
|
|
|
25
25
|
export interface OllamaCloudConfig {
|
|
26
26
|
/** When false, ollama_web_search and ollama_web_fetch tools are not registered. Default: true. */
|
|
27
27
|
webTools?: boolean;
|
|
28
|
+
/** When true, the footer usage status bar is shown. Default: false (opt-in; enable with /ollama-usage-status). */
|
|
29
|
+
usageStatus?: boolean;
|
|
28
30
|
}
|
|
29
31
|
|
|
30
32
|
// --- Defaults ---
|
|
31
33
|
|
|
32
34
|
const DEFAULT_CONFIG: OllamaCloudConfig = {
|
|
33
35
|
webTools: true,
|
|
36
|
+
usageStatus: false,
|
|
34
37
|
};
|
|
35
38
|
|
|
36
39
|
// --- Validation ---
|
|
@@ -38,6 +41,7 @@ const DEFAULT_CONFIG: OllamaCloudConfig = {
|
|
|
38
41
|
/** Allowed config keys and their expected types for runtime validation. */
|
|
39
42
|
const CONFIG_SCHEMA: Record<keyof OllamaCloudConfig, "boolean"> = {
|
|
40
43
|
webTools: "boolean",
|
|
44
|
+
usageStatus: "boolean",
|
|
41
45
|
};
|
|
42
46
|
|
|
43
47
|
/**
|
package/index.ts
CHANGED
|
@@ -24,12 +24,29 @@
|
|
|
24
24
|
* Only models with "tools" capability are registered.
|
|
25
25
|
*/
|
|
26
26
|
|
|
27
|
-
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
27
|
+
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
28
28
|
import { loadConfig, resolveWebToolsEnv } from "./config.ts";
|
|
29
29
|
import { GENERATED_MODELS } from "./models.generated.ts";
|
|
30
30
|
import { OLLAMA_BASE, refreshOllamaCatalog } from "./models.ts";
|
|
31
|
+
import { fetchUsage, formatUsage, formatUsageStatusColored } from "./usage.ts";
|
|
32
|
+
import { getCloudApiKey } from "./utils.ts";
|
|
31
33
|
import { registerWebFetchTool, registerWebSearchTool } from "./web-tools.ts";
|
|
32
34
|
|
|
35
|
+
/**
|
|
36
|
+
* Resolve the new enabled state for /ollama-usage-status from its argument.
|
|
37
|
+
* Exported for unit testing.
|
|
38
|
+
*/
|
|
39
|
+
export function resolveUsageStatusToggle(arg: string, current: boolean): { enabled: boolean; error?: string } {
|
|
40
|
+
const a = arg.trim().toLowerCase();
|
|
41
|
+
if (a === "on" || a === "enable") return { enabled: true };
|
|
42
|
+
if (a === "off" || a === "disable") return { enabled: false };
|
|
43
|
+
if (a === "") return { enabled: !current };
|
|
44
|
+
return {
|
|
45
|
+
enabled: current,
|
|
46
|
+
error: `Unknown argument "${arg.trim()}". Usage: /ollama-usage-status [on|off|enable|disable]`,
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
|
|
33
50
|
// --- Main ---
|
|
34
51
|
|
|
35
52
|
export default async function (pi: ExtensionAPI) {
|
|
@@ -82,21 +99,25 @@ export default async function (pi: ExtensionAPI) {
|
|
|
82
99
|
}
|
|
83
100
|
}
|
|
84
101
|
|
|
85
|
-
//
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
//
|
|
89
|
-
|
|
102
|
+
// Config is read once per extension factory invocation (on the first
|
|
103
|
+
// session_start). The factory is re-invoked on /new, /fork, /resume, and
|
|
104
|
+
// /reload, so runtime toggles (e.g. /ollama-webtools, /ollama-usage-status)
|
|
105
|
+
// reset to the config default on each session restart. Restart pi or /reload
|
|
106
|
+
// to pick up config file changes.
|
|
107
|
+
let configLoaded = false;
|
|
90
108
|
let webToolsEnabled = false;
|
|
109
|
+
let usageStatusEnabled = false;
|
|
91
110
|
|
|
92
111
|
pi.on("session_start", async (_event, ctx) => {
|
|
93
|
-
if (!
|
|
94
|
-
|
|
112
|
+
if (!configLoaded) {
|
|
113
|
+
configLoaded = true;
|
|
95
114
|
const config = loadConfig(ctx.cwd);
|
|
96
115
|
if (config.webTools !== false) {
|
|
97
116
|
webToolsEnabled = true;
|
|
98
117
|
ensureWebToolsRegistered();
|
|
99
118
|
}
|
|
119
|
+
// The status bar is opt-in: enabled only when the config explicitly sets it true.
|
|
120
|
+
usageStatusEnabled = config.usageStatus === true;
|
|
100
121
|
}
|
|
101
122
|
// On every session start (including resume/fork/new), re-apply the
|
|
102
123
|
// runtime state. Tools may have been unregistered during teardown.
|
|
@@ -104,6 +125,126 @@ export default async function (pi: ExtensionAPI) {
|
|
|
104
125
|
ensureWebToolsRegistered();
|
|
105
126
|
setWebToolsActive(true);
|
|
106
127
|
}
|
|
128
|
+
// Start the usage status bar when ollama-cloud is the active provider.
|
|
129
|
+
if (usageStatusEnabled && isOllamaCloud(ctx)) {
|
|
130
|
+
startUsageStatus(ctx);
|
|
131
|
+
}
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
// --- Usage Command ---
|
|
135
|
+
|
|
136
|
+
pi.registerCommand("ollama-cloud-usage", {
|
|
137
|
+
description: "Show Ollama Cloud monthly usage limits.",
|
|
138
|
+
handler: async (_args, ctx) => {
|
|
139
|
+
const apiKey = await getCloudApiKey(ctx);
|
|
140
|
+
if (!apiKey) {
|
|
141
|
+
ctx.ui.notify("No Ollama Cloud API key configured. Set OLLAMA_API_KEY or add to auth.json.", "error");
|
|
142
|
+
return;
|
|
143
|
+
}
|
|
144
|
+
try {
|
|
145
|
+
const data = await fetchUsage(apiKey);
|
|
146
|
+
ctx.ui.notify(formatUsage(data), "info");
|
|
147
|
+
} catch (err) {
|
|
148
|
+
ctx.ui.notify(err instanceof Error ? err.message : String(err), "error");
|
|
149
|
+
}
|
|
150
|
+
},
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
// --- Usage Status Bar ---
|
|
154
|
+
|
|
155
|
+
// Footer status showing live monthly usage while ollama-cloud is the
|
|
156
|
+
// active provider. Refreshes on a 5-minute timer; agent_end also triggers a
|
|
157
|
+
// refresh but is throttled to the same cooldown so a turn never hammers the
|
|
158
|
+
// undocumented /api/usage endpoint. The quota-bar concept is inspired by
|
|
159
|
+
// @entelligentsia/pi-ollama-cloud-usage-tracker.
|
|
160
|
+
const USAGE_STATUS_KEY = "ollama-usage";
|
|
161
|
+
const USAGE_REFRESH_MS = 5 * 60_000;
|
|
162
|
+
let usageTimer: ReturnType<typeof setInterval> | null = null;
|
|
163
|
+
let usageActive = false;
|
|
164
|
+
// Timestamp (ms) of the most recent refresh attempt; gates the agent_end
|
|
165
|
+
// refresh so it fires at most once per cooldown. Set when a fetch starts, so
|
|
166
|
+
// a failing endpoint is also throttled, not just a successful one.
|
|
167
|
+
let lastRefreshAt = 0;
|
|
168
|
+
|
|
169
|
+
async function refreshUsageStatus(ctx: ExtensionContext) {
|
|
170
|
+
try {
|
|
171
|
+
const apiKey = await getCloudApiKey(ctx);
|
|
172
|
+
if (!apiKey) {
|
|
173
|
+
ctx.ui.setStatus(USAGE_STATUS_KEY, undefined);
|
|
174
|
+
return;
|
|
175
|
+
}
|
|
176
|
+
lastRefreshAt = Date.now();
|
|
177
|
+
const data = await fetchUsage(apiKey);
|
|
178
|
+
ctx.ui.setStatus(USAGE_STATUS_KEY, formatUsageStatusColored(ctx.ui.theme, data));
|
|
179
|
+
} catch {
|
|
180
|
+
// Transient errors (undocumented endpoint, network) should not spam the
|
|
181
|
+
// footer; clear the status and retry on the next refresh.
|
|
182
|
+
ctx.ui.setStatus(USAGE_STATUS_KEY, undefined);
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
function startUsageStatus(ctx: ExtensionContext) {
|
|
187
|
+
if (usageActive) return;
|
|
188
|
+
// The status bar is TUI-only; skip the fetch and timer in print/json/rpc.
|
|
189
|
+
if (ctx.mode !== "tui") return;
|
|
190
|
+
usageActive = true;
|
|
191
|
+
refreshUsageStatus(ctx);
|
|
192
|
+
usageTimer = setInterval(() => refreshUsageStatus(ctx), USAGE_REFRESH_MS);
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function stopUsageStatus(ctx: ExtensionContext) {
|
|
196
|
+
usageActive = false;
|
|
197
|
+
if (usageTimer) {
|
|
198
|
+
clearInterval(usageTimer);
|
|
199
|
+
usageTimer = null;
|
|
200
|
+
}
|
|
201
|
+
ctx.ui.setStatus(USAGE_STATUS_KEY, undefined);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function isOllamaCloud(ctx: ExtensionContext): boolean {
|
|
205
|
+
return ctx.model?.provider === "ollama-cloud";
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
pi.on("model_select", async (_event, ctx) => {
|
|
209
|
+
if (usageStatusEnabled && isOllamaCloud(ctx)) {
|
|
210
|
+
startUsageStatus(ctx);
|
|
211
|
+
} else {
|
|
212
|
+
stopUsageStatus(ctx);
|
|
213
|
+
}
|
|
214
|
+
});
|
|
215
|
+
|
|
216
|
+
pi.on("agent_end", async (_event, ctx) => {
|
|
217
|
+
// Throttle the after-turn refresh to the same cooldown as the timer so a
|
|
218
|
+
// burst of turns never exceeds one /api/usage call per 5 minutes.
|
|
219
|
+
if (usageActive && isOllamaCloud(ctx) && Date.now() - lastRefreshAt >= USAGE_REFRESH_MS) {
|
|
220
|
+
await refreshUsageStatus(ctx);
|
|
221
|
+
}
|
|
222
|
+
});
|
|
223
|
+
|
|
224
|
+
pi.on("session_shutdown", async (_event, ctx) => {
|
|
225
|
+
stopUsageStatus(ctx);
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
pi.registerCommand("ollama-usage-status", {
|
|
229
|
+
description:
|
|
230
|
+
"Enable or disable the Ollama Cloud usage status bar. " +
|
|
231
|
+
"Accepts optional argument: on/off/enable/disable. Without argument, toggles.",
|
|
232
|
+
handler: async (args, ctx) => {
|
|
233
|
+
const { enabled, error } = resolveUsageStatusToggle(args, usageStatusEnabled);
|
|
234
|
+
if (error) {
|
|
235
|
+
ctx.ui.notify(error, "error");
|
|
236
|
+
return;
|
|
237
|
+
}
|
|
238
|
+
usageStatusEnabled = enabled;
|
|
239
|
+
|
|
240
|
+
if (usageStatusEnabled && isOllamaCloud(ctx)) {
|
|
241
|
+
startUsageStatus(ctx);
|
|
242
|
+
} else {
|
|
243
|
+
stopUsageStatus(ctx);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
ctx.ui.notify(`Ollama Cloud usage status: ${usageStatusEnabled ? "enabled" : "disabled"}`, "info");
|
|
247
|
+
},
|
|
107
248
|
});
|
|
108
249
|
|
|
109
250
|
// Only register the runtime toggle command when the env var doesn't force tools off.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
// Auto-generated by scripts/generate-limits.ts
|
|
2
|
+
// Do not edit manually.
|
|
3
|
+
// Probed models: 19 (0 failed)
|
|
4
|
+
|
|
5
|
+
export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
6
|
+
"deepseek-v4-flash:0731": 65536,
|
|
7
|
+
"deepseek-v4-pro:0813": 65536,
|
|
8
|
+
"gemma4:31b": 262144,
|
|
9
|
+
"glm-5.1": 131072,
|
|
10
|
+
"glm-5.2": 131072,
|
|
11
|
+
"glm-5.3": 524288,
|
|
12
|
+
"glm-5.3-flash": 524288,
|
|
13
|
+
"gpt-oss:120b": 131072,
|
|
14
|
+
"gpt-oss:20b": 131072,
|
|
15
|
+
"kimi-k2.6": 262144,
|
|
16
|
+
"kimi-k2.7-code": 262144,
|
|
17
|
+
"kimi-k3": 524288,
|
|
18
|
+
"minimax-m2.7": 131072,
|
|
19
|
+
"minimax-m3": 131072,
|
|
20
|
+
"mistral-large-3:675b": 262144,
|
|
21
|
+
"nemotron-3-nano:30b": 131072,
|
|
22
|
+
"nemotron-3-super": 65536,
|
|
23
|
+
"nemotron-3-ultra": 65536,
|
|
24
|
+
"qwen3.5:397b": 65536,
|
|
25
|
+
};
|
package/models.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-models.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-03T10:12:02.243Z
|
|
4
|
+
// Model count: 19
|
|
5
5
|
|
|
6
6
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
7
7
|
|
|
@@ -29,13 +29,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
29
29
|
},
|
|
30
30
|
contextWindow: 1048576,
|
|
31
31
|
cost: {
|
|
32
|
-
cacheRead: 0.
|
|
32
|
+
cacheRead: 0.014,
|
|
33
33
|
cacheWrite: 0,
|
|
34
|
-
input: 0.
|
|
35
|
-
output:
|
|
34
|
+
input: 0.44,
|
|
35
|
+
output: 1.32,
|
|
36
36
|
},
|
|
37
37
|
input: ["text"],
|
|
38
|
-
maxTokens:
|
|
38
|
+
maxTokens: 65536,
|
|
39
39
|
reasoning: true,
|
|
40
40
|
thinkingLevelMap: {
|
|
41
41
|
high: "high",
|
|
@@ -47,8 +47,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
47
47
|
},
|
|
48
48
|
},
|
|
49
49
|
{
|
|
50
|
-
id: "deepseek-v4-
|
|
51
|
-
name: "deepseek-v4-
|
|
50
|
+
id: "deepseek-v4-pro:0813",
|
|
51
|
+
name: "deepseek-v4-pro:0813",
|
|
52
52
|
compat: {
|
|
53
53
|
maxTokensField: "max_tokens",
|
|
54
54
|
openRouterRouting: {},
|
|
@@ -69,13 +69,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
69
69
|
},
|
|
70
70
|
contextWindow: 1048576,
|
|
71
71
|
cost: {
|
|
72
|
-
cacheRead: 0.
|
|
72
|
+
cacheRead: 0.044,
|
|
73
73
|
cacheWrite: 0,
|
|
74
|
-
input:
|
|
75
|
-
output:
|
|
74
|
+
input: 1.32,
|
|
75
|
+
output: 3.96,
|
|
76
76
|
},
|
|
77
77
|
input: ["text"],
|
|
78
|
-
maxTokens:
|
|
78
|
+
maxTokens: 65536,
|
|
79
79
|
reasoning: true,
|
|
80
80
|
thinkingLevelMap: {
|
|
81
81
|
high: "high",
|
|
@@ -87,8 +87,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
87
87
|
},
|
|
88
88
|
},
|
|
89
89
|
{
|
|
90
|
-
id: "
|
|
91
|
-
name: "
|
|
90
|
+
id: "gemma4:31b",
|
|
91
|
+
name: "gemma4:31b",
|
|
92
92
|
compat: {
|
|
93
93
|
maxTokensField: "max_tokens",
|
|
94
94
|
openRouterRouting: {},
|
|
@@ -107,15 +107,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
107
107
|
vercelGatewayRouting: {},
|
|
108
108
|
zaiToolStream: false,
|
|
109
109
|
},
|
|
110
|
-
contextWindow:
|
|
110
|
+
contextWindow: 262144,
|
|
111
111
|
cost: {
|
|
112
|
-
cacheRead: 0.
|
|
112
|
+
cacheRead: 0.05,
|
|
113
113
|
cacheWrite: 0,
|
|
114
|
-
input: 0.
|
|
115
|
-
output: 0.
|
|
114
|
+
input: 0.14,
|
|
115
|
+
output: 0.4,
|
|
116
116
|
},
|
|
117
|
-
input: ["text"],
|
|
118
|
-
maxTokens:
|
|
117
|
+
input: ["text", "image"],
|
|
118
|
+
maxTokens: 262144,
|
|
119
119
|
reasoning: true,
|
|
120
120
|
thinkingLevelMap: {
|
|
121
121
|
high: "high",
|
|
@@ -127,8 +127,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
127
127
|
},
|
|
128
128
|
},
|
|
129
129
|
{
|
|
130
|
-
id: "
|
|
131
|
-
name: "
|
|
130
|
+
id: "glm-5.1",
|
|
131
|
+
name: "glm-5.1",
|
|
132
132
|
compat: {
|
|
133
133
|
maxTokensField: "max_tokens",
|
|
134
134
|
openRouterRouting: {},
|
|
@@ -147,15 +147,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
147
147
|
vercelGatewayRouting: {},
|
|
148
148
|
zaiToolStream: false,
|
|
149
149
|
},
|
|
150
|
-
contextWindow:
|
|
150
|
+
contextWindow: 202752,
|
|
151
151
|
cost: {
|
|
152
|
-
cacheRead: 0.
|
|
152
|
+
cacheRead: 0.2,
|
|
153
153
|
cacheWrite: 0,
|
|
154
|
-
input:
|
|
155
|
-
output:
|
|
154
|
+
input: 1,
|
|
155
|
+
output: 3.2,
|
|
156
156
|
},
|
|
157
|
-
input: ["text"
|
|
158
|
-
maxTokens:
|
|
157
|
+
input: ["text"],
|
|
158
|
+
maxTokens: 131072,
|
|
159
159
|
reasoning: true,
|
|
160
160
|
thinkingLevelMap: {
|
|
161
161
|
high: "high",
|
|
@@ -167,8 +167,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
167
167
|
},
|
|
168
168
|
},
|
|
169
169
|
{
|
|
170
|
-
id: "glm-5.
|
|
171
|
-
name: "glm-5.
|
|
170
|
+
id: "glm-5.2",
|
|
171
|
+
name: "glm-5.2",
|
|
172
172
|
compat: {
|
|
173
173
|
maxTokensField: "max_tokens",
|
|
174
174
|
openRouterRouting: {},
|
|
@@ -187,7 +187,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
187
187
|
vercelGatewayRouting: {},
|
|
188
188
|
zaiToolStream: false,
|
|
189
189
|
},
|
|
190
|
-
contextWindow:
|
|
190
|
+
contextWindow: 1048576,
|
|
191
191
|
cost: {
|
|
192
192
|
cacheRead: 0.26,
|
|
193
193
|
cacheWrite: 0,
|
|
@@ -195,20 +195,20 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
195
195
|
output: 4.4,
|
|
196
196
|
},
|
|
197
197
|
input: ["text"],
|
|
198
|
-
maxTokens:
|
|
198
|
+
maxTokens: 131072,
|
|
199
199
|
reasoning: true,
|
|
200
200
|
thinkingLevelMap: {
|
|
201
201
|
high: "high",
|
|
202
|
-
low:
|
|
203
|
-
medium:
|
|
202
|
+
low: null,
|
|
203
|
+
medium: null,
|
|
204
204
|
minimal: null,
|
|
205
205
|
off: "none",
|
|
206
206
|
xhigh: "max",
|
|
207
207
|
},
|
|
208
208
|
},
|
|
209
209
|
{
|
|
210
|
-
id: "glm-5.
|
|
211
|
-
name: "glm-5.
|
|
210
|
+
id: "glm-5.3",
|
|
211
|
+
name: "glm-5.3",
|
|
212
212
|
compat: {
|
|
213
213
|
maxTokensField: "max_tokens",
|
|
214
214
|
openRouterRouting: {},
|
|
@@ -227,7 +227,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
227
227
|
vercelGatewayRouting: {},
|
|
228
228
|
zaiToolStream: false,
|
|
229
229
|
},
|
|
230
|
-
contextWindow:
|
|
230
|
+
contextWindow: 1048576,
|
|
231
231
|
cost: {
|
|
232
232
|
cacheRead: 0.26,
|
|
233
233
|
cacheWrite: 0,
|
|
@@ -235,12 +235,52 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
235
235
|
output: 4.4,
|
|
236
236
|
},
|
|
237
237
|
input: ["text"],
|
|
238
|
-
maxTokens:
|
|
238
|
+
maxTokens: 524288,
|
|
239
239
|
reasoning: true,
|
|
240
240
|
thinkingLevelMap: {
|
|
241
241
|
high: "high",
|
|
242
|
-
low:
|
|
243
|
-
medium:
|
|
242
|
+
low: "low",
|
|
243
|
+
medium: "medium",
|
|
244
|
+
minimal: null,
|
|
245
|
+
off: "none",
|
|
246
|
+
xhigh: "max",
|
|
247
|
+
},
|
|
248
|
+
},
|
|
249
|
+
{
|
|
250
|
+
id: "glm-5.3-flash",
|
|
251
|
+
name: "glm-5.3-flash",
|
|
252
|
+
compat: {
|
|
253
|
+
maxTokensField: "max_tokens",
|
|
254
|
+
openRouterRouting: {},
|
|
255
|
+
requiresAssistantAfterToolResult: false,
|
|
256
|
+
requiresReasoningContentOnAssistantMessages: false,
|
|
257
|
+
requiresThinkingAsText: false,
|
|
258
|
+
requiresToolResultName: false,
|
|
259
|
+
sendSessionAffinityHeaders: false,
|
|
260
|
+
supportsDeveloperRole: false,
|
|
261
|
+
supportsLongCacheRetention: false,
|
|
262
|
+
supportsReasoningEffort: true,
|
|
263
|
+
supportsStore: false,
|
|
264
|
+
supportsStrictMode: false,
|
|
265
|
+
supportsUsageInStreaming: true,
|
|
266
|
+
thinkingFormat: "openai",
|
|
267
|
+
vercelGatewayRouting: {},
|
|
268
|
+
zaiToolStream: false,
|
|
269
|
+
},
|
|
270
|
+
contextWindow: 1048576,
|
|
271
|
+
cost: {
|
|
272
|
+
cacheRead: 0.03,
|
|
273
|
+
cacheWrite: 0,
|
|
274
|
+
input: 0.15,
|
|
275
|
+
output: 0.5,
|
|
276
|
+
},
|
|
277
|
+
input: ["text", "image"],
|
|
278
|
+
maxTokens: 524288,
|
|
279
|
+
reasoning: true,
|
|
280
|
+
thinkingLevelMap: {
|
|
281
|
+
high: "high",
|
|
282
|
+
low: "low",
|
|
283
|
+
medium: "medium",
|
|
244
284
|
minimal: null,
|
|
245
285
|
off: "none",
|
|
246
286
|
xhigh: "max",
|
|
@@ -269,13 +309,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
269
309
|
},
|
|
270
310
|
contextWindow: 131072,
|
|
271
311
|
cost: {
|
|
272
|
-
cacheRead: 0,
|
|
312
|
+
cacheRead: 0.014,
|
|
273
313
|
cacheWrite: 0,
|
|
274
|
-
input: 0.
|
|
275
|
-
output: 0.
|
|
314
|
+
input: 0.15,
|
|
315
|
+
output: 0.6,
|
|
276
316
|
},
|
|
277
317
|
input: ["text"],
|
|
278
|
-
maxTokens:
|
|
318
|
+
maxTokens: 131072,
|
|
279
319
|
reasoning: true,
|
|
280
320
|
thinkingLevelMap: {
|
|
281
321
|
high: "high",
|
|
@@ -309,13 +349,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
309
349
|
},
|
|
310
350
|
contextWindow: 131072,
|
|
311
351
|
cost: {
|
|
312
|
-
cacheRead: 0.
|
|
352
|
+
cacheRead: 0.035,
|
|
313
353
|
cacheWrite: 0,
|
|
314
|
-
input: 0.
|
|
315
|
-
output: 0.
|
|
354
|
+
input: 0.07,
|
|
355
|
+
output: 0.3,
|
|
316
356
|
},
|
|
317
357
|
input: ["text"],
|
|
318
|
-
maxTokens:
|
|
358
|
+
maxTokens: 131072,
|
|
319
359
|
reasoning: true,
|
|
320
360
|
thinkingLevelMap: {
|
|
321
361
|
high: "high",
|
|
@@ -355,7 +395,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
355
395
|
output: 4,
|
|
356
396
|
},
|
|
357
397
|
input: ["text", "image"],
|
|
358
|
-
maxTokens:
|
|
398
|
+
maxTokens: 262144,
|
|
359
399
|
reasoning: true,
|
|
360
400
|
thinkingLevelMap: {
|
|
361
401
|
high: "high",
|
|
@@ -395,7 +435,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
395
435
|
output: 4,
|
|
396
436
|
},
|
|
397
437
|
input: ["text", "image"],
|
|
398
|
-
maxTokens:
|
|
438
|
+
maxTokens: 262144,
|
|
399
439
|
reasoning: true,
|
|
400
440
|
thinkingLevelMap: {
|
|
401
441
|
high: "high",
|
|
@@ -435,7 +475,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
435
475
|
output: 15,
|
|
436
476
|
},
|
|
437
477
|
input: ["text", "image"],
|
|
438
|
-
maxTokens:
|
|
478
|
+
maxTokens: 524288,
|
|
439
479
|
reasoning: true,
|
|
440
480
|
thinkingLevelMap: {
|
|
441
481
|
high: "high",
|
|
@@ -470,12 +510,12 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
470
510
|
contextWindow: 196608,
|
|
471
511
|
cost: {
|
|
472
512
|
cacheRead: 0.06,
|
|
473
|
-
cacheWrite: 0
|
|
513
|
+
cacheWrite: 0,
|
|
474
514
|
input: 0.3,
|
|
475
515
|
output: 1.2,
|
|
476
516
|
},
|
|
477
517
|
input: ["text"],
|
|
478
|
-
maxTokens:
|
|
518
|
+
maxTokens: 131072,
|
|
479
519
|
reasoning: true,
|
|
480
520
|
thinkingLevelMap: {
|
|
481
521
|
high: "high",
|
|
@@ -507,15 +547,15 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
507
547
|
vercelGatewayRouting: {},
|
|
508
548
|
zaiToolStream: false,
|
|
509
549
|
},
|
|
510
|
-
contextWindow:
|
|
550
|
+
contextWindow: 512000,
|
|
511
551
|
cost: {
|
|
512
|
-
cacheRead: 0.
|
|
552
|
+
cacheRead: 0.12,
|
|
513
553
|
cacheWrite: 0,
|
|
514
|
-
input: 0.
|
|
515
|
-
output:
|
|
554
|
+
input: 0.6,
|
|
555
|
+
output: 2.4,
|
|
516
556
|
},
|
|
517
557
|
input: ["text", "image"],
|
|
518
|
-
maxTokens:
|
|
558
|
+
maxTokens: 131072,
|
|
519
559
|
reasoning: true,
|
|
520
560
|
thinkingLevelMap: {
|
|
521
561
|
high: "high",
|
|
@@ -549,13 +589,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
549
589
|
},
|
|
550
590
|
contextWindow: 262144,
|
|
551
591
|
cost: {
|
|
552
|
-
cacheRead: 0,
|
|
592
|
+
cacheRead: 0.5,
|
|
553
593
|
cacheWrite: 0,
|
|
554
594
|
input: 0.5,
|
|
555
595
|
output: 1.5,
|
|
556
596
|
},
|
|
557
597
|
input: ["text", "image"],
|
|
558
|
-
maxTokens:
|
|
598
|
+
maxTokens: 262144,
|
|
559
599
|
reasoning: false,
|
|
560
600
|
},
|
|
561
601
|
{
|
|
@@ -581,13 +621,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
581
621
|
},
|
|
582
622
|
contextWindow: 262144,
|
|
583
623
|
cost: {
|
|
584
|
-
cacheRead: 0.
|
|
624
|
+
cacheRead: 0.06,
|
|
585
625
|
cacheWrite: 0,
|
|
586
|
-
input: 0.
|
|
587
|
-
output: 0.
|
|
626
|
+
input: 0.06,
|
|
627
|
+
output: 0.24,
|
|
588
628
|
},
|
|
589
629
|
input: ["text"],
|
|
590
|
-
maxTokens:
|
|
630
|
+
maxTokens: 131072,
|
|
591
631
|
reasoning: true,
|
|
592
632
|
thinkingLevelMap: {
|
|
593
633
|
high: "high",
|
|
@@ -621,13 +661,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
621
661
|
},
|
|
622
662
|
contextWindow: 262144,
|
|
623
663
|
cost: {
|
|
624
|
-
cacheRead: 0,
|
|
664
|
+
cacheRead: 0.015,
|
|
625
665
|
cacheWrite: 0,
|
|
626
|
-
input: 0.
|
|
627
|
-
output: 0.
|
|
666
|
+
input: 0.015,
|
|
667
|
+
output: 0.6,
|
|
628
668
|
},
|
|
629
669
|
input: ["text"],
|
|
630
|
-
maxTokens:
|
|
670
|
+
maxTokens: 65536,
|
|
631
671
|
reasoning: true,
|
|
632
672
|
thinkingLevelMap: {
|
|
633
673
|
high: "high",
|
|
@@ -661,13 +701,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
661
701
|
},
|
|
662
702
|
contextWindow: 262144,
|
|
663
703
|
cost: {
|
|
664
|
-
cacheRead: 0.
|
|
704
|
+
cacheRead: 0.1,
|
|
665
705
|
cacheWrite: 0,
|
|
666
|
-
input: 0.
|
|
667
|
-
output:
|
|
706
|
+
input: 0.1,
|
|
707
|
+
output: 3,
|
|
668
708
|
},
|
|
669
709
|
input: ["text"],
|
|
670
|
-
maxTokens:
|
|
710
|
+
maxTokens: 65536,
|
|
671
711
|
reasoning: true,
|
|
672
712
|
thinkingLevelMap: {
|
|
673
713
|
high: "high",
|
|
@@ -701,13 +741,13 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
701
741
|
},
|
|
702
742
|
contextWindow: 262144,
|
|
703
743
|
cost: {
|
|
704
|
-
cacheRead: 0,
|
|
744
|
+
cacheRead: 0.6,
|
|
705
745
|
cacheWrite: 0,
|
|
706
746
|
input: 0.6,
|
|
707
747
|
output: 3.6,
|
|
708
748
|
},
|
|
709
749
|
input: ["text", "image"],
|
|
710
|
-
maxTokens:
|
|
750
|
+
maxTokens: 65536,
|
|
711
751
|
reasoning: true,
|
|
712
752
|
thinkingLevelMap: {
|
|
713
753
|
high: null,
|
package/models.ts
CHANGED
|
@@ -1,22 +1,34 @@
|
|
|
1
1
|
import type { RefreshModelsContext } from "@earendil-works/pi-ai";
|
|
2
2
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { MODEL_MAX_OUTPUT_TOKENS } from "./limits.generated.ts";
|
|
3
4
|
import { GENERATED_MODELS } from "./models.generated.ts";
|
|
4
5
|
import { MODEL_PRICING, type ModelPrice } from "./pricing.generated.ts";
|
|
5
6
|
import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
|
|
6
7
|
import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
|
|
7
8
|
|
|
8
9
|
// --- Pricing ---
|
|
9
|
-
//
|
|
10
|
+
// Per-1M-token prices are generated from the ollama.com/pricing model table by
|
|
10
11
|
// scripts/generate-pricing.ts (see pricing.generated.ts, do not edit by hand).
|
|
11
12
|
// Ollama Cloud is subscription-billed; these are equivalent pay-as-you-go
|
|
12
|
-
//
|
|
13
|
+
// rates so /cost shows comparable usage, not actual charges.
|
|
13
14
|
|
|
14
|
-
/** Resolve the
|
|
15
|
+
/** Resolve the price for an Ollama Cloud model ID. Exact match only;
|
|
15
16
|
* unmapped models return zero. */
|
|
16
17
|
function resolvePrice(id: string): ModelPrice {
|
|
17
18
|
return MODEL_PRICING[id] ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
|
|
18
19
|
}
|
|
19
20
|
|
|
21
|
+
// --- Max output tokens ---
|
|
22
|
+
// Per-model limits are probed against the live API by scripts/generate-limits.ts
|
|
23
|
+
// (see limits.generated.ts, do not edit by hand). /api/show does not expose the
|
|
24
|
+
// limit, so models without a probed value fall back to 32768.
|
|
25
|
+
|
|
26
|
+
/** Resolve the max output tokens for an Ollama Cloud model ID. Exact match only;
|
|
27
|
+
* unmapped models fall back to 32768. */
|
|
28
|
+
function resolveMaxTokens(id: string): number {
|
|
29
|
+
return MODEL_MAX_OUTPUT_TOKENS[id] ?? 32768;
|
|
30
|
+
}
|
|
31
|
+
|
|
20
32
|
// --- Constants ---
|
|
21
33
|
const FETCH_TIMEOUT_MS = 10000;
|
|
22
34
|
// How long a stored catalog is considered fresh before the next network refresh
|
|
@@ -127,9 +139,7 @@ export function assembleModels(raw: Record<string, OllamaShowResponse>): Provide
|
|
|
127
139
|
input: (data.capabilities?.includes("vision") ? ["text", "image"] : ["text"]) as ("text" | "image")[],
|
|
128
140
|
cost: resolvePrice(id),
|
|
129
141
|
contextWindow: getContextLength(data.model_info ?? {}),
|
|
130
|
-
|
|
131
|
-
// https://github.com/ollama/ollama/issues/7222). 32768 matches most Ollama Cloud context windows.
|
|
132
|
-
maxTokens: 32768,
|
|
142
|
+
maxTokens: resolveMaxTokens(id),
|
|
133
143
|
compat: buildCompat(),
|
|
134
144
|
}));
|
|
135
145
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-ollama-cloud",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package"
|
|
@@ -8,10 +8,12 @@
|
|
|
8
8
|
"files": [
|
|
9
9
|
"index.ts",
|
|
10
10
|
"config.ts",
|
|
11
|
+
"limits.generated.ts",
|
|
11
12
|
"models.ts",
|
|
12
13
|
"models.generated.ts",
|
|
13
14
|
"pricing.generated.ts",
|
|
14
15
|
"thinking-levels.ts",
|
|
16
|
+
"usage.ts",
|
|
15
17
|
"utils.ts",
|
|
16
18
|
"web-tools.ts",
|
|
17
19
|
"CHANGELOG.md",
|
|
@@ -29,7 +31,8 @@
|
|
|
29
31
|
"format": "biome format --write .",
|
|
30
32
|
"test": "vitest run",
|
|
31
33
|
"smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
|
|
32
|
-
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts"
|
|
34
|
+
"generate-models": "tsx scripts/generate-pricing.ts && tsx scripts/generate-models.ts && biome format --write models.generated.ts pricing.generated.ts",
|
|
35
|
+
"generate-limits": "tsx scripts/generate-limits.ts && biome format --write limits.generated.ts"
|
|
33
36
|
},
|
|
34
37
|
"pi": {
|
|
35
38
|
"extensions": [
|
package/pricing.generated.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-pricing.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
4
|
-
// Model count:
|
|
3
|
+
// Generated: 2026-09-03T10:12:00.135Z
|
|
4
|
+
// Model count: 19
|
|
5
5
|
|
|
6
6
|
export interface ModelPrice {
|
|
7
7
|
input: number;
|
|
@@ -11,22 +11,23 @@ export interface ModelPrice {
|
|
|
11
11
|
}
|
|
12
12
|
|
|
13
13
|
export const MODEL_PRICING: Record<string, ModelPrice> = {
|
|
14
|
-
"deepseek-v4-flash:0731": { input: 0.
|
|
15
|
-
"deepseek-v4-
|
|
16
|
-
"
|
|
17
|
-
"
|
|
18
|
-
"glm-5.1": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
14
|
+
"deepseek-v4-flash:0731": { input: 0.44, output: 1.32, cacheRead: 0.014, cacheWrite: 0 },
|
|
15
|
+
"deepseek-v4-pro:0813": { input: 1.32, output: 3.96, cacheRead: 0.044, cacheWrite: 0 },
|
|
16
|
+
"gemma4:31b": { input: 0.14, output: 0.4, cacheRead: 0.05, cacheWrite: 0 },
|
|
17
|
+
"glm-5.1": { input: 1, output: 3.2, cacheRead: 0.2, cacheWrite: 0 },
|
|
19
18
|
"glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
20
|
-
"
|
|
21
|
-
"
|
|
19
|
+
"glm-5.3": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
20
|
+
"glm-5.3-flash": { input: 0.15, output: 0.5, cacheRead: 0.03, cacheWrite: 0 },
|
|
21
|
+
"gpt-oss:120b": { input: 0.15, output: 0.6, cacheRead: 0.014, cacheWrite: 0 },
|
|
22
|
+
"gpt-oss:20b": { input: 0.07, output: 0.3, cacheRead: 0.035, cacheWrite: 0 },
|
|
22
23
|
"kimi-k2.6": { input: 0.95, output: 4, cacheRead: 0.16, cacheWrite: 0 },
|
|
23
24
|
"kimi-k2.7-code": { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 },
|
|
24
25
|
"kimi-k3": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 },
|
|
25
|
-
"minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0
|
|
26
|
-
"minimax-m3": { input: 0.
|
|
27
|
-
"mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0, cacheWrite: 0 },
|
|
28
|
-
"nemotron-3-nano:30b": { input: 0.
|
|
29
|
-
"nemotron-3-super": { input: 0.
|
|
30
|
-
"nemotron-3-ultra": { input: 0.
|
|
31
|
-
"qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0, cacheWrite: 0 },
|
|
26
|
+
"minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
|
|
27
|
+
"minimax-m3": { input: 0.6, output: 2.4, cacheRead: 0.12, cacheWrite: 0 },
|
|
28
|
+
"mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0.5, cacheWrite: 0 },
|
|
29
|
+
"nemotron-3-nano:30b": { input: 0.06, output: 0.24, cacheRead: 0.06, cacheWrite: 0 },
|
|
30
|
+
"nemotron-3-super": { input: 0.015, output: 0.6, cacheRead: 0.015, cacheWrite: 0 },
|
|
31
|
+
"nemotron-3-ultra": { input: 0.1, output: 3, cacheRead: 0.1, cacheWrite: 0 },
|
|
32
|
+
"qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0.6, cacheWrite: 0 },
|
|
32
33
|
};
|
package/usage.ts
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ollama Cloud usage data plane: fetch and format /api/usage.
|
|
3
|
+
*
|
|
4
|
+
* Self-contained module. Depends on:
|
|
5
|
+
* - models.ts - only for OLLAMA_BASE URL constant
|
|
6
|
+
* - utils.ts - fetchJsonWithTimeout
|
|
7
|
+
* Does NOT depend on provider registration, model fetching, or API key
|
|
8
|
+
* resolution (the caller resolves the key and passes it in).
|
|
9
|
+
*
|
|
10
|
+
* The /api/usage endpoint is undocumented and could change or disappear. The
|
|
11
|
+
* fetch degrades gracefully: distinct HTTP statuses map to distinct
|
|
12
|
+
* user-facing errors, and a malformed body raises a clear error rather than
|
|
13
|
+
* crashing.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import type { Theme } from "@earendil-works/pi-coding-agent";
|
|
17
|
+
import { OLLAMA_BASE } from "./models.ts";
|
|
18
|
+
import { fetchJsonWithTimeout, httpError } from "./utils.ts";
|
|
19
|
+
|
|
20
|
+
// --- Types ---
|
|
21
|
+
|
|
22
|
+
export interface UsageModel {
|
|
23
|
+
name: string;
|
|
24
|
+
request_count: number;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export interface UsageLimit {
|
|
28
|
+
/** Fraction of the plan's cap, 0-1 (not tokens). */
|
|
29
|
+
usage: number;
|
|
30
|
+
/** Per-model request counts (not token counts). */
|
|
31
|
+
models: UsageModel[];
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export interface UsageActivity {
|
|
35
|
+
cost?: string;
|
|
36
|
+
period?: {
|
|
37
|
+
type?: string;
|
|
38
|
+
starting_at?: string;
|
|
39
|
+
ending_at?: string;
|
|
40
|
+
};
|
|
41
|
+
models?: UsageModel[];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export interface UsageData {
|
|
45
|
+
limits: {
|
|
46
|
+
monthly: UsageLimit;
|
|
47
|
+
};
|
|
48
|
+
activity?: UsageActivity;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
// --- Constants ---
|
|
52
|
+
|
|
53
|
+
const USAGE_TIMEOUT_MS = 10000;
|
|
54
|
+
|
|
55
|
+
// --- Validation ---
|
|
56
|
+
|
|
57
|
+
/** Validate a single usage limit: a 0-1 fraction plus per-model request counts. */
|
|
58
|
+
export function isUsageLimit(data: unknown): data is UsageLimit {
|
|
59
|
+
if (data == null || typeof data !== "object") return false;
|
|
60
|
+
const d = data as UsageLimit;
|
|
61
|
+
return (
|
|
62
|
+
typeof d.usage === "number" &&
|
|
63
|
+
Array.isArray(d.models) &&
|
|
64
|
+
d.models.every(
|
|
65
|
+
(m) =>
|
|
66
|
+
m != null &&
|
|
67
|
+
typeof m === "object" &&
|
|
68
|
+
typeof (m as UsageModel).name === "string" &&
|
|
69
|
+
typeof (m as UsageModel).request_count === "number",
|
|
70
|
+
)
|
|
71
|
+
);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Validate a parsed /api/usage response: must have a monthly limit. */
|
|
75
|
+
export function isUsageResponse(data: unknown): data is UsageData {
|
|
76
|
+
if (data == null || typeof data !== "object") return false;
|
|
77
|
+
const d = data as UsageData;
|
|
78
|
+
return d.limits != null && typeof d.limits === "object" && isUsageLimit(d.limits.monthly);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// --- Fetch ---
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Fetch Ollama Cloud usage from the undocumented /api/usage endpoint.
|
|
85
|
+
* The caller resolves the API key and passes it in.
|
|
86
|
+
*/
|
|
87
|
+
export async function fetchUsage(apiKey: string, externalSignal?: AbortSignal): Promise<UsageData> {
|
|
88
|
+
const res = await fetchJsonWithTimeout<UsageData>(
|
|
89
|
+
`${OLLAMA_BASE}/api/usage`,
|
|
90
|
+
{
|
|
91
|
+
method: "GET",
|
|
92
|
+
headers: { Authorization: `Bearer ${apiKey}` },
|
|
93
|
+
},
|
|
94
|
+
USAGE_TIMEOUT_MS,
|
|
95
|
+
externalSignal,
|
|
96
|
+
);
|
|
97
|
+
|
|
98
|
+
if (!res.ok) {
|
|
99
|
+
// The 404 case is specific to this undocumented endpoint: it may have
|
|
100
|
+
// changed or disappeared, so surface that distinctly before the shared
|
|
101
|
+
// status mapping.
|
|
102
|
+
if (res.status === 404) {
|
|
103
|
+
throw new Error(
|
|
104
|
+
"Ollama Cloud usage failed: the /api/usage endpoint is unavailable (status 404). " +
|
|
105
|
+
"It is undocumented and may have changed.",
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
httpError("usage", res.status, res.error);
|
|
109
|
+
}
|
|
110
|
+
if (!isUsageResponse(res.data)) {
|
|
111
|
+
throw new Error("Ollama Cloud usage failed: unexpected response shape from the API.");
|
|
112
|
+
}
|
|
113
|
+
return res.data;
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// --- Formatting ---
|
|
117
|
+
|
|
118
|
+
/** Clamp a 0-1 usage fraction to a 0-100 percentage for display. */
|
|
119
|
+
function usagePercent(usage: number): number {
|
|
120
|
+
if (!Number.isFinite(usage)) return 0;
|
|
121
|
+
return Math.min(Math.max(Math.round(usage * 100), 0), 100);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Format usage for the /ollama-cloud-usage command output. */
|
|
125
|
+
export function formatUsage(data: UsageData): string {
|
|
126
|
+
const lines: string[] = ["Ollama Cloud usage:"];
|
|
127
|
+
|
|
128
|
+
const monthlyPct = usagePercent(data.limits.monthly.usage);
|
|
129
|
+
lines.push(` Monthly (30d): ${monthlyPct}%`);
|
|
130
|
+
for (const m of data.limits.monthly.models) {
|
|
131
|
+
lines.push(` - ${m.name}: ${m.request_count} request${m.request_count === 1 ? "" : "s"}`);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
if (typeof data.activity?.cost === "string") {
|
|
135
|
+
lines.push(` Activity (4wk): $${data.activity.cost}`);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return lines.join("\n");
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** Render a 10-character quota bar for a 0-100 percentage. */
|
|
142
|
+
function quotaBar(pct: number): string {
|
|
143
|
+
const filled = Math.min(Math.max(Math.floor(pct / 10), 0), 10);
|
|
144
|
+
return `▕${"█".repeat(filled)}${"░".repeat(10 - filled)}▏`;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** Color a single usage segment by how close it is to the cap. */
|
|
148
|
+
function colorSegment(theme: Theme, label: string, pct: number): string {
|
|
149
|
+
const color = pct >= 80 ? "error" : pct >= 60 ? "warning" : "success";
|
|
150
|
+
return theme.fg(color, `${label} ${quotaBar(pct)} ${pct}%`);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Compact one-line usage for the footer status bar, colored by usage level.
|
|
155
|
+
* The endpoint exposes a monthly reset period but not an exact timestamp, so
|
|
156
|
+
* the color reflects the usage fraction rather than pace.
|
|
157
|
+
*/
|
|
158
|
+
export function formatUsageStatusColored(theme: Theme, data: UsageData): string {
|
|
159
|
+
const monthly = usagePercent(data.limits.monthly.usage);
|
|
160
|
+
return colorSegment(theme, "30d", monthly);
|
|
161
|
+
}
|
package/utils.ts
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
|
|
1
3
|
export async function fetchJsonWithTimeout<T>(
|
|
2
4
|
url: string,
|
|
3
5
|
init: RequestInit,
|
|
@@ -75,3 +77,37 @@ export function getContextLength(modelInfo: Record<string, unknown>): number {
|
|
|
75
77
|
}
|
|
76
78
|
return 128000;
|
|
77
79
|
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Resolve the Ollama Cloud API key for a tool execution or command.
|
|
83
|
+
*
|
|
84
|
+
* Prefers the canonical provider auth chain (ctx.modelRegistry.getApiKeyForProvider),
|
|
85
|
+
* which honors runtime/CLI key overrides, the registered
|
|
86
|
+
* apiKey: "$OLLAMA_API_KEY" config, and stored auth.json credentials. Falls back
|
|
87
|
+
* to the OLLAMA_API_KEY env var for the case where the provider is not yet
|
|
88
|
+
* registered at call time.
|
|
89
|
+
*/
|
|
90
|
+
export async function getCloudApiKey(ctx: Pick<ExtensionContext, "modelRegistry">): Promise<string | undefined> {
|
|
91
|
+
return (await ctx.modelRegistry.getApiKeyForProvider("ollama-cloud")) ?? process.env.OLLAMA_API_KEY;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Throw a user-facing error for a non-ok Ollama Cloud HTTP response, mapping
|
|
96
|
+
* distinct status codes. Shared by the web tools and the usage command.
|
|
97
|
+
*/
|
|
98
|
+
export function httpError(op: string, status: number, error?: string): never {
|
|
99
|
+
if (status === 401 || status === 403) {
|
|
100
|
+
throw new Error(
|
|
101
|
+
`Ollama Cloud ${op} failed: authentication error. Check your API key in OLLAMA_API_KEY or auth.json.`,
|
|
102
|
+
);
|
|
103
|
+
}
|
|
104
|
+
if (status === 429) {
|
|
105
|
+
throw new Error(`Ollama Cloud ${op} failed: rate limited. Try again shortly.`);
|
|
106
|
+
}
|
|
107
|
+
if (status >= 500) {
|
|
108
|
+
throw new Error(`Ollama Cloud ${op} failed: server error (status ${status}). Try again shortly.`);
|
|
109
|
+
}
|
|
110
|
+
throw new Error(
|
|
111
|
+
`Ollama Cloud ${op} failed: unexpected response (status ${status}${error ? `: ${error}` : ""}). Try again shortly.`,
|
|
112
|
+
);
|
|
113
|
+
}
|
package/web-tools.ts
CHANGED
|
@@ -19,7 +19,6 @@
|
|
|
19
19
|
import {
|
|
20
20
|
type AgentToolResult,
|
|
21
21
|
type ExtensionAPI,
|
|
22
|
-
type ExtensionContext,
|
|
23
22
|
keyHint,
|
|
24
23
|
type Theme,
|
|
25
24
|
type ToolRenderResultOptions,
|
|
@@ -28,7 +27,7 @@ import {
|
|
|
28
27
|
import { type Component, Text, truncateToWidth } from "@earendil-works/pi-tui";
|
|
29
28
|
import { Type } from "@sinclair/typebox";
|
|
30
29
|
import { OLLAMA_BASE } from "./models.ts";
|
|
31
|
-
import { fetchJsonWithTimeout } from "./utils.ts";
|
|
30
|
+
import { fetchJsonWithTimeout, getCloudApiKey, httpError } from "./utils.ts";
|
|
32
31
|
|
|
33
32
|
// --- Types ---
|
|
34
33
|
|
|
@@ -50,62 +49,11 @@ interface FetchResponse {
|
|
|
50
49
|
|
|
51
50
|
const WEB_TOOLS_TIMEOUT_MS = 15000;
|
|
52
51
|
|
|
53
|
-
/**
|
|
54
|
-
* Resolve the Ollama Cloud API key for a tool execution.
|
|
55
|
-
*
|
|
56
|
-
* Prefers the canonical provider auth chain (ctx.modelRegistry.getApiKeyForProvider),
|
|
57
|
-
* which honors runtime/CLI key overrides, the registered
|
|
58
|
-
* apiKey: "$OLLAMA_API_KEY" config, and stored auth.json credentials. Falls back
|
|
59
|
-
* to the OLLAMA_API_KEY env var for the case where the provider is not yet
|
|
60
|
-
* registered at tool-call time.
|
|
61
|
-
*
|
|
62
|
-
* Exported for unit testing.
|
|
63
|
-
*/
|
|
64
|
-
export async function getCloudApiKey(ctx: Pick<ExtensionContext, "modelRegistry">): Promise<string | undefined> {
|
|
65
|
-
return (await ctx.modelRegistry.getApiKeyForProvider("ollama-cloud")) ?? process.env.OLLAMA_API_KEY;
|
|
66
|
-
}
|
|
67
|
-
|
|
68
52
|
/** Throw a no-API-key error. */
|
|
69
53
|
function noApiKeyError(): never {
|
|
70
54
|
throw new Error("No Ollama Cloud API key configured. Set OLLAMA_API_KEY or add to auth.json.");
|
|
71
55
|
}
|
|
72
56
|
|
|
73
|
-
/** Throw a search error for a non-ok result, mapping distinct status codes. */
|
|
74
|
-
function searchError(status: number, error?: string): never {
|
|
75
|
-
if (status === 401 || status === 403) {
|
|
76
|
-
throw new Error(
|
|
77
|
-
"Ollama Cloud search failed: authentication error. " + "Check your API key in OLLAMA_API_KEY or auth.json.",
|
|
78
|
-
);
|
|
79
|
-
}
|
|
80
|
-
if (status === 429) {
|
|
81
|
-
throw new Error("Ollama Cloud search failed: rate limited. Try again shortly.");
|
|
82
|
-
}
|
|
83
|
-
if (status >= 500) {
|
|
84
|
-
throw new Error(`Ollama Cloud search failed: server error (status ${status}). Try again shortly.`);
|
|
85
|
-
}
|
|
86
|
-
throw new Error(
|
|
87
|
-
`Ollama Cloud search failed: unexpected response (status ${status}${error ? `: ${error}` : ""}). Try again shortly.`,
|
|
88
|
-
);
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
/** Throw a fetch error for a non-ok result, mapping distinct status codes. */
|
|
92
|
-
function fetchError(status: number, error?: string): never {
|
|
93
|
-
if (status === 401 || status === 403) {
|
|
94
|
-
throw new Error(
|
|
95
|
-
"Ollama Cloud fetch failed: authentication error. " + "Check your API key in OLLAMA_API_KEY or auth.json.",
|
|
96
|
-
);
|
|
97
|
-
}
|
|
98
|
-
if (status === 429) {
|
|
99
|
-
throw new Error("Ollama Cloud fetch failed: rate limited. Try again shortly.");
|
|
100
|
-
}
|
|
101
|
-
if (status >= 500) {
|
|
102
|
-
throw new Error(`Ollama Cloud fetch failed: server error (status ${status}). Try again shortly.`);
|
|
103
|
-
}
|
|
104
|
-
throw new Error(
|
|
105
|
-
`Ollama Cloud fetch failed: unexpected response (status ${status}${error ? `: ${error}` : ""}). Try again shortly.`,
|
|
106
|
-
);
|
|
107
|
-
}
|
|
108
|
-
|
|
109
57
|
const PREVIEW_LINES = 8;
|
|
110
58
|
|
|
111
59
|
/**
|
|
@@ -232,7 +180,7 @@ export function registerWebSearchTool(pi: ExtensionAPI) {
|
|
|
232
180
|
);
|
|
233
181
|
|
|
234
182
|
if (!res.ok) {
|
|
235
|
-
|
|
183
|
+
httpError("search", res.status, res.error);
|
|
236
184
|
}
|
|
237
185
|
if (!isSearchResponse(res.data)) {
|
|
238
186
|
throw new Error("Web search failed: unexpected response shape from the API.");
|
|
@@ -287,7 +235,7 @@ export function registerWebFetchTool(pi: ExtensionAPI) {
|
|
|
287
235
|
);
|
|
288
236
|
|
|
289
237
|
if (!res.ok) {
|
|
290
|
-
|
|
238
|
+
httpError("fetch", res.status, res.error);
|
|
291
239
|
}
|
|
292
240
|
if (!isFetchResponse(res.data)) {
|
|
293
241
|
throw new Error("Web fetch failed: unexpected response shape from the API.");
|