pi-ollama-cloud 0.3.1 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,20 @@
2
2
 
3
3
  All notable changes to this project will be documented in this file.
4
4
 
5
+ ## [0.4.1] - 2026-05-07
6
+
7
+ - Add `renderCall` to `ollama_web_search` and `ollama_web_fetch` tools so the TUI displays the query/URL in the tool call header instead of just the bare tool name. (#12)
8
+
9
+ ## [0.4.0] - 2026-05-06
10
+
11
+ - Fix `/api/chat` requests not disabling thinking when Pi's thinking level is set to `off`. Maps Pi `off` to `reasoning_effort: "none"` on models where the API respects it, hides the `off` level on models where it doesn't (gpt-oss, kimi-k2-thinking, minimax, qwen3-vl). (#6)
12
+ - Add `thinking-levels.ts` with curated per-model thinking level maps (DEFAULT, GPT_OSS, QWEN3, NO_OFF), validated against all 24 thinking-capable models via automated experiment (see docs/think-experiment.md).
13
+ - Fix system prompt (AGENTS.md content) not being read by GLM models by setting `supportsDeveloperRole: false` on all registered models.
14
+ - Rename smoke test to test, add lint step.
15
+ - Treat stale local model caches as usable for immediate startup while triggering the same visible refresh flow as `/ollama-cloud-refresh` on `session_start`; use fallback models only when the cache is missing or invalid.
16
+ - Add a single-line `/ollama-cloud-refresh` progress widget showing the current stage, count, percentage, failures, and progress bar.
17
+ - Add thinking on/off assertions to the CI test workflow.
18
+
5
19
  ## [0.3.1] - 2026-05-05
6
20
 
7
21
  - Fix `OLLAMA_API_KEY` env var not being respected by `fetchModels` and web tools. pi-ai does not know about the `ollama-cloud` provider ID, so `AuthStorage.getApiKey()` alone misses the env var. Added explicit `process.env.OLLAMA_API_KEY` fallback.
package/README.md CHANGED
@@ -2,14 +2,15 @@
2
2
 
3
3
  Ollama Cloud provider plugin for [Pi](https://github.com/badlogic/pi-mono) coding agent.
4
4
 
5
- Registers Ollama Cloud as a model provider with dynamically fetched models, and provides `ollama_web_search` and `ollama_web_fetch` tools that use the [Ollama Cloud web search API](https://docs.ollama.com/capabilities/web-search) — no local Ollama server required.
5
+ Registers Ollama Cloud as a model provider with dynamically fetched models, and provides `ollama_web_search` and `ollama_web_fetch` tools that use the [Ollama Cloud web search API](https://docs.ollama.com/capabilities/web-search) - no local Ollama server required.
6
6
 
7
7
  ## Features
8
8
 
9
9
  - **Dynamic model discovery** - Fetches the full model list from `ollama.com/v1/models`, then fetches per-model details via `/api/show` to determine capabilities, context length, and tool support.
10
+ - **Curated thinking levels** - Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible `reasoning_effort` values via `thinking-levels.ts`, with per-model exceptions based on API testing.
10
11
  - **Persistent cache** - Raw API responses are cached at `~/.pi/agent/cache/ollama-cloud-models.json` so models are available immediately on startup without hitting the network.
11
- - **Cold cache fallback** - When no cache exists, a small set of hardcoded models is used until `/ollama-cloud-refresh` is run.
12
- - **`/ollama-cloud-refresh` command** - Re-fetches the model list from the API and updates the cache and provider registration live (no restart needed).
12
+ - **Startup refresh** - When the local cache is stale, the plugin uses it immediately and then runs the same visible refresh flow as `/ollama-cloud-refresh` once the Pi session UI is available. Missing/invalid caches use a small fallback list until refresh completes.
13
+ - **`/ollama-cloud-refresh` command** - Re-fetches the model list and updates the cache and provider registration live (no restart needed).
13
14
  - **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
14
15
  - **`ollama_web_fetch` tool** - Fetch and extract text content from a web page URL using Ollama Cloud's `/api/web_fetch` endpoint. Returns page title, content, and links.
15
16
  - **Zero cost tracking** - All models are registered with zero costs since Ollama Cloud uses a flat subscription model (Free, Pro, Max) rather than per-token billing. Per-request costs don't apply, so Pi's cost tracker always shows zero. See [ollama.com/pricing](https://ollama.com/pricing) for plan details.
@@ -93,13 +94,13 @@ Accepted disabling values are `0`, `false`, `no`, `off`, or an empty string. Whe
93
94
 
94
95
  ### 4. Fetch models
95
96
 
96
- On first launch the plugin will use a small set of fallback models. Run:
97
+ On first launch the plugin registers a small hardcoded fallback list, then refreshes model metadata automatically with the same progress widget used by the manual command. If an existing cache is merely stale, that cached model list remains active while refresh runs. You can also run:
97
98
 
98
99
  ```
99
100
  /ollama-cloud-refresh
100
101
  ```
101
102
 
102
- This fetches the full model list from the Ollama Cloud API and caches it locally.
103
+ This fetches the full model list from the Ollama Cloud API and overwrites the local cache.
103
104
 
104
105
  ### 5. Select a model
105
106
 
@@ -114,18 +115,40 @@ The plugin uses two Ollama Cloud API endpoints to build the model list:
114
115
 
115
116
  Only models with the `tools` capability are registered - these are the ones Pi can use for tool-calling.
116
117
 
117
- The raw `/api/show` responses are cached at `~/.pi/agent/cache/ollama-cloud-models.json`. This cache **never expires** - run `/ollama-cloud-refresh` to update it.
118
+ The raw `/api/show` responses are cached at `~/.pi/agent/cache/ollama-cloud-models.json` with a top-level `timestamp` value. If that local cache is older than 30 days, the plugin keeps using it immediately and triggers the visible refresh flow on `session_start`. If the cache is missing or invalid, the plugin registers a small hardcoded model list until refresh succeeds. If no key is available or refresh fails, the current registered list remains active until `/ollama-cloud-refresh` succeeds.
118
119
 
119
120
  Model metadata is derived from the cached data:
120
121
 
121
122
  | Field | Source |
122
123
  |---|---|
123
124
  | `reasoning` | `capabilities` includes `"thinking"` |
125
+ | `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 4 maps (DEFAULT, GPT_OSS, QWEN3, NO_OFF) based on API testing |
124
126
  | `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
125
127
  | `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
126
128
  | `maxTokens` | Fixed at 32768 |
127
129
  | `cost` | All zeros (Ollama Cloud uses subscription plans, not per-token billing - see [pricing](https://ollama.com/pricing)) |
128
130
 
131
+ ### Thinking level mapping
132
+
133
+ Pi's thinking levels are mapped to Ollama Cloud's OpenAI-compatible `reasoning_effort` parameter in [`thinking-levels.ts`](thinking-levels.ts). The API accepts `none`, `low`, `medium`, `high`, and `max`. Effects of `max` over `high` vary by model and prompt difficulty - see [`docs/think-experiment.md`](docs/think-experiment.md) for details.
134
+
135
+ | Map | Models | Levels exposed | Notes |
136
+ |---|---|---|---|
137
+ | `DEFAULT` | Most thinking models | off, low, medium, high, xhigh | `minimal` hidden (duplicate of low) |
138
+ | `GPT_OSS` | `gpt-oss*` | low, medium, high | Can't disable thinking, no off or xhigh |
139
+ | `QWEN3` | `qwen3*` (except `qwen3-vl*`) | off, medium | Binary-only (think/nothink), no gradation |
140
+ | `NO_OFF` | `qwen3-vl*`, `kimi-k2-thinking`, `minimax*` | low, medium, high, xhigh | "none" doesn't disable thinking on these models |
141
+
142
+ See [docs/think-experiment.md](docs/think-experiment.md) for the testing methodology and results.
143
+
144
+ Refresh from inside Pi:
145
+
146
+ ```text
147
+ /ollama-cloud-refresh
148
+ ```
149
+
150
+ That command updates `~/.pi/agent/cache/ollama-cloud-models.json` with a new `timestamp` and re-registers the provider live, so no restart is required.
151
+
129
152
  ## Tools
130
153
 
131
154
  | Tool | Description |
@@ -171,7 +194,7 @@ The project uses [Biome](https://biomejs.dev/) for linting and formatting (2-spa
171
194
 
172
195
  **You can use both at the same time.** The providers live under different names (`ollama` vs `ollama-cloud`), so you can switch between them with `/model` or `Ctrl+L`. For example, use your local `ollama` provider for low-latency work on smaller models, and `ollama-cloud` for direct access to the full catalog of cloud models without needing a local server.
173
196
 
174
- > **Note:** The [`@ollama/pi-web-search`](https://www.npmjs.com/package/@ollama/pi-web-search) package (installed automatically by `ollama launch pi`) calls the **local** Ollama server's `/api/experimental/web_search` and `/api/experimental/web_fetch` endpoints and authenticates via `ollama signin`. This extension's `ollama_web_search` and `ollama_web_fetch` tools use the **cloud** API at `ollama.com/api/web_search` and `ollama.com/api/web_fetch` instead — same API key, no local server required. Both can coexist: the local tools register as `web_search`/`web_fetch` and these register as `ollama_web_search`/`ollama_web_fetch` to avoid name conflicts.
197
+ > **Note:** The [`@ollama/pi-web-search`](https://www.npmjs.com/package/@ollama/pi-web-search) package (installed automatically by `ollama launch pi`) calls the **local** Ollama server's `/api/experimental/web_search` and `/api/experimental/web_fetch` endpoints and authenticates via `ollama signin`. This extension's `ollama_web_search` and `ollama_web_fetch` tools use the **cloud** API at `ollama.com/api/web_search` and `ollama.com/api/web_fetch` instead - same API key, no local server required. Both can coexist: the local tools register as `web_search`/`web_fetch` and these register as `ollama_web_search`/`ollama_web_fetch` to avoid name conflicts.
175
198
 
176
199
  ## Releasing
177
200
 
@@ -184,9 +207,9 @@ npm version minor # or patch, or major
184
207
  git push --tags
185
208
  ```
186
209
 
187
- The tag version must match the version in `package.json` — `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
210
+ The tag version must match the version in `package.json` - `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
188
211
 
189
- The workflow uses npm's [trusted publishing](https://docs.npmjs.com/trusted-publishers/) (OIDC) — no tokens stored as secrets. To set it up:
212
+ The workflow uses npm's [trusted publishing](https://docs.npmjs.com/trusted-publishers/) (OIDC) - no tokens stored as secrets. To set it up:
190
213
 
191
214
  1. Go to [npmjs.com](https://www.npmjs.com) → your avatar → **Packages** → `pi-ollama-cloud` → **Settings** → **Trusted publishing**
192
215
  2. Click **GitHub Actions** and enter:
@@ -197,5 +220,5 @@ Each publish also gets automatic [provenance attestation](https://docs.npmjs.com
197
220
 
198
221
  ## Notes
199
222
 
200
- - Some Ollama Cloud models may reject the `developer` message role, causing a `400` error. If you encounter this, the model may need `compat: { supportsDeveloperRole: false }`. You can edit `index.ts` to add this for specific models, or open an issue to track it.
223
+ - The extension sets `supportsDeveloperRole: false` on all models so the system prompt always uses `role: "system"`. Without this, pi sends the prompt as `role: "developer"` for thinking-capable models, which some models (e.g. GLM-5.1) ignore entirely - the prompt simply isn't read.
201
224
  - The fetch timeout is 10 seconds per request. On slow connections, some model detail fetches may time out - the plugin reports how many succeeded vs failed.
package/index.ts CHANGED
@@ -7,7 +7,7 @@
7
7
  * 1. Get an API key from https://ollama.com
8
8
  * 2. Add to auth.json in the agent config dir (~/.pi/agent/auth.json, or set PI_CODING_AGENT_DIR):
9
9
  * { "ollama-cloud": { "type": "api_key", "key": "your-key" } }
10
- * 3. Run /ollama-cloud-refresh to fetch models (uses cache or fallback on boot)
10
+ * 3. Run /ollama-cloud-refresh to fetch model metadata
11
11
  * 4. Use /model or ctrl+l to select an Ollama Cloud model
12
12
  *
13
13
  * Two endpoints are used to build the model list:
@@ -17,14 +17,22 @@
17
17
  * Raw /api/show responses are cached at <agentDir>/cache/ollama-cloud-models.json
18
18
  * so the provider assembly can be debugged and re-derived without re-fetching.
19
19
  *
20
- * Cache never expires -- run /ollama-cloud-refresh to update.
21
- * Cold cache falls back to a small set of hardcoded models.
20
+ * Local cache entries include timestamp. Stale local caches are used immediately while a visible startup
21
+ * refresh runs; missing/invalid caches use a small hardcoded model list until refresh completes.
22
22
  *
23
23
  * Only models with "tools" capability are registered.
24
24
  */
25
25
 
26
26
  import type { ExtensionAPI, ExtensionCommandContext, ProviderModelConfig } from "@mariozechner/pi-coding-agent";
27
- import { assembleModels, FALLBACK_MODELS, fetchModels, OLLAMA_BASE, readCache, writeCache } from "./models.ts";
27
+ import {
28
+ assembleModels,
29
+ FALLBACK_MODELS,
30
+ fetchModels,
31
+ OLLAMA_BASE,
32
+ type RefreshProgress,
33
+ readCacheState,
34
+ writeCache,
35
+ } from "./models.ts";
28
36
  import { registerWebFetchTool, registerWebSearchTool } from "./web-tools.ts";
29
37
 
30
38
  /**
@@ -48,30 +56,66 @@ function registerProvider(pi: ExtensionAPI, models: ProviderModelConfig[]) {
48
56
  });
49
57
  }
50
58
 
51
- function registerRefreshCommand(pi: ExtensionAPI) {
52
- pi.registerCommand("ollama-cloud-refresh", {
53
- description: "Refresh Ollama Cloud models from the API",
54
- handler: async (_args: string, ctx: ExtensionCommandContext) => {
55
- ctx.ui.setWorkingMessage("Refreshing Ollama Cloud models...");
59
+ function renderProgressBar(current: number, total: number, width = 15): string {
60
+ if (total <= 0) return `[${"░".repeat(width)}]`;
61
+ const ratio = Math.max(0, Math.min(1, current / total));
62
+ const filled = Math.round(ratio * width);
63
+ return `[${"█".repeat(filled)}${"░".repeat(width - filled)}]`;
64
+ }
56
65
 
57
- const raw = await fetchModels(ctx);
58
- if (!raw) {
59
- ctx.ui.setWorkingMessage();
60
- return;
61
- }
66
+ function createRefreshProgressUi(ctx: Pick<ExtensionCommandContext, "ui">) {
67
+ const key = "ollama-cloud-refresh";
68
+ return {
69
+ update(progress: RefreshProgress) {
70
+ const current = progress.current ?? 0;
71
+ const total = progress.total ?? 0;
72
+ const percent = total > 0 ? Math.round((current / total) * 100) : 0;
73
+ const failed = progress.failed ? `, ${progress.failed} failed` : "";
74
+ const stage =
75
+ progress.stage === "list"
76
+ ? "Discovering models"
77
+ : progress.stage === "details"
78
+ ? "Fetching model details"
79
+ : "Done";
80
+ const summary = total > 0 ? `${current}/${total} (${percent}%${failed})` : progress.message;
81
+ const line = `☁ Ollama Cloud - ${stage} — ${summary} ${renderProgressBar(current, total)}`;
62
82
 
63
- writeCache(raw);
64
- const newModels = assembleModels(raw);
83
+ ctx.ui.setWorkingMessage(`Refreshing Ollama Cloud models - ${stage.toLowerCase()}`);
84
+ ctx.ui.setWidget(key, [line], { placement: "belowEditor" });
85
+ },
86
+ clear() {
87
+ ctx.ui.setWidget(key, undefined);
88
+ ctx.ui.setStatus(key, undefined);
89
+ ctx.ui.setWorkingMessage();
90
+ },
91
+ };
92
+ }
65
93
 
66
- // NOTE: Some models may trigger errors like:
67
- // Error: 400 "developer is not one of ['system', 'assistant', 'user', 'tool', 'function']"
68
- // If that comes up, consider setting `supportsDeveloperRole: false` in the compat field
69
- // for the provider or specific models, e.g.:
70
- // compat: { supportsDeveloperRole: false }
71
- registerProvider(pi, newModels);
94
+ async function runRefresh(pi: ExtensionAPI, ctx: Pick<ExtensionCommandContext, "ui">) {
95
+ const progressUi = createRefreshProgressUi(ctx);
96
+ try {
97
+ progressUi.update({ stage: "list", message: "Starting refresh..." });
72
98
 
73
- ctx.ui.notify(`Registered ${newModels.length} Ollama Cloud models`, "info");
74
- ctx.ui.setWorkingMessage();
99
+ const raw = await fetchModels(ctx, (progress) => progressUi.update(progress));
100
+ if (!raw) return false;
101
+
102
+ writeCache(raw);
103
+ const newModels = assembleModels(raw);
104
+
105
+ registerProvider(pi, newModels);
106
+
107
+ ctx.ui.notify(`Registered ${newModels.length} Ollama Cloud models`, "info");
108
+ return true;
109
+ } finally {
110
+ progressUi.clear();
111
+ }
112
+ }
113
+
114
+ function registerRefreshCommand(pi: ExtensionAPI) {
115
+ pi.registerCommand("ollama-cloud-refresh", {
116
+ description: "Refresh Ollama Cloud models from the API",
117
+ handler: async (_args: string, ctx: ExtensionCommandContext) => {
118
+ await runRefresh(pi, ctx);
75
119
  },
76
120
  });
77
121
  }
@@ -79,12 +123,22 @@ function registerRefreshCommand(pi: ExtensionAPI) {
79
123
  // --- Main ---
80
124
 
81
125
  export default async function (pi: ExtensionAPI) {
82
- const cached = readCache();
83
- const models = cached ? assembleModels(cached) : FALLBACK_MODELS;
126
+ const cacheState = readCacheState();
127
+ const needsStartupRefresh = cacheState.status !== "fresh";
128
+ const models = cacheState.status === "missing" ? FALLBACK_MODELS : assembleModels(cacheState.models);
84
129
 
85
130
  registerProvider(pi, models);
86
131
  registerRefreshCommand(pi);
87
132
 
133
+ if (needsStartupRefresh) {
134
+ let started = false;
135
+ pi.on("session_start", async (_event, ctx) => {
136
+ if (started) return;
137
+ started = true;
138
+ await runRefresh(pi, ctx);
139
+ });
140
+ }
141
+
88
142
  if (!WEB_TOOLS_DISABLED) {
89
143
  registerWebSearchTool(pi);
90
144
  registerWebFetchTool(pi);
package/models.ts CHANGED
@@ -1,19 +1,20 @@
1
- import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
1
+ import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
- import { getModels, getProviders } from "@mariozechner/pi-ai";
4
3
  import {
5
4
  AuthStorage,
6
5
  type ExtensionCommandContext,
7
6
  getAgentDir,
8
7
  type ProviderModelConfig,
9
8
  } from "@mariozechner/pi-coding-agent";
9
+ import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
10
+ import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
10
11
 
11
12
  // --- Constants ---
12
13
  const CACHE_DIR = join(getAgentDir(), "cache");
13
14
  const CACHE_FILE = join(CACHE_DIR, "ollama-cloud-models.json");
15
+ const CACHE_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000;
14
16
  const FETCH_TIMEOUT_MS = 10000;
15
17
 
16
- // --- API fetch ---
17
18
  export const OLLAMA_BASE = (process.env.OLLAMA_API_BASE || "https://ollama.com").replace(/\/+$/, "");
18
19
 
19
20
  // Initialize AuthStorage
@@ -21,7 +22,7 @@ const authStorage = AuthStorage.create();
21
22
 
22
23
  // --- Raw API types ---
23
24
  /** Response from POST /api/show */
24
- export interface OllamaShowResponse {
25
+ interface OllamaShowResponse {
25
26
  details: {
26
27
  parent_model: string;
27
28
  format: string;
@@ -35,82 +36,39 @@ export interface OllamaShowResponse {
35
36
  modified_at: string;
36
37
  }
37
38
 
38
- /** On-disk cache: raw /api/show responses keyed by model ID */
39
- interface CachedData {
40
- timestamp: number;
41
- models: Record<string, OllamaShowResponse>;
42
- }
39
+ type CachedOllamaModel = OllamaShowResponse;
43
40
 
44
- // --- Assembly: raw API data -> ProviderModelConfig[] ---
45
- function getContextLength(modelInfo: Record<string, unknown>): number {
46
- for (const [key, value] of Object.entries(modelInfo)) {
47
- if (key.endsWith(".context_length") && typeof value === "number") {
48
- return value;
49
- }
50
- }
51
- return 128000;
41
+ /** On-disk cache: raw /api/show responses keyed by model ID. */
42
+ interface CachedData {
43
+ /** Unix epoch milliseconds used to decide when the generated metadata is stale. */
44
+ timestamp?: number;
45
+ models: Record<string, CachedOllamaModel>;
52
46
  }
53
47
 
54
- // --- Built-in model knowledge index ---
55
- // Build a lookup of model ID -> thinkingLevelMap from pi's built-in models.
56
- // This avoids hardcoding model-family mappings: when pi-mono updates its
57
- // model definitions (e.g. DeepSeek V4's thinking levels), the extension
58
- // picks up the changes automatically.
59
- const BUILTIN_THINKING_MAP: Record<string, ProviderModelConfig["thinkingLevelMap"]> = {};
60
- // Fallback: family stem -> [stem, thinkingLevelMap] pairs for models whose Ollama Cloud
61
- // ID doesn't match exactly. The stem is derived by stripping provider prefixes and
62
- // non-alphanumeric characters (e.g. "gemma-4-31b-it" -> "gemma431bit").
63
- // When looking up an Ollama model by its details.family field, we search for a pi stem
64
- // that starts with the family stem (e.g. family "gemma4" -> pi "gemma431bit").
65
- // Entries are sorted longest-first so the most specific match wins.
66
- const BUILTIN_FAMILY_ENTRIES: [string, NonNullable<ProviderModelConfig["thinkingLevelMap"]>][] = [];
67
- for (const provider of getProviders()) {
68
- for (const model of getModels(provider as any)) {
69
- if (model.thinkingLevelMap) {
70
- BUILTIN_THINKING_MAP[model.id] = model.thinkingLevelMap;
71
- const stem = model.id
72
- .replace(/^[a-z0-9-]+\//, "") // strip provider prefix (e.g. "zai/", "deepseek/")
73
- .replace(/[^a-zA-Z0-9]/g, "") // strip non-alphanumeric
74
- .toLowerCase();
75
- BUILTIN_FAMILY_ENTRIES.push([stem, model.thinkingLevelMap]);
76
- }
77
- }
78
- }
79
- // Longest stems first so a more specific match (e.g. "gemma431bit") wins over a generic one (e.g. "gemma4").
80
- BUILTIN_FAMILY_ENTRIES.sort((a, b) => b[0].length - a[0].length);
81
-
82
- function resolveThinkingLevelMap(modelId: string, data: OllamaShowResponse): ProviderModelConfig["thinkingLevelMap"] {
83
- // 1. Exact ID match (e.g. "deepseek-v4-pro")
84
- const exact = BUILTIN_THINKING_MAP[modelId];
85
- if (exact) return exact;
86
-
87
- // 2. Family-based fallback: match Ollama's details.family against pi model stems
88
- if (data.capabilities?.includes("thinking")) {
89
- const familyStem = data.details?.family?.replace(/[^a-zA-Z0-9]/g, "").toLowerCase() ?? "";
90
- if (familyStem) {
91
- for (const [stem, tlm] of BUILTIN_FAMILY_ENTRIES) {
92
- if (stem.startsWith(familyStem)) {
93
- return tlm;
94
- }
95
- }
96
- }
97
- }
48
+ type RefreshProgressStage = "list" | "details" | "done";
98
49
 
99
- return undefined;
50
+ export interface RefreshProgress {
51
+ stage: RefreshProgressStage;
52
+ current?: number;
53
+ total?: number;
54
+ failed?: number;
55
+ message: string;
100
56
  }
101
57
 
102
- export function assembleModels(raw: Record<string, OllamaShowResponse>): ProviderModelConfig[] {
58
+ // --- Assembly: raw API data -> ProviderModelConfig[] ---
59
+ export function assembleModels(raw: Record<string, CachedOllamaModel>): ProviderModelConfig[] {
103
60
  return Object.entries(raw)
104
61
  .filter(([, data]) => data.capabilities?.includes("tools"))
105
62
  .map(([id, data]) => ({
106
63
  id,
107
64
  name: id,
108
65
  reasoning: data.capabilities?.includes("thinking") ?? false,
109
- thinkingLevelMap: resolveThinkingLevelMap(id, data),
66
+ thinkingLevelMap: resolveThinkingLevelMap(id, data.capabilities ?? []),
110
67
  input: (data.capabilities?.includes("vision") ? ["text", "image"] : ["text"]) as ("text" | "image")[],
111
68
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
112
69
  contextWindow: getContextLength(data.model_info ?? {}),
113
70
  maxTokens: 32768,
71
+ compat: { supportsDeveloperRole: false },
114
72
  }));
115
73
  }
116
74
 
@@ -124,6 +82,7 @@ export const FALLBACK_MODELS: ProviderModelConfig[] = [
124
82
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
125
83
  contextWindow: 202752,
126
84
  maxTokens: 32768,
85
+ compat: { supportsDeveloperRole: false },
127
86
  },
128
87
  {
129
88
  id: "gemma4:31b",
@@ -133,102 +92,216 @@ export const FALLBACK_MODELS: ProviderModelConfig[] = [
133
92
  cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
134
93
  contextWindow: 262144,
135
94
  maxTokens: 32768,
95
+ compat: { supportsDeveloperRole: false },
96
+ },
97
+ {
98
+ id: "deepseek-v4-pro",
99
+ name: "DeepSeek V4 Pro",
100
+ reasoning: true,
101
+ input: ["text"],
102
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
103
+ contextWindow: 1000000,
104
+ maxTokens: 32768,
105
+ compat: { supportsDeveloperRole: false },
106
+ },
107
+ {
108
+ id: "qwen3.5",
109
+ name: "Qwen 3.5",
110
+ reasoning: true,
111
+ input: ["text"],
112
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
113
+ contextWindow: 131072,
114
+ maxTokens: 32768,
115
+ compat: { supportsDeveloperRole: false },
116
+ },
117
+ {
118
+ id: "kimi-k2.6",
119
+ name: "Kimi K2.6",
120
+ reasoning: true,
121
+ input: ["text"],
122
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
123
+ contextWindow: 131072,
124
+ maxTokens: 32768,
125
+ compat: { supportsDeveloperRole: false },
136
126
  },
137
127
  ];
138
128
 
139
129
  // --- Cache I/O ---
140
- export function readCache(): Record<string, OllamaShowResponse> | null {
130
+ type CacheState =
131
+ | { status: "fresh"; models: Record<string, CachedOllamaModel> }
132
+ | { status: "stale"; models: Record<string, CachedOllamaModel> }
133
+ | { status: "missing" };
134
+
135
+ function createCacheData(models: Record<string, CachedOllamaModel>, now = new Date()): CachedData {
136
+ return { timestamp: now.getTime(), models };
137
+ }
138
+
139
+ function readCacheData(path: string): CachedData | null {
141
140
  try {
142
- if (!existsSync(CACHE_FILE)) return null;
143
- const data: CachedData = JSON.parse(readFileSync(CACHE_FILE, "utf-8"));
141
+ const data: CachedData = JSON.parse(readFileSync(path, "utf-8"));
144
142
  if (!data.models || Object.keys(data.models).length === 0) return null;
145
- return data.models;
143
+ return data;
146
144
  } catch {
147
145
  return null;
148
146
  }
149
147
  }
150
148
 
151
- export function writeCache(models: Record<string, OllamaShowResponse>): void {
149
+ function isFreshGeneratedCache(data: CachedData): boolean {
150
+ if (typeof data.timestamp !== "number" || !Number.isFinite(data.timestamp)) return false;
151
+ return Date.now() - data.timestamp <= CACHE_MAX_AGE_MS;
152
+ }
153
+
154
+ export function readCacheState(): CacheState {
155
+ if (!existsSync(CACHE_FILE)) return { status: "missing" };
156
+
157
+ const data = readCacheData(CACHE_FILE);
158
+ if (!data) {
159
+ try {
160
+ rmSync(CACHE_FILE, { force: true });
161
+ } catch {
162
+ // Ignore cache delete errors.
163
+ }
164
+ return { status: "missing" };
165
+ }
166
+
167
+ return isFreshGeneratedCache(data)
168
+ ? { status: "fresh", models: data.models }
169
+ : { status: "stale", models: data.models };
170
+ }
171
+
172
+ export function writeCache(models: Record<string, CachedOllamaModel>): void {
152
173
  try {
153
174
  mkdirSync(CACHE_DIR, { recursive: true });
154
- writeFileSync(CACHE_FILE, JSON.stringify({ timestamp: Date.now(), models } satisfies CachedData, null, 2));
175
+ writeFileSync(CACHE_FILE, JSON.stringify(createCacheData(models), null, 2));
155
176
  } catch {
156
177
  // Ignore cache write errors
157
178
  }
158
179
  }
159
180
 
160
181
  // --- Fetch Models ---
161
- export async function fetchModels(ctx: ExtensionCommandContext): Promise<Record<string, OllamaShowResponse> | null> {
162
- const apiKey = (await authStorage.getApiKey("ollama-cloud")) ?? process.env.OLLAMA_API_KEY;
163
-
164
- if (!apiKey) {
165
- ctx.ui.notify(
166
- "No Ollama Cloud API key found. \n" +
167
- "Please ensure your API key is set in: \n" +
168
- "- auth.json file (at ~/.pi/agent/auth.json) under 'ollama-cloud' key,\n" +
169
- "- or via the CLI --api-key flag.\n" +
170
- "Example auth.json entry: \n" +
171
- '{ "ollama-cloud": { "type": "api_key", "key": "YOUR_API_KEY" } }',
172
- "error",
173
- );
174
- return null;
175
- }
182
+ async function fetchModelIds(apiKey: string, timeoutMs = FETCH_TIMEOUT_MS): Promise<string[]> {
183
+ const res = await fetchJsonWithTimeout<{ data: { id: string }[] }>(
184
+ `${OLLAMA_BASE}/v1/models`,
185
+ { headers: { Authorization: `Bearer ${apiKey}` } },
186
+ timeoutMs,
187
+ );
188
+ if (!res.ok || !res.data)
189
+ throw new Error(`Failed to fetch model list: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
190
+ return res.data.data.map((m) => m.id);
191
+ }
176
192
 
177
- // 1. Fetch model list from /v1/models
178
- let modelIds: string[];
179
- const listController = new AbortController();
180
- const listTimeout = setTimeout(() => listController.abort(), FETCH_TIMEOUT_MS);
181
- try {
182
- const res = await fetch(`${OLLAMA_BASE}/v1/models`, {
183
- headers: { Authorization: `Bearer ${apiKey}` },
184
- signal: listController.signal,
185
- });
186
- if (!res.ok) {
187
- ctx.ui.notify(`Failed to fetch model list: ${res.status}`, "error");
188
- return null;
193
+ async function fetchModelDetails(apiKey: string, id: string, timeoutMs = FETCH_TIMEOUT_MS): Promise<CachedOllamaModel> {
194
+ const res = await fetchJsonWithTimeout<OllamaShowResponse>(
195
+ `${OLLAMA_BASE}/api/show`,
196
+ {
197
+ method: "POST",
198
+ headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json" },
199
+ body: JSON.stringify({ model: id }),
200
+ },
201
+ timeoutMs,
202
+ );
203
+ if (!res.ok || !res.data)
204
+ throw new Error(`Failed to fetch /api/show for ${id}: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
205
+ return res.data;
206
+ }
207
+
208
+ async function refreshOllamaCloudModels(params: {
209
+ apiKey: string;
210
+ notify?: (message: string, level?: "info" | "error") => void;
211
+ onProgress?: (progress: RefreshProgress) => void;
212
+ workers?: number;
213
+ }): Promise<Record<string, CachedOllamaModel>> {
214
+ const notify = params.notify ?? (() => undefined);
215
+ const onProgress = params.onProgress ?? (() => undefined);
216
+ onProgress({ stage: "list", message: "Fetching model list..." });
217
+ const modelIds = await fetchModelIds(params.apiKey);
218
+ notify(`Found ${modelIds.length} models, fetching details...`);
219
+ onProgress({ stage: "details", current: 0, total: modelIds.length, failed: 0, message: "Fetching model details" });
220
+
221
+ let detailsDone = 0;
222
+ let detailsFailed = 0;
223
+ const detailResults = await concurrentMap(modelIds, params.workers ?? 8, async (id) => {
224
+ try {
225
+ return [id, await fetchModelDetails(params.apiKey, id)] as const;
226
+ } catch (error) {
227
+ detailsFailed++;
228
+ throw error;
229
+ } finally {
230
+ detailsDone++;
231
+ onProgress({
232
+ stage: "details",
233
+ current: detailsDone,
234
+ total: modelIds.length,
235
+ failed: detailsFailed,
236
+ message: "Fetching model details",
237
+ });
238
+ }
239
+ });
240
+ const models: Record<string, CachedOllamaModel> = {};
241
+ for (const result of detailResults) {
242
+ if (result.status === "fulfilled") {
243
+ const [id, data] = result.value;
244
+ models[id] = data;
189
245
  }
190
- const data = (await res.json()) as { data: { id: string }[] };
191
- modelIds = data.data.map((m) => m.id);
192
- ctx.ui.notify(`Found ${modelIds.length} models, fetching details...`);
193
- } catch {
194
- ctx.ui.notify("Failed to fetch Ollama Cloud models", "error");
195
- return null;
196
- } finally {
197
- clearTimeout(listTimeout);
198
246
  }
247
+ const succeeded = Object.keys(models).length;
248
+ if (succeeded === 0)
249
+ throw new Error(`Failed to fetch model details${detailsFailed ? ` (${detailsFailed} failed)` : ""}`);
250
+ notify(`Fetched ${succeeded} model details${detailsFailed ? ` (${detailsFailed} failed)` : ""}`, "info");
199
251
 
200
- // 2. Fetch /api/show for each model in parallel
201
- const results: Record<string, OllamaShowResponse> = {};
202
- await Promise.allSettled(
203
- modelIds.map(async (id) => {
204
- const showController = new AbortController();
205
- const showTimeout = setTimeout(() => showController.abort(), FETCH_TIMEOUT_MS);
206
- try {
207
- const res = await fetch(`${OLLAMA_BASE}/api/show`, {
208
- method: "POST",
209
- headers: {
210
- Authorization: `Bearer ${apiKey}`,
211
- "Content-Type": "application/json",
212
- },
213
- body: JSON.stringify({ model: id }),
214
- signal: showController.signal,
215
- });
216
- if (!res.ok) throw new Error(`status ${res.status}`);
217
- const showData = (await res.json()) as OllamaShowResponse;
218
- results[id] = showData;
219
- } finally {
220
- clearTimeout(showTimeout);
221
- }
222
- }),
223
- );
252
+ onProgress({
253
+ stage: "done",
254
+ current: Object.keys(models).length,
255
+ total: Object.keys(models).length,
256
+ message: "Done",
257
+ });
258
+ return models;
259
+ }
260
+
261
+ async function getOllamaCloudApiKey(): Promise<string | undefined> {
262
+ return (await authStorage.getApiKey("ollama-cloud")) ?? process.env.OLLAMA_API_KEY;
263
+ }
224
264
 
225
- const succeeded = Object.keys(results).length;
226
- const failed = modelIds.length - succeeded;
227
- if (succeeded === 0) {
228
- ctx.ui.notify(`Failed to fetch model details${failed ? ` (${failed} failed)` : ""}`, "error");
265
+ async function refreshModelsFromAuth(
266
+ params: {
267
+ notify?: (message: string, level?: "info" | "error") => void;
268
+ onProgress?: (progress: RefreshProgress) => void;
269
+ } = {},
270
+ ): Promise<Record<string, CachedOllamaModel> | null> {
271
+ const apiKey = await getOllamaCloudApiKey();
272
+ if (!apiKey) return null;
273
+
274
+ return refreshOllamaCloudModels({
275
+ apiKey,
276
+ notify: params.notify,
277
+ onProgress: params.onProgress,
278
+ });
279
+ }
280
+
281
+ export async function fetchModels(
282
+ ctx: Pick<ExtensionCommandContext, "ui">,
283
+ onProgress?: (progress: RefreshProgress) => void,
284
+ ): Promise<Record<string, CachedOllamaModel> | null> {
285
+ try {
286
+ const result = await refreshModelsFromAuth({
287
+ notify: (message, level) => ctx.ui.notify(message, level),
288
+ onProgress,
289
+ });
290
+ if (!result) {
291
+ ctx.ui.notify(
292
+ "No Ollama Cloud API key found. \n" +
293
+ "Please ensure your API key is set in either: \n" +
294
+ "- OLLAMA_API_KEY environment variable,\n" +
295
+ "- auth.json file (at ~/.pi/agent/auth.json) under 'ollama-cloud' key,\n" +
296
+ "- or via the CLI --api-key flag.\n" +
297
+ "Example auth.json entry: \n" +
298
+ '{ "ollama-cloud": { "type": "api_key", "key": "YOUR_API_KEY" } }',
299
+ "error",
300
+ );
301
+ }
302
+ return result;
303
+ } catch (error) {
304
+ ctx.ui.notify(error instanceof Error ? error.message : String(error), "error");
229
305
  return null;
230
306
  }
231
- ctx.ui.notify(`Fetched ${succeeded} model details${failed ? ` (${failed} failed)` : ""}`, "info");
232
-
233
- return results;
234
307
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-ollama-cloud",
3
- "version": "0.3.1",
3
+ "version": "0.4.1",
4
4
  "type": "module",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -8,6 +8,8 @@
8
8
  "files": [
9
9
  "index.ts",
10
10
  "models.ts",
11
+ "thinking-levels.ts",
12
+ "utils.ts",
11
13
  "web-tools.ts",
12
14
  "CHANGELOG.md",
13
15
  "README.md",
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Thinking level mapping for Ollama Cloud models.
3
+ *
4
+ * Maps Pi's thinking levels to Ollama Cloud's OpenAI-compatible
5
+ * `reasoning_effort` values. The API accepts "none", "low", "medium",
6
+ * "high", and "max". On simple prompts, "max" can be a no-op over
7
+ * "high", but on harder prompts it can increase thinking substantially
8
+ * (e.g. deepseek-v4-pro: ~32k tokens on high vs ~55k on max).
9
+ *
10
+ * A `null` value means the level is hidden in Pi's UI.
11
+ *
12
+ * Model-specific behavior discovered through testing (see docs/think-experiment.md):
13
+ * - Most models: all levels work, "none" disables thinking
14
+ * - GPT-OSS: no off mode, only low/medium/high
15
+ * - Qwen 3.x (non-VL): binary-only (think/nothink) - off works
16
+ * - Qwen 3 VL: "none" doesn't disable thinking - off is hidden
17
+ * - Kimi K2 Thinking: "none" doesn't disable thinking - off is hidden
18
+ * - MiniMax M2.x: "none" doesn't disable thinking - off is hidden
19
+ *
20
+ * Reference: https://docs.ollama.com/api/openai-compatibility
21
+ */
22
+
23
+ import type { ProviderModelConfig } from "@mariozechner/pi-coding-agent";
24
+
25
+ export type ThinkingLevelMap = NonNullable<ProviderModelConfig["thinkingLevelMap"]>;
26
+
27
+ /** Default: off/low/medium/high/xhigh with minimal hidden. */
28
+ export const DEFAULT: ThinkingLevelMap = {
29
+ off: "none",
30
+ minimal: null,
31
+ low: "low",
32
+ medium: "medium",
33
+ high: "high",
34
+ xhigh: "max",
35
+ };
36
+
37
+ /** GPT-OSS: can't disable thinking, only low/medium/high.
38
+ * https://ollama.com/library/gpt-oss */
39
+ export const GPT_OSS: ThinkingLevelMap = {
40
+ off: null,
41
+ minimal: null,
42
+ low: "low",
43
+ medium: "medium",
44
+ high: "high",
45
+ xhigh: null,
46
+ };
47
+
48
+ /** Qwen 3.x: binary-only (think/nothink), no gradation.
49
+ * https://docs.ollama.com/capabilities/thinking */
50
+ export const QWEN3: ThinkingLevelMap = {
51
+ off: "none",
52
+ minimal: null,
53
+ low: null,
54
+ medium: "medium",
55
+ high: null,
56
+ xhigh: null,
57
+ };
58
+
59
+ /** "none" doesn't disable thinking - off is hidden.
60
+ * Used by kimi and minimax families. */
61
+ export const NO_OFF: ThinkingLevelMap = {
62
+ off: null,
63
+ minimal: null,
64
+ low: "low",
65
+ medium: "medium",
66
+ high: "high",
67
+ xhigh: "max",
68
+ };
69
+
70
+ /**
71
+ * Resolve the thinking level map for a model.
72
+ * Matches by model ID prefix (case-sensitive, checks first chars).
73
+ */
74
+ export function resolve(id: string, capabilities: string[]): ThinkingLevelMap | undefined {
75
+ if (!capabilities.includes("thinking")) return undefined;
76
+
77
+ if (id.startsWith("gpt-oss")) return GPT_OSS;
78
+ if (id.startsWith("qwen3-vl")) return NO_OFF;
79
+ if (id.startsWith("qwen3")) return QWEN3;
80
+ if (id === "kimi-k2-thinking") return NO_OFF;
81
+ if (id.startsWith("minimax")) return NO_OFF;
82
+
83
+ return DEFAULT;
84
+ }
package/utils.ts ADDED
@@ -0,0 +1,60 @@
1
+ export async function fetchJsonWithTimeout<T>(
2
+ url: string,
3
+ init: RequestInit,
4
+ timeoutMs: number,
5
+ ): Promise<{ ok: boolean; status: number; data: T | null; error?: string }> {
6
+ const controller = new AbortController();
7
+ const timeout = setTimeout(() => controller.abort(), timeoutMs);
8
+ try {
9
+ const res = await fetch(url, { ...init, signal: controller.signal });
10
+ const text = await res.text();
11
+ let data: T | null = null;
12
+ try {
13
+ data = text ? (JSON.parse(text) as T) : null;
14
+ } catch {
15
+ // Keep data null and report text below.
16
+ }
17
+ const error =
18
+ data && typeof data === "object" && "error" in data
19
+ ? typeof (data as { error: unknown }).error === "object"
20
+ ? JSON.stringify((data as { error: unknown }).error)
21
+ : String((data as { error: unknown }).error)
22
+ : text;
23
+ return { ok: res.ok, status: res.status, data, error: res.ok ? undefined : error };
24
+ } catch (error) {
25
+ return { ok: false, status: 0, data: null, error: error instanceof Error ? error.message : String(error) };
26
+ } finally {
27
+ clearTimeout(timeout);
28
+ }
29
+ }
30
+
31
+ export async function concurrentMap<T, R>(
32
+ items: T[],
33
+ workers: number,
34
+ fn: (item: T) => Promise<R>,
35
+ ): Promise<PromiseSettledResult<R>[]> {
36
+ const results: PromiseSettledResult<R>[] = new Array(items.length);
37
+ let next = 0;
38
+ await Promise.all(
39
+ Array.from({ length: Math.max(1, workers) }, async () => {
40
+ while (next < items.length) {
41
+ const index = next++;
42
+ try {
43
+ results[index] = { status: "fulfilled", value: await fn(items[index]) };
44
+ } catch (reason) {
45
+ results[index] = { status: "rejected", reason };
46
+ }
47
+ }
48
+ }),
49
+ );
50
+ return results;
51
+ }
52
+
53
+ export function getContextLength(modelInfo: Record<string, unknown>): number {
54
+ for (const [key, value] of Object.entries(modelInfo)) {
55
+ if (key.endsWith(".context_length") && typeof value === "number") {
56
+ return value;
57
+ }
58
+ }
59
+ return 128000;
60
+ }
package/web-tools.ts CHANGED
@@ -162,6 +162,10 @@ export function registerWebSearchTool(pi: ExtensionAPI) {
162
162
  };
163
163
  }
164
164
  },
165
+ renderCall(args, theme, _context) {
166
+ const display = args.query ? `ollama_web_search("${args.query}")` : "ollama_web_search";
167
+ return new Text(theme.fg("toolTitle", display), 0, 0);
168
+ },
165
169
  renderResult: createRenderResult(),
166
170
  });
167
171
  }
@@ -222,6 +226,10 @@ export function registerWebFetchTool(pi: ExtensionAPI) {
222
226
  };
223
227
  }
224
228
  },
229
+ renderCall(args, theme, _context) {
230
+ const display = args.url ? `ollama_web_fetch("${args.url}")` : "ollama_web_fetch";
231
+ return new Text(theme.fg("toolTitle", display), 0, 0);
232
+ },
225
233
  renderResult: createRenderResult(),
226
234
  });
227
235
  }