pi-ollama-cloud 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -1
- package/README.md +113 -37
- package/config.ts +4 -0
- package/index.ts +163 -120
- package/models.generated.ts +56 -56
- package/models.ts +162 -135
- package/package.json +8 -3
- package/pricing.generated.ts +6 -6
- package/usage.ts +170 -0
- package/utils.ts +36 -0
- package/web-tools.ts +42 -102
package/models.generated.ts
CHANGED
|
@@ -1,14 +1,54 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-models.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
3
|
+
// Generated: 2026-08-08T04:19:46.959Z
|
|
4
4
|
// Model count: 18
|
|
5
5
|
|
|
6
6
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
7
7
|
|
|
8
8
|
export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
9
9
|
{
|
|
10
|
-
id: "deepseek-v4-flash",
|
|
11
|
-
name: "deepseek-v4-flash",
|
|
10
|
+
id: "deepseek-v4-flash:0731",
|
|
11
|
+
name: "deepseek-v4-flash:0731",
|
|
12
|
+
compat: {
|
|
13
|
+
maxTokensField: "max_tokens",
|
|
14
|
+
openRouterRouting: {},
|
|
15
|
+
requiresAssistantAfterToolResult: false,
|
|
16
|
+
requiresReasoningContentOnAssistantMessages: false,
|
|
17
|
+
requiresThinkingAsText: false,
|
|
18
|
+
requiresToolResultName: false,
|
|
19
|
+
sendSessionAffinityHeaders: false,
|
|
20
|
+
supportsDeveloperRole: false,
|
|
21
|
+
supportsLongCacheRetention: false,
|
|
22
|
+
supportsReasoningEffort: true,
|
|
23
|
+
supportsStore: false,
|
|
24
|
+
supportsStrictMode: false,
|
|
25
|
+
supportsUsageInStreaming: true,
|
|
26
|
+
thinkingFormat: "openai",
|
|
27
|
+
vercelGatewayRouting: {},
|
|
28
|
+
zaiToolStream: false,
|
|
29
|
+
},
|
|
30
|
+
contextWindow: 1048576,
|
|
31
|
+
cost: {
|
|
32
|
+
cacheRead: 0.0028,
|
|
33
|
+
cacheWrite: 0,
|
|
34
|
+
input: 0.14,
|
|
35
|
+
output: 0.28,
|
|
36
|
+
},
|
|
37
|
+
input: ["text"],
|
|
38
|
+
maxTokens: 32768,
|
|
39
|
+
reasoning: true,
|
|
40
|
+
thinkingLevelMap: {
|
|
41
|
+
high: "high",
|
|
42
|
+
low: "low",
|
|
43
|
+
medium: "medium",
|
|
44
|
+
minimal: null,
|
|
45
|
+
off: "none",
|
|
46
|
+
xhigh: "max",
|
|
47
|
+
},
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
id: "deepseek-v4-flash:preview",
|
|
51
|
+
name: "deepseek-v4-flash:preview",
|
|
12
52
|
compat: {
|
|
13
53
|
maxTokensField: "max_tokens",
|
|
14
54
|
openRouterRouting: {},
|
|
@@ -109,10 +149,10 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
109
149
|
},
|
|
110
150
|
contextWindow: 262144,
|
|
111
151
|
cost: {
|
|
112
|
-
cacheRead: 0.
|
|
152
|
+
cacheRead: 0.1,
|
|
113
153
|
cacheWrite: 0,
|
|
114
|
-
input: 0.
|
|
115
|
-
output: 0.
|
|
154
|
+
input: 0.1,
|
|
155
|
+
output: 0.34,
|
|
116
156
|
},
|
|
117
157
|
input: ["text", "image"],
|
|
118
158
|
maxTokens: 32768,
|
|
@@ -286,46 +326,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
286
326
|
xhigh: null,
|
|
287
327
|
},
|
|
288
328
|
},
|
|
289
|
-
{
|
|
290
|
-
id: "kimi-k2.5",
|
|
291
|
-
name: "kimi-k2.5",
|
|
292
|
-
compat: {
|
|
293
|
-
maxTokensField: "max_tokens",
|
|
294
|
-
openRouterRouting: {},
|
|
295
|
-
requiresAssistantAfterToolResult: false,
|
|
296
|
-
requiresReasoningContentOnAssistantMessages: false,
|
|
297
|
-
requiresThinkingAsText: false,
|
|
298
|
-
requiresToolResultName: false,
|
|
299
|
-
sendSessionAffinityHeaders: false,
|
|
300
|
-
supportsDeveloperRole: false,
|
|
301
|
-
supportsLongCacheRetention: false,
|
|
302
|
-
supportsReasoningEffort: true,
|
|
303
|
-
supportsStore: false,
|
|
304
|
-
supportsStrictMode: false,
|
|
305
|
-
supportsUsageInStreaming: true,
|
|
306
|
-
thinkingFormat: "openai",
|
|
307
|
-
vercelGatewayRouting: {},
|
|
308
|
-
zaiToolStream: false,
|
|
309
|
-
},
|
|
310
|
-
contextWindow: 262144,
|
|
311
|
-
cost: {
|
|
312
|
-
cacheRead: 0.1,
|
|
313
|
-
cacheWrite: 0,
|
|
314
|
-
input: 0.6,
|
|
315
|
-
output: 3,
|
|
316
|
-
},
|
|
317
|
-
input: ["text", "image"],
|
|
318
|
-
maxTokens: 32768,
|
|
319
|
-
reasoning: true,
|
|
320
|
-
thinkingLevelMap: {
|
|
321
|
-
high: "high",
|
|
322
|
-
low: "low",
|
|
323
|
-
medium: "medium",
|
|
324
|
-
minimal: null,
|
|
325
|
-
off: "none",
|
|
326
|
-
xhigh: "max",
|
|
327
|
-
},
|
|
328
|
-
},
|
|
329
329
|
{
|
|
330
330
|
id: "kimi-k2.6",
|
|
331
331
|
name: "kimi-k2.6",
|
|
@@ -407,8 +407,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
407
407
|
},
|
|
408
408
|
},
|
|
409
409
|
{
|
|
410
|
-
id: "
|
|
411
|
-
name: "
|
|
410
|
+
id: "kimi-k3",
|
|
411
|
+
name: "kimi-k3",
|
|
412
412
|
compat: {
|
|
413
413
|
maxTokensField: "max_tokens",
|
|
414
414
|
openRouterRouting: {},
|
|
@@ -427,14 +427,14 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
427
427
|
vercelGatewayRouting: {},
|
|
428
428
|
zaiToolStream: false,
|
|
429
429
|
},
|
|
430
|
-
contextWindow:
|
|
430
|
+
contextWindow: 1048576,
|
|
431
431
|
cost: {
|
|
432
|
-
cacheRead: 0.
|
|
433
|
-
cacheWrite: 0
|
|
434
|
-
input:
|
|
435
|
-
output:
|
|
432
|
+
cacheRead: 0.3,
|
|
433
|
+
cacheWrite: 0,
|
|
434
|
+
input: 3,
|
|
435
|
+
output: 15,
|
|
436
436
|
},
|
|
437
|
-
input: ["text"],
|
|
437
|
+
input: ["text", "image"],
|
|
438
438
|
maxTokens: 32768,
|
|
439
439
|
reasoning: true,
|
|
440
440
|
thinkingLevelMap: {
|
|
@@ -442,7 +442,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
442
442
|
low: "low",
|
|
443
443
|
medium: "medium",
|
|
444
444
|
minimal: null,
|
|
445
|
-
off:
|
|
445
|
+
off: "none",
|
|
446
446
|
xhigh: "max",
|
|
447
447
|
},
|
|
448
448
|
},
|
|
@@ -581,7 +581,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
|
|
|
581
581
|
},
|
|
582
582
|
contextWindow: 262144,
|
|
583
583
|
cost: {
|
|
584
|
-
cacheRead: 0,
|
|
584
|
+
cacheRead: 0.03,
|
|
585
585
|
cacheWrite: 0,
|
|
586
586
|
input: 0.05,
|
|
587
587
|
output: 0.2,
|
package/models.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import {
|
|
1
|
+
import type { RefreshModelsContext } from "@earendil-works/pi-ai";
|
|
2
|
+
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { GENERATED_MODELS } from "./models.generated.ts";
|
|
4
4
|
import { MODEL_PRICING, type ModelPrice } from "./pricing.generated.ts";
|
|
5
5
|
import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
|
|
6
6
|
import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
|
|
@@ -18,10 +18,10 @@ function resolvePrice(id: string): ModelPrice {
|
|
|
18
18
|
}
|
|
19
19
|
|
|
20
20
|
// --- Constants ---
|
|
21
|
-
const CACHE_DIR = join(getAgentDir(), "cache");
|
|
22
|
-
const CACHE_FILE = join(CACHE_DIR, "ollama-cloud-models.json");
|
|
23
|
-
const CACHE_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000;
|
|
24
21
|
const FETCH_TIMEOUT_MS = 10000;
|
|
22
|
+
// How long a stored catalog is considered fresh before the next network refresh
|
|
23
|
+
// (mirrors pi-mono's REMOTE_CATALOG_REFRESH_INTERVAL_MS in remote-catalog-provider.ts).
|
|
24
|
+
const REFRESH_COOLDOWN_MS = 4 * 60 * 60 * 1000;
|
|
25
25
|
|
|
26
26
|
// The cloud extension always targets ollama.com; local Ollama daemons (typically
|
|
27
27
|
// pointed at via OLLAMA_API_BASE for the local CLI) are a different product and
|
|
@@ -67,25 +67,6 @@ interface OllamaShowResponse {
|
|
|
67
67
|
modified_at: string;
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
-
type CachedOllamaModel = OllamaShowResponse;
|
|
71
|
-
|
|
72
|
-
/** On-disk cache: raw /api/show responses keyed by model ID. */
|
|
73
|
-
interface CachedData {
|
|
74
|
-
/** Unix epoch milliseconds used to decide when the generated metadata is stale. */
|
|
75
|
-
timestamp?: number;
|
|
76
|
-
models: Record<string, CachedOllamaModel>;
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
type RefreshProgressStage = "list" | "details" | "done";
|
|
80
|
-
|
|
81
|
-
export interface RefreshProgress {
|
|
82
|
-
stage: RefreshProgressStage;
|
|
83
|
-
current?: number;
|
|
84
|
-
total?: number;
|
|
85
|
-
failed?: number;
|
|
86
|
-
message: string;
|
|
87
|
-
}
|
|
88
|
-
|
|
89
70
|
// --- Assembly: raw API data -> ProviderModelConfig[] ---
|
|
90
71
|
|
|
91
72
|
/**
|
|
@@ -135,7 +116,7 @@ function buildCompat(): ProviderModelConfig["compat"] {
|
|
|
135
116
|
};
|
|
136
117
|
}
|
|
137
118
|
|
|
138
|
-
export function assembleModels(raw: Record<string,
|
|
119
|
+
export function assembleModels(raw: Record<string, OllamaShowResponse>): ProviderModelConfig[] {
|
|
139
120
|
return Object.entries(raw)
|
|
140
121
|
.filter(([, data]) => data.capabilities?.includes("tools"))
|
|
141
122
|
.map(([id, data]) => ({
|
|
@@ -153,60 +134,8 @@ export function assembleModels(raw: Record<string, CachedOllamaModel>): Provider
|
|
|
153
134
|
}));
|
|
154
135
|
}
|
|
155
136
|
|
|
156
|
-
// --- Cache I/O ---
|
|
157
|
-
type CacheState =
|
|
158
|
-
| { status: "fresh"; models: Record<string, CachedOllamaModel> }
|
|
159
|
-
| { status: "stale"; models: Record<string, CachedOllamaModel> }
|
|
160
|
-
| { status: "missing" };
|
|
161
|
-
|
|
162
|
-
function createCacheData(models: Record<string, CachedOllamaModel>, now = new Date()): CachedData {
|
|
163
|
-
return { timestamp: now.getTime(), models };
|
|
164
|
-
}
|
|
165
|
-
|
|
166
|
-
function readCacheData(path: string): CachedData | null {
|
|
167
|
-
try {
|
|
168
|
-
const data: CachedData = JSON.parse(readFileSync(path, "utf-8"));
|
|
169
|
-
if (!data.models || Object.keys(data.models).length === 0) return null;
|
|
170
|
-
return data;
|
|
171
|
-
} catch {
|
|
172
|
-
return null;
|
|
173
|
-
}
|
|
174
|
-
}
|
|
175
|
-
|
|
176
|
-
function isFreshGeneratedCache(data: CachedData): boolean {
|
|
177
|
-
if (typeof data.timestamp !== "number" || !Number.isFinite(data.timestamp)) return false;
|
|
178
|
-
return Date.now() - data.timestamp <= CACHE_MAX_AGE_MS;
|
|
179
|
-
}
|
|
180
|
-
|
|
181
|
-
export function readCacheState(): CacheState {
|
|
182
|
-
if (!existsSync(CACHE_FILE)) return { status: "missing" };
|
|
183
|
-
|
|
184
|
-
const data = readCacheData(CACHE_FILE);
|
|
185
|
-
if (!data) {
|
|
186
|
-
try {
|
|
187
|
-
rmSync(CACHE_FILE, { force: true });
|
|
188
|
-
} catch {
|
|
189
|
-
// Ignore cache delete errors.
|
|
190
|
-
}
|
|
191
|
-
return { status: "missing" };
|
|
192
|
-
}
|
|
193
|
-
|
|
194
|
-
return isFreshGeneratedCache(data)
|
|
195
|
-
? { status: "fresh", models: data.models }
|
|
196
|
-
: { status: "stale", models: data.models };
|
|
197
|
-
}
|
|
198
|
-
|
|
199
|
-
export function writeCache(models: Record<string, CachedOllamaModel>): void {
|
|
200
|
-
try {
|
|
201
|
-
mkdirSync(CACHE_DIR, { recursive: true });
|
|
202
|
-
writeFileSync(CACHE_FILE, JSON.stringify(createCacheData(models), null, 2));
|
|
203
|
-
} catch {
|
|
204
|
-
// Ignore cache write errors
|
|
205
|
-
}
|
|
206
|
-
}
|
|
207
|
-
|
|
208
137
|
// --- Fetch Models ---
|
|
209
|
-
export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<string[]> {
|
|
138
|
+
export async function fetchModelIds(signal?: AbortSignal, timeoutMs = FETCH_TIMEOUT_MS): Promise<string[]> {
|
|
210
139
|
const headers: Record<string, string> = {};
|
|
211
140
|
const apiKey = process.env.OLLAMA_API_KEY;
|
|
212
141
|
if (apiKey) {
|
|
@@ -217,10 +146,11 @@ export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<strin
|
|
|
217
146
|
`${OLLAMA_BASE}/v1/models`,
|
|
218
147
|
{ headers },
|
|
219
148
|
timeoutMs,
|
|
149
|
+
signal,
|
|
220
150
|
);
|
|
221
151
|
|
|
222
152
|
if (res.status === 429) {
|
|
223
|
-
throw new Error("Ollama Cloud rate limited. Try again shortly.");
|
|
153
|
+
throw new Error("Ollama Cloud model list fetch rate limited. Try again shortly.");
|
|
224
154
|
}
|
|
225
155
|
if (!res.ok || !res.data) {
|
|
226
156
|
throw new Error(`Failed to fetch model list: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
|
|
@@ -229,7 +159,11 @@ export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<strin
|
|
|
229
159
|
return res.data.data.map((m) => m.id);
|
|
230
160
|
}
|
|
231
161
|
|
|
232
|
-
export async function fetchModelDetails(
|
|
162
|
+
export async function fetchModelDetails(
|
|
163
|
+
id: string,
|
|
164
|
+
signal?: AbortSignal,
|
|
165
|
+
timeoutMs = FETCH_TIMEOUT_MS,
|
|
166
|
+
): Promise<OllamaShowResponse> {
|
|
233
167
|
const headers: Record<string, string> = { "Content-Type": "application/json" };
|
|
234
168
|
const apiKey = process.env.OLLAMA_API_KEY;
|
|
235
169
|
if (apiKey) {
|
|
@@ -244,10 +178,11 @@ export async function fetchModelDetails(id: string, timeoutMs = FETCH_TIMEOUT_MS
|
|
|
244
178
|
body: JSON.stringify({ model: id }),
|
|
245
179
|
},
|
|
246
180
|
timeoutMs,
|
|
181
|
+
signal,
|
|
247
182
|
);
|
|
248
183
|
|
|
249
184
|
if (res.status === 429) {
|
|
250
|
-
throw new Error("Ollama Cloud rate limited. Try again shortly.");
|
|
185
|
+
throw new Error("Ollama Cloud /api/show rate limited. Try again shortly.");
|
|
251
186
|
}
|
|
252
187
|
if (!res.ok || !res.data) {
|
|
253
188
|
throw new Error(`Failed to fetch /api/show for ${id}: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
|
|
@@ -256,69 +191,161 @@ export async function fetchModelDetails(id: string, timeoutMs = FETCH_TIMEOUT_MS
|
|
|
256
191
|
return res.data;
|
|
257
192
|
}
|
|
258
193
|
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
let detailsDone = 0;
|
|
272
|
-
let detailsFailed = 0;
|
|
273
|
-
const detailResults = await concurrentMap(modelIds, params.workers ?? 8, async (id) => {
|
|
274
|
-
try {
|
|
275
|
-
return [id, await fetchModelDetails(id)] as const;
|
|
276
|
-
} catch (error) {
|
|
277
|
-
detailsFailed++;
|
|
278
|
-
throw error;
|
|
279
|
-
} finally {
|
|
280
|
-
detailsDone++;
|
|
281
|
-
onProgress({
|
|
282
|
-
stage: "details",
|
|
283
|
-
current: detailsDone,
|
|
284
|
-
total: modelIds.length,
|
|
285
|
-
failed: detailsFailed,
|
|
286
|
-
message: "Fetching model details",
|
|
287
|
-
});
|
|
288
|
-
}
|
|
194
|
+
/**
|
|
195
|
+
* Fetch per-model /api/show details for a list of model IDs, 8 workers at a time.
|
|
196
|
+
* Returns the models that succeeded. Throws when every detail request fails
|
|
197
|
+
* (the zero-succeeded case), so the caller can surface a real failure instead
|
|
198
|
+
* of an empty catalog.
|
|
199
|
+
*/
|
|
200
|
+
export async function refreshOllamaCloudModels(
|
|
201
|
+
modelIds: string[],
|
|
202
|
+
signal?: AbortSignal,
|
|
203
|
+
): Promise<{ models: Record<string, OllamaShowResponse>; failed: number }> {
|
|
204
|
+
const detailResults = await concurrentMap(modelIds, 8, async (id) => {
|
|
205
|
+
return [id, await fetchModelDetails(id, signal)] as const;
|
|
289
206
|
});
|
|
290
|
-
const models: Record<string,
|
|
207
|
+
const models: Record<string, OllamaShowResponse> = {};
|
|
208
|
+
let failed = 0;
|
|
291
209
|
for (const result of detailResults) {
|
|
292
210
|
if (result.status === "fulfilled") {
|
|
293
211
|
const [id, data] = result.value;
|
|
294
212
|
models[id] = data;
|
|
213
|
+
} else {
|
|
214
|
+
failed++;
|
|
295
215
|
}
|
|
296
216
|
}
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
onProgress({
|
|
303
|
-
stage: "done",
|
|
304
|
-
current: Object.keys(models).length,
|
|
305
|
-
total: Object.keys(models).length,
|
|
306
|
-
message: "Done",
|
|
307
|
-
});
|
|
308
|
-
return models;
|
|
217
|
+
if (Object.keys(models).length === 0) {
|
|
218
|
+
throw new Error(`Failed to fetch model details${failed ? ` (${failed} failed)` : ""}`);
|
|
219
|
+
}
|
|
220
|
+
return { models, failed };
|
|
309
221
|
}
|
|
310
222
|
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
223
|
+
// --- refreshModels callback ---
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* The `refreshModels` callback pi invokes for the "ollama-cloud" provider.
|
|
227
|
+
* Pi calls it twice per refresh: a restore phase (`allowNetwork: false`) before
|
|
228
|
+
* auth resolution, then a network phase (`allowNetwork: true`) only when a
|
|
229
|
+
* credential resolves. The composer swaps the return value into the model list
|
|
230
|
+
* on every invocation, so this must never return `[]`.
|
|
231
|
+
*
|
|
232
|
+
* The model fetch itself is keyless (public `/v1/models` and `/api/show`
|
|
233
|
+
* endpoints; `Authorization` is only added when `OLLAMA_API_KEY` is set). But
|
|
234
|
+
* pi only invokes this network phase when a credential resolves, so a
|
|
235
|
+
* credentialless user stays on `GENERATED_MODELS` until they configure a key.
|
|
236
|
+
* That is a non-issue in practice because a credentialless user cannot run
|
|
237
|
+
* models anyway.
|
|
238
|
+
*/
|
|
239
|
+
export async function refreshOllamaCatalog(context: RefreshModelsContext): Promise<ProviderModelConfig[]> {
|
|
240
|
+
// The fallback list: the persisted snapshot (copied) when non-empty, else the
|
|
241
|
+
// baked-in list. Guards against a stored empty catalog (e.g. a prior bad
|
|
242
|
+
// refresh) propagating [] across sessions. A mutable copy is returned because
|
|
243
|
+
// the stored list is `readonly` and the return type is a mutable array.
|
|
244
|
+
const fallback = context.stored?.models.length ? [...context.stored.models] : GENERATED_MODELS;
|
|
245
|
+
|
|
246
|
+
// Restore phase. Rehydrate from the persisted snapshot so removals stick
|
|
247
|
+
// across sessions; fall back to the baked-in list on first launch. Also the
|
|
248
|
+
// early-out for an already-aborted signal.
|
|
249
|
+
if (!context.allowNetwork || context.signal.aborted) {
|
|
250
|
+
return fallback;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Cooldown: skip the network fetch when the stored catalog was checked within
|
|
254
|
+
// the freshness window and the refresh isn't forced (mirrors pi-mono's
|
|
255
|
+
// remote-catalog-provider). A forced refresh (pi update --models) always fetches.
|
|
256
|
+
if (
|
|
257
|
+
!context.force &&
|
|
258
|
+
context.stored?.checkedAt !== undefined &&
|
|
259
|
+
Date.now() - context.stored.checkedAt < REFRESH_COOLDOWN_MS
|
|
260
|
+
) {
|
|
261
|
+
return fallback;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// Network phase. The /v1/models and /api/show endpoints are publicly
|
|
265
|
+
// accessible and do not require authentication, so context.credential is
|
|
266
|
+
// intentionally not threaded into fetchModelIds/fetchModelDetails. Only
|
|
267
|
+
// the web tools (search, fetch) require an API key.
|
|
268
|
+
//
|
|
269
|
+
// Pi's model-selector aborts a catalog refresh after 15s
|
|
270
|
+
// (packages/coding-agent/src/modes/interactive/components/model-selector.ts
|
|
271
|
+
// in pi-mono), so a cold refresh must stay under that budget or the in-memory
|
|
272
|
+
// list won't update on the first picker-open. With ~18 models and 8 workers
|
|
273
|
+
// this is ~1s today; revisit if the catalog grows or the API slows.
|
|
274
|
+
let modelIds: string[];
|
|
275
|
+
let raw: Record<string, OllamaShowResponse>;
|
|
276
|
+
let failed = 0;
|
|
315
277
|
try {
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
}
|
|
278
|
+
modelIds = await fetchModelIds(context.signal);
|
|
279
|
+
if (context.signal.aborted) {
|
|
280
|
+
return fallback;
|
|
281
|
+
}
|
|
282
|
+
const result = await refreshOllamaCloudModels(modelIds, context.signal);
|
|
283
|
+
raw = result.models;
|
|
284
|
+
failed = result.failed;
|
|
320
285
|
} catch (error) {
|
|
321
|
-
|
|
322
|
-
|
|
286
|
+
// Abort mid-flight returns the current baseline; any other error propagates
|
|
287
|
+
// and pi keeps the last good catalog (no publish was reached).
|
|
288
|
+
if (context.signal.aborted) {
|
|
289
|
+
return fallback;
|
|
290
|
+
}
|
|
291
|
+
throw error;
|
|
292
|
+
}
|
|
293
|
+
if (context.signal.aborted) {
|
|
294
|
+
return fallback;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
const models = assembleModels(raw);
|
|
298
|
+
// Guard: an empty assembled list (e.g. the live API returned no tools-capable
|
|
299
|
+
// models) must not be persisted or swapped in, or it would kill the provider
|
|
300
|
+
// for the cooldown window. Keep the last good catalog instead.
|
|
301
|
+
if (models.length === 0) {
|
|
302
|
+
return fallback;
|
|
323
303
|
}
|
|
304
|
+
|
|
305
|
+
// The store is typed to pi-ai's internal Model shape, so rehydrate the live
|
|
306
|
+
// list with the provider identity fields. A Model is structurally assignable
|
|
307
|
+
// to ProviderModelConfig, so the same list is returned to pi (the composer
|
|
308
|
+
// overrides provider/api/baseUrl on the in-memory swap regardless).
|
|
309
|
+
const persisted = models.map((model) => ({
|
|
310
|
+
...model,
|
|
311
|
+
provider: "ollama-cloud",
|
|
312
|
+
api: "openai-completions",
|
|
313
|
+
baseUrl: `${OLLAMA_BASE}/v1`,
|
|
314
|
+
}));
|
|
315
|
+
|
|
316
|
+
// Best-effort persistence into pi's FileModelsStore. The in-memory list swap
|
|
317
|
+
// happens automatically from the return value, so a failed store write must
|
|
318
|
+
// not prevent returning the fresh catalog.
|
|
319
|
+
if (failed === 0) {
|
|
320
|
+
// Fully successful: persist the fresh catalog.
|
|
321
|
+
try {
|
|
322
|
+
const published = await context.publish({ persist: { models: persisted, checkedAt: Date.now() } });
|
|
323
|
+
if (!published) {
|
|
324
|
+
console.warn("[pi-ollama-cloud] Catalog persist rejected (generation check failed or refresh superseded).");
|
|
325
|
+
}
|
|
326
|
+
} catch {
|
|
327
|
+
// Persistence failure is non-fatal.
|
|
328
|
+
}
|
|
329
|
+
} else {
|
|
330
|
+
// Partial failure: keep the last-good catalog (if any) but advance checkedAt
|
|
331
|
+
// so the cooldown applies and a flaky catalog isn't re-fetched on every
|
|
332
|
+
// /model open, then surface the incomplete refresh. Mirrors pi-mono's
|
|
333
|
+
// remote-catalog-provider, which persists then throws on a transient failure;
|
|
334
|
+
// pi keeps the last-good catalog and reports the error.
|
|
335
|
+
if (context.stored?.models.length) {
|
|
336
|
+
try {
|
|
337
|
+
const published = await context.publish({ persist: { ...context.stored, checkedAt: Date.now() } });
|
|
338
|
+
if (!published) {
|
|
339
|
+
console.warn(
|
|
340
|
+
"[pi-ollama-cloud] Partial-failure persist rejected (generation check failed or refresh superseded).",
|
|
341
|
+
);
|
|
342
|
+
}
|
|
343
|
+
} catch {
|
|
344
|
+
// Persistence failure is non-fatal.
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
throw new Error(`Ollama Cloud catalog refresh incomplete: ${failed} model(s) failed`);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
return persisted;
|
|
324
351
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-ollama-cloud",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package"
|
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
"models.generated.ts",
|
|
13
13
|
"pricing.generated.ts",
|
|
14
14
|
"thinking-levels.ts",
|
|
15
|
+
"usage.ts",
|
|
15
16
|
"utils.ts",
|
|
16
17
|
"web-tools.ts",
|
|
17
18
|
"CHANGELOG.md",
|
|
@@ -20,11 +21,12 @@
|
|
|
20
21
|
],
|
|
21
22
|
"repository": {
|
|
22
23
|
"type": "git",
|
|
23
|
-
"url": "https://github.com/fgrehm/pi-ollama-cloud"
|
|
24
|
+
"url": "git+https://github.com/fgrehm/pi-ollama-cloud.git"
|
|
24
25
|
},
|
|
25
26
|
"scripts": {
|
|
26
|
-
"check": "biome check --write .",
|
|
27
|
+
"check": "biome check --write . && tsgo --noEmit",
|
|
27
28
|
"lint": "biome check .",
|
|
29
|
+
"typecheck": "tsgo --noEmit",
|
|
28
30
|
"format": "biome format --write .",
|
|
29
31
|
"test": "vitest run",
|
|
30
32
|
"smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
|
|
@@ -36,12 +38,15 @@
|
|
|
36
38
|
]
|
|
37
39
|
},
|
|
38
40
|
"peerDependencies": {
|
|
41
|
+
"@earendil-works/pi-ai": "*",
|
|
39
42
|
"@earendil-works/pi-coding-agent": "*",
|
|
40
43
|
"@earendil-works/pi-tui": "*",
|
|
41
44
|
"@sinclair/typebox": "*"
|
|
42
45
|
},
|
|
43
46
|
"devDependencies": {
|
|
44
47
|
"@biomejs/biome": "2",
|
|
48
|
+
"@types/node": "^26.1.2",
|
|
49
|
+
"@typescript/native-preview": "7.0.0-dev.20260707.2",
|
|
45
50
|
"tsx": "^4.19.0",
|
|
46
51
|
"vitest": "^4.1.6"
|
|
47
52
|
}
|
package/pricing.generated.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
// Auto-generated by scripts/generate-pricing.ts
|
|
2
2
|
// Do not edit manually.
|
|
3
|
-
// Generated: 2026-
|
|
3
|
+
// Generated: 2026-08-08T04:19:45.594Z
|
|
4
4
|
// Model count: 18
|
|
5
5
|
|
|
6
6
|
export interface ModelPrice {
|
|
@@ -11,21 +11,21 @@ export interface ModelPrice {
|
|
|
11
11
|
}
|
|
12
12
|
|
|
13
13
|
export const MODEL_PRICING: Record<string, ModelPrice> = {
|
|
14
|
-
"deepseek-v4-flash": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
|
|
14
|
+
"deepseek-v4-flash:0731": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
|
|
15
|
+
"deepseek-v4-flash:preview": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
|
|
15
16
|
"deepseek-v4-pro": { input: 0.435, output: 0.87, cacheRead: 0.003625, cacheWrite: 0 },
|
|
16
|
-
"gemma4:31b": { input: 0.
|
|
17
|
+
"gemma4:31b": { input: 0.1, output: 0.34, cacheRead: 0.1, cacheWrite: 0 },
|
|
17
18
|
"glm-5.1": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
18
19
|
"glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
|
|
19
20
|
"gpt-oss:120b": { input: 0.037, output: 0.17, cacheRead: 0, cacheWrite: 0 },
|
|
20
21
|
"gpt-oss:20b": { input: 0.03, output: 0.13, cacheRead: 0.03, cacheWrite: 0 },
|
|
21
|
-
"kimi-k2.5": { input: 0.6, output: 3, cacheRead: 0.1, cacheWrite: 0 },
|
|
22
22
|
"kimi-k2.6": { input: 0.95, output: 4, cacheRead: 0.16, cacheWrite: 0 },
|
|
23
23
|
"kimi-k2.7-code": { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 },
|
|
24
|
-
"
|
|
24
|
+
"kimi-k3": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 },
|
|
25
25
|
"minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0.375 },
|
|
26
26
|
"minimax-m3": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
|
|
27
27
|
"mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0, cacheWrite: 0 },
|
|
28
|
-
"nemotron-3-nano:30b": { input: 0.05, output: 0.2, cacheRead: 0, cacheWrite: 0 },
|
|
28
|
+
"nemotron-3-nano:30b": { input: 0.05, output: 0.2, cacheRead: 0.03, cacheWrite: 0 },
|
|
29
29
|
"nemotron-3-super": { input: 0.2, output: 0.8, cacheRead: 0, cacheWrite: 0 },
|
|
30
30
|
"nemotron-3-ultra": { input: 0.5, output: 2.5, cacheRead: 0.15, cacheWrite: 0 },
|
|
31
31
|
"qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0, cacheWrite: 0 },
|