pi-ollama-cloud 0.7.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,54 @@
1
1
  // Auto-generated by scripts/generate-models.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-07-17T19:55:27.551Z
3
+ // Generated: 2026-08-08T04:19:46.959Z
4
4
  // Model count: 18
5
5
 
6
6
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
7
7
 
8
8
  export const GENERATED_MODELS: ProviderModelConfig[] = [
9
9
  {
10
- id: "deepseek-v4-flash",
11
- name: "deepseek-v4-flash",
10
+ id: "deepseek-v4-flash:0731",
11
+ name: "deepseek-v4-flash:0731",
12
+ compat: {
13
+ maxTokensField: "max_tokens",
14
+ openRouterRouting: {},
15
+ requiresAssistantAfterToolResult: false,
16
+ requiresReasoningContentOnAssistantMessages: false,
17
+ requiresThinkingAsText: false,
18
+ requiresToolResultName: false,
19
+ sendSessionAffinityHeaders: false,
20
+ supportsDeveloperRole: false,
21
+ supportsLongCacheRetention: false,
22
+ supportsReasoningEffort: true,
23
+ supportsStore: false,
24
+ supportsStrictMode: false,
25
+ supportsUsageInStreaming: true,
26
+ thinkingFormat: "openai",
27
+ vercelGatewayRouting: {},
28
+ zaiToolStream: false,
29
+ },
30
+ contextWindow: 1048576,
31
+ cost: {
32
+ cacheRead: 0.0028,
33
+ cacheWrite: 0,
34
+ input: 0.14,
35
+ output: 0.28,
36
+ },
37
+ input: ["text"],
38
+ maxTokens: 32768,
39
+ reasoning: true,
40
+ thinkingLevelMap: {
41
+ high: "high",
42
+ low: "low",
43
+ medium: "medium",
44
+ minimal: null,
45
+ off: "none",
46
+ xhigh: "max",
47
+ },
48
+ },
49
+ {
50
+ id: "deepseek-v4-flash:preview",
51
+ name: "deepseek-v4-flash:preview",
12
52
  compat: {
13
53
  maxTokensField: "max_tokens",
14
54
  openRouterRouting: {},
@@ -109,10 +149,10 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
109
149
  },
110
150
  contextWindow: 262144,
111
151
  cost: {
112
- cacheRead: 0.12,
152
+ cacheRead: 0.1,
113
153
  cacheWrite: 0,
114
- input: 0.22,
115
- output: 0.55,
154
+ input: 0.1,
155
+ output: 0.34,
116
156
  },
117
157
  input: ["text", "image"],
118
158
  maxTokens: 32768,
@@ -286,46 +326,6 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
286
326
  xhigh: null,
287
327
  },
288
328
  },
289
- {
290
- id: "kimi-k2.5",
291
- name: "kimi-k2.5",
292
- compat: {
293
- maxTokensField: "max_tokens",
294
- openRouterRouting: {},
295
- requiresAssistantAfterToolResult: false,
296
- requiresReasoningContentOnAssistantMessages: false,
297
- requiresThinkingAsText: false,
298
- requiresToolResultName: false,
299
- sendSessionAffinityHeaders: false,
300
- supportsDeveloperRole: false,
301
- supportsLongCacheRetention: false,
302
- supportsReasoningEffort: true,
303
- supportsStore: false,
304
- supportsStrictMode: false,
305
- supportsUsageInStreaming: true,
306
- thinkingFormat: "openai",
307
- vercelGatewayRouting: {},
308
- zaiToolStream: false,
309
- },
310
- contextWindow: 262144,
311
- cost: {
312
- cacheRead: 0.1,
313
- cacheWrite: 0,
314
- input: 0.6,
315
- output: 3,
316
- },
317
- input: ["text", "image"],
318
- maxTokens: 32768,
319
- reasoning: true,
320
- thinkingLevelMap: {
321
- high: "high",
322
- low: "low",
323
- medium: "medium",
324
- minimal: null,
325
- off: "none",
326
- xhigh: "max",
327
- },
328
- },
329
329
  {
330
330
  id: "kimi-k2.6",
331
331
  name: "kimi-k2.6",
@@ -407,8 +407,8 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
407
407
  },
408
408
  },
409
409
  {
410
- id: "minimax-m2.5",
411
- name: "minimax-m2.5",
410
+ id: "kimi-k3",
411
+ name: "kimi-k3",
412
412
  compat: {
413
413
  maxTokensField: "max_tokens",
414
414
  openRouterRouting: {},
@@ -427,14 +427,14 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
427
427
  vercelGatewayRouting: {},
428
428
  zaiToolStream: false,
429
429
  },
430
- contextWindow: 196608,
430
+ contextWindow: 1048576,
431
431
  cost: {
432
- cacheRead: 0.03,
433
- cacheWrite: 0.375,
434
- input: 0.3,
435
- output: 1.2,
432
+ cacheRead: 0.3,
433
+ cacheWrite: 0,
434
+ input: 3,
435
+ output: 15,
436
436
  },
437
- input: ["text"],
437
+ input: ["text", "image"],
438
438
  maxTokens: 32768,
439
439
  reasoning: true,
440
440
  thinkingLevelMap: {
@@ -442,7 +442,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
442
442
  low: "low",
443
443
  medium: "medium",
444
444
  minimal: null,
445
- off: null,
445
+ off: "none",
446
446
  xhigh: "max",
447
447
  },
448
448
  },
@@ -581,7 +581,7 @@ export const GENERATED_MODELS: ProviderModelConfig[] = [
581
581
  },
582
582
  contextWindow: 262144,
583
583
  cost: {
584
- cacheRead: 0,
584
+ cacheRead: 0.03,
585
585
  cacheWrite: 0,
586
586
  input: 0.05,
587
587
  output: 0.2,
package/models.ts CHANGED
@@ -1,6 +1,6 @@
1
- import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
2
- import { join } from "node:path";
3
- import { type ExtensionCommandContext, getAgentDir, type ProviderModelConfig } from "@earendil-works/pi-coding-agent";
1
+ import type { RefreshModelsContext } from "@earendil-works/pi-ai";
2
+ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
3
+ import { GENERATED_MODELS } from "./models.generated.ts";
4
4
  import { MODEL_PRICING, type ModelPrice } from "./pricing.generated.ts";
5
5
  import { resolve as resolveThinkingLevelMap } from "./thinking-levels.ts";
6
6
  import { concurrentMap, fetchJsonWithTimeout, getContextLength } from "./utils.ts";
@@ -18,10 +18,10 @@ function resolvePrice(id: string): ModelPrice {
18
18
  }
19
19
 
20
20
  // --- Constants ---
21
- const CACHE_DIR = join(getAgentDir(), "cache");
22
- const CACHE_FILE = join(CACHE_DIR, "ollama-cloud-models.json");
23
- const CACHE_MAX_AGE_MS = 30 * 24 * 60 * 60 * 1000;
24
21
  const FETCH_TIMEOUT_MS = 10000;
22
+ // How long a stored catalog is considered fresh before the next network refresh
23
+ // (mirrors pi-mono's REMOTE_CATALOG_REFRESH_INTERVAL_MS in remote-catalog-provider.ts).
24
+ const REFRESH_COOLDOWN_MS = 4 * 60 * 60 * 1000;
25
25
 
26
26
  // The cloud extension always targets ollama.com; local Ollama daemons (typically
27
27
  // pointed at via OLLAMA_API_BASE for the local CLI) are a different product and
@@ -67,25 +67,6 @@ interface OllamaShowResponse {
67
67
  modified_at: string;
68
68
  }
69
69
 
70
- type CachedOllamaModel = OllamaShowResponse;
71
-
72
- /** On-disk cache: raw /api/show responses keyed by model ID. */
73
- interface CachedData {
74
- /** Unix epoch milliseconds used to decide when the generated metadata is stale. */
75
- timestamp?: number;
76
- models: Record<string, CachedOllamaModel>;
77
- }
78
-
79
- type RefreshProgressStage = "list" | "details" | "done";
80
-
81
- export interface RefreshProgress {
82
- stage: RefreshProgressStage;
83
- current?: number;
84
- total?: number;
85
- failed?: number;
86
- message: string;
87
- }
88
-
89
70
  // --- Assembly: raw API data -> ProviderModelConfig[] ---
90
71
 
91
72
  /**
@@ -135,7 +116,7 @@ function buildCompat(): ProviderModelConfig["compat"] {
135
116
  };
136
117
  }
137
118
 
138
- export function assembleModels(raw: Record<string, CachedOllamaModel>): ProviderModelConfig[] {
119
+ export function assembleModels(raw: Record<string, OllamaShowResponse>): ProviderModelConfig[] {
139
120
  return Object.entries(raw)
140
121
  .filter(([, data]) => data.capabilities?.includes("tools"))
141
122
  .map(([id, data]) => ({
@@ -153,60 +134,8 @@ export function assembleModels(raw: Record<string, CachedOllamaModel>): Provider
153
134
  }));
154
135
  }
155
136
 
156
- // --- Cache I/O ---
157
- type CacheState =
158
- | { status: "fresh"; models: Record<string, CachedOllamaModel> }
159
- | { status: "stale"; models: Record<string, CachedOllamaModel> }
160
- | { status: "missing" };
161
-
162
- function createCacheData(models: Record<string, CachedOllamaModel>, now = new Date()): CachedData {
163
- return { timestamp: now.getTime(), models };
164
- }
165
-
166
- function readCacheData(path: string): CachedData | null {
167
- try {
168
- const data: CachedData = JSON.parse(readFileSync(path, "utf-8"));
169
- if (!data.models || Object.keys(data.models).length === 0) return null;
170
- return data;
171
- } catch {
172
- return null;
173
- }
174
- }
175
-
176
- function isFreshGeneratedCache(data: CachedData): boolean {
177
- if (typeof data.timestamp !== "number" || !Number.isFinite(data.timestamp)) return false;
178
- return Date.now() - data.timestamp <= CACHE_MAX_AGE_MS;
179
- }
180
-
181
- export function readCacheState(): CacheState {
182
- if (!existsSync(CACHE_FILE)) return { status: "missing" };
183
-
184
- const data = readCacheData(CACHE_FILE);
185
- if (!data) {
186
- try {
187
- rmSync(CACHE_FILE, { force: true });
188
- } catch {
189
- // Ignore cache delete errors.
190
- }
191
- return { status: "missing" };
192
- }
193
-
194
- return isFreshGeneratedCache(data)
195
- ? { status: "fresh", models: data.models }
196
- : { status: "stale", models: data.models };
197
- }
198
-
199
- export function writeCache(models: Record<string, CachedOllamaModel>): void {
200
- try {
201
- mkdirSync(CACHE_DIR, { recursive: true });
202
- writeFileSync(CACHE_FILE, JSON.stringify(createCacheData(models), null, 2));
203
- } catch {
204
- // Ignore cache write errors
205
- }
206
- }
207
-
208
137
  // --- Fetch Models ---
209
- export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<string[]> {
138
+ export async function fetchModelIds(signal?: AbortSignal, timeoutMs = FETCH_TIMEOUT_MS): Promise<string[]> {
210
139
  const headers: Record<string, string> = {};
211
140
  const apiKey = process.env.OLLAMA_API_KEY;
212
141
  if (apiKey) {
@@ -217,10 +146,11 @@ export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<strin
217
146
  `${OLLAMA_BASE}/v1/models`,
218
147
  { headers },
219
148
  timeoutMs,
149
+ signal,
220
150
  );
221
151
 
222
152
  if (res.status === 429) {
223
- throw new Error("Ollama Cloud rate limited. Try again shortly.");
153
+ throw new Error("Ollama Cloud model list fetch rate limited. Try again shortly.");
224
154
  }
225
155
  if (!res.ok || !res.data) {
226
156
  throw new Error(`Failed to fetch model list: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
@@ -229,7 +159,11 @@ export async function fetchModelIds(timeoutMs = FETCH_TIMEOUT_MS): Promise<strin
229
159
  return res.data.data.map((m) => m.id);
230
160
  }
231
161
 
232
- export async function fetchModelDetails(id: string, timeoutMs = FETCH_TIMEOUT_MS): Promise<CachedOllamaModel> {
162
+ export async function fetchModelDetails(
163
+ id: string,
164
+ signal?: AbortSignal,
165
+ timeoutMs = FETCH_TIMEOUT_MS,
166
+ ): Promise<OllamaShowResponse> {
233
167
  const headers: Record<string, string> = { "Content-Type": "application/json" };
234
168
  const apiKey = process.env.OLLAMA_API_KEY;
235
169
  if (apiKey) {
@@ -244,10 +178,11 @@ export async function fetchModelDetails(id: string, timeoutMs = FETCH_TIMEOUT_MS
244
178
  body: JSON.stringify({ model: id }),
245
179
  },
246
180
  timeoutMs,
181
+ signal,
247
182
  );
248
183
 
249
184
  if (res.status === 429) {
250
- throw new Error("Ollama Cloud rate limited. Try again shortly.");
185
+ throw new Error("Ollama Cloud /api/show rate limited. Try again shortly.");
251
186
  }
252
187
  if (!res.ok || !res.data) {
253
188
  throw new Error(`Failed to fetch /api/show for ${id}: ${res.status}${res.error ? ` - ${res.error}` : ""}`);
@@ -256,69 +191,161 @@ export async function fetchModelDetails(id: string, timeoutMs = FETCH_TIMEOUT_MS
256
191
  return res.data;
257
192
  }
258
193
 
259
- export async function refreshOllamaCloudModels(params: {
260
- notify?: (message: string, level?: "info" | "error") => void;
261
- onProgress?: (progress: RefreshProgress) => void;
262
- workers?: number;
263
- }): Promise<Record<string, CachedOllamaModel>> {
264
- const notify = params.notify ?? (() => undefined);
265
- const onProgress = params.onProgress ?? (() => undefined);
266
- onProgress({ stage: "list", message: "Fetching model list..." });
267
- const modelIds = await fetchModelIds();
268
- notify(`Found ${modelIds.length} models, fetching details...`);
269
- onProgress({ stage: "details", current: 0, total: modelIds.length, failed: 0, message: "Fetching model details" });
270
-
271
- let detailsDone = 0;
272
- let detailsFailed = 0;
273
- const detailResults = await concurrentMap(modelIds, params.workers ?? 8, async (id) => {
274
- try {
275
- return [id, await fetchModelDetails(id)] as const;
276
- } catch (error) {
277
- detailsFailed++;
278
- throw error;
279
- } finally {
280
- detailsDone++;
281
- onProgress({
282
- stage: "details",
283
- current: detailsDone,
284
- total: modelIds.length,
285
- failed: detailsFailed,
286
- message: "Fetching model details",
287
- });
288
- }
194
+ /**
195
+ * Fetch per-model /api/show details for a list of model IDs, 8 workers at a time.
196
+ * Returns the models that succeeded. Throws when every detail request fails
197
+ * (the zero-succeeded case), so the caller can surface a real failure instead
198
+ * of an empty catalog.
199
+ */
200
+ export async function refreshOllamaCloudModels(
201
+ modelIds: string[],
202
+ signal?: AbortSignal,
203
+ ): Promise<{ models: Record<string, OllamaShowResponse>; failed: number }> {
204
+ const detailResults = await concurrentMap(modelIds, 8, async (id) => {
205
+ return [id, await fetchModelDetails(id, signal)] as const;
289
206
  });
290
- const models: Record<string, CachedOllamaModel> = {};
207
+ const models: Record<string, OllamaShowResponse> = {};
208
+ let failed = 0;
291
209
  for (const result of detailResults) {
292
210
  if (result.status === "fulfilled") {
293
211
  const [id, data] = result.value;
294
212
  models[id] = data;
213
+ } else {
214
+ failed++;
295
215
  }
296
216
  }
297
- const succeeded = Object.keys(models).length;
298
- if (succeeded === 0)
299
- throw new Error(`Failed to fetch model details${detailsFailed ? ` (${detailsFailed} failed)` : ""}`);
300
- notify(`Fetched ${succeeded} model details${detailsFailed ? ` (${detailsFailed} failed)` : ""}`, "info");
301
-
302
- onProgress({
303
- stage: "done",
304
- current: Object.keys(models).length,
305
- total: Object.keys(models).length,
306
- message: "Done",
307
- });
308
- return models;
217
+ if (Object.keys(models).length === 0) {
218
+ throw new Error(`Failed to fetch model details${failed ? ` (${failed} failed)` : ""}`);
219
+ }
220
+ return { models, failed };
309
221
  }
310
222
 
311
- export async function fetchModels(
312
- ctx: Pick<ExtensionCommandContext, "ui">,
313
- onProgress?: (progress: RefreshProgress) => void,
314
- ): Promise<Record<string, CachedOllamaModel> | null> {
223
+ // --- refreshModels callback ---
224
+
225
+ /**
226
+ * The `refreshModels` callback pi invokes for the "ollama-cloud" provider.
227
+ * Pi calls it twice per refresh: a restore phase (`allowNetwork: false`) before
228
+ * auth resolution, then a network phase (`allowNetwork: true`) only when a
229
+ * credential resolves. The composer swaps the return value into the model list
230
+ * on every invocation, so this must never return `[]`.
231
+ *
232
+ * The model fetch itself is keyless (public `/v1/models` and `/api/show`
233
+ * endpoints; `Authorization` is only added when `OLLAMA_API_KEY` is set). But
234
+ * pi only invokes this network phase when a credential resolves, so a
235
+ * credentialless user stays on `GENERATED_MODELS` until they configure a key.
236
+ * That is a non-issue in practice because a credentialless user cannot run
237
+ * models anyway.
238
+ */
239
+ export async function refreshOllamaCatalog(context: RefreshModelsContext): Promise<ProviderModelConfig[]> {
240
+ // The fallback list: the persisted snapshot (copied) when non-empty, else the
241
+ // baked-in list. Guards against a stored empty catalog (e.g. a prior bad
242
+ // refresh) propagating [] across sessions. A mutable copy is returned because
243
+ // the stored list is `readonly` and the return type is a mutable array.
244
+ const fallback = context.stored?.models.length ? [...context.stored.models] : GENERATED_MODELS;
245
+
246
+ // Restore phase. Rehydrate from the persisted snapshot so removals stick
247
+ // across sessions; fall back to the baked-in list on first launch. Also the
248
+ // early-out for an already-aborted signal.
249
+ if (!context.allowNetwork || context.signal.aborted) {
250
+ return fallback;
251
+ }
252
+
253
+ // Cooldown: skip the network fetch when the stored catalog was checked within
254
+ // the freshness window and the refresh isn't forced (mirrors pi-mono's
255
+ // remote-catalog-provider). A forced refresh (pi update --models) always fetches.
256
+ if (
257
+ !context.force &&
258
+ context.stored?.checkedAt !== undefined &&
259
+ Date.now() - context.stored.checkedAt < REFRESH_COOLDOWN_MS
260
+ ) {
261
+ return fallback;
262
+ }
263
+
264
+ // Network phase. The /v1/models and /api/show endpoints are publicly
265
+ // accessible and do not require authentication, so context.credential is
266
+ // intentionally not threaded into fetchModelIds/fetchModelDetails. Only
267
+ // the web tools (search, fetch) require an API key.
268
+ //
269
+ // Pi's model-selector aborts a catalog refresh after 15s
270
+ // (packages/coding-agent/src/modes/interactive/components/model-selector.ts
271
+ // in pi-mono), so a cold refresh must stay under that budget or the in-memory
272
+ // list won't update on the first picker-open. With ~18 models and 8 workers
273
+ // this is ~1s today; revisit if the catalog grows or the API slows.
274
+ let modelIds: string[];
275
+ let raw: Record<string, OllamaShowResponse>;
276
+ let failed = 0;
315
277
  try {
316
- return await refreshOllamaCloudModels({
317
- notify: (message, level) => ctx.ui.notify(message, level),
318
- onProgress,
319
- });
278
+ modelIds = await fetchModelIds(context.signal);
279
+ if (context.signal.aborted) {
280
+ return fallback;
281
+ }
282
+ const result = await refreshOllamaCloudModels(modelIds, context.signal);
283
+ raw = result.models;
284
+ failed = result.failed;
320
285
  } catch (error) {
321
- ctx.ui.notify(error instanceof Error ? error.message : String(error), "error");
322
- return null;
286
+ // Abort mid-flight returns the current baseline; any other error propagates
287
+ // and pi keeps the last good catalog (no publish was reached).
288
+ if (context.signal.aborted) {
289
+ return fallback;
290
+ }
291
+ throw error;
292
+ }
293
+ if (context.signal.aborted) {
294
+ return fallback;
295
+ }
296
+
297
+ const models = assembleModels(raw);
298
+ // Guard: an empty assembled list (e.g. the live API returned no tools-capable
299
+ // models) must not be persisted or swapped in, or it would kill the provider
300
+ // for the cooldown window. Keep the last good catalog instead.
301
+ if (models.length === 0) {
302
+ return fallback;
323
303
  }
304
+
305
+ // The store is typed to pi-ai's internal Model shape, so rehydrate the live
306
+ // list with the provider identity fields. A Model is structurally assignable
307
+ // to ProviderModelConfig, so the same list is returned to pi (the composer
308
+ // overrides provider/api/baseUrl on the in-memory swap regardless).
309
+ const persisted = models.map((model) => ({
310
+ ...model,
311
+ provider: "ollama-cloud",
312
+ api: "openai-completions",
313
+ baseUrl: `${OLLAMA_BASE}/v1`,
314
+ }));
315
+
316
+ // Best-effort persistence into pi's FileModelsStore. The in-memory list swap
317
+ // happens automatically from the return value, so a failed store write must
318
+ // not prevent returning the fresh catalog.
319
+ if (failed === 0) {
320
+ // Fully successful: persist the fresh catalog.
321
+ try {
322
+ const published = await context.publish({ persist: { models: persisted, checkedAt: Date.now() } });
323
+ if (!published) {
324
+ console.warn("[pi-ollama-cloud] Catalog persist rejected (generation check failed or refresh superseded).");
325
+ }
326
+ } catch {
327
+ // Persistence failure is non-fatal.
328
+ }
329
+ } else {
330
+ // Partial failure: keep the last-good catalog (if any) but advance checkedAt
331
+ // so the cooldown applies and a flaky catalog isn't re-fetched on every
332
+ // /model open, then surface the incomplete refresh. Mirrors pi-mono's
333
+ // remote-catalog-provider, which persists then throws on a transient failure;
334
+ // pi keeps the last-good catalog and reports the error.
335
+ if (context.stored?.models.length) {
336
+ try {
337
+ const published = await context.publish({ persist: { ...context.stored, checkedAt: Date.now() } });
338
+ if (!published) {
339
+ console.warn(
340
+ "[pi-ollama-cloud] Partial-failure persist rejected (generation check failed or refresh superseded).",
341
+ );
342
+ }
343
+ } catch {
344
+ // Persistence failure is non-fatal.
345
+ }
346
+ }
347
+ throw new Error(`Ollama Cloud catalog refresh incomplete: ${failed} model(s) failed`);
348
+ }
349
+
350
+ return persisted;
324
351
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-ollama-cloud",
3
- "version": "0.7.0",
3
+ "version": "0.9.0",
4
4
  "type": "module",
5
5
  "keywords": [
6
6
  "pi-package"
@@ -12,6 +12,7 @@
12
12
  "models.generated.ts",
13
13
  "pricing.generated.ts",
14
14
  "thinking-levels.ts",
15
+ "usage.ts",
15
16
  "utils.ts",
16
17
  "web-tools.ts",
17
18
  "CHANGELOG.md",
@@ -20,11 +21,12 @@
20
21
  ],
21
22
  "repository": {
22
23
  "type": "git",
23
- "url": "https://github.com/fgrehm/pi-ollama-cloud"
24
+ "url": "git+https://github.com/fgrehm/pi-ollama-cloud.git"
24
25
  },
25
26
  "scripts": {
26
- "check": "biome check --write .",
27
+ "check": "biome check --write . && tsgo --noEmit",
27
28
  "lint": "biome check .",
29
+ "typecheck": "tsgo --noEmit",
28
30
  "format": "biome format --write .",
29
31
  "test": "vitest run",
30
32
  "smoke:web-tools": "tsx scripts/smoke-web-tools.ts",
@@ -36,12 +38,15 @@
36
38
  ]
37
39
  },
38
40
  "peerDependencies": {
41
+ "@earendil-works/pi-ai": "*",
39
42
  "@earendil-works/pi-coding-agent": "*",
40
43
  "@earendil-works/pi-tui": "*",
41
44
  "@sinclair/typebox": "*"
42
45
  },
43
46
  "devDependencies": {
44
47
  "@biomejs/biome": "2",
48
+ "@types/node": "^26.1.2",
49
+ "@typescript/native-preview": "7.0.0-dev.20260707.2",
45
50
  "tsx": "^4.19.0",
46
51
  "vitest": "^4.1.6"
47
52
  }
@@ -1,6 +1,6 @@
1
1
  // Auto-generated by scripts/generate-pricing.ts
2
2
  // Do not edit manually.
3
- // Generated: 2026-07-17T19:55:25.388Z
3
+ // Generated: 2026-08-08T04:19:45.594Z
4
4
  // Model count: 18
5
5
 
6
6
  export interface ModelPrice {
@@ -11,21 +11,21 @@ export interface ModelPrice {
11
11
  }
12
12
 
13
13
  export const MODEL_PRICING: Record<string, ModelPrice> = {
14
- "deepseek-v4-flash": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
14
+ "deepseek-v4-flash:0731": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
15
+ "deepseek-v4-flash:preview": { input: 0.14, output: 0.28, cacheRead: 0.0028, cacheWrite: 0 },
15
16
  "deepseek-v4-pro": { input: 0.435, output: 0.87, cacheRead: 0.003625, cacheWrite: 0 },
16
- "gemma4:31b": { input: 0.22, output: 0.55, cacheRead: 0.12, cacheWrite: 0 },
17
+ "gemma4:31b": { input: 0.1, output: 0.34, cacheRead: 0.1, cacheWrite: 0 },
17
18
  "glm-5.1": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
18
19
  "glm-5.2": { input: 1.4, output: 4.4, cacheRead: 0.26, cacheWrite: 0 },
19
20
  "gpt-oss:120b": { input: 0.037, output: 0.17, cacheRead: 0, cacheWrite: 0 },
20
21
  "gpt-oss:20b": { input: 0.03, output: 0.13, cacheRead: 0.03, cacheWrite: 0 },
21
- "kimi-k2.5": { input: 0.6, output: 3, cacheRead: 0.1, cacheWrite: 0 },
22
22
  "kimi-k2.6": { input: 0.95, output: 4, cacheRead: 0.16, cacheWrite: 0 },
23
23
  "kimi-k2.7-code": { input: 0.95, output: 4, cacheRead: 0.19, cacheWrite: 0 },
24
- "minimax-m2.5": { input: 0.3, output: 1.2, cacheRead: 0.03, cacheWrite: 0.375 },
24
+ "kimi-k3": { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 0 },
25
25
  "minimax-m2.7": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0.375 },
26
26
  "minimax-m3": { input: 0.3, output: 1.2, cacheRead: 0.06, cacheWrite: 0 },
27
27
  "mistral-large-3:675b": { input: 0.5, output: 1.5, cacheRead: 0, cacheWrite: 0 },
28
- "nemotron-3-nano:30b": { input: 0.05, output: 0.2, cacheRead: 0, cacheWrite: 0 },
28
+ "nemotron-3-nano:30b": { input: 0.05, output: 0.2, cacheRead: 0.03, cacheWrite: 0 },
29
29
  "nemotron-3-super": { input: 0.2, output: 0.8, cacheRead: 0, cacheWrite: 0 },
30
30
  "nemotron-3-ultra": { input: 0.5, output: 2.5, cacheRead: 0.15, cacheWrite: 0 },
31
31
  "qwen3.5:397b": { input: 0.6, output: 3.6, cacheRead: 0, cacheWrite: 0 },