pi-ollama-cloud 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/README.md +56 -7
- package/cache.ts +181 -0
- package/index.ts +2 -2
- package/limits.generated.ts +25 -0
- package/models.generated.ts +114 -74
- package/models.ts +16 -6
- package/package.json +5 -2
- package/pricing.generated.ts +17 -16
- package/usage.ts +49 -22
- package/utils.ts +6 -0
- package/web-tools.ts +262 -57
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [Unreleased]
|
|
6
|
+
|
|
7
|
+
- Fix `/ollama-cloud-usage` and the usage status bar failing with "unexpected response shape" after the undocumented `/api/usage` endpoint flipped between a single `limits.monthly` bucket and `limits.session` plus `limits.weekly` (the shape has flip-flopped repeatedly as of 2026-09). Any bucket present (`monthly`, `session`, `weekly`) is accepted alone or in combination, and whichever are present are displayed as `5h`/`7d`/`30d` segments. Thanks @johanngyger (#56).
|
|
8
|
+
- Cache `ollama_web_search` results (24h) and `ollama_web_fetch` pages (24h success / 15 min failure) on disk under the pi agent home, so repeated queries and page reads cost 0 API calls. Expired entries are pruned on write, the cache is capped at 500 entries per kind (oldest evicted beyond the cap; `PI_OLLAMA_SEARCH_MAX_ENTRIES`), a partially corrupted cache file is validated per entry and degrades to "no cache" instead of crashing tool calls, and the file is written with `0600` permissions since it stores page content and URLs that can embed credentials. Tune with `PI_OLLAMA_SEARCH_TTL_HOURS`, `PI_OLLAMA_SEARCH_FAIL_TTL_MINUTES`, `PI_OLLAMA_SEARCH_MAX_ENTRIES`, and `PI_OLLAMA_SEARCH_CACHE_PATH`.
|
|
9
|
+
- Bound web tool context usage: search snippets truncate to 500 chars with `[truncated]`/`[complete]` markers, `expand=<index>` returns a truncated result's full content from the cached search (0 extra API calls), and `ollama_web_fetch` pages long pages in 3000-char chunks via `offset`/`full` with a `Continue:` hint for the next offset (`PI_OLLAMA_SEARCH_SNIPPET_CHARS`/`PI_OLLAMA_SEARCH_CHUNK_CHARS` to tune).
|
|
10
|
+
- Add `refresh=true` to both web tools to bypass the cache (including a cached failure) and re-call the API; the fresh result replaces the cache entry.
|
|
11
|
+
- Failed page fetches are negative-cached for 15 min with a diagnostic message (likely cause + next steps) instead of a bare error. Auth (401/403), rate-limit (429), transport (timeout/abort/network), server (5xx), and unexpected-response-shape failures are never cached — search reports transport errors as `transport error` instead of a confusing `status 0` — so retrying after a fixed key, an expired rate-limit window, or a transient server blip re-calls the API immediately.
|
|
12
|
+
- Note: `ollama_web_fetch` tool results now carry `details: { title, totalChars, links }` (previously `{ title, content, links }`); paged content is read via the tool output text, not `details.content`.
|
|
13
|
+
|
|
14
|
+
## [0.10.0] - 2026-09-03
|
|
15
|
+
|
|
16
|
+
- **Breaking:** Adapt to the changed `/api/usage` response shape. The endpoint now returns a single `limits.monthly` bucket (replacing `limits.session` and `limits.weekly`) and adds an `activity.models` array. `UsageData`, `isUsageResponse`, `formatUsage`, and `formatUsageStatusColored` now read the monthly limit; the status bar shows a single `30d` segment instead of `5h`/`7d`.
|
|
17
|
+
- Refreshed the generated catalog from the live API: added `deepseek-v4-pro:0813`, `glm-5.3`, and `glm-5.3-flash`; removed `deepseek-v4-flash:preview` and `deepseek-v4-pro`, which are no longer listed.
|
|
18
|
+
- Source per-token pricing from the official model table on ollama.com/pricing instead of models.dev estimates, which no longer track Ollama's published rates (up to ~13x off per model). `scripts/generate-pricing.ts` now scrapes the pricing page (the table is server-rendered; no JSON endpoint exists) and matches catalog IDs to pricing rows by exact or `:tag`-family match, replacing the `OLLAMA_TO_MODELSDEV` mapping. Regenerated `pricing.generated.ts` with the official rates, including new models (`glm-5.3`, `glm-5.3-flash`, `deepseek-v4-pro:0813`). Fixes #51. Thanks @Hackbard (#52).
|
|
19
|
+
- Probe per-model max output tokens against the live API via `scripts/generate-limits.ts` into `limits.generated.ts`, replacing the fixed 32768 default for known models (unprobed models still fall back to 32768). The probe timeout is 60s so slow first-token models are not dropped from the table. Thanks @f440 (#49).
|
|
20
|
+
|
|
5
21
|
## [0.9.0] - 2026-08-11
|
|
6
22
|
|
|
7
23
|
- Add `/ollama-cloud-usage` command to show Ollama Cloud session (5h) and weekly (7d) usage limits, per-model request counts, and the 4-week activity cost, fetched from the undocumented `/api/usage` endpoint with the already-resolved API key.
|
package/README.md
CHANGED
|
@@ -12,7 +12,7 @@ Registers Ollama Cloud as a model provider with dynamically fetched models, and
|
|
|
12
12
|
- **Automatic model refresh** - On startup, `/model` open, and `pi update --models`, pi calls the extension's `refreshModels` callback to fetch the latest models from the API and persists them through pi's own model store. No manual refresh command.
|
|
13
13
|
- **`ollama_web_search` tool** - Search the web for real-time information using Ollama Cloud's `/api/web_search` endpoint. Returns titles, URLs, and content snippets.
|
|
14
14
|
- **`ollama_web_fetch` tool** - Fetch and extract text content from a web page URL using Ollama Cloud's `/api/web_fetch` endpoint. Returns page title, content, and links.
|
|
15
|
-
- **
|
|
15
|
+
- **Per-token cost tracking** - Models are registered with the official per-token prices from [ollama.com/pricing](https://ollama.com/pricing), so Pi's `/cost` shows comparable usage. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges.
|
|
16
16
|
|
|
17
17
|
## Prerequisites
|
|
18
18
|
|
|
@@ -135,8 +135,14 @@ Model metadata is derived from the `/api/show` response:
|
|
|
135
135
|
| `thinkingLevelMap` | [`thinking-levels.ts`](thinking-levels.ts) with 5 maps (DEFAULT, GPT_OSS, QWEN3, GLM_52, NO_OFF) based on API testing |
|
|
136
136
|
| `input` | `["text", "image"]` if `capabilities` includes `"vision"`, else `["text"]` |
|
|
137
137
|
| `contextWindow` | `model_info.*.context_length` (falls back to 128000) |
|
|
138
|
-
| `maxTokens` |
|
|
139
|
-
| `cost` |
|
|
138
|
+
| `maxTokens` | Probed per-model limits from [`limits.generated.ts`](limits.generated.ts), generated by `scripts/generate-limits.ts` (requires `OLLAMA_API_KEY`). Models without a probed limit fall back to 32768. |
|
|
139
|
+
| `cost` | Official per-1M-token prices from the [ollama.com/pricing](https://ollama.com/pricing) model table, generated by `scripts/generate-pricing.ts` into `pricing.generated.ts`. Ollama Cloud is subscription-billed, so these are equivalent pay-as-you-go rates, not actual charges. Catalog IDs with no matching pricing row default to zero. Prices are pinned to the installed package version and only update on a new release, so newly added models register with zero cost until then. |
|
|
140
|
+
|
|
141
|
+
The per-model max output token table (`limits.generated.ts`) is probed against the live API by `scripts/generate-limits.ts`, since `/api/show` does not expose the limit. It needs an API key: `OLLAMA_API_KEY=<key> npm run generate-limits`. Limits ship with the package, so regenerated values take effect on the next release.
|
|
142
|
+
|
|
143
|
+
The API itself returns no cost data: completion responses report only token counts (`prompt_tokens`/`completion_tokens`/`total_tokens`, including the final usage chunk when streaming), and `/api/show` exposes no pricing fields. The prices above come from the static `/pricing` page table and are only as fresh as the last regeneration.
|
|
144
|
+
|
|
145
|
+
Cache pricing is informational only: the `/pricing` page lists a "Cached input" column, but the completion API does not report cache token usage (there is no `prompt_tokens_details.cached_tokens` or equivalent in any response, verified against the live API in September 2026), so pi never sees cache hits and `/cost` estimates do not reflect them. `cacheWrite` is always zero because the pricing table has no cache-write column.
|
|
140
146
|
|
|
141
147
|
### Thinking level mapping
|
|
142
148
|
|
|
@@ -161,19 +167,60 @@ See [docs/think-experiment.md](docs/think-experiment.md) for the testing methodo
|
|
|
161
167
|
|
|
162
168
|
Both tools use the same Ollama Cloud API key configured for the provider. No local Ollama server is needed.
|
|
163
169
|
|
|
170
|
+
### Caching
|
|
171
|
+
|
|
172
|
+
Both tools cache results on disk (under the pi agent home, `~/.pi/agent/cache/pi-ollama-cloud/cache.json`). A repeated search query or page fetch within the TTL is served from cache and costs 0 API calls:
|
|
173
|
+
|
|
174
|
+
- Successful searches and pages: cached for 24h
|
|
175
|
+
- Failed page fetches: negative-cached for 15 min, so retrying a dead page does not re-call the API. Auth (401/403), rate-limit (429), transport (timeouts, aborts, network errors), and server (5xx) failures are not cached — fixing the key, waiting out the limit, or a transient blip lets a retry through immediately
|
|
176
|
+
- Expired entries are pruned on write and the cache is capped at 500 entries per kind (searches/pages), evicting the oldest first. This bounds entry count, not file size: full page and search content can still make `cache.json` large, and loading it parses the whole file
|
|
177
|
+
- The cache file is written with `0600` permissions. It stores full page content and raw URLs, which can embed credentials in query strings — avoid fetching URLs that carry secrets in the query string, or set a custom `PI_OLLAMA_SEARCH_CACHE_PATH`
|
|
178
|
+
- Concurrent pi processes share the cache file on a last-writer-wins basis (no cross-process locking): one process's save can drop another's fresh entries, at the cost of a redundant API call
|
|
179
|
+
- `refresh=true` on either tool bypasses the cache (including a cached failure) and re-calls the API; the fresh result replaces the cache entry
|
|
180
|
+
|
|
181
|
+
### `ollama_web_search`
|
|
182
|
+
|
|
183
|
+
Returns up to 5 results by default (`max_results`, max 10; title, URL, 500-char snippet). Snippets are marked `[truncated]` when the source is longer than the snippet. Output ends with `# live query` or `# from cache` to show whether the API was called.
|
|
184
|
+
|
|
185
|
+
The search API returns each result's full content; it is cached in full, so a truncated result can be expanded without a separate fetch:
|
|
186
|
+
|
|
187
|
+
- `expand=<index>` — return the full content of that result (1-based) from the cached search, 0 extra API calls. The cache key includes `max_results`, so expanding hits the cache only when the query was searched with the same `max_results`; otherwise the search runs live first.
|
|
188
|
+
- Use `ollama_web_fetch` only when the search result's content is not enough (e.g. you need a different page, or the search excerpt is shorter than the full page).
|
|
189
|
+
|
|
190
|
+
### `ollama_web_fetch`
|
|
191
|
+
|
|
192
|
+
Returns the page title, a 3000-char slice of the content, and links. Long pages are read in chunks to keep the context window small:
|
|
193
|
+
|
|
194
|
+
- `offset=N` — continue reading from character N (the output tells you the next offset)
|
|
195
|
+
- `full=true` — return all remaining content from `offset` in one call
|
|
196
|
+
|
|
197
|
+
A failed fetch throws a diagnostic message (likely cause + next steps) instead of a bare error.
|
|
198
|
+
|
|
199
|
+
### Tuning
|
|
200
|
+
|
|
201
|
+
| Env var | Default | Meaning |
|
|
202
|
+
|---|---|---|
|
|
203
|
+
| `PI_OLLAMA_SEARCH_TTL_HOURS` | `24` | Success cache TTL |
|
|
204
|
+
| `PI_OLLAMA_SEARCH_FAIL_TTL_MINUTES` | `15` | Failure (negative) cache TTL |
|
|
205
|
+
| `PI_OLLAMA_SEARCH_CACHE_PATH` | `<pi agent home>/cache/pi-ollama-cloud/cache.json` | Cache file location |
|
|
206
|
+
| `PI_OLLAMA_SEARCH_MAX_ENTRIES` | `500` | Max cached entries per kind (searches/pages); oldest evicted beyond the cap |
|
|
207
|
+
| `PI_OLLAMA_SEARCH_SNIPPET_CHARS` | `500` | Search snippet length |
|
|
208
|
+
| `PI_OLLAMA_SEARCH_CHUNK_CHARS` | `3000` | Fetch chunk size |
|
|
209
|
+
|
|
164
210
|
## Commands
|
|
165
211
|
|
|
166
212
|
| Command | Description |
|
|
167
213
|
|---|---|
|
|
168
214
|
| `/ollama-webtools [on\|off\|enable\|disable]` | Enable or disable the `ollama_web_search` and `ollama_web_fetch` tools. Toggles if no argument given. |
|
|
169
|
-
| `/ollama-cloud-usage` | Show Ollama Cloud
|
|
215
|
+
| `/ollama-cloud-usage` | Show Ollama Cloud usage limits (one section per limit bucket the API reports), per-model request counts, and the 4-week activity cost. |
|
|
170
216
|
| `/ollama-usage-status [on\|off\|enable\|disable]` | Enable or disable the footer usage status bar. Toggles if no argument given. |
|
|
171
217
|
|
|
172
218
|
## Usage status bar
|
|
173
219
|
|
|
174
220
|
While an `ollama-cloud` model is the active provider, the footer shows a compact
|
|
175
|
-
live usage readout
|
|
176
|
-
|
|
221
|
+
live usage readout with one segment per limit bucket the API reports
|
|
222
|
+
(`5h ▕███░░░░░░░▏ 34% 7d ▕█░░░░░░░░░▏ 7%`, or a single `30d` segment) that
|
|
223
|
+
refreshes every 5 minutes and after each agent turn (but no more often than every 5 minutes). It is colored by how close
|
|
177
224
|
it is to the cap: green below 60%, yellow at 60-79%, red at 80%+. It reads the
|
|
178
225
|
same undocumented `/api/usage` endpoint as `/ollama-cloud-usage` and clears
|
|
179
226
|
itself on transient errors or when you switch to a non-Ollama-Cloud provider.
|
|
@@ -227,6 +274,7 @@ npm run check # lint + format + type-check (auto-fix)
|
|
|
227
274
|
npm run lint # lint only (no fixes)
|
|
228
275
|
npm run typecheck # type-check only (tsgo --noEmit)
|
|
229
276
|
npm run format # format only
|
|
277
|
+
OLLAMA_API_KEY=<key> npm run generate-limits # probe max output tokens (writes limits.generated.ts)
|
|
230
278
|
```
|
|
231
279
|
|
|
232
280
|
The project uses [Biome](https://biomejs.dev/) for linting and formatting (2-space indent, line width 120) and [tsgo](https://github.com/microsoft/typescript-go) for type-checking.
|
|
@@ -295,7 +343,8 @@ git push --tags
|
|
|
295
343
|
Because the model catalog refreshes automatically at runtime, a release is **not** needed to ship new models. Publish only when:
|
|
296
344
|
|
|
297
345
|
- A model is retired and still listed by the API: add it to `RETIRED_MODEL_IDS` in `scripts/generate-models.ts` (check https://docs.ollama.com/cloud#retirements, then regenerate `models.generated.ts`).
|
|
298
|
-
- Pricing changes:
|
|
346
|
+
- Pricing changes: Ollama updates the model pricing table, or a new model needs a pricing row (regenerate `pricing.generated.ts`).
|
|
347
|
+
- Max output token limits changed: run `OLLAMA_API_KEY=<key> npm run generate-limits` locally and commit.
|
|
299
348
|
|
|
300
349
|
The tag version must match the version in `package.json` - `npm version` handles this automatically. The workflow at `.github/workflows/publish.yml` verifies the match before publishing to npm.
|
|
301
350
|
|
package/cache.ts
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
import { createHash, randomBytes } from "node:crypto";
|
|
2
|
+
import { chmodSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
import { getAgentDir } from "@earendil-works/pi-coding-agent";
|
|
5
|
+
import { envInt } from "./utils.ts";
|
|
6
|
+
|
|
7
|
+
export const CACHE_PATH =
|
|
8
|
+
process.env.PI_OLLAMA_SEARCH_CACHE_PATH ?? join(getAgentDir(), "cache", "pi-ollama-cloud", "cache.json");
|
|
9
|
+
export const CACHE_TTL_MS = envInt("PI_OLLAMA_SEARCH_TTL_HOURS", 24) * 60 * 60 * 1000;
|
|
10
|
+
export const FAIL_TTL_MS = envInt("PI_OLLAMA_SEARCH_FAIL_TTL_MINUTES", 15) * 60 * 1000;
|
|
11
|
+
/** Max entries per map (searches/pages); oldest-ts entries are evicted beyond this. */
|
|
12
|
+
export const MAX_ENTRIES = envInt("PI_OLLAMA_SEARCH_MAX_ENTRIES", 500);
|
|
13
|
+
|
|
14
|
+
export interface SearchResult {
|
|
15
|
+
title: string;
|
|
16
|
+
url: string;
|
|
17
|
+
content: string;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export interface SearchCacheEntry {
|
|
21
|
+
ts: number;
|
|
22
|
+
q: string;
|
|
23
|
+
results: SearchResult[];
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface PageCacheEntry {
|
|
27
|
+
ts: number;
|
|
28
|
+
status?: number;
|
|
29
|
+
title?: string;
|
|
30
|
+
content?: string;
|
|
31
|
+
links?: string[] | null;
|
|
32
|
+
error?: string;
|
|
33
|
+
errorType?: "response-shape";
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export interface CacheData {
|
|
37
|
+
searches: Record<string, SearchCacheEntry>;
|
|
38
|
+
pages: Record<string, PageCacheEntry>;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
interface CacheOptions {
|
|
42
|
+
path: string;
|
|
43
|
+
ttlMs: number;
|
|
44
|
+
failTtlMs: number;
|
|
45
|
+
maxEntries: number;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
function isRecord(v: unknown): v is Record<string, unknown> {
|
|
49
|
+
return typeof v === "object" && v !== null && !Array.isArray(v);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Keys that must never come from a parsed JSON file (prototype pollution). */
|
|
53
|
+
const UNSAFE_KEYS = new Set(["__proto__", "constructor", "prototype"]);
|
|
54
|
+
|
|
55
|
+
export function isSafeKey(key: string): boolean {
|
|
56
|
+
return !UNSAFE_KEYS.has(key);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** Shallow shape checks so a partially corrupted cache file degrades instead of crashing tool calls. */
|
|
60
|
+
function isSearchEntry(v: unknown): v is SearchCacheEntry {
|
|
61
|
+
return (
|
|
62
|
+
isRecord(v) &&
|
|
63
|
+
typeof v.ts === "number" &&
|
|
64
|
+
typeof v.q === "string" &&
|
|
65
|
+
Array.isArray(v.results) &&
|
|
66
|
+
v.results.every(
|
|
67
|
+
(r) => isRecord(r) && typeof r.title === "string" && typeof r.url === "string" && typeof r.content === "string",
|
|
68
|
+
)
|
|
69
|
+
);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function isPageEntry(v: unknown): v is PageCacheEntry {
|
|
73
|
+
if (!isRecord(v) || typeof v.ts !== "number") return false;
|
|
74
|
+
const fieldsValid =
|
|
75
|
+
(v.status === undefined || typeof v.status === "number") &&
|
|
76
|
+
(v.title === undefined || typeof v.title === "string") &&
|
|
77
|
+
(v.content === undefined || typeof v.content === "string") &&
|
|
78
|
+
(v.links === null ||
|
|
79
|
+
v.links === undefined ||
|
|
80
|
+
(Array.isArray(v.links) && v.links.every((l) => typeof l === "string"))) &&
|
|
81
|
+
(v.error === undefined || (typeof v.error === "string" && v.error !== "")) &&
|
|
82
|
+
(v.errorType === undefined || v.errorType === "response-shape");
|
|
83
|
+
if (!fieldsValid) return false;
|
|
84
|
+
// Must be either a real failure or a real success; anything else (e.g. an
|
|
85
|
+
// entry with neither content nor a non-empty error) would render as a fake
|
|
86
|
+
// empty success.
|
|
87
|
+
return (
|
|
88
|
+
(typeof v.error === "string" && v.error !== "") || (typeof v.title === "string" && typeof v.content === "string")
|
|
89
|
+
);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export interface CacheStore {
|
|
93
|
+
loadCache(): CacheData;
|
|
94
|
+
saveCache(): void;
|
|
95
|
+
isFresh(entry: { ts: number; error?: string } | undefined): boolean;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export function createCache(options: Partial<CacheOptions> = {}): CacheStore {
|
|
99
|
+
const path = options.path ?? CACHE_PATH;
|
|
100
|
+
const ttlMs = options.ttlMs ?? CACHE_TTL_MS;
|
|
101
|
+
const failTtlMs = options.failTtlMs ?? FAIL_TTL_MS;
|
|
102
|
+
const maxEntries = options.maxEntries ?? MAX_ENTRIES;
|
|
103
|
+
let cacheData: CacheData | null = null;
|
|
104
|
+
|
|
105
|
+
function loadCache(): CacheData {
|
|
106
|
+
if (cacheData) return cacheData;
|
|
107
|
+
try {
|
|
108
|
+
const raw: unknown = JSON.parse(readFileSync(path, "utf8"));
|
|
109
|
+
if (isRecord(raw) && isRecord(raw.searches) && isRecord(raw.pages)) {
|
|
110
|
+
// Per-entry validation: drop poisoned entries so a partially corrupt
|
|
111
|
+
// file degrades to "those entries are gone" instead of crashing calls.
|
|
112
|
+
cacheData = { searches: {}, pages: {} };
|
|
113
|
+
for (const [key, entry] of Object.entries(raw.searches)) {
|
|
114
|
+
if (isSafeKey(key) && isSearchEntry(entry)) cacheData.searches[key] = entry;
|
|
115
|
+
}
|
|
116
|
+
for (const [key, entry] of Object.entries(raw.pages)) {
|
|
117
|
+
if (isSafeKey(key) && isPageEntry(entry)) cacheData.pages[key] = entry;
|
|
118
|
+
}
|
|
119
|
+
return cacheData;
|
|
120
|
+
}
|
|
121
|
+
} catch {
|
|
122
|
+
// First run or corrupt file, start fresh.
|
|
123
|
+
}
|
|
124
|
+
cacheData = { searches: {}, pages: {} };
|
|
125
|
+
return cacheData;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function isFresh(entry: { ts: number; error?: string } | undefined): boolean {
|
|
129
|
+
if (!entry) return false;
|
|
130
|
+
// A future ts (hand-edited file) would otherwise be fresh forever; treat as stale.
|
|
131
|
+
if (entry.ts > Date.now()) return false;
|
|
132
|
+
return Date.now() - entry.ts < (entry.error ? failTtlMs : ttlMs);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function evictOldest(map: Record<string, { ts: number }>): void {
|
|
136
|
+
const keys = Object.keys(map);
|
|
137
|
+
if (keys.length <= maxEntries) return;
|
|
138
|
+
const overflow = keys.sort((a, b) => map[a].ts - map[b].ts).slice(0, keys.length - maxEntries);
|
|
139
|
+
for (const key of overflow) delete map[key];
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function saveCache(): void {
|
|
143
|
+
const data = loadCache();
|
|
144
|
+
for (const [key, entry] of Object.entries(data.searches)) if (!isFresh(entry)) delete data.searches[key];
|
|
145
|
+
for (const [key, entry] of Object.entries(data.pages)) if (!isFresh(entry)) delete data.pages[key];
|
|
146
|
+
// TTL bounds entry age; the cap bounds entry count so an aggressive session
|
|
147
|
+
// cannot grow the file without limit.
|
|
148
|
+
evictOldest(data.searches);
|
|
149
|
+
evictOldest(data.pages);
|
|
150
|
+
try {
|
|
151
|
+
mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
|
|
152
|
+
// Use a unique temporary path so concurrent processes cannot overwrite
|
|
153
|
+
// one another's in-progress writes.
|
|
154
|
+
const tmp = `${path}.${process.pid}.${randomBytes(8).toString("hex")}.tmp`;
|
|
155
|
+
try {
|
|
156
|
+
// 0o600: the cache stores full page content and URLs, which can embed
|
|
157
|
+
// credentials in query strings; it should not be world-readable.
|
|
158
|
+
writeFileSync(tmp, JSON.stringify(data), { mode: 0o600 });
|
|
159
|
+
renameSync(tmp, path);
|
|
160
|
+
// Fix perms of a file written by a pre-0600 version.
|
|
161
|
+
chmodSync(path, 0o600);
|
|
162
|
+
} catch (error) {
|
|
163
|
+
rmSync(tmp, { force: true });
|
|
164
|
+
throw error;
|
|
165
|
+
}
|
|
166
|
+
} catch {
|
|
167
|
+
// Cache is best-effort; a failed write must not break the tool call.
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
return { loadCache, saveCache, isFresh };
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
export const defaultCache = createCache();
|
|
175
|
+
export const loadCache = defaultCache.loadCache;
|
|
176
|
+
export const saveCache = defaultCache.saveCache;
|
|
177
|
+
export const isFresh = defaultCache.isFresh;
|
|
178
|
+
|
|
179
|
+
export function searchCacheKey(query: string, maxResults: number): string {
|
|
180
|
+
return createHash("sha1").update(`${query}\n${maxResults}`).digest("hex");
|
|
181
|
+
}
|
package/index.ts
CHANGED
|
@@ -134,7 +134,7 @@ export default async function (pi: ExtensionAPI) {
|
|
|
134
134
|
// --- Usage Command ---
|
|
135
135
|
|
|
136
136
|
pi.registerCommand("ollama-cloud-usage", {
|
|
137
|
-
description: "Show Ollama Cloud
|
|
137
|
+
description: "Show Ollama Cloud usage limits.",
|
|
138
138
|
handler: async (_args, ctx) => {
|
|
139
139
|
const apiKey = await getCloudApiKey(ctx);
|
|
140
140
|
if (!apiKey) {
|
|
@@ -152,7 +152,7 @@ export default async function (pi: ExtensionAPI) {
|
|
|
152
152
|
|
|
153
153
|
// --- Usage Status Bar ---
|
|
154
154
|
|
|
155
|
-
// Footer status showing live
|
|
155
|
+
// Footer status showing live usage while ollama-cloud is the
|
|
156
156
|
// active provider. Refreshes on a 5-minute timer; agent_end also triggers a
|
|
157
157
|
// refresh but is throttled to the same cooldown so a turn never hammers the
|
|
158
158
|
// undocumented /api/usage endpoint. The quota-bar concept is inspired by
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
// Auto-generated by scripts/generate-limits.ts
|
|
2
|
+
// Do not edit manually.
|
|
3
|
+
// Probed models: 19 (0 failed)
|
|
4
|
+
|
|
5
|
+
export const MODEL_MAX_OUTPUT_TOKENS: Record<string, number> = {
|
|
6
|
+
"deepseek-v4-flash:0731": 65536,
|
|
7
|
+
"deepseek-v4-pro:0813": 65536,
|
|
8
|
+
"gemma4:31b": 262144,
|
|
9
|
+
"glm-5.1": 131072,
|
|
10
|
+
"glm-5.2": 131072,
|
|
11
|
+
"glm-5.3": 524288,
|
|
12
|
+
"glm-5.3-flash": 524288,
|
|
13
|
+
"gpt-oss:120b": 131072,
|
|
14
|
+
"gpt-oss:20b": 131072,
|
|
15
|
+
"kimi-k2.6": 262144,
|
|
16
|
+
"kimi-k2.7-code": 262144,
|
|
17
|
+
"kimi-k3": 524288,
|
|
18
|
+
"minimax-m2.7": 131072,
|
|
19
|
+
"minimax-m3": 131072,
|
|
20
|
+
"mistral-large-3:675b": 262144,
|
|
21
|
+
"nemotron-3-nano:30b": 131072,
|
|
22
|
+
"nemotron-3-super": 65536,
|
|
23
|
+
"nemotron-3-ultra": 65536,
|
|
24
|
+
"qwen3.5:397b": 65536,
|
|
25
|
+
};
|