@earendil-works/pi-coding-agent 0.87.0 → 0.87.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +25 -711
  3. package/dist/bundle/chunks/{anthropic-messages-MYU5ZMRF.js → anthropic-messages-J5WXPPPC.js} +1 -1
  4. package/dist/bundle/chunks/{chunk-GV2E3GBU.js → chunk-65HAU2C5.js} +1 -1
  5. package/dist/bundle/chunks/{chunk-4DKZACXI.js → chunk-OJP47DM6.js} +13 -13
  6. package/dist/bundle/chunks/github-copilot.js +1 -1
  7. package/dist/bundle/chunks/{openai-completions-XHML6MTL.js → openai-completions-OBX42CLD.js} +1 -1
  8. package/dist/bundle/chunks/{virtual-modules-BNWPZYDH.js → virtual-modules-VHMJYYWQ.js} +1 -1
  9. package/dist/bundle/cli-runtime.js +1 -1
  10. package/dist/bundle/index.js +1 -1
  11. package/dist/bundle/rpc-entry.js +1 -1
  12. package/dist/cli/args.d.ts.map +1 -1
  13. package/dist/cli/args.js +14 -4
  14. package/dist/cli/args.js.map +1 -1
  15. package/dist/core/compaction/compaction.d.ts.map +1 -1
  16. package/dist/core/compaction/compaction.js +9 -9
  17. package/dist/core/compaction/compaction.js.map +1 -1
  18. package/dist/core/model-resolver.d.ts.map +1 -1
  19. package/dist/core/model-resolver.js +1 -1
  20. package/dist/core/model-resolver.js.map +1 -1
  21. package/docs/cli-integration.md +106 -0
  22. package/docs/cli.md +268 -0
  23. package/docs/compaction.md +22 -22
  24. package/docs/configuration.md +45 -0
  25. package/docs/containerization.md +109 -82
  26. package/docs/custom-provider.md +132 -784
  27. package/docs/docs.json +139 -99
  28. package/docs/environment-variables.md +3 -5
  29. package/docs/extensions.md +134 -3020
  30. package/docs/how-pi-works.md +49 -0
  31. package/docs/images/interactive-mode.png +0 -0
  32. package/docs/index.md +24 -69
  33. package/docs/json.md +193 -65
  34. package/docs/keybindings.md +57 -102
  35. package/docs/llama-cpp.md +3 -3
  36. package/docs/message-types.md +261 -0
  37. package/docs/models.md +55 -565
  38. package/docs/packages.md +66 -167
  39. package/docs/prompt-templates.md +31 -68
  40. package/docs/providers.md +102 -240
  41. package/docs/quickstart.md +61 -106
  42. package/docs/rpc-commands.md +854 -0
  43. package/docs/rpc-extension-ui.md +200 -0
  44. package/docs/rpc.md +129 -1556
  45. package/docs/sdk.md +76 -1171
  46. package/docs/security.md +70 -32
  47. package/docs/session-format.md +10 -214
  48. package/docs/sessions.md +35 -141
  49. package/docs/settings.md +109 -387
  50. package/docs/shell-aliases.md +85 -5
  51. package/docs/skills.md +51 -190
  52. package/docs/slash-commands.md +60 -0
  53. package/docs/terminal-setup.md +105 -78
  54. package/docs/termux.md +74 -83
  55. package/docs/themes.md +68 -280
  56. package/docs/tmux.md +31 -39
  57. package/docs/tui.md +69 -923
  58. package/docs/usage.md +54 -272
  59. package/docs/windows.md +43 -17
  60. package/examples/README.md +13 -2
  61. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  62. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  63. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  64. package/examples/extensions/gondolin/package-lock.json +2 -2
  65. package/examples/extensions/gondolin/package.json +1 -1
  66. package/examples/extensions/sandbox/package-lock.json +2 -2
  67. package/examples/extensions/sandbox/package.json +1 -1
  68. package/examples/extensions/with-deps/package-lock.json +2 -2
  69. package/examples/extensions/with-deps/package.json +1 -1
  70. package/examples/rpc-client.ts +35 -0
  71. package/examples/rpc-extension-ui.ts +25 -5
  72. package/examples/sdk/README.md +1 -1
  73. package/npm-shrinkwrap.json +20 -20
  74. package/package.json +8 -8
  75. package/docs/development.md +0 -90
package/docs/models.md CHANGED
@@ -1,247 +1,73 @@
1
- # Custom Models
1
+ # Choose a Model
2
2
 
3
- Add custom providers and models (Ollama, vLLM, LM Studio, proxies) via `~/.pi/agent/models.json`.
3
+ For a built-in provider, start with `/login`, then choose a model with `/model`. Use custom model configuration only when Pi does not already include the provider or endpoint you need.
4
4
 
5
- ## Table of Contents
5
+ ## Choose a connection
6
6
 
7
- - [Minimal Example](#minimal-example)
8
- - [Full Example](#full-example)
9
- - [Supported APIs](#supported-apis)
10
- - [Provider Configuration](#provider-configuration)
11
- - [Model Configuration](#model-configuration)
12
- - [Prompt Cache Lifetimes](#prompt-cache-lifetimes)
13
- - [Overriding Built-in Providers](#overriding-built-in-providers)
14
- - [Per-model Overrides](#per-model-overrides)
15
- - [Anthropic Messages Compatibility](#anthropic-messages-compatibility)
16
- - [OpenAI Compatibility](#openai-compatibility)
7
+ | What you have | Recommended setup |
8
+ |---|---|
9
+ | A supported subscription | Sign in through `/login` |
10
+ | A provider API key | Store it through `/login` or set its environment variable |
11
+ | A local GGUF model | Connect Pi to the llama.cpp router |
12
+ | An OpenAI-, Anthropic-, or Google-compatible endpoint | Add it to `models.json` |
13
+ | A provider with a custom protocol or authentication flow | Build or install a provider extension |
17
14
 
18
- ## Minimal Example
15
+ Browse the [model catalog](https://pi.dev/models) for current providers, model IDs, capabilities, context limits, and pricing. Pi starts with its bundled catalog and can overlay newer catalog data from pi.dev. Cached catalog data remains available offline; run `pi update --models` to force a refresh.
19
16
 
20
- For local models (Ollama, LM Studio, vLLM), only `id` is required per model:
17
+ ## Authenticate
21
18
 
22
- ```json
23
- {
24
- "providers": {
25
- "ollama": {
26
- "baseUrl": "http://localhost:11434/v1",
27
- "api": "openai-completions",
28
- "apiKey": "ollama",
29
- "models": [
30
- { "id": "llama3.1:8b" },
31
- { "id": "qwen2.5-coder:7b" }
32
- ]
33
- }
34
- }
35
- }
36
- ```
37
-
38
- The `apiKey` value is a placeholder because Ollama ignores it. pi still treats models as requiring auth before they appear in `/model`, so keyless local servers should keep a dummy value, save a key for that provider with `/login`, or pass `--api-key` when selecting the model.
39
-
40
- Some OpenAI-compatible servers do not understand the `developer` role used for reasoning-capable models. For those providers, set `compat.supportsDeveloperRole` to `false` so pi sends the system prompt as a `system` message instead. If the server also does not support `reasoning_effort`, set `compat.supportsReasoningEffort` to `false` too.
41
-
42
- You can set `compat` at the provider level to apply to all models, or at the model level to override a specific model. This commonly applies to Ollama, vLLM, SGLang, and similar OpenAI-compatible servers.
43
-
44
- ```json
45
- {
46
- "providers": {
47
- "ollama": {
48
- "baseUrl": "http://localhost:11434/v1",
49
- "api": "openai-completions",
50
- "apiKey": "ollama",
51
- "compat": {
52
- "supportsDeveloperRole": false,
53
- "supportsReasoningEffort": false
54
- },
55
- "models": [
56
- {
57
- "id": "gpt-oss:20b",
58
- "reasoning": true
59
- }
60
- ]
61
- }
62
- }
63
- }
64
- ```
65
-
66
- ## Full Example
67
-
68
- Override defaults when you need specific values:
69
-
70
- ```json
71
- {
72
- "providers": {
73
- "ollama": {
74
- "baseUrl": "http://localhost:11434/v1",
75
- "api": "openai-completions",
76
- "apiKey": "ollama",
77
- "models": [
78
- {
79
- "id": "llama3.1:8b",
80
- "name": "Llama 3.1 8B (Local)",
81
- "reasoning": false,
82
- "input": ["text"],
83
- "contextWindow": 128000,
84
- "maxTokens": 32000,
85
- "cost": { "input": 0, "output": 0, "cacheRead": 0, "cacheWrite": 0 }
86
- }
87
- ]
88
- }
89
- }
90
- }
91
- ```
92
-
93
- The file reloads each time you open `/model`. Edit during session; no restart needed.
94
-
95
- ## Google AI Studio Example
96
-
97
- Use `google-generative-ai` with a `baseUrl` to add models from Google AI Studio, including custom Gemma 4 entries:
98
-
99
- ```json
100
- {
101
- "providers": {
102
- "my-google": {
103
- "baseUrl": "https://generativelanguage.googleapis.com/v1beta",
104
- "api": "google-generative-ai",
105
- "apiKey": "$GEMINI_API_KEY",
106
- "models": [
107
- {
108
- "id": "gemma-4-31b-it",
109
- "name": "Gemma 4 31B",
110
- "input": ["text", "image"],
111
- "contextWindow": 262144,
112
- "reasoning": true
113
- }
114
- ]
115
- }
116
- }
117
- }
118
- ```
19
+ Run `/login` and select a provider. Pi stores credentials in [`auth.json`](configuration.md#agent-directory). Run `/logout` to remove stored credentials for a provider.
119
20
 
120
- The `baseUrl` is required when adding custom models to the `google-generative-ai` API type.
21
+ You can instead provide an API key through the provider's environment variable. This is useful in CI and other environments where Pi should not write credentials. [Provider Authentication](providers.md) lists the variables and cloud-provider setup.
121
22
 
122
- ## Supported APIs
23
+ When several credential sources are configured, Pi uses a runtime `--api-key` first, then a stored `auth.json` credential, an `apiKey` from `models.json`, and finally the provider's environment variables or ambient cloud credentials. Provider extensions can define their own authentication behavior.
123
24
 
124
- | API | Description |
125
- |-----|-------------|
126
- | `openai-completions` | OpenAI Chat Completions (most compatible) |
127
- | `openai-responses` | OpenAI Responses API |
128
- | `anthropic-messages` | Anthropic Messages API |
129
- | `google-generative-ai` | Google Generative AI |
25
+ Keep `auth.json` and any credential commands private. Project settings and extensions can execute inside the Pi process after you trust a project. Review [Security](security.md) before loading configuration from an untrusted directory.
130
26
 
131
- Set `api` at provider level (default for all models) or model level (override per model).
27
+ ## Select a model
132
28
 
133
- ## Provider Configuration
29
+ Run `/model` to search available models. The picker shows models whose providers have usable authentication. Press `Ctrl+S` on a model to save it as the default for new sessions.
134
30
 
135
- | Field | Description |
136
- |-------|-------------|
137
- | `baseUrl` | API endpoint URL |
138
- | `api` | API type (see above) |
139
- | `apiKey` | Optional API key config (see value resolution below). Omit it when auth is provided by `/login`/`auth.json` or CLI `--api-key`. |
140
- | `oauth` | Dynamic OAuth provider type. Currently supports `"radius"`; requires the gateway `baseUrl`. |
141
- | `headers` | Custom headers (see value resolution below) |
142
- | `authHeader` | Set `true` to add `Authorization: Bearer <apiKey>` automatically |
143
- | `models` | Array of model configurations |
144
- | `modelOverrides` | Per-model overrides for built-in or extension-registered models on this provider |
31
+ Run `/thinking` to select the thinking level for the current model. Press `Ctrl+S` there to save the startup level. Pi limits the choices to levels supported by the selected model.
145
32
 
146
- For providers with `models`, non-built-in provider configs need `baseUrl` and an `api` value at either provider or model level. `apiKey` is not required to load the file: models become available when auth is configured through `/login`/`auth.json`, CLI `--api-key`, or provider `apiKey`. If no auth is configured, the models load but stay unavailable in `/model` and `--list-models`.
33
+ `Ctrl+P` cycles through available models. Use `/scoped-models` to control that cycle and save the selection, or configure model patterns through [Settings](settings.md#model-cycling).
147
34
 
148
- ### Value Resolution
35
+ A session records model and thinking-level changes. Resuming the session restores them without changing defaults for new sessions.
149
36
 
150
- The `apiKey` and `headers` fields support command execution, environment interpolation, and literals:
37
+ ## Connect local models
151
38
 
152
- - **Shell command:** `"!command"` at the start executes the whole value as a command and uses stdout
153
- ```json
154
- "apiKey": "!security find-generic-password -ws 'anthropic'"
155
- "apiKey": "!op read 'op://vault/item/credential'"
156
- ```
157
- - **Environment interpolation:** `"$ENV_VAR"` or `"${ENV_VAR}"` uses the value of the named variable. Interpolation works inside larger literals.
158
- ```json
159
- "apiKey": "$MY_API_KEY"
160
- "apiKey": "${KEY_PREFIX}_${KEY_SUFFIX}"
161
- ```
162
- `$FOO_BAR` is the variable `FOO_BAR`; use `${FOO}_BAR` when `BAR` is literal text. Missing environment variables make the value unresolved.
163
- - **Escapes:** `"$$"` emits a literal `"$"`; `"$!"` emits a literal `"!"` without triggering command execution.
164
- ```json
165
- "apiKey": "$$literal-dollar-prefix"
166
- "apiKey": "$!literal-bang-prefix"
167
- ```
168
- - **Literal value:** Used directly. Plain uppercase strings such as `MY_API_KEY` are literals; use `$MY_API_KEY` for environment variables.
169
- ```json
170
- "apiKey": "sk-..."
171
- ```
39
+ Pi integrates directly with the llama.cpp router. The router discovers GGUF files and loads models on demand. Pi's `/llama` command manages the router, while `/model` selects one of its loaded models.
172
40
 
173
- For `models.json`, shell commands are resolved at request time. pi intentionally does not apply built-in TTL, stale reuse, or recovery logic for arbitrary commands. Different commands need different caching and failure strategies, and pi cannot infer the right one.
41
+ Follow [Local Models with llama.cpp](llama-cpp.md) for server startup, model layout, downloads, and connection troubleshooting.
174
42
 
175
- If your command is slow, expensive, rate-limited, or should keep using a previous value on transient failures, wrap it in your own script or command that implements the caching or TTL behavior you want.
43
+ For Ollama, LM Studio, vLLM, SGLang, and other compatible servers, [configure a compatible endpoint](#configure-a-compatible-endpoint) in `models.json`.
176
44
 
177
- `/model` availability checks use configured auth presence and do not execute shell commands.
45
+ ## Configure a compatible endpoint
178
46
 
179
- ### Custom Headers
47
+ Use [`models.json`](configuration.md#agent-directory) when an endpoint speaks an API Pi already supports. This includes most Ollama, LM Studio, vLLM, SGLang, and proxy deployments.
180
48
 
181
49
  ```json
182
50
  {
183
51
  "providers": {
184
- "custom-proxy": {
185
- "baseUrl": "https://proxy.example.com/v1",
186
- "apiKey": "$MY_API_KEY",
187
- "api": "anthropic-messages",
188
- "headers": {
189
- "x-portkey-api-key": "$PORTKEY_API_KEY",
190
- "x-secret": "!op read 'op://vault/item/secret'"
191
- },
192
- "models": [...]
52
+ "ollama": {
53
+ "baseUrl": "http://localhost:11434/v1",
54
+ "api": "openai-completions",
55
+ "apiKey": "ollama",
56
+ "models": [
57
+ { "id": "qwen2.5-coder:7b" }
58
+ ]
193
59
  }
194
60
  }
195
61
  }
196
62
  ```
197
63
 
198
- ## Model Configuration
199
-
200
- | Field | Required | Default | Description |
201
- |-------|----------|---------|-------------|
202
- | `id` | Yes | — | Model identifier (passed to the API) |
203
- | `name` | No | `id` | Human-readable model label. Used for matching (`--model` patterns) and shown as secondary model detail text. |
204
- | `api` | No | provider's `api` | Override provider's API for this model |
205
- | `reasoning` | No | `false` | Supports extended thinking |
206
- | `thinkingLevelMap` | No | omitted | Maps pi thinking levels to provider values and marks unsupported levels (see below) |
207
- | `input` | No | `["text"]` | Input types: `["text"]` or `["text", "image"]` |
208
- | `inputLimits` | No | omitted | Request limits and image preprocessing for this model (see below) |
209
- | `contextWindow` | No | `128000` | Context window size in tokens |
210
- | `maxTokens` | No | `16384` | Maximum output tokens |
211
- | `samplingParams` | No | omitted | Sampling parameters merged verbatim into every request body (see below) |
212
- | `cost` | No | all zeros | Per-million-token rates with optional request-wide input pricing tiers |
213
- | `promptCache` | No | omitted | Best-effort prompt cache lifetime in seconds per retention tier (see below) |
214
- | `compat` | No | provider `compat` | Provider compatibility overrides. Merged with provider-level `compat` when both are set. |
215
-
216
- A cost tier supplies a complete alternate rate set and applies to the full request when total input usage (`input + cacheRead + cacheWrite`) exceeds `inputTokensAbove`. When multiple tiers match, the highest threshold wins.
64
+ The dummy key makes the model available to Pi; Ollama ignores it. For an authenticated endpoint, `apiKey` and header values can use `$NAME` or `${NAME}` environment interpolation, a literal value, or a leading `!command`. Commands in `models.json` run at request time and are not cached by Pi.
217
65
 
218
- ```json
219
- {
220
- "cost": {
221
- "input": 5,
222
- "output": 30,
223
- "cacheRead": 0.5,
224
- "cacheWrite": 6.25,
225
- "tiers": [
226
- {
227
- "inputTokensAbove": 272000,
228
- "input": 10,
229
- "output": 45,
230
- "cacheRead": 1,
231
- "cacheWrite": 12.5
232
- }
233
- ]
234
- }
235
- }
236
- ```
237
-
238
- Current behavior:
239
- - `/model`, `--list-models`, and the interactive footer display entries by model `id`.
240
- - The configured `name` is used for model matching and secondary model detail text. It does not replace the footer/status-bar model id.
66
+ Opening `/model` reloads the file. A `models` entry adds or replaces a model with the same ID on that provider. Use `modelOverrides` to change metadata for an existing built-in or extension-provided model without replacing the provider's model list. Unknown override IDs are ignored.
241
67
 
242
- ### Image Input Limits
68
+ ### Describe model input and caching
243
69
 
244
- Use `inputLimits.images.resize` to configure how new images are encoded before they enter conversation history:
70
+ Use `inputLimits.images.resize` to control how Pi encodes new image attachments, `read` results, and tool-result images before storing them in conversation history:
245
71
 
246
72
  ```json
247
73
  {
@@ -260,374 +86,38 @@ Use `inputLimits.images.resize` to configure how new images are encoded before t
260
86
  }
261
87
  ```
262
88
 
263
- `maxBytes` is the maximum base64-encoded payload size. Omitted resize fields use pi's conservative defaults: 2000×2000, 4.5 MiB encoded, and JPEG quality 80. Built-in vision models carry that profile explicitly so unknown gateways never receive larger images than before.
264
-
265
- Pi applies the selected model's resize profile to `@file` attachments, the `read` tool, and images returned by tools. Images are encoded once before they enter history; changing models does not rewrite historical images or invalidate the cached conversation prefix. The `images.autoResize` setting can disable resizing globally.
266
-
267
- The catalog can also record `inputLimits.maxRequestBytes`, `images.maxPerMessage`, and `images.maxPerRequest`. These fields describe hard provider limits; this initial implementation does not yet rewrite or reject conversation history based on them.
268
-
269
- ### Prompt Cache Lifetimes
270
-
271
- `promptCache` states how long the provider keeps a prompt cache entry alive for each retention tier pi can request (`short` is the default tier; `long` is used when `PI_CACHE_RETENTION=long`). Values are seconds and are estimates: providers publish ranges, so pick the conservative end.
272
-
273
- ```json
274
- {
275
- "id": "claude-sonnet-5",
276
- "promptCache": { "short": 300, "long": 3600 }
277
- }
278
- ```
279
-
280
- The built-in catalog fills this in for direct Anthropic (5 min / 1 h). Other providers, including direct OpenAI, have no built-in lifetime until their cache-expiry and replay behavior has been validated for warming. A model without a value for the tier a request used is never warmed; custom models and provider overrides can opt in when the backing cache behavior is known. See [Cache Warming](settings.md#cache-warming).
281
-
282
- ### Sampling Parameters
283
-
284
- `samplingParams` is a free-form object merged verbatim into every request body for the model, after the fields pi sets itself, so its keys win. Use it to send sampling parameters pi does not model — including server-specific ones like llama.cpp's `min_p` or vLLM's `top_k`:
285
-
286
- ```json
287
- {
288
- "id": "deepseek-v4-flash",
289
- "samplingParams": {
290
- "temperature": 1.0,
291
- "top_p": 0.95,
292
- "top_k": 0,
293
- "min_p": 0.0
294
- }
295
- }
296
- ```
297
-
298
- Only OpenAI-compatible APIs apply it (`openai-completions`, `openai-responses`, `azure-openai-responses`); other APIs ignore it. Keys override pi's named request fields (for example a `temperature` key here beats the request-level temperature), so prefer it as the single source of sampling truth for a model. In `modelOverrides`, `samplingParams` merges per key with the base model's value.
299
-
300
- A constant thinking-token cap can go here too, but it will not follow `thinkingBudgets` or leave room for the answer. Prefer `compat.thinkingTokenBudgetField` (or the `supportsThinkingTokenBudget` alias) for that.
301
-
302
- ### Thinking Level Map
303
-
304
- Use `thinkingLevelMap` on a model to describe model-specific thinking controls. Keys are pi thinking levels: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Maps may contain holes; for example, a model can expose `high` and `max` without exposing `xhigh`.
305
-
306
- Values are tristate:
307
-
308
- | Value | Meaning |
309
- |-------|---------|
310
- | omitted | Standard levels through `high` use the provider's default mapping; extended `xhigh` and `max` levels are unsupported |
311
- | string | Level is supported and this value is sent to the provider |
312
- | `null` | Level is unsupported and hidden/skipped/clamped away |
313
-
314
- Example for a model that only supports off, high, and max reasoning:
315
-
316
- ```json
317
- {
318
- "id": "deepseek-v4-pro",
319
- "reasoning": true,
320
- "thinkingLevelMap": {
321
- "minimal": null,
322
- "low": null,
323
- "medium": null,
324
- "high": "high",
325
- "xhigh": null,
326
- "max": "max"
327
- }
328
- }
329
- ```
330
-
331
- Example for a model where thinking cannot be disabled:
332
-
333
- ```json
334
- {
335
- "id": "always-thinking-model",
336
- "reasoning": true,
337
- "thinkingLevelMap": {
338
- "off": null
339
- }
340
- }
341
- ```
342
-
343
- Migration: older configs that used `compat.reasoningEffortMap` should move that mapping to model-level `thinkingLevelMap`. Use `null` for levels that should not appear in the UI.
344
-
345
- ## Overriding Built-in Providers
346
-
347
- Route a built-in provider through a proxy without redefining models:
348
-
349
- ```json
350
- {
351
- "providers": {
352
- "anthropic": {
353
- "baseUrl": "https://my-proxy.example.com/v1"
354
- }
355
- }
356
- }
357
- ```
358
-
359
- All built-in Anthropic models remain available. Existing OAuth or API key auth continues to work.
360
-
361
- To merge custom models into a built-in provider, include the `models` array:
362
-
363
- ```json
364
- {
365
- "providers": {
366
- "anthropic": {
367
- "baseUrl": "https://my-proxy.example.com/v1",
368
- "apiKey": "$ANTHROPIC_API_KEY",
369
- "api": "anthropic-messages",
370
- "models": [...]
371
- }
372
- }
373
- }
374
- ```
375
-
376
- Merge semantics:
377
- - Built-in models are kept.
378
- - Custom models are upserted by `id` within the provider.
379
- - If a custom model `id` matches a built-in model `id`, the custom model replaces that built-in model.
380
- - If a custom model `id` is new, it is added alongside built-in models.
381
-
382
- ## Per-model Overrides
383
-
384
- Use `modelOverrides` to customize built-in models and matching extension-registered models without replacing the provider's full model list.
385
-
386
- ```json
387
- {
388
- "providers": {
389
- "openrouter": {
390
- "modelOverrides": {
391
- "anthropic/claude-sonnet-4": {
392
- "name": "Claude Sonnet 4 (Bedrock Route)",
393
- "compat": {
394
- "openRouterRouting": {
395
- "only": ["amazon-bedrock"]
396
- }
397
- }
398
- }
399
- }
400
- }
401
- }
402
- }
403
- ```
404
-
405
- `modelOverrides` supports these fields per model: `name`, `reasoning`, `thinkingLevelMap`, `input`, `inputLimits` (deep-merged), `cost` (partial), `promptCache` (merged per tier), `contextWindow`, `maxTokens`, `samplingParams` (merged per key), `headers`, `compat`.
406
-
407
- Use a `promptCache` override to enable cache warming through a proxy whose backing cache you know, for example OpenRouter routed to Anthropic:
89
+ `maxBytes` limits the base64-encoded payload. Omitted resize fields use conservative defaults of 2000 by 2000 pixels, 4.5 MiB encoded, and JPEG quality 80. Images are encoded once; changing models does not rewrite historical images. The catalog can also describe hard request limits with `inputLimits.maxRequestBytes`, `images.maxPerMessage`, and `images.maxPerRequest`, but Pi does not yet rewrite or reject history based on them.
408
90
 
409
- ```json
410
- {
411
- "providers": {
412
- "openrouter": {
413
- "modelOverrides": {
414
- "anthropic/claude-sonnet-4": {
415
- "promptCache": { "short": 300 }
416
- }
417
- }
418
- }
419
- }
420
- }
421
- ```
91
+ <a id="prompt-cache-lifetimes"></a>
422
92
 
423
- Direct OpenAI GPT-5.6 Sol, Terra, and Luna default to a `272000` context window so requests remain within OpenAI's short-context pricing tier. To opt into OpenAI's 1.05M context window, increase it for each model you use:
93
+ Use `promptCache` to declare the provider's best-effort cache lifetime in seconds for the `short` or `long` retention tier:
424
94
 
425
95
  ```json
426
- {
427
- "providers": {
428
- "openai": {
429
- "modelOverrides": {
430
- "gpt-5.6-sol": {
431
- "contextWindow": 1050000
432
- }
433
- }
434
- }
435
- }
436
- }
96
+ { "id": "claude-sonnet-5", "promptCache": { "short": 300, "long": 3600 } }
437
97
  ```
438
98
 
439
- The override preserves the built-in pricing metadata. Requests with more than 272K total input tokens use GPT-5.6's long-context rates for the entire request. Apply the same override to `gpt-5.6-terra` or `gpt-5.6-luna` when needed.
440
-
441
- Behavior notes:
442
- - `modelOverrides` are applied to built-in provider models and matching extension-registered provider models.
443
- - Unknown model IDs are ignored.
444
- - You can combine provider-level `baseUrl`/`headers` with `modelOverrides`.
445
- - Overriding `name` changes model matching and secondary detail text only; the footer and primary model lists continue to show the model `id`.
446
- - If `models` is also defined for a provider, custom models are merged after built-in overrides. A custom model with the same `id` replaces the overridden built-in model entry.
447
-
448
- ## Anthropic Messages Compatibility
99
+ Choose the conservative end of any published range. A model without a lifetime for the active tier is not eligible for cache warming. A `modelOverrides` entry can set `inputLimits` or `promptCache` for a built-in or extension model, including a model accessed through a validated proxy. See [`cacheWarming`](settings.md#model-and-thinking).
449
100
 
450
- For providers or proxies using `api: "anthropic-messages"`, use `compat` to control Anthropic-specific request compatibility.
101
+ Compatibility settings should describe verified differences in the endpoint's request or response behavior. Do not enable them based only on an endpoint advertising OpenAI or Anthropic compatibility.
451
102
 
452
- By default pi sends per-tool `eager_input_streaming: true`. If a proxy or Anthropic-compatible backend rejects that field, set `supportsEagerToolInputStreaming` to `false`. Pi will omit `tools[].eager_input_streaming` and send the legacy `fine-grained-tool-streaming-2025-05-14` beta header for tool-enabled requests instead.
103
+ ## Add a custom provider
453
104
 
454
- Some Anthropic models require adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`) instead of the legacy budget-based thinking payload. Built-in models set this automatically. For custom providers or aliases that route to those models, set `forceAdaptiveThinking` to `true`.
105
+ Use an extension when the provider needs custom streaming, model discovery, or authentication behavior. See [Custom Providers](custom-provider.md) for the extension workflow.
455
106
 
456
- Claude models with per-turn effort support use `supportsMidConvoEffort`. Pi then persists each response's provider effort, reconstructs effort-only system messages on later requests, and sends thinking binding controls with `prefix_mismatch_behavior: "drop_block"` to avoid stale signed-thinking prefixes causing persistent 400 responses. Set this only for the exact supported Claude model on a faithful Anthropic Messages transport; do not enable it for APIs that merely imitate the Messages shape.
107
+ ## Troubleshooting
457
108
 
458
- Some Anthropic-compatible providers emit thinking blocks with empty signatures and still expect them on replay. Set `allowEmptySignature` to `true` only for those providers; real Anthropic rejects empty thinking signatures.
109
+ ### A model does not appear
459
110
 
460
- Built-in Anthropic models enable `supportsStrictTools` in their model metadata. Custom Anthropic-compatible models must set it to `true` when their endpoint accepts strict JSON-schema tool definitions.
111
+ Confirm that its provider has usable authentication. Custom models can load from `models.json` but remain unavailable in `/model` until Pi can resolve credentials. For llama.cpp, only models currently loaded by the router appear.
461
112
 
462
- ```json
463
- {
464
- "providers": {
465
- "anthropic-proxy": {
466
- "baseUrl": "https://proxy.example.com",
467
- "api": "anthropic-messages",
468
- "apiKey": "$ANTHROPIC_PROXY_KEY",
469
- "compat": {
470
- "supportsEagerToolInputStreaming": false,
471
- "supportsLongCacheRetention": true,
472
- "forceAdaptiveThinking": true,
473
- "allowEmptySignature": true
474
- },
475
- "models": [
476
- {
477
- "id": "claude-opus-4-7",
478
- "reasoning": true,
479
- "input": ["text", "image"]
480
- }
481
- ]
482
- }
483
- }
484
- }
485
- ```
113
+ ### Authentication works in one shell only
486
114
 
487
- | Field | Description |
488
- |-------|-------------|
489
- | `supportsEagerToolInputStreaming` | Whether the provider accepts per-tool `eager_input_streaming`. Default: `true`. Set to `false` to omit that field and use the legacy fine-grained tool streaming beta header on tool-enabled requests. |
490
- | `supportsLongCacheRetention` | Whether the provider accepts Anthropic long cache retention (`cache_control.ttl: "1h"`) when cache retention is `long`. Default: `true`. |
491
- | `sendSessionAffinityHeaders` | Whether to send `x-session-affinity` from the session id when caching is enabled. Default: auto-detected for known providers. |
492
- | `supportsCacheControlOnTools` | Whether the provider accepts Anthropic-style `cache_control` markers on tool definitions. Default: `true`. |
493
- | `forceAdaptiveThinking` | Whether to send adaptive thinking (`thinking.type: "adaptive"` plus `output_config.effort`) for this model. Built-in adaptive models set this automatically. Default: `false`. |
494
- | `supportsMidConvoEffort` | Whether the exact Claude model transport supports per-turn effort system messages and thinking binding controls. Pi persists native effort levels and always sends `drop_block` when enabled. Default: `false`. |
495
- | `allowEmptySignature` | Whether to replay empty thinking signatures as `signature: ""` instead of converting thinking to text. Default: `false`. |
496
- | `supportsStrictTools` | Whether the provider accepts strict JSON-schema tool definitions. Default: `false`; built-in Anthropic models enable it in generated metadata. |
497
- | `allowedFallbackModels` | Up to three server-side fallback models, each with `provider`, `model`, and complete `cost` metadata. An empty array disables fallback. |
115
+ Check whether the key came from an environment variable rather than `auth.json`. Environment variables must be present in the process that starts Pi.
498
116
 
499
- ## OpenAI Compatibility
117
+ ### Sign-in opens a browser on a remote machine
500
118
 
501
- For providers with partial OpenAI compatibility, use the `compat` field.
119
+ Complete the provider's headless authentication flow when available. Some providers let you paste the final redirect URL or authorization code back into Pi. See [Authenticate interactively](providers.md#authenticate-interactively).
502
120
 
503
- - Provider-level `compat` applies defaults to all models under that provider.
504
- - Model-level `compat` overrides provider-level values for that model.
121
+ ### A compatible endpoint rejects requests
505
122
 
506
- ```json
507
- {
508
- "providers": {
509
- "local-llm": {
510
- "baseUrl": "http://localhost:8080/v1",
511
- "api": "openai-completions",
512
- "compat": {
513
- "supportsUsageInStreaming": false,
514
- "maxTokensField": "max_tokens"
515
- },
516
- "models": [...]
517
- }
518
- }
519
- }
520
- ```
521
-
522
- | Field | Description |
523
- |-------|-------------|
524
- | `supportsStore` | Provider supports `store` field |
525
- | `supportsDeveloperRole` | Use `developer` vs `system` role |
526
- | `supportsReasoningEffort` | Support for `reasoning_effort` parameter |
527
- | `supportsUsageInStreaming` | Supports `stream_options: { include_usage: true }` (default: `true`) |
528
- | `supportsFinishReason` | Whether streamed responses include `finish_reason`. When `false`, pi infers `stop` or `toolUse` when the stream ends. Default: `true`. |
529
- | `maxTokensField` | Use `max_completion_tokens` or `max_tokens` |
530
- | `requiresToolResultName` | Include `name` on tool result messages |
531
- | `requiresAssistantAfterToolResult` | Insert an assistant message before a user message after tool results |
532
- | `requiresThinkingAsText` | Convert thinking blocks to plain text |
533
- | `requiresReasoningContentOnAssistantMessages` | Include empty `reasoning_content` on all replayed assistant messages when reasoning is enabled |
534
- | `thinkingFormat` | Use `reasoning_effort`, `openrouter`, `deepseek`, `together`, `baseten`, `zai`, `qwen`, `chat-template`, or `qwen-chat-template` thinking parameters |
535
- | `chatTemplateKwargs` | `chat_template_kwargs` values for `thinkingFormat: "chat-template"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values |
536
- | `chatTemplateArgs` | `chat_template_args` values for `thinkingFormat: "baseten"`; use `{ "$var": "thinking.enabled" }`, `{ "$var": "thinking.effort" }`, or `{ "$var": "thinking.budget" }` for pi-controlled thinking values |
537
- | `thinkingTokenBudgetField` | Top-level request field used to cap reasoning tokens from `thinkingBudgets`, clamped so at least 1024 tokens remain for the answer. `"thinking_token_budget"` (vLLM), `"thinking_budget"` (Qwen/DashScope/SGLang), `"thinking_budget_tokens"` (llama.cpp). Off by default; not set on the generated catalog. |
538
- | `supportsThinkingTokenBudget` | Alias for `thinkingTokenBudgetField: "thinking_token_budget"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: `false`. |
539
- | `cacheControlFormat` | Use Anthropic-style `cache_control` markers on the system prompt, last tool definition, and last user, assistant, or tool-result text content. Currently only `anthropic` is supported. |
540
- | `sendSessionAffinityHeaders` | For `openai-completions`, send session-affinity headers from the session id when caching is enabled. Default: `false`. |
541
- | `sessionAffinityFormat` | For `openai-completions` and `openai-responses`, the session-affinity header format: `openai` sends `session_id`/`x-client-request-id` (completions also `x-session-affinity`), `openai-nosession` omits the underscore-containing `session_id` header, `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param. Default: auto-detected. |
542
- | `supportsStrictMode` | Whether the provider accepts strict JSON-schema function tool definitions. Defaults depend on the API; built-in OpenAI models carry explicit capability metadata. |
543
- | `supportsOpenAIGrammarTools` | Whether OpenAI-compatible APIs emit custom Lark/regex grammar tools. When `false`, grammar-constrained tools fall back to normal function tools. Default: `false`; the built-in model catalog enables it for GPT-5+ models on OpenAI, OpenAI Codex, Azure OpenAI, GitHub Copilot, opencode, and Cloudflare AI Gateway. |
544
- | `supportsLongCacheRetention` | Whether the provider accepts long cache retention when cache retention is `long`: `prompt_cache_options.ttl: "30m"` for GPT-5.6+ Responses models, `prompt_cache_retention: "24h"` for earlier OpenAI models, or `cache_control.ttl: "1h"` when `cacheControlFormat` is `anthropic`. Default: `true`. |
545
- | `openRouterRouting` | OpenRouter provider routing preferences. This object is sent as-is in the `provider` field of the [OpenRouter API request](https://openrouter.ai/docs/guides/routing/provider-selection). |
546
- | `vercelGatewayRouting` | Vercel AI Gateway routing config for provider selection (`only`, `order`) |
547
-
548
- `openrouter` uses `reasoning: { effort }`. `together` uses `reasoning: { enabled }` and also `reasoning_effort` when `supportsReasoningEffort` is enabled. `qwen` uses top-level `enable_thinking`. Use `qwen-chat-template` for local Qwen-compatible servers that require `chat_template_kwargs.enable_thinking` and `preserve_thinking`. Use `chat-template` for vLLM/Hugging Face chat templates that need configurable `chat_template_kwargs`, such as `chatTemplateKwargs: { "thinking": { "$var": "thinking.enabled" } }` for DeepSeek V3.x templates. Use `thinkingFormat: "baseten"` with `chatTemplateArgs` for providers that expose toggle controls through `chat_template_args` and optionally support top-level `reasoning_effort`.
549
-
550
- `thinkingTokenBudgetField` is independent of `thinkingFormat`. Do not enable it on the generated Qwen catalog: those models already send `reasoning_effort`, and DashScope rejects `thinking_budget` together with `reasoning_effort`.
551
-
552
- `cacheControlFormat: "anthropic"` is for OpenAI-compatible providers that expose Anthropic-style prompt caching through `cache_control` markers on text content and tool definitions.
553
-
554
- Example:
555
-
556
- ```json
557
- {
558
- "providers": {
559
- "openrouter": {
560
- "baseUrl": "https://openrouter.ai/api/v1",
561
- "apiKey": "$OPENROUTER_API_KEY",
562
- "api": "openai-completions",
563
- "models": [
564
- {
565
- "id": "openrouter/anthropic/claude-3.5-sonnet",
566
- "name": "OpenRouter Claude 3.5 Sonnet",
567
- "compat": {
568
- "openRouterRouting": {
569
- "allow_fallbacks": true,
570
- "require_parameters": false,
571
- "data_collection": "deny",
572
- "zdr": true,
573
- "enforce_distillable_text": false,
574
- "order": ["anthropic", "amazon-bedrock", "google-vertex"],
575
- "only": ["anthropic", "amazon-bedrock"],
576
- "ignore": ["gmicloud", "friendli"],
577
- "quantizations": ["fp16", "bf16"],
578
- "sort": {
579
- "by": "price",
580
- "partition": "model"
581
- },
582
- "max_price": {
583
- "prompt": 10,
584
- "completion": 20
585
- },
586
- "preferred_min_throughput": {
587
- "p50": 100,
588
- "p90": 50
589
- },
590
- "preferred_max_latency": {
591
- "p50": 1,
592
- "p90": 3,
593
- "p99": 5
594
- }
595
- }
596
- }
597
- }
598
- ]
599
- }
600
- }
601
- }
602
- ```
603
-
604
- Vercel AI Gateway example:
605
-
606
- ```json
607
- {
608
- "providers": {
609
- "vercel-ai-gateway": {
610
- "baseUrl": "https://ai-gateway.vercel.sh/v1",
611
- "apiKey": "$AI_GATEWAY_API_KEY",
612
- "api": "openai-completions",
613
- "models": [
614
- {
615
- "id": "moonshotai/kimi-k2.5",
616
- "name": "Kimi K2.5 (Fireworks via Vercel)",
617
- "reasoning": true,
618
- "input": ["text", "image"],
619
- "cost": { "input": 0.6, "output": 3, "cacheRead": 0, "cacheWrite": 0 },
620
- "contextWindow": 262144,
621
- "maxTokens": 262144,
622
- "compat": {
623
- "vercelGatewayRouting": {
624
- "only": ["fireworks", "novita"],
625
- "order": ["fireworks", "novita"]
626
- }
627
- }
628
- }
629
- ]
630
- }
631
- }
632
- }
633
- ```
123
+ Check its API type and compatibility settings in `models.json`. The upstream server must support the corresponding request fields and behavior.