pi-llama-cpp 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +187 -10
- package/package.json +6 -6
- package/src/api/client.ts +76 -32
- package/src/constants.ts +15 -0
- package/src/index.ts +3 -5
- package/src/interfaces/events.ts +14 -3
- package/src/interfaces/server.ts +27 -0
- package/src/interfaces/settings.ts +76 -2
- package/src/managers/command.ts +369 -50
- package/src/managers/events.ts +28 -10
- package/src/managers/server.ts +90 -16
- package/src/managers/settings.ts +181 -44
- package/src/models/baseModel.ts +37 -15
- package/src/models/routerModel.ts +2 -1
- package/src/server.ts +103 -28
- package/src/sse/client.ts +28 -16
- package/src/sse/manager.ts +26 -13
- package/src/ui/dialog.ts +287 -0
- package/src/ui/overrideEntryEditor.ts +119 -0
- package/src/ui/overrideSettingsList.ts +682 -0
- package/src/ui/serverListEditor.ts +32 -0
- package/src/ui/serverSettingsList.ts +466 -0
- package/src/ui/strings.ts +127 -0
- package/src/utils/errors.ts +5 -0
- package/src/utils/settingsStore.ts +56 -0
- package/src/utils/urls.ts +16 -0
- package/tests/commandManager.test.ts +346 -11
- package/tests/dialog.test.ts +186 -0
- package/tests/events.test.ts +120 -88
- package/tests/legacyModel.test.ts +4 -19
- package/tests/mocks.ts +149 -32
- package/tests/overrides.test.ts +352 -0
- package/tests/server.test.ts +42 -40
- package/tests/serverManager.test.ts +264 -55
- package/tests/settings.test.ts +654 -51
- package/tests/settingsStore.test.ts +190 -0
- package/tests/singleModel.test.ts +32 -0
- package/tests/sseManager.test.ts +88 -11
- package/src/interfaces/auth.ts +0 -6
- package/src/utils/cache.ts +0 -39
- package/src/utils/mutex.ts +0 -24
package/README.md
CHANGED
|
@@ -61,6 +61,18 @@ The recommended way to configure the extension is using the `llamaSettings` key.
|
|
|
61
61
|
|
|
62
62
|
Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global):
|
|
63
63
|
|
|
64
|
+
#### Minimal configuration
|
|
65
|
+
|
|
66
|
+
```json
|
|
67
|
+
{
|
|
68
|
+
"llamaSettings": {
|
|
69
|
+
"servers": [{ "url": "http://127.0.0.1:8080" }]
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
#### Full configuration
|
|
75
|
+
|
|
64
76
|
```json
|
|
65
77
|
{
|
|
66
78
|
"llamaSettings": {
|
|
@@ -68,7 +80,8 @@ Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (g
|
|
|
68
80
|
{
|
|
69
81
|
"url": "http://127.0.0.1:8080",
|
|
70
82
|
"id": "local",
|
|
71
|
-
"name": "Local Server"
|
|
83
|
+
"name": "Local Server",
|
|
84
|
+
"overrides": {}
|
|
72
85
|
},
|
|
73
86
|
{
|
|
74
87
|
"url": "http://10.0.0.5:8080",
|
|
@@ -77,6 +90,7 @@ Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (g
|
|
|
77
90
|
],
|
|
78
91
|
"reactToModelSelect": true,
|
|
79
92
|
"autoloadOnMessage": false,
|
|
93
|
+
"sortBy": "asc",
|
|
80
94
|
"pollingTimeout": 60000,
|
|
81
95
|
"serverTimeout": 1000
|
|
82
96
|
}
|
|
@@ -91,7 +105,7 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
|
|
|
91
105
|
| ------ | ------ | -------- | ---------------------------------------------------------------------------- |
|
|
92
106
|
| `url` | string | Yes | The URL of the llama.cpp server |
|
|
93
107
|
| `id` | string | No | Custom provider ID (used for API key auth). Defaults to `llama-server=<url>` |
|
|
94
|
-
| `name` | string | No | Display name for the server in the UI (shown as `Llama.cpp
|
|
108
|
+
| `name` | string | No | Display name for the server in the UI (shown as `Llama.cpp (<name>)`) |
|
|
95
109
|
|
|
96
110
|
> **Note:** If you set a custom `id`, you can use it in `~/.pi/agent/auth.json`. The extension will also fall back to the URL-based ID if no key is found for the custom `id`.
|
|
97
111
|
|
|
@@ -101,11 +115,32 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
|
|
|
101
115
|
| -------------------- | ------- | ------- | ------------------------------------------------------------- |
|
|
102
116
|
| `reactToModelSelect` | boolean | `true` | Load the model when you switch via Pi's model picker. |
|
|
103
117
|
| `autoloadOnMessage` | boolean | `false` | Automatically load an unloaded model before sending a message |
|
|
118
|
+
| `sortBy` | string | `"asc"` | Sort order for models (see below) |
|
|
104
119
|
| `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
|
|
105
120
|
| `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
|
|
106
121
|
|
|
107
122
|
> **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
|
|
108
123
|
|
|
124
|
+
#### In-session settings menu
|
|
125
|
+
|
|
126
|
+
Run `/models settings` to edit the scalar settings above without hand-editing JSON. Changes are written to the **project** `.pi/settings.json` if it exists, otherwise to **global** `~/.pi/agent/settings.json`. Boolean and sort changes apply immediately; timeout changes apply on the next model load. The `servers` list is edited with `/models servers` (see below), and per-server model overrides with `/models overrides` (see [Model Overrides](#model-overrides)).
|
|
127
|
+
|
|
128
|
+
#### Server list editor
|
|
129
|
+
|
|
130
|
+
Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
|
|
131
|
+
without hand-editing JSON. Each change is written immediately to the **project**
|
|
132
|
+
`.pi/settings.json` if it exists, otherwise to **global**
|
|
133
|
+
`~/.pi/agent/settings.json`.
|
|
134
|
+
|
|
135
|
+
Changes take effect immediately after closing the editor: new servers
|
|
136
|
+
register their providers, removed ones leave pi's registry right away,
|
|
137
|
+
and edited ones are re-registered with the fresh config — no restart or
|
|
138
|
+
`/models` needed.
|
|
139
|
+
|
|
140
|
+
Limitation: a model already loading in the background on a removed or
|
|
141
|
+
edited server finishes loading, but its progress notifications stop;
|
|
142
|
+
re-select it from the (new) provider afterwards.
|
|
143
|
+
|
|
109
144
|
#### Environment variable
|
|
110
145
|
|
|
111
146
|
For a quick setup, you can use the `LLAMA_SERVER_URL` environment variable instead of the JSON config:
|
|
@@ -200,6 +235,7 @@ llama-server --model path/to/model.gguf ...
|
|
|
200
235
|
|
|
201
236
|
The extension determines the context size as follows:
|
|
202
237
|
|
|
238
|
+
- A per-model `contextSize` override (see [Model Overrides](#model-overrides)) takes precedence over everything below
|
|
203
239
|
- **Router mode**
|
|
204
240
|
- When loaded, reads `meta.n_ctx` from the `/v1/models` endpoint
|
|
205
241
|
- When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
|
|
@@ -209,11 +245,14 @@ The extension determines the context size as follows:
|
|
|
209
245
|
|
|
210
246
|
### Commands
|
|
211
247
|
|
|
212
|
-
| Command
|
|
213
|
-
|
|
|
214
|
-
| `/models`
|
|
215
|
-
| `/models info`
|
|
216
|
-
| `/models unload`
|
|
248
|
+
| Command | Description |
|
|
249
|
+
| ------------------- | --------------------------------------------------------------------------------------- |
|
|
250
|
+
| `/models` | Browse your models with live status. Select a model to load, switch, or unload it. |
|
|
251
|
+
| `/models info` | Show detailed information for all available models at once. |
|
|
252
|
+
| `/models unload` | Unload all loaded models at once. |
|
|
253
|
+
| `/models settings` | Open a menu to edit the scalar `llamaSettings` fields. |
|
|
254
|
+
| `/models servers` | Add, edit or remove llama.cpp server URLs via a TUI editor. |
|
|
255
|
+
| `/models overrides` | Edit per-server model overrides (`llamaSettings.servers[].overrides`) via a TUI editor. |
|
|
217
256
|
|
|
218
257
|
> **Note:** When a llama.cpp server is slow to respond, it will be skipped at startup with a warning. Run `/models` to retry without timeout and see all models.
|
|
219
258
|
|
|
@@ -221,6 +260,19 @@ The extension determines the context size as follows:
|
|
|
221
260
|
|
|
222
261
|
> **Note:** The `/models unload` command only makes sense in router mode.
|
|
223
262
|
|
|
263
|
+
#### Model sorting
|
|
264
|
+
|
|
265
|
+
The order of models in the `/models` menu is controlled by the `sortBy` setting.
|
|
266
|
+
Servers maintain their order from `llamaSettings`; sorting applies **within each server**:
|
|
267
|
+
|
|
268
|
+
| Value | Description |
|
|
269
|
+
| ------------- | ------------------------------------------------------------------------------------------------------------------------- |
|
|
270
|
+
| `"asc"` | Sort by model ID ascending (default) |
|
|
271
|
+
| `"desc"` | Sort by model ID descending |
|
|
272
|
+
| `"asc-name"` | Sort by model name ascending (ties broken by ID) |
|
|
273
|
+
| `"desc-name"` | Sort by model name descending (ties broken by ID) |
|
|
274
|
+
| `"api"` | No sorting — models appear in the order returned by each server's `/v1/models` endpoint, servers in `llamaSettings` order |
|
|
275
|
+
|
|
224
276
|
### Model Actions
|
|
225
277
|
|
|
226
278
|
When browsing models via the `/models` command, you can:
|
|
@@ -266,6 +318,129 @@ User-defined budgets can override the defaults by adding a `thinkingBudgets` obj
|
|
|
266
318
|
Only `minimal`, `low`, `medium`, `high` and `xhigh` are configurable — `off` (0) and `max` (-1, unlimited) are fixed.
|
|
267
319
|
The extension automatically injects the appropriate `thinking_budget_tokens` into each request payload based on the selected level.
|
|
268
320
|
|
|
321
|
+
### Model Overrides
|
|
322
|
+
|
|
323
|
+
A locally-run `llama.cpp` server is free, but you can simulate costs for budgeting, experimentation, or comparison purposes — and fine-tune what the extension reports about each model.
|
|
324
|
+
|
|
325
|
+
This extension supports **per-model, per-server configuration** via the `overrides` key inside each server entry of `llamaSettings.servers`. Each entry can override the model's `cost`, `capabilities`, `reasoning`, `contextSize`, `maxTokens`, and `compat`, regardless of what the server reports.
|
|
326
|
+
|
|
327
|
+
Add overrides to your server configuration:
|
|
328
|
+
|
|
329
|
+
```json
|
|
330
|
+
{
|
|
331
|
+
"llamaSettings": {
|
|
332
|
+
"servers": [
|
|
333
|
+
{
|
|
334
|
+
"url": "http://127.0.0.1:8080",
|
|
335
|
+
"overrides": {
|
|
336
|
+
"qwen-3.8-27b": {
|
|
337
|
+
"cost": { "input": 0.42, "output": 3.0, "cacheRead": 0.085 }
|
|
338
|
+
},
|
|
339
|
+
"glm-5.3-flash": {
|
|
340
|
+
"cost": { "input": 0.15, "output": 0.5, "cacheRead": 0.03 },
|
|
341
|
+
"capabilities": ["text"],
|
|
342
|
+
"reasoning": false,
|
|
343
|
+
"contextSize": 32768,
|
|
344
|
+
"maxTokens": 4096
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
]
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
```
|
|
352
|
+
|
|
353
|
+
Every field of an override is optional — absent fields fall back to what the extension detects (`capabilities`) or to its defaults (`reasoning: true`, zeroed cost).
|
|
354
|
+
|
|
355
|
+
#### Override editor
|
|
356
|
+
|
|
357
|
+
Run `/models overrides` to edit a server's override entries without hand-editing
|
|
358
|
+
JSON. It opens a settings menu (same UX as `/models settings`).
|
|
359
|
+
|
|
360
|
+
Each change is written immediately to the **project**
|
|
361
|
+
`.pi/settings.json` if it exists, otherwise to **global**
|
|
362
|
+
`~/.pi/agent/settings.json`.
|
|
363
|
+
|
|
364
|
+
Overrides take effect on the next provider request after closing the editor —
|
|
365
|
+
no `/reload` needed.
|
|
366
|
+
|
|
367
|
+
#### Cost Fields
|
|
368
|
+
|
|
369
|
+
Inside an override, the `cost` object accepts:
|
|
370
|
+
|
|
371
|
+
| Field | Type | Description |
|
|
372
|
+
| ------------ | ------ | ----------------------------------- |
|
|
373
|
+
| `input` | number | Cost per million input tokens |
|
|
374
|
+
| `output` | number | Cost per million output tokens |
|
|
375
|
+
| `cacheRead` | number | Cost per million cache read tokens |
|
|
376
|
+
| `cacheWrite` | number | Cost per million cache write tokens |
|
|
377
|
+
|
|
378
|
+
All four fields are optional — unspecified fields default to zero.
|
|
379
|
+
In the override editor, entering `0` (or leaving a field empty) removes the field from the settings — and the `cost` object itself once no fields remain — with the same effect as leaving it unset.
|
|
380
|
+
|
|
381
|
+
#### Other Fields
|
|
382
|
+
|
|
383
|
+
| Field | Type | Description |
|
|
384
|
+
| -------------- | ---------------- | ---------------------------------------------------------------------------------------------- |
|
|
385
|
+
| `capabilities` | array of strings | Pi capabilities for the model (`"text"`, `"image"`). Fully replaces the detected capabilities. |
|
|
386
|
+
| `reasoning` | boolean | Whether the model is a reasoning model. Defaults to `true` when absent. |
|
|
387
|
+
| `contextSize` | number | Override the model's context size in tokens. Falls back to autodetection when absent or `0`. |
|
|
388
|
+
| `maxTokens` | number | Override max generation tokens. Falls back to context size when absent or `0`. |
|
|
389
|
+
| `compat` | object | OpenAI-compatible provider compatibility settings (see below). |
|
|
390
|
+
|
|
391
|
+
#### Compatibility (`compat`)
|
|
392
|
+
|
|
393
|
+
The `compat` field accepts any subset of [OpenAI-compatible provider compatibility settings](https://github.com/earendil-works/pi/blob/main/packages/ai/src/types.ts) used by the `openai-completions` API. These control how the extension talks to your server — for example, disabling `developer` role support, choosing the thinking format, enabling Anthropic-style cache control, or setting thinking token budgets.
|
|
394
|
+
|
|
395
|
+
Example:
|
|
396
|
+
|
|
397
|
+
```json
|
|
398
|
+
{
|
|
399
|
+
"llamaSettings": {
|
|
400
|
+
"servers": [
|
|
401
|
+
{
|
|
402
|
+
"url": "http://127.0.0.1:8080",
|
|
403
|
+
"overrides": {
|
|
404
|
+
"llama-3": {
|
|
405
|
+
"compat": {
|
|
406
|
+
"supportsDeveloperRole": false,
|
|
407
|
+
"thinkingFormat": "openai",
|
|
408
|
+
"thinkingTokenBudgetField": "thinking_budget_tokens"
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
]
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
```
|
|
417
|
+
|
|
418
|
+
### Prefix matching
|
|
419
|
+
|
|
420
|
+
Override keys are treated as **prefix filters** — a model ID matches if it starts with the key. When multiple patterns match, the **longest (most specific) match wins**. This lets you define broad patterns at the top of your overrides and override them with more specific ones below.
|
|
421
|
+
|
|
422
|
+
Example:
|
|
423
|
+
|
|
424
|
+
```json
|
|
425
|
+
{
|
|
426
|
+
"llama": { "cost": { "input": 0.01, "output": 0.02 } },
|
|
427
|
+
"llama-3": { "reasoning": false },
|
|
428
|
+
"llama-3-8b": { "cost": { "input": 0.2, "output": 0.6 } }
|
|
429
|
+
}
|
|
430
|
+
```
|
|
431
|
+
|
|
432
|
+
| Model ID | Matching keys | Winner (longest) | Effective override |
|
|
433
|
+
| ------------- | -------------------------------- | ---------------- | ------------------------------------------------------ |
|
|
434
|
+
| `llama-3-8b` | `llama`, `llama-3`, `llama-3-8b` | `llama-3-8b` | `{ cost: { input: 0.2, output: 0.6 } }` |
|
|
435
|
+
| `llama-3-70b` | `llama`, `llama-3` | `llama-3` | `{ reasoning: false }` |
|
|
436
|
+
| `mistral-7b` | `llama` (no) | none | defaults (zero cost, detected caps, `reasoning: true`) |
|
|
437
|
+
|
|
438
|
+
> **Note:** Exact model IDs still work — they are simply the longest possible prefix for themselves. Empty keys are silently ignored.
|
|
439
|
+
|
|
440
|
+
Model matching uses this prefix system — the model ID must start with the override key for a match.
|
|
441
|
+
|
|
442
|
+
> **Note:** Overrides are resolved through the same settings merge logic (project overrides global), so they follow the same precedence chain as other server settings. If the same URL appears multiple times with different `overrides`, only the first one's overrides will be used (consistent with existing dedup behavior).
|
|
443
|
+
|
|
269
444
|
### Model Selection Event
|
|
270
445
|
|
|
271
446
|
When you switch models via Pi's model picker (instead of using the `/models` command), the extension listens for the `model_select` event, which also loads the requested model before the conversation begins.
|
|
@@ -288,9 +463,11 @@ If loading takes longer than **60 seconds** (configurable via `pollingTimeout`),
|
|
|
288
463
|
|
|
289
464
|
Each model exposed to Pi includes the following defaults:
|
|
290
465
|
|
|
291
|
-
- **`
|
|
292
|
-
- **`
|
|
293
|
-
- **`
|
|
466
|
+
- **`contextWindow`** — detected from llama-server (see how the extension determines the context size above); can be overridden per-model via `llamaSettings.servers[].overrides` (see [Model Overrides](#model-overrides))
|
|
467
|
+
- **`maxTokens`** — dynamically set to the model's context window (detected from llama-server); can be overridden per-model via `llamaSettings.servers[].overrides` (see [Model Overrides](#model-overrides))
|
|
468
|
+
- **`reasoning`** — `true` by default (llama.cpp's `/v1/models` endpoint does not expose it); can be overridden per-model via `llamaSettings.servers[].overrides` (see [Model Overrides](#model-overrides))
|
|
469
|
+
- **`cost`** — all zero by default; can be customized per-model via `llamaSettings.servers[].overrides` (see [Model Overrides](#model-overrides))
|
|
470
|
+
- **`compat`** — OpenAI-compatible provider compatibility settings; can be set per-model via `llamaSettings.servers[].overrides` (see [Model Overrides](#model-overrides))
|
|
294
471
|
|
|
295
472
|
## Dependencies
|
|
296
473
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.12.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -32,14 +32,14 @@
|
|
|
32
32
|
]
|
|
33
33
|
},
|
|
34
34
|
"peerDependencies": {
|
|
35
|
-
"@earendil-works/pi-ai": ">=0.
|
|
36
|
-
"@earendil-works/pi-coding-agent": ">=0.
|
|
37
|
-
"@earendil-works/pi-tui": ">=0.
|
|
35
|
+
"@earendil-works/pi-ai": ">=0.85.1",
|
|
36
|
+
"@earendil-works/pi-coding-agent": ">=0.85.1",
|
|
37
|
+
"@earendil-works/pi-tui": ">=0.85.1"
|
|
38
38
|
},
|
|
39
39
|
"type": "module",
|
|
40
40
|
"devDependencies": {
|
|
41
|
-
"@types/node": "^26.4.
|
|
41
|
+
"@types/node": "^26.4.1",
|
|
42
42
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
43
|
-
"vitest": "^
|
|
43
|
+
"vitest": "^5.0.0"
|
|
44
44
|
}
|
|
45
45
|
}
|
package/src/api/client.ts
CHANGED
|
@@ -1,13 +1,29 @@
|
|
|
1
1
|
import { POLLING_INTERVAL } from "../constants";
|
|
2
|
-
import { Cache } from "../utils/cache";
|
|
3
|
-
import { Mutex } from "../utils/mutex";
|
|
4
2
|
|
|
5
3
|
/**
|
|
6
|
-
*
|
|
4
|
+
* How long GET responses stay cached: half the polling interval, so a poll
|
|
5
|
+
* tick always reaches the server while multiple reads within one tick
|
|
6
|
+
* (fan-outs like `toProviderConfig`, concurrent model polls) do not.
|
|
7
|
+
*/
|
|
8
|
+
const CACHE_TTL = POLLING_INTERVAL / 2;
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* HTTP client for llama-server with GET caching and request deduplication.
|
|
12
|
+
*
|
|
13
|
+
* Two complementary mechanisms keep repeated reads from reaching the server:
|
|
14
|
+
* - the TTL cache absorbs time-spaced repeats (poll loops, sequential reads);
|
|
15
|
+
* - the in-flight map absorbs simultaneous bursts: concurrent callers for the
|
|
16
|
+
* same key share one request's promise.
|
|
17
|
+
*
|
|
18
|
+
* POST requests are deduplicated but never cached — they are not idempotent
|
|
19
|
+
* reads, and the only caller (`Server.postRequest()`) clears the cache around
|
|
20
|
+
* them. The dedup matters: independent load triggers (command, auto-load,
|
|
21
|
+
* model-select) can race, and merging their duplicate `POST /models/load`
|
|
22
|
+
* into one server call avoids load-state glitches.
|
|
7
23
|
*/
|
|
8
24
|
export class ApiClient {
|
|
9
|
-
private cache = new
|
|
10
|
-
private
|
|
25
|
+
private cache = new Map<string, { data: unknown; timestamp: number }>();
|
|
26
|
+
private inflight = new Map<string, Promise<unknown>>();
|
|
11
27
|
|
|
12
28
|
/**
|
|
13
29
|
* Creates a new ApiClient.
|
|
@@ -22,40 +38,34 @@ export class ApiClient {
|
|
|
22
38
|
|
|
23
39
|
/**
|
|
24
40
|
* Makes a cached, deduplicated GET request to the llama-server.
|
|
25
|
-
* Results are cached for half the polling interval and in-flight requests are deduplicated.
|
|
26
41
|
*
|
|
27
42
|
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
28
43
|
* @returns The parsed JSON response from the server
|
|
29
44
|
*/
|
|
30
45
|
async get<T>(endpoint: string): Promise<T> {
|
|
31
|
-
const cached = this.
|
|
46
|
+
const cached = this.cacheGet<T>(endpoint);
|
|
32
47
|
if (cached !== undefined) return cached;
|
|
33
48
|
|
|
34
|
-
return this.
|
|
35
|
-
const data =
|
|
36
|
-
this.
|
|
49
|
+
return this.dedupe(endpoint, async () => {
|
|
50
|
+
const data = await this.do_get<T>(endpoint);
|
|
51
|
+
this.cacheSet(endpoint, data);
|
|
37
52
|
return data;
|
|
38
53
|
});
|
|
39
54
|
}
|
|
40
55
|
|
|
41
56
|
/**
|
|
42
|
-
* Makes a
|
|
43
|
-
*
|
|
57
|
+
* Makes a deduplicated POST request to the llama-server.
|
|
58
|
+
* Concurrent duplicate requests share one server call; responses are
|
|
59
|
+
* never cached.
|
|
44
60
|
*
|
|
45
61
|
* @param endpoint The endpoint path to post to
|
|
46
62
|
* @param body The optional request body
|
|
47
63
|
* @returns The parsed JSON response from the server
|
|
48
64
|
*/
|
|
49
65
|
async post<T>(endpoint: string, body?: Record<string, unknown>): Promise<T> {
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
return this.mutex.getOrCreate(key, async () => {
|
|
55
|
-
const data = (await this.do_post<T>(endpoint, body)) as T;
|
|
56
|
-
this.cache.set(key, data);
|
|
57
|
-
return data;
|
|
58
|
-
});
|
|
66
|
+
return this.dedupe(this.cacheKey(endpoint, body), async () =>
|
|
67
|
+
this.do_post<T>(endpoint, body),
|
|
68
|
+
);
|
|
59
69
|
}
|
|
60
70
|
|
|
61
71
|
/**
|
|
@@ -65,6 +75,51 @@ export class ApiClient {
|
|
|
65
75
|
this.cache.clear();
|
|
66
76
|
}
|
|
67
77
|
|
|
78
|
+
/**
|
|
79
|
+
* Runs `fn` for the given key, or returns an existing in-flight promise.
|
|
80
|
+
* Concurrent callers for the same key share the same promise.
|
|
81
|
+
*/
|
|
82
|
+
private dedupe<T>(key: string, fn: () => Promise<T>): Promise<T> {
|
|
83
|
+
const existing = this.inflight.get(key);
|
|
84
|
+
if (existing) return existing as Promise<T>;
|
|
85
|
+
|
|
86
|
+
const promise = fn().finally(() => {
|
|
87
|
+
this.inflight.delete(key);
|
|
88
|
+
});
|
|
89
|
+
this.inflight.set(key, promise);
|
|
90
|
+
return promise;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Gets a cached GET response. Returns `undefined` if missing or expired.
|
|
95
|
+
*/
|
|
96
|
+
private cacheGet<T>(key: string): T | undefined {
|
|
97
|
+
const entry = this.cache.get(key);
|
|
98
|
+
if (!entry) return undefined;
|
|
99
|
+
if (Date.now() - entry.timestamp > CACHE_TTL) {
|
|
100
|
+
this.cache.delete(key);
|
|
101
|
+
return undefined;
|
|
102
|
+
}
|
|
103
|
+
return entry.data as T;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Stores a GET response in the cache with the current timestamp.
|
|
108
|
+
*/
|
|
109
|
+
private cacheSet(key: string, data: unknown): void {
|
|
110
|
+
this.cache.set(key, { data, timestamp: Date.now() });
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Builds the dedup key for a POST request.
|
|
115
|
+
*
|
|
116
|
+
* @param endpoint The endpoint path to post to
|
|
117
|
+
* @param body The optional request body
|
|
118
|
+
*/
|
|
119
|
+
private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
|
|
120
|
+
return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
|
|
121
|
+
}
|
|
122
|
+
|
|
68
123
|
/**
|
|
69
124
|
* Makes a raw GET request to the llama-server.
|
|
70
125
|
* This bypasses caching and deduplication.
|
|
@@ -107,15 +162,4 @@ export class ApiClient {
|
|
|
107
162
|
|
|
108
163
|
return res.json();
|
|
109
164
|
}
|
|
110
|
-
|
|
111
|
-
/**
|
|
112
|
-
* Sets a cache key
|
|
113
|
-
*
|
|
114
|
-
* @param endpoint The endpoint path to post to
|
|
115
|
-
* @param body The optional request body
|
|
116
|
-
* @returns The cache key
|
|
117
|
-
*/
|
|
118
|
-
private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
|
|
119
|
-
return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
|
|
120
|
-
}
|
|
121
165
|
}
|
package/src/constants.ts
CHANGED
|
@@ -3,6 +3,11 @@
|
|
|
3
3
|
*/
|
|
4
4
|
export const PROVIDER_PREFIX = "llama-server";
|
|
5
5
|
|
|
6
|
+
/**
|
|
7
|
+
* The settings key used in project/global settings.
|
|
8
|
+
*/
|
|
9
|
+
export const SETTINGS_KEY = "llamaSettings";
|
|
10
|
+
|
|
6
11
|
/**
|
|
7
12
|
* This provider's name
|
|
8
13
|
*/
|
|
@@ -58,6 +63,16 @@ export const REACT_TO_MODEL_SELECT = true;
|
|
|
58
63
|
*/
|
|
59
64
|
export const AUTOLOAD_ON_MESSAGE = false;
|
|
60
65
|
|
|
66
|
+
/**
|
|
67
|
+
* Default sort order for model lists.
|
|
68
|
+
*/
|
|
69
|
+
export const SORT_BY = "asc";
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Sort order options for model lists.
|
|
73
|
+
*/
|
|
74
|
+
export type SortBy = "asc" | "desc" | "asc-name" | "desc-name" | "api";
|
|
75
|
+
|
|
61
76
|
/**
|
|
62
77
|
* Thinking budgets to send to the server, depending on user-selected level in Pi.
|
|
63
78
|
*/
|
package/src/index.ts
CHANGED
|
@@ -14,11 +14,9 @@ import { ServerManager } from "./managers/server";
|
|
|
14
14
|
import { settings } from "./managers/settings";
|
|
15
15
|
|
|
16
16
|
export default async function (pi: ExtensionAPI) {
|
|
17
|
-
const
|
|
18
|
-
|
|
19
|
-
const
|
|
20
|
-
const serverManager = new ServerManager(servers);
|
|
21
|
-
const commandManager = new CommandManager(serverManager);
|
|
17
|
+
const serverManager = new ServerManager(settings);
|
|
18
|
+
const eventManager = new EventManager(serverManager, settings);
|
|
19
|
+
const commandManager = new CommandManager(serverManager, settings);
|
|
22
20
|
|
|
23
21
|
// Register providers once at startup
|
|
24
22
|
await serverManager.initialize(pi);
|
package/src/interfaces/events.ts
CHANGED
|
@@ -1,3 +1,14 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
import type { ExtensionEvent } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* pi-coding-agent does not re-export its `ModelSelectEvent` from the public
|
|
5
|
+
* API (it exists in the internal extensions/index but is omitted from the
|
|
6
|
+
* root re-export list, and the package's `exports` map blocks deep imports).
|
|
7
|
+
* It is therefore derived from the exported `ExtensionEvent` union, which
|
|
8
|
+
* yields pi's real event shape (`model: Model<any>`, `previousModel`,
|
|
9
|
+
* `source`) and tracks the API automatically.
|
|
10
|
+
*/
|
|
11
|
+
export type ModelSelectEvent = Extract<
|
|
12
|
+
ExtensionEvent,
|
|
13
|
+
{ type: "model_select" }
|
|
14
|
+
>;
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import type { ModelOverride } from "./settings";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Identity of a llama.cpp server endpoint.
|
|
5
|
+
*
|
|
6
|
+
* Persisted counterpart: {@link LlamaServer} (`llamaSettings.servers`),
|
|
7
|
+
* whose `url`/`id`/`name` map onto these fields.
|
|
8
|
+
*/
|
|
9
|
+
export interface ServerOptions {
|
|
10
|
+
/**
|
|
11
|
+
* Base URL of the llama.cpp server (e.g. "http://127.0.0.1:8080").
|
|
12
|
+
*/
|
|
13
|
+
baseUrl: string;
|
|
14
|
+
/**
|
|
15
|
+
* Custom provider ID; falls back to a URL-based one.
|
|
16
|
+
*/
|
|
17
|
+
customId?: string;
|
|
18
|
+
/**
|
|
19
|
+
* Custom provider name suffix; falls back to the base URL.
|
|
20
|
+
*/
|
|
21
|
+
customName?: string;
|
|
22
|
+
/**
|
|
23
|
+
* Per-model overrides resolved from `llamaSettings.servers`. See
|
|
24
|
+
* {@link ModelOverride} for the fallback semantics of each field.
|
|
25
|
+
*/
|
|
26
|
+
overrides?: Record<string, ModelOverride>;
|
|
27
|
+
}
|
|
@@ -1,7 +1,47 @@
|
|
|
1
|
+
import type { ModelCost, OpenAICompletionsCompat } from "@earendil-works/pi-ai";
|
|
2
|
+
import type { SortBy } from "../constants";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Per-model overrides applied on top of what llama-server reports.
|
|
6
|
+
* Every field is optional — absent fields fall back to detection/defaults.
|
|
7
|
+
*/
|
|
8
|
+
export interface ModelOverride {
|
|
9
|
+
/**
|
|
10
|
+
* Per-model token pricing. All four cost fields are optional —
|
|
11
|
+
* unspecified fields default to zero.
|
|
12
|
+
*/
|
|
13
|
+
cost?: Partial<ModelCost>;
|
|
14
|
+
/**
|
|
15
|
+
* Pi capabilities for the model. When present, **fully replaces** the
|
|
16
|
+
* capabilities detected from the server (no merging).
|
|
17
|
+
*/
|
|
18
|
+
capabilities?: ("text" | "image")[];
|
|
19
|
+
/**
|
|
20
|
+
* Whether the model is a reasoning model. When absent, defaults to `true`.
|
|
21
|
+
*/
|
|
22
|
+
reasoning?: boolean;
|
|
23
|
+
/**
|
|
24
|
+
* Override the model's context size (in tokens), replacing the value
|
|
25
|
+
* autodetected from the server. When absent or `0`, falls back to
|
|
26
|
+
* detection (then `FALLBACK_CTX`).
|
|
27
|
+
*/
|
|
28
|
+
contextSize?: number;
|
|
29
|
+
/**
|
|
30
|
+
* Override the maximum number of tokens the model can generate. When
|
|
31
|
+
* absent, falls back to the context size detected from the server.
|
|
32
|
+
*/
|
|
33
|
+
maxTokens?: number;
|
|
34
|
+
/**
|
|
35
|
+
* OpenAI-compatible provider compatibility settings. Merged with any
|
|
36
|
+
* provider-level compat when the model is registered with Pi.
|
|
37
|
+
*/
|
|
38
|
+
compat?: Partial<OpenAICompletionsCompat>;
|
|
39
|
+
}
|
|
40
|
+
|
|
1
41
|
/**
|
|
2
42
|
* A description of a server in the "llamaSettings" key
|
|
3
43
|
*/
|
|
4
|
-
interface LlamaServer {
|
|
44
|
+
export interface LlamaServer {
|
|
5
45
|
/**
|
|
6
46
|
* The URL of the llama.cpp server.
|
|
7
47
|
*/
|
|
@@ -14,6 +54,33 @@ interface LlamaServer {
|
|
|
14
54
|
* Custom display name for this server.
|
|
15
55
|
*/
|
|
16
56
|
name?: string;
|
|
57
|
+
/**
|
|
58
|
+
* Per-model overrides for this server. Keys are **prefix filters** —
|
|
59
|
+
* a model ID matches if it starts with the key. When multiple patterns
|
|
60
|
+
* match, the **longest (most specific) match wins**.
|
|
61
|
+
*
|
|
62
|
+
* All fields of an override are optional — absent fields fall back to
|
|
63
|
+
* detection (`capabilities`) or defaults (`reasoning: true`, zero costs).
|
|
64
|
+
*
|
|
65
|
+
* Example:
|
|
66
|
+
* ```json
|
|
67
|
+
* {
|
|
68
|
+
* "llama": { "cost": { "input": 0.01, "output": 0.02 } },
|
|
69
|
+
* "llama-3": { "reasoning": false },
|
|
70
|
+
* "llama-3-8b": {
|
|
71
|
+
* "cost": { "input": 0.2, "output": 0.6, "cacheRead": 0.01 },
|
|
72
|
+
* "capabilities": ["text", "image"]
|
|
73
|
+
* }
|
|
74
|
+
* }
|
|
75
|
+
* ```
|
|
76
|
+
*
|
|
77
|
+
* For model `"llama-3-8b"`:
|
|
78
|
+
* - `"llama"` matches → cost `{ input: 0.01, output: 0.02 }`
|
|
79
|
+
* - `"llama-3"` matches → reasoning `false`
|
|
80
|
+
* - `"llama-3-8b"` matches → cost + capabilities fully replaced
|
|
81
|
+
* - **Winner**: `"llama-3-8b"` (longest match)
|
|
82
|
+
*/
|
|
83
|
+
overrides?: Record<string, ModelOverride>;
|
|
17
84
|
}
|
|
18
85
|
|
|
19
86
|
/**
|
|
@@ -29,14 +96,16 @@ interface LlamaServer {
|
|
|
29
96
|
* }],
|
|
30
97
|
* "reactToModelSelect": true
|
|
31
98
|
* "autoloadOnMessage": false
|
|
99
|
+
* "sortBy": "asc"
|
|
32
100
|
* }
|
|
33
101
|
}
|
|
34
102
|
*/
|
|
35
103
|
export interface LlamaSettings {
|
|
36
104
|
/**
|
|
37
105
|
* List of servers to connect to.
|
|
106
|
+
* @default []
|
|
38
107
|
*/
|
|
39
|
-
servers
|
|
108
|
+
servers?: LlamaServer[];
|
|
40
109
|
/**
|
|
41
110
|
* Whether to react to model selection events by loading the model.
|
|
42
111
|
* @default true
|
|
@@ -57,4 +126,9 @@ export interface LlamaSettings {
|
|
|
57
126
|
* @default 1000
|
|
58
127
|
*/
|
|
59
128
|
serverTimeout?: number;
|
|
129
|
+
/**
|
|
130
|
+
* How to sort models in the /models command.
|
|
131
|
+
* @default "asc"
|
|
132
|
+
*/
|
|
133
|
+
sortBy?: SortBy;
|
|
60
134
|
}
|