pi-llama-cpp 0.9.2 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +163 -27
- package/package.json +4 -3
- package/src/api/client.ts +76 -32
- package/src/constants.ts +28 -3
- package/src/index.ts +4 -12
- package/src/interfaces/events.ts +14 -3
- package/src/interfaces/server.ts +20 -0
- package/src/interfaces/settings.ts +69 -0
- package/src/managers/command.ts +285 -43
- package/src/managers/events.ts +50 -12
- package/src/managers/server.ts +90 -17
- package/src/managers/settings.ts +268 -0
- package/src/models/baseModel.ts +32 -17
- package/src/models/legacyModel.ts +2 -2
- package/src/models/routerModel.ts +3 -3
- package/src/server.ts +86 -21
- package/src/sse/client.ts +28 -16
- package/src/sse/manager.ts +23 -10
- package/src/ui/serverListEditor.ts +481 -0
- package/src/utils/errors.ts +5 -0
- package/src/utils/settingsStore.ts +60 -0
- package/src/utils/urls.ts +16 -0
- package/tests/commandManager.test.ts +256 -11
- package/tests/events.test.ts +229 -64
- package/tests/mocks.ts +145 -32
- package/tests/server.test.ts +54 -6
- package/tests/serverListEditor.test.ts +637 -0
- package/tests/serverManager.test.ts +282 -39
- package/tests/settings.test.ts +793 -0
- package/tests/settingsStore.test.ts +209 -0
- package/tests/sseManager.test.ts +98 -0
- package/src/interfaces/auth.ts +0 -6
- package/src/resolver.ts +0 -149
- package/src/utils/cache.ts +0 -39
- package/src/utils/mutex.ts +0 -24
- package/tests/resolver.test.ts +0 -184
package/README.md
CHANGED
|
@@ -9,9 +9,9 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
|
|
|
9
9
|
- **Load / unload / switch** — manage models directly from the Pi command palette
|
|
10
10
|
- **Multi-model router support** — works with both single-model and multi-model llama.cpp server configurations
|
|
11
11
|
- **Image capabilities detection** — detects multimodal models automatically
|
|
12
|
-
- **Flexible URL resolution** — configures the server
|
|
12
|
+
- **Flexible URL resolution** — configures the server via `llamaSettings` (project/global), environment variable, or legacy `llamaServerUrl`
|
|
13
13
|
- **Auth support** — allows to login into a llama.cpp server that was secured with an API key
|
|
14
|
-
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously
|
|
14
|
+
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously via `llamaSettings.servers` or semicolon-separated URLs
|
|
15
15
|
- **Thinking budget support** — configurable token budgets for model reasoning/thinking, mapped to Pi's thinking levels
|
|
16
16
|
- **Real-time progress tracking** — live loading progress via SSE (falls back to polling)
|
|
17
17
|
|
|
@@ -48,34 +48,154 @@ pi install https://github.com/gsanhueza/pi-llama-cpp
|
|
|
48
48
|
|
|
49
49
|
## Configuration
|
|
50
50
|
|
|
51
|
-
The extension resolves the llama.cpp server
|
|
51
|
+
The extension resolves the llama.cpp server configuration using the following priority order:
|
|
52
52
|
|
|
53
|
-
1. **
|
|
53
|
+
1. **Environment variable** — `LLAMA_SERVER_URL`
|
|
54
|
+
2. **`llamaSettings`** — Main configuration format in `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global)
|
|
55
|
+
3. **`llamaServerUrl`** — Legacy format in `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global)
|
|
56
|
+
4. **Default** — `http://127.0.0.1:8080`
|
|
57
|
+
|
|
58
|
+
### Server configuration
|
|
54
59
|
|
|
55
|
-
|
|
56
|
-
{
|
|
57
|
-
"llamaServerUrl": "http://127.0.0.1:8080"
|
|
58
|
-
}
|
|
59
|
-
```
|
|
60
|
+
The recommended way to configure the extension is using the `llamaSettings` key. This provides a structured way to define multiple servers with custom names and IDs, plus additional behavior options.
|
|
60
61
|
|
|
61
|
-
|
|
62
|
+
Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global):
|
|
62
63
|
|
|
63
|
-
|
|
64
|
+
```json
|
|
65
|
+
{
|
|
66
|
+
"llamaSettings": {
|
|
67
|
+
"servers": [
|
|
68
|
+
{
|
|
69
|
+
"url": "http://127.0.0.1:8080",
|
|
70
|
+
"id": "local",
|
|
71
|
+
"name": "Local Server"
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"url": "http://10.0.0.5:8080",
|
|
75
|
+
"name": "Remote Server"
|
|
76
|
+
}
|
|
77
|
+
],
|
|
78
|
+
"reactToModelSelect": true,
|
|
79
|
+
"autoloadOnMessage": false,
|
|
80
|
+
"sortBy": "asc",
|
|
81
|
+
"pollingTimeout": 60000,
|
|
82
|
+
"serverTimeout": 1000
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
```
|
|
64
86
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
87
|
+
With this config, the servers will appear in Pi as **Llama.cpp (Local Server)** and **Llama.cpp (Remote Server)**.
|
|
88
|
+
|
|
89
|
+
#### Server Options
|
|
90
|
+
|
|
91
|
+
| Option | Type | Required | Description |
|
|
92
|
+
| ------ | ------ | -------- | ---------------------------------------------------------------------------- |
|
|
93
|
+
| `url` | string | Yes | The URL of the llama.cpp server |
|
|
94
|
+
| `id` | string | No | Custom provider ID (used for API key auth). Defaults to `llama-server=<url>` |
|
|
95
|
+
| `name` | string | No | Display name for the server in the UI (shown as `Llama.cpp — <name>`) |
|
|
96
|
+
|
|
97
|
+
> **Note:** If you set a custom `id`, you can use it in `~/.pi/agent/auth.json`. The extension will also fall back to the URL-based ID if no key is found for the custom `id`.
|
|
98
|
+
|
|
99
|
+
#### Settings Options
|
|
100
|
+
|
|
101
|
+
| Option | Type | Default | Description |
|
|
102
|
+
| -------------------- | ------- | ------- | ------------------------------------------------------------- |
|
|
103
|
+
| `reactToModelSelect` | boolean | `true` | Load the model when you switch via Pi's model picker. |
|
|
104
|
+
| `autoloadOnMessage` | boolean | `false` | Automatically load an unloaded model before sending a message |
|
|
105
|
+
| `sortBy` | string | `"asc"` | Sort order for models (see below) |
|
|
106
|
+
| `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
|
|
107
|
+
| `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
|
|
108
|
+
|
|
109
|
+
> **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
|
|
110
|
+
|
|
111
|
+
#### In-session settings menu
|
|
112
|
+
|
|
113
|
+
Run `/models settings` to edit the scalar settings above without hand-editing JSON:
|
|
114
|
+
|
|
115
|
+
- **Enter/Space** cycles the value under the cursor; **Esc** closes the menu.
|
|
116
|
+
- Booleans toggle `on`/`off`, `sortBy` cycles through the sort orders, and the
|
|
117
|
+
timeouts cycle through presets (`pollingTimeout`: 15s/30s/60s/120s/300s,
|
|
118
|
+
`serverTimeout`: 500ms/1s/2s/5s/10s).
|
|
119
|
+
- Changes are written to the **global** `~/.pi/agent/settings.json` only. If a
|
|
120
|
+
project `.pi/settings.json` defines the same key, its value keeps winning in
|
|
121
|
+
the merged view until you remove it there.
|
|
122
|
+
- Boolean and sort changes apply immediately; timeout changes apply on the next
|
|
123
|
+
model load.
|
|
124
|
+
- The `servers` list is edited with `/models servers` (see below).
|
|
125
|
+
|
|
126
|
+
#### Server list editor
|
|
127
|
+
|
|
128
|
+
Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
|
|
129
|
+
without hand-editing JSON:
|
|
130
|
+
|
|
131
|
+
- **↑/↓** moves the cursor, **Enter/e** edits the selected URL, **i** edits
|
|
132
|
+
its `id`, **n** its `name`, **a** adds a new entry, **d** deletes it
|
|
133
|
+
(after an "Are you sure?" confirmation — only **y** confirms;
|
|
134
|
+
**Enter** is ignored, **Esc/n** cancels), **Esc** closes the editor.
|
|
135
|
+
- One URL per entry (`http://host:port`). Trailing slashes are stripped on
|
|
136
|
+
save; `;`-separated values are rejected — use separate entries instead.
|
|
137
|
+
- Each change is written immediately to the **global**
|
|
138
|
+
`~/.pi/agent/settings.json`. If a project `.pi/settings.json` defines
|
|
139
|
+
`servers`, its list keeps winning in the merged view until you remove it
|
|
140
|
+
there.
|
|
141
|
+
- Changes apply the next time providers are scanned — run `/models` to see
|
|
142
|
+
them. Additions, removals, and URL/`id`/`name` edits all take effect on
|
|
143
|
+
the next `/models`: new servers register their providers, removed ones
|
|
144
|
+
leave pi's registry immediately, and edited ones are re-registered with
|
|
145
|
+
the fresh config — no restart needed.
|
|
146
|
+
- Limitation: a model already loading in the background on a removed or
|
|
147
|
+
edited server finishes loading, but its progress notifications stop;
|
|
148
|
+
re-select it from the (new) provider afterwards.
|
|
149
|
+
- The editor shows a warning when the `LLAMA_SERVER_URL` environment variable
|
|
150
|
+
is set, since it overrides the configured servers.
|
|
151
|
+
- Per-server `id`/`name` overrides can be edited with **i**/**n**; saving an
|
|
152
|
+
empty value clears the override. The list shows them as a
|
|
153
|
+
`(<id> - <name>)` suffix, falling back to the auto-detected
|
|
154
|
+
`llama-server=<url>` id when no custom `id` is set.
|
|
155
|
+
|
|
156
|
+
#### Environment variable
|
|
157
|
+
|
|
158
|
+
For a quick setup, you can use the `LLAMA_SERVER_URL` environment variable instead of the JSON config:
|
|
70
159
|
|
|
71
|
-
|
|
160
|
+
```bash
|
|
161
|
+
export LLAMA_SERVER_URL="http://127.0.0.1:8080"
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
This is equivalent to defining a single server in `llamaSettings.servers` with just a URL.
|
|
165
|
+
|
|
166
|
+
### Legacy configuration
|
|
72
167
|
|
|
73
|
-
|
|
168
|
+
For a simpler setup, you can use the legacy `llamaServerUrl` key:
|
|
74
169
|
|
|
75
|
-
|
|
170
|
+
```json
|
|
171
|
+
{
|
|
172
|
+
"llamaServerUrl": "http://127.0.0.1:8080"
|
|
173
|
+
}
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
This is equivalent to defining a single server in `llamaSettings.servers` with just a URL.
|
|
177
|
+
|
|
178
|
+
### Multiple servers
|
|
179
|
+
|
|
180
|
+
To connect to multiple llama.cpp servers simultaneously:
|
|
181
|
+
|
|
182
|
+
**Using `llamaSettings` (recommended):**
|
|
183
|
+
|
|
184
|
+
```json
|
|
185
|
+
{
|
|
186
|
+
"llamaSettings": {
|
|
187
|
+
"servers": [
|
|
188
|
+
{ "url": "http://127.0.0.1:8080" },
|
|
189
|
+
{ "url": "http://127.0.0.1:8081" },
|
|
190
|
+
{ "url": "http://10.0.0.5:8080" }
|
|
191
|
+
]
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
**Using the environment variable:**
|
|
76
197
|
|
|
77
198
|
```bash
|
|
78
|
-
# Example for env, but you can use any of the other methods
|
|
79
199
|
LLAMA_SERVER_URL="http://127.0.0.1:8080;http://127.0.0.1:8081;http://10.0.0.5:8080"
|
|
80
200
|
```
|
|
81
201
|
|
|
@@ -86,7 +206,7 @@ Each server gets its own provider (e.g., **Llama.cpp (http://127.0.0.1:8080)**)
|
|
|
86
206
|
If your llama.cpp server requires authentication, use `/login` in Pi, select the "API key" option, and choose the provider from the list that correlates with the server needing the API key.
|
|
87
207
|
|
|
88
208
|
Alternatively, configure the API key in `~/.pi/agent/auth.json`:
|
|
89
|
-
Use the provider ID `llama-server=<url
|
|
209
|
+
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`):
|
|
90
210
|
|
|
91
211
|
```json
|
|
92
212
|
{
|
|
@@ -136,11 +256,13 @@ The extension determines the context size as follows:
|
|
|
136
256
|
|
|
137
257
|
### Commands
|
|
138
258
|
|
|
139
|
-
| Command
|
|
140
|
-
|
|
|
141
|
-
| `/models`
|
|
142
|
-
| `/models info`
|
|
143
|
-
| `/models unload`
|
|
259
|
+
| Command | Description |
|
|
260
|
+
| ------------------ | ---------------------------------------------------------------------------------- |
|
|
261
|
+
| `/models` | Browse your models with live status. Select a model to load, switch, or unload it. |
|
|
262
|
+
| `/models info` | Show detailed information for all available models at once. |
|
|
263
|
+
| `/models unload` | Unload all loaded models at once. |
|
|
264
|
+
| `/models servers` | Add, edit or remove llama.cpp server URLs via a TUI editor. |
|
|
265
|
+
| `/models settings` | Open a menu to edit the scalar `llamaSettings` fields. |
|
|
144
266
|
|
|
145
267
|
> **Note:** When a llama.cpp server is slow to respond, it will be skipped at startup with a warning. Run `/models` to retry without timeout and see all models.
|
|
146
268
|
|
|
@@ -148,6 +270,18 @@ The extension determines the context size as follows:
|
|
|
148
270
|
|
|
149
271
|
> **Note:** The `/models unload` command only makes sense in router mode.
|
|
150
272
|
|
|
273
|
+
#### Model sorting
|
|
274
|
+
|
|
275
|
+
The order of models in the `/models` menu is controlled by the `sortBy` setting:
|
|
276
|
+
|
|
277
|
+
| Value | Description |
|
|
278
|
+
| ------------- | --------------------------------------------------------------------------------------- |
|
|
279
|
+
| `"asc"` | Sort by model ID ascending (default) |
|
|
280
|
+
| `"desc"` | Sort by model ID descending |
|
|
281
|
+
| `"asc-name"` | Sort by model name ascending (ties broken by ID) |
|
|
282
|
+
| `"desc-name"` | Sort by model name descending (ties broken by ID) |
|
|
283
|
+
| `"api"` | No sorting — models appear in the order returned by each server's `/v1/models` endpoint |
|
|
284
|
+
|
|
151
285
|
### Model Actions
|
|
152
286
|
|
|
153
287
|
When browsing models via the `/models` command, you can:
|
|
@@ -199,13 +333,15 @@ When you switch models via Pi's model picker (instead of using the `/models` com
|
|
|
199
333
|
|
|
200
334
|
This keeps the server in sync with the active model in Pi, regardless of how the switch was initiated — you don't need to manually load models before using them.
|
|
201
335
|
|
|
336
|
+
You can disable this behavior by setting `reactToModelSelect` to `false` in `llamaSettings`.
|
|
337
|
+
|
|
202
338
|
> **Note:** If you switch sessions while a model load is in-flight, you'll see a warning, but the load continues in the background. Use `/models` in the new session to verify the model status.
|
|
203
339
|
|
|
204
340
|
### Loading Models
|
|
205
341
|
|
|
206
342
|
When you trigger a load, switch, or retry action, the extension uses SSE (Server-Sent Events) to receive real-time progress updates from the server. If SSE is not available, it falls back to polling.
|
|
207
343
|
|
|
208
|
-
If loading takes longer than **60 seconds
|
|
344
|
+
If loading takes longer than **60 seconds** (configurable via `pollingTimeout`), the operation times out with an error.
|
|
209
345
|
|
|
210
346
|
> **Note:** The timeout only applies to the progress detection. The model might still be loading in the background.
|
|
211
347
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -36,9 +36,10 @@
|
|
|
36
36
|
"@earendil-works/pi-coding-agent": ">=0.84.0",
|
|
37
37
|
"@earendil-works/pi-tui": ">=0.84.0"
|
|
38
38
|
},
|
|
39
|
+
"type": "module",
|
|
39
40
|
"devDependencies": {
|
|
40
|
-
"@types/node": "^26.
|
|
41
|
+
"@types/node": "^26.4.1",
|
|
41
42
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
42
|
-
"vitest": "^
|
|
43
|
+
"vitest": "^5.0.0"
|
|
43
44
|
}
|
|
44
45
|
}
|
package/src/api/client.ts
CHANGED
|
@@ -1,13 +1,29 @@
|
|
|
1
1
|
import { POLLING_INTERVAL } from "../constants";
|
|
2
|
-
import { Cache } from "../utils/cache";
|
|
3
|
-
import { Mutex } from "../utils/mutex";
|
|
4
2
|
|
|
5
3
|
/**
|
|
6
|
-
*
|
|
4
|
+
* How long GET responses stay cached: half the polling interval, so a poll
|
|
5
|
+
* tick always reaches the server while multiple reads within one tick
|
|
6
|
+
* (fan-outs like `toProviderConfig`, concurrent model polls) do not.
|
|
7
|
+
*/
|
|
8
|
+
const CACHE_TTL = POLLING_INTERVAL / 2;
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* HTTP client for llama-server with GET caching and request deduplication.
|
|
12
|
+
*
|
|
13
|
+
* Two complementary mechanisms keep repeated reads from reaching the server:
|
|
14
|
+
* - the TTL cache absorbs time-spaced repeats (poll loops, sequential reads);
|
|
15
|
+
* - the in-flight map absorbs simultaneous bursts: concurrent callers for the
|
|
16
|
+
* same key share one request's promise.
|
|
17
|
+
*
|
|
18
|
+
* POST requests are deduplicated but never cached — they are not idempotent
|
|
19
|
+
* reads, and the only caller (`Server.postRequest()`) clears the cache around
|
|
20
|
+
* them. The dedup matters: independent load triggers (command, auto-load,
|
|
21
|
+
* model-select) can race, and merging their duplicate `POST /models/load`
|
|
22
|
+
* into one server call avoids load-state glitches.
|
|
7
23
|
*/
|
|
8
24
|
export class ApiClient {
|
|
9
|
-
private cache = new
|
|
10
|
-
private
|
|
25
|
+
private cache = new Map<string, { data: unknown; timestamp: number }>();
|
|
26
|
+
private inflight = new Map<string, Promise<unknown>>();
|
|
11
27
|
|
|
12
28
|
/**
|
|
13
29
|
* Creates a new ApiClient.
|
|
@@ -22,40 +38,34 @@ export class ApiClient {
|
|
|
22
38
|
|
|
23
39
|
/**
|
|
24
40
|
* Makes a cached, deduplicated GET request to the llama-server.
|
|
25
|
-
* Results are cached for half the polling interval and in-flight requests are deduplicated.
|
|
26
41
|
*
|
|
27
42
|
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
28
43
|
* @returns The parsed JSON response from the server
|
|
29
44
|
*/
|
|
30
45
|
async get<T>(endpoint: string): Promise<T> {
|
|
31
|
-
const cached = this.
|
|
46
|
+
const cached = this.cacheGet<T>(endpoint);
|
|
32
47
|
if (cached !== undefined) return cached;
|
|
33
48
|
|
|
34
|
-
return this.
|
|
35
|
-
const data =
|
|
36
|
-
this.
|
|
49
|
+
return this.dedupe(endpoint, async () => {
|
|
50
|
+
const data = await this.do_get<T>(endpoint);
|
|
51
|
+
this.cacheSet(endpoint, data);
|
|
37
52
|
return data;
|
|
38
53
|
});
|
|
39
54
|
}
|
|
40
55
|
|
|
41
56
|
/**
|
|
42
|
-
* Makes a
|
|
43
|
-
*
|
|
57
|
+
* Makes a deduplicated POST request to the llama-server.
|
|
58
|
+
* Concurrent duplicate requests share one server call; responses are
|
|
59
|
+
* never cached.
|
|
44
60
|
*
|
|
45
61
|
* @param endpoint The endpoint path to post to
|
|
46
62
|
* @param body The optional request body
|
|
47
63
|
* @returns The parsed JSON response from the server
|
|
48
64
|
*/
|
|
49
65
|
async post<T>(endpoint: string, body?: Record<string, unknown>): Promise<T> {
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
return this.mutex.getOrCreate(key, async () => {
|
|
55
|
-
const data = (await this.do_post<T>(endpoint, body)) as T;
|
|
56
|
-
this.cache.set(key, data);
|
|
57
|
-
return data;
|
|
58
|
-
});
|
|
66
|
+
return this.dedupe(this.cacheKey(endpoint, body), async () =>
|
|
67
|
+
this.do_post<T>(endpoint, body),
|
|
68
|
+
);
|
|
59
69
|
}
|
|
60
70
|
|
|
61
71
|
/**
|
|
@@ -65,6 +75,51 @@ export class ApiClient {
|
|
|
65
75
|
this.cache.clear();
|
|
66
76
|
}
|
|
67
77
|
|
|
78
|
+
/**
|
|
79
|
+
* Runs `fn` for the given key, or returns an existing in-flight promise.
|
|
80
|
+
* Concurrent callers for the same key share the same promise.
|
|
81
|
+
*/
|
|
82
|
+
private dedupe<T>(key: string, fn: () => Promise<T>): Promise<T> {
|
|
83
|
+
const existing = this.inflight.get(key);
|
|
84
|
+
if (existing) return existing as Promise<T>;
|
|
85
|
+
|
|
86
|
+
const promise = fn().finally(() => {
|
|
87
|
+
this.inflight.delete(key);
|
|
88
|
+
});
|
|
89
|
+
this.inflight.set(key, promise);
|
|
90
|
+
return promise;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Gets a cached GET response. Returns `undefined` if missing or expired.
|
|
95
|
+
*/
|
|
96
|
+
private cacheGet<T>(key: string): T | undefined {
|
|
97
|
+
const entry = this.cache.get(key);
|
|
98
|
+
if (!entry) return undefined;
|
|
99
|
+
if (Date.now() - entry.timestamp > CACHE_TTL) {
|
|
100
|
+
this.cache.delete(key);
|
|
101
|
+
return undefined;
|
|
102
|
+
}
|
|
103
|
+
return entry.data as T;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Stores a GET response in the cache with the current timestamp.
|
|
108
|
+
*/
|
|
109
|
+
private cacheSet(key: string, data: unknown): void {
|
|
110
|
+
this.cache.set(key, { data, timestamp: Date.now() });
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Builds the dedup key for a POST request.
|
|
115
|
+
*
|
|
116
|
+
* @param endpoint The endpoint path to post to
|
|
117
|
+
* @param body The optional request body
|
|
118
|
+
*/
|
|
119
|
+
private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
|
|
120
|
+
return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
|
|
121
|
+
}
|
|
122
|
+
|
|
68
123
|
/**
|
|
69
124
|
* Makes a raw GET request to the llama-server.
|
|
70
125
|
* This bypasses caching and deduplication.
|
|
@@ -107,15 +162,4 @@ export class ApiClient {
|
|
|
107
162
|
|
|
108
163
|
return res.json();
|
|
109
164
|
}
|
|
110
|
-
|
|
111
|
-
/**
|
|
112
|
-
* Sets a cache key
|
|
113
|
-
*
|
|
114
|
-
* @param endpoint The endpoint path to post to
|
|
115
|
-
* @param body The optional request body
|
|
116
|
-
* @returns The cache key
|
|
117
|
-
*/
|
|
118
|
-
private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
|
|
119
|
-
return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
|
|
120
|
-
}
|
|
121
165
|
}
|
package/src/constants.ts
CHANGED
|
@@ -3,6 +3,11 @@
|
|
|
3
3
|
*/
|
|
4
4
|
export const PROVIDER_PREFIX = "llama-server";
|
|
5
5
|
|
|
6
|
+
/**
|
|
7
|
+
* The settings key used in project/global settings.
|
|
8
|
+
*/
|
|
9
|
+
export const SETTINGS_KEY = "llamaSettings";
|
|
10
|
+
|
|
6
11
|
/**
|
|
7
12
|
* This provider's name
|
|
8
13
|
*/
|
|
@@ -21,12 +26,12 @@ export const API_KEY_PLACEHOLDER = "sk-placeholder";
|
|
|
21
26
|
/**
|
|
22
27
|
* The default URL if the resolver couldn't find it
|
|
23
28
|
*/
|
|
24
|
-
export const
|
|
29
|
+
export const LLAMA_SERVER_URL = "http://127.0.0.1:8080";
|
|
25
30
|
|
|
26
31
|
/**
|
|
27
32
|
* The default context if the server didn't expose it
|
|
28
33
|
*/
|
|
29
|
-
export const
|
|
34
|
+
export const FALLBACK_CTX = 128000;
|
|
30
35
|
|
|
31
36
|
/**
|
|
32
37
|
* Polling interval (ms) for checking model load status
|
|
@@ -48,10 +53,30 @@ export const READABLE_TIMEOUT = 15000;
|
|
|
48
53
|
*/
|
|
49
54
|
export const SERVER_TIMEOUT = 1000;
|
|
50
55
|
|
|
56
|
+
/**
|
|
57
|
+
* Default value for reactToModelSelect setting.
|
|
58
|
+
*/
|
|
59
|
+
export const REACT_TO_MODEL_SELECT = true;
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Default value for autoloadOnMessage setting.
|
|
63
|
+
*/
|
|
64
|
+
export const AUTOLOAD_ON_MESSAGE = false;
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Default sort order for model lists.
|
|
68
|
+
*/
|
|
69
|
+
export const SORT_BY = "asc";
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Sort order options for model lists.
|
|
73
|
+
*/
|
|
74
|
+
export type SortBy = "asc" | "desc" | "asc-name" | "desc-name" | "api";
|
|
75
|
+
|
|
51
76
|
/**
|
|
52
77
|
* Thinking budgets to send to the server, depending on user-selected level in Pi.
|
|
53
78
|
*/
|
|
54
|
-
export const
|
|
79
|
+
export const THINKING_BUDGETS = {
|
|
55
80
|
off: 0,
|
|
56
81
|
minimal: 1024,
|
|
57
82
|
low: 2048,
|
package/src/index.ts
CHANGED
|
@@ -11,17 +11,12 @@ import { ModelSelectEvent } from "./interfaces/events";
|
|
|
11
11
|
import { CommandManager } from "./managers/command";
|
|
12
12
|
import { EventManager } from "./managers/events";
|
|
13
13
|
import { ServerManager } from "./managers/server";
|
|
14
|
-
import {
|
|
15
|
-
import { Server } from "./server";
|
|
14
|
+
import { settings } from "./managers/settings";
|
|
16
15
|
|
|
17
16
|
export default async function (pi: ExtensionAPI) {
|
|
18
|
-
const
|
|
19
|
-
const
|
|
20
|
-
const
|
|
21
|
-
|
|
22
|
-
const eventManager = new EventManager(servers);
|
|
23
|
-
const serverManager = new ServerManager(servers);
|
|
24
|
-
const commandManager = new CommandManager(serverManager);
|
|
17
|
+
const serverManager = new ServerManager(settings);
|
|
18
|
+
const eventManager = new EventManager(serverManager, settings);
|
|
19
|
+
const commandManager = new CommandManager(serverManager, settings);
|
|
25
20
|
|
|
26
21
|
// Register providers once at startup
|
|
27
22
|
await serverManager.initialize(pi);
|
|
@@ -40,9 +35,6 @@ export default async function (pi: ExtensionAPI) {
|
|
|
40
35
|
if (event.reason !== "startup") return;
|
|
41
36
|
for (const warning of serverManager.getWarnings())
|
|
42
37
|
ctx.ui.notify(warning, "warning");
|
|
43
|
-
|
|
44
|
-
for (const warning of resolver.getWarnings())
|
|
45
|
-
ctx.ui.notify(warning, "warning");
|
|
46
38
|
});
|
|
47
39
|
|
|
48
40
|
pi.on(
|
package/src/interfaces/events.ts
CHANGED
|
@@ -1,3 +1,14 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
import type { ExtensionEvent } from "@earendil-works/pi-coding-agent";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* pi-coding-agent does not re-export its `ModelSelectEvent` from the public
|
|
5
|
+
* API (it exists in the internal extensions/index but is omitted from the
|
|
6
|
+
* root re-export list, and the package's `exports` map blocks deep imports).
|
|
7
|
+
* It is therefore derived from the exported `ExtensionEvent` union, which
|
|
8
|
+
* yields pi's real event shape (`model: Model<any>`, `previousModel`,
|
|
9
|
+
* `source`) and tracks the API automatically.
|
|
10
|
+
*/
|
|
11
|
+
export type ModelSelectEvent = Extract<
|
|
12
|
+
ExtensionEvent,
|
|
13
|
+
{ type: "model_select" }
|
|
14
|
+
>;
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Identity of a llama.cpp server endpoint.
|
|
3
|
+
*
|
|
4
|
+
* Persisted counterpart: {@link LlamaServer} (`llamaSettings.servers`),
|
|
5
|
+
* whose `url`/`id`/`name` map onto these fields.
|
|
6
|
+
*/
|
|
7
|
+
export interface ServerOptions {
|
|
8
|
+
/**
|
|
9
|
+
* Base URL of the llama.cpp server (e.g. "http://127.0.0.1:8080").
|
|
10
|
+
*/
|
|
11
|
+
baseUrl: string;
|
|
12
|
+
/**
|
|
13
|
+
* Custom provider ID; falls back to a URL-based one.
|
|
14
|
+
*/
|
|
15
|
+
customId?: string;
|
|
16
|
+
/**
|
|
17
|
+
* Custom provider name suffix; falls back to the base URL.
|
|
18
|
+
*/
|
|
19
|
+
customName?: string;
|
|
20
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import type { SortBy } from "../constants";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* A description of a server in the "llamaSettings" key
|
|
5
|
+
*/
|
|
6
|
+
export interface LlamaServer {
|
|
7
|
+
/**
|
|
8
|
+
* The URL of the llama.cpp server.
|
|
9
|
+
*/
|
|
10
|
+
url: string;
|
|
11
|
+
/**
|
|
12
|
+
* Custom provider ID for this server.
|
|
13
|
+
*/
|
|
14
|
+
id?: string;
|
|
15
|
+
/**
|
|
16
|
+
* Custom display name for this server.
|
|
17
|
+
*/
|
|
18
|
+
name?: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* The main configuration interface for this extension
|
|
23
|
+
*
|
|
24
|
+
* E.g.:
|
|
25
|
+
*
|
|
26
|
+
* {
|
|
27
|
+
* "servers": [{
|
|
28
|
+
* "id": "server-a",
|
|
29
|
+
* "name": "Server A",
|
|
30
|
+
* "url": "http://localhost:8080"
|
|
31
|
+
* }],
|
|
32
|
+
* "reactToModelSelect": true
|
|
33
|
+
* "autoloadOnMessage": false
|
|
34
|
+
* "sortBy": "asc"
|
|
35
|
+
* }
|
|
36
|
+
}
|
|
37
|
+
*/
|
|
38
|
+
export interface LlamaSettings {
|
|
39
|
+
/**
|
|
40
|
+
* List of servers to connect to.
|
|
41
|
+
* @default []
|
|
42
|
+
*/
|
|
43
|
+
servers?: LlamaServer[];
|
|
44
|
+
/**
|
|
45
|
+
* Whether to react to model selection events by loading the model.
|
|
46
|
+
* @default true
|
|
47
|
+
*/
|
|
48
|
+
reactToModelSelect?: boolean;
|
|
49
|
+
/**
|
|
50
|
+
* Whether to auto-load models when a message is sent.
|
|
51
|
+
* @default false
|
|
52
|
+
*/
|
|
53
|
+
autoloadOnMessage?: boolean;
|
|
54
|
+
/**
|
|
55
|
+
* Maximum time (ms) to wait for model loading before giving up.
|
|
56
|
+
* @default 60000
|
|
57
|
+
*/
|
|
58
|
+
pollingTimeout?: number;
|
|
59
|
+
/**
|
|
60
|
+
* Timeout (ms) for server verification and SSE support probe.
|
|
61
|
+
* @default 1000
|
|
62
|
+
*/
|
|
63
|
+
serverTimeout?: number;
|
|
64
|
+
/**
|
|
65
|
+
* How to sort models in the /models command.
|
|
66
|
+
* @default "asc"
|
|
67
|
+
*/
|
|
68
|
+
sortBy?: SortBy;
|
|
69
|
+
}
|