pi-llama-cpp 0.14.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -13
- package/package.json +3 -3
- package/src/api/client.ts +59 -22
- package/src/constants.ts +5 -0
- package/src/enums/mode.ts +1 -0
- package/src/index.ts +2 -1
- package/src/interfaces/endpoints/models.ts +6 -0
- package/src/interfaces/endpoints/props.ts +14 -0
- package/src/interfaces/settings.ts +6 -0
- package/src/managers/command/models.ts +12 -6
- package/src/managers/command.ts +6 -1
- package/src/managers/server.ts +30 -38
- package/src/managers/settings.ts +23 -3
- package/src/models/legacyModel.ts +3 -3
- package/src/models/llamaSwapModel.ts +90 -0
- package/src/models/routerModel.ts +2 -2
- package/src/models/singleModel.ts +2 -2
- package/src/server.ts +42 -15
- package/src/sse/manager.ts +2 -2
- package/src/ui/dialog/confirm.ts +1 -1
- package/src/ui/dialog/input.ts +1 -1
- package/src/ui/editors/editorOptions.ts +8 -1
- package/src/ui/editors/itemBuilder.ts +2 -6
- package/src/ui/editors/listEditor.ts +24 -15
- package/src/ui/editors/override/entryEditor.ts +105 -23
- package/src/ui/editors/override/fields/base.ts +53 -0
- package/src/ui/editors/override/fields/capabilities.ts +36 -0
- package/src/ui/editors/override/fields/cost.ts +64 -0
- package/src/ui/editors/override/fields/index.ts +68 -0
- package/src/ui/editors/override/fields/numeric.ts +55 -0
- package/src/ui/editors/override/fields/pattern.ts +24 -0
- package/src/ui/editors/override/fields/reasoning.ts +33 -0
- package/src/ui/editors/override/itemBuilder.ts +4 -4
- package/src/ui/editors/server/fields.ts +2 -2
- package/src/ui/editors/server/itemBuilder.ts +2 -3
- package/src/ui/editors/server/serverEditor.ts +58 -22
- package/src/ui/editors/server/utils.ts +50 -0
- package/src/ui/settings/index.ts +11 -0
- package/src/ui/strings.ts +17 -0
- package/src/utils/credentialResolver.ts +110 -0
- package/src/utils/health.ts +2 -2
- package/src/utils/urlResolver.ts +1 -1
- package/tests/{commandManager.test.ts → command/commandManager.test.ts} +80 -9
- package/tests/{events.test.ts → events/events.test.ts} +6 -6
- package/tests/mocks.ts +4 -0
- package/tests/{legacyModel.test.ts → models/legacyModel.test.ts} +5 -5
- package/tests/models/llamaSwapModel.test.ts +327 -0
- package/tests/{routerModel.test.ts → models/routerModel.test.ts} +5 -5
- package/tests/{singleModel.test.ts → models/singleModel.test.ts} +5 -5
- package/tests/{health.test.ts → server/health.test.ts} +4 -4
- package/tests/{server.test.ts → server/server.test.ts} +57 -7
- package/tests/{serverManager.test.ts → server/serverManager.test.ts} +4 -4
- package/tests/{settings.test.ts → settings/settings.test.ts} +198 -355
- package/tests/{settingsStore.test.ts → settings/settingsStore.test.ts} +1 -1
- package/tests/{sseManager.test.ts → sse/sseManager.test.ts} +2 -2
- package/tests/{dialog.test.ts → ui/dialog.test.ts} +5 -4
- package/tests/{overrides.test.ts → ui/overrides.test.ts} +9 -42
- package/tests/utils/credentialResolver.test.ts +66 -0
- package/tests/utils/urlResolver.test.ts +116 -0
- package/src/ui/editors/override/fields.ts +0 -294
- package/src/ui/editors/override/handlers.ts +0 -118
- package/src/ui/editors/server/handlers.ts +0 -32
package/README.md
CHANGED
|
@@ -12,6 +12,7 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
|
|
|
12
12
|
- **Flexible URL resolution** — configures the server via `llamaSettings` (project/global), environment variable, or legacy `llamaServerUrl`
|
|
13
13
|
- **Auth support** — allows to login into a llama.cpp server that was secured with an API key
|
|
14
14
|
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously via `llamaSettings.servers` or semicolon-separated URLs
|
|
15
|
+
- **Basic llama-swap support** — auto-detects and provides basic integration with [llama-swap](https://github.com/mostlygeek/llama-swap) gateways
|
|
15
16
|
- **Thinking budget support** — configurable token budgets for model reasoning/thinking, mapped to Pi's thinking levels
|
|
16
17
|
- **Real-time progress tracking** — live loading progress via SSE (falls back to polling)
|
|
17
18
|
|
|
@@ -35,11 +36,12 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
|
|
|
35
36
|
|
|
36
37
|
When browsing servers via `/models servers`, each server URL is prefixed with a health indicator:
|
|
37
38
|
|
|
38
|
-
| Icon | Status
|
|
39
|
-
| ---- |
|
|
40
|
-
| 🟢 | Healthy
|
|
41
|
-
| 🟡 | Timeout
|
|
42
|
-
| 🔴 | Unreachable
|
|
39
|
+
| Icon | Status | Description |
|
|
40
|
+
| ---- | ------------ | --------------------------------------------- |
|
|
41
|
+
| 🟢 | Healthy | Server responded successfully to health check |
|
|
42
|
+
| 🟡 | Timeout | Server health check timed out |
|
|
43
|
+
| 🔴 | Unreachable | Server could not be reached |
|
|
44
|
+
| ⛔ | Unauthorized | Server requires an API key |
|
|
43
45
|
|
|
44
46
|
## Installation
|
|
45
47
|
|
|
@@ -127,6 +129,7 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
|
|
|
127
129
|
| `sortBy` | string | `"asc"` | Sort order for models (see below) |
|
|
128
130
|
| `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
|
|
129
131
|
| `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
|
|
132
|
+
| `showServerUrls` | boolean | `true` | Show `[Server: <url>]` suffix in the /models model list |
|
|
130
133
|
|
|
131
134
|
> **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
|
|
132
135
|
|
|
@@ -137,11 +140,11 @@ Run `/models settings` to edit the scalar settings above without hand-editing JS
|
|
|
137
140
|
#### Server list editor
|
|
138
141
|
|
|
139
142
|
Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
|
|
140
|
-
without hand-editing JSON. Each server URL is prefixed with a
|
|
141
|
-
(🟢 healthy, 🟡 timeout, 🔴 unreachable) that reflects the
|
|
142
|
-
health check against the server. Each change is
|
|
143
|
-
`.pi/settings.json` if it exists,
|
|
144
|
-
`~/.pi/agent/settings.json`.
|
|
143
|
+
without hand-editing JSON. Each server URL is prefixed with a status indicator
|
|
144
|
+
(🟢 healthy, 🟡 timeout, 🔴 unreachable, ⛔ unauthorized) that reflects the
|
|
145
|
+
result of a health check and an auth probe against the server. Each change is
|
|
146
|
+
written immediately to the **project** `.pi/settings.json` if it exists,
|
|
147
|
+
otherwise to **global** `~/.pi/agent/settings.json`.
|
|
145
148
|
|
|
146
149
|
Changes take effect immediately after closing the editor: new servers
|
|
147
150
|
register their providers, removed ones leave pi's registry right away,
|
|
@@ -205,17 +208,48 @@ Each server gets its own provider (e.g., **Llama.cpp (http://127.0.0.1:8080)**)
|
|
|
205
208
|
If your llama.cpp server requires authentication, use `/login` in Pi, select the "API key" option, and choose the provider from the list that correlates with the server needing the API key.
|
|
206
209
|
|
|
207
210
|
Alternatively, configure the API key in `~/.pi/agent/auth.json`:
|
|
208
|
-
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`)
|
|
211
|
+
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`).
|
|
212
|
+
|
|
213
|
+
The `key` field supports several formats:
|
|
214
|
+
|
|
215
|
+
| Format | Example | Description |
|
|
216
|
+
| ----------------- | -------------------------------------------- | ------------------------------------------ |
|
|
217
|
+
| **Literal** | `"sk-abc123"` | API key stored directly |
|
|
218
|
+
| **Env ref** | `"$OPENAI_API_KEY"` or `"${OPENAI_API_KEY}"` | Resolved from `process.env` or `env` field |
|
|
219
|
+
| **Shell command** | `"!cat ~/.secrets/llama-key"` | Stdout of the command is used |
|
|
220
|
+
| **Escape** | `"$$literal"` | `$$` → literal `$`, `$!` → literal `!` |
|
|
209
221
|
|
|
210
222
|
```json
|
|
211
223
|
{
|
|
212
224
|
"llama-server=http://127.0.0.1:8080": {
|
|
213
225
|
"type": "api_key",
|
|
214
|
-
"key": "
|
|
226
|
+
"key": "sk-abc123"
|
|
215
227
|
},
|
|
216
228
|
"llama-server=https://some-url-for-llama-cpp": {
|
|
217
229
|
"type": "api_key",
|
|
218
|
-
"key": "
|
|
230
|
+
"key": "$LLAMA_API_KEY"
|
|
231
|
+
},
|
|
232
|
+
"llama-server=https://secure-server": {
|
|
233
|
+
"type": "api_key",
|
|
234
|
+
"key": "!cat ~/.secrets/llama-key"
|
|
235
|
+
},
|
|
236
|
+
"llama-server=https://braced-ref": {
|
|
237
|
+
"type": "api_key",
|
|
238
|
+
"key": "${API_KEY}"
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
For env ref formats, you can also store the variable value alongside the key using the `env` field:
|
|
244
|
+
|
|
245
|
+
```json
|
|
246
|
+
{
|
|
247
|
+
"llama-server=http://127.0.0.1:8080": {
|
|
248
|
+
"type": "api_key",
|
|
249
|
+
"key": "$MY_KEY",
|
|
250
|
+
"env": {
|
|
251
|
+
"MY_KEY": "sk-abc123"
|
|
252
|
+
}
|
|
219
253
|
}
|
|
220
254
|
}
|
|
221
255
|
```
|
|
@@ -244,6 +278,10 @@ llama-server --model path/to/model.gguf ...
|
|
|
244
278
|
|
|
245
279
|
> **Note:** The ik_llama.cpp fork is not legacy at all, but it uses an old way of describing models compared to llama.cpp.
|
|
246
280
|
|
|
281
|
+
- For llama-swap mode, point the extension at a running [llama-swap](https://github.com/mostlygeek/llama-swap) instance instead of a raw llama.cpp server. The extension auto-detects this mode via the `src` field in the server props response.
|
|
282
|
+
|
|
283
|
+
> **Note:** llama-swap support is basic — only model listing, status, load/unload, and capability detection are implemented.
|
|
284
|
+
|
|
247
285
|
The extension determines the context size as follows:
|
|
248
286
|
|
|
249
287
|
- A per-model `contextSize` override (see [Model Overrides](#model-overrides)) takes precedence over everything below
|
|
@@ -252,6 +290,7 @@ The extension determines the context size as follows:
|
|
|
252
290
|
- When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
|
|
253
291
|
- **Single mode** — reads `meta.n_ctx` from the `/v1/models` endpoint
|
|
254
292
|
- **Legacy mode** — reads `max_model_len` from `/v1/models`, falling back to `n_ctx` from `/props`
|
|
293
|
+
- **Llama-swap mode** — reads `meta.n_ctx` from the llama-swap server via `/v1/models`
|
|
255
294
|
- Falls back to `128000` if not available
|
|
256
295
|
|
|
257
296
|
### Commands
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.16.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -38,8 +38,8 @@
|
|
|
38
38
|
},
|
|
39
39
|
"type": "module",
|
|
40
40
|
"devDependencies": {
|
|
41
|
-
"@types/node": "^26.6.
|
|
41
|
+
"@types/node": "^26.6.3",
|
|
42
42
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
43
|
-
"vitest": "^5.0.
|
|
43
|
+
"vitest": "^5.0.2"
|
|
44
44
|
}
|
|
45
45
|
}
|
package/src/api/client.ts
CHANGED
|
@@ -81,6 +81,30 @@ export class ApiClient {
|
|
|
81
81
|
);
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
/**
|
|
85
|
+
* Makes a raw GET request that does not parse JSON. Useful for endpoints
|
|
86
|
+
* that return empty or non-JSON responses.
|
|
87
|
+
*
|
|
88
|
+
* @param endpoint The endpoint path to fetch
|
|
89
|
+
*/
|
|
90
|
+
async rawGet(endpoint: string): Promise<void> {
|
|
91
|
+
return this.do_request<void>(endpoint, "GET", undefined, false);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Makes a raw POST request that does not parse JSON. Useful for endpoints
|
|
96
|
+
* that return empty or non-JSON responses.
|
|
97
|
+
*
|
|
98
|
+
* @param endpoint The endpoint path to post to
|
|
99
|
+
* @param body The optional request body
|
|
100
|
+
*/
|
|
101
|
+
async rawPost(
|
|
102
|
+
endpoint: string,
|
|
103
|
+
body?: Record<string, unknown>,
|
|
104
|
+
): Promise<void> {
|
|
105
|
+
return this.do_request<void>(endpoint, "POST", body, false);
|
|
106
|
+
}
|
|
107
|
+
|
|
84
108
|
/**
|
|
85
109
|
* Clears the entire cache.
|
|
86
110
|
*/
|
|
@@ -134,26 +158,54 @@ export class ApiClient {
|
|
|
134
158
|
}
|
|
135
159
|
|
|
136
160
|
/**
|
|
137
|
-
* Makes a raw
|
|
161
|
+
* Makes a raw request to the llama-server.
|
|
138
162
|
* This bypasses caching and deduplication.
|
|
139
163
|
*
|
|
140
|
-
* @param endpoint The endpoint path
|
|
141
|
-
* @
|
|
164
|
+
* @param endpoint The endpoint path
|
|
165
|
+
* @param method The HTTP method
|
|
166
|
+
* @param body The optional request body
|
|
167
|
+
* @param parseJson Whether to parse the response as JSON (default: true)
|
|
168
|
+
* @returns The parsed JSON response, or void if parseJson is false
|
|
142
169
|
* @throws ApiError with status 401 when the server rejects the request
|
|
143
170
|
* due to authentication (invalid/missing API key).
|
|
144
171
|
*/
|
|
145
|
-
private async
|
|
172
|
+
private async do_request<T>(
|
|
173
|
+
endpoint: string,
|
|
174
|
+
method: "GET" | "POST",
|
|
175
|
+
body?: Record<string, unknown>,
|
|
176
|
+
parseJson = true,
|
|
177
|
+
): Promise<T | void> {
|
|
146
178
|
const url = `${this.baseUrl}${endpoint}`;
|
|
147
179
|
|
|
148
180
|
const res = await fetch(url, {
|
|
149
|
-
|
|
181
|
+
method,
|
|
182
|
+
headers: {
|
|
183
|
+
...(method === "POST"
|
|
184
|
+
? { "Content-Type": "application/json" }
|
|
185
|
+
: undefined),
|
|
186
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
187
|
+
},
|
|
188
|
+
...(body ? { body: JSON.stringify(body) } : undefined),
|
|
150
189
|
});
|
|
151
190
|
|
|
152
191
|
if (res.status === 401) {
|
|
153
192
|
throw new ApiError("authentication", res.status);
|
|
154
193
|
}
|
|
155
194
|
|
|
156
|
-
return res.json();
|
|
195
|
+
return parseJson ? ((await res.json()) as T) : (undefined as T);
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Makes a raw GET request to the llama-server.
|
|
200
|
+
* This bypasses caching and deduplication.
|
|
201
|
+
*
|
|
202
|
+
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
203
|
+
* @returns The parsed JSON response from the server
|
|
204
|
+
* @throws ApiError with status 401 when the server rejects the request
|
|
205
|
+
* due to authentication (invalid/missing API key).
|
|
206
|
+
*/
|
|
207
|
+
private async do_get<T>(endpoint: string): Promise<T> {
|
|
208
|
+
return this.do_request<T>(endpoint, "GET") as Promise<T>;
|
|
157
209
|
}
|
|
158
210
|
|
|
159
211
|
/**
|
|
@@ -170,21 +222,6 @@ export class ApiClient {
|
|
|
170
222
|
endpoint: string,
|
|
171
223
|
body?: Record<string, unknown>,
|
|
172
224
|
): Promise<T> {
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
const res = await fetch(url, {
|
|
176
|
-
method: "POST",
|
|
177
|
-
headers: {
|
|
178
|
-
"Content-Type": "application/json",
|
|
179
|
-
Authorization: `Bearer ${this.apiKey}`,
|
|
180
|
-
},
|
|
181
|
-
body: body ? JSON.stringify(body) : undefined,
|
|
182
|
-
});
|
|
183
|
-
|
|
184
|
-
if (res.status === 401) {
|
|
185
|
-
throw new ApiError("authentication", res.status);
|
|
186
|
-
}
|
|
187
|
-
|
|
188
|
-
return res.json();
|
|
225
|
+
return this.do_request<T>(endpoint, "POST", body) as Promise<T>;
|
|
189
226
|
}
|
|
190
227
|
}
|
package/src/constants.ts
CHANGED
|
@@ -76,6 +76,11 @@ export const AUTOLOAD_ON_MESSAGE = false;
|
|
|
76
76
|
*/
|
|
77
77
|
export const SORT_BY = "asc";
|
|
78
78
|
|
|
79
|
+
/**
|
|
80
|
+
* Default for showServerUrls — show server URLs in /models by default.
|
|
81
|
+
*/
|
|
82
|
+
export const SHOW_SERVER_URLS = true;
|
|
83
|
+
|
|
79
84
|
/**
|
|
80
85
|
* Thinking budgets to send to the server, depending on user-selected level in Pi.
|
|
81
86
|
*/
|
package/src/enums/mode.ts
CHANGED
package/src/index.ts
CHANGED
|
@@ -11,9 +11,10 @@ import { ModelSelectEvent } from "./interfaces/events";
|
|
|
11
11
|
import { CommandManager } from "./managers/command";
|
|
12
12
|
import { EventManager } from "./managers/events";
|
|
13
13
|
import { ServerManager } from "./managers/server";
|
|
14
|
-
import {
|
|
14
|
+
import { createSettingsManager } from "./managers/settings";
|
|
15
15
|
|
|
16
16
|
export default async function (pi: ExtensionAPI) {
|
|
17
|
+
const settings = createSettingsManager();
|
|
17
18
|
const serverManager = new ServerManager(settings);
|
|
18
19
|
const eventManager = new EventManager(serverManager, settings);
|
|
19
20
|
const commandManager = new CommandManager(serverManager, settings);
|
|
@@ -13,6 +13,20 @@ export interface PropsEndpoint {
|
|
|
13
13
|
cors_proxy_enabled: boolean;
|
|
14
14
|
}
|
|
15
15
|
|
|
16
|
+
/**
|
|
17
|
+
* llama-swap returns a different shape for /props (without a model ID):
|
|
18
|
+
* an error response with a `src` that we can use as a trick to detect it.
|
|
19
|
+
*/
|
|
20
|
+
export interface LlamaSwapPropsError {
|
|
21
|
+
src: "llama-swap";
|
|
22
|
+
error: {
|
|
23
|
+
message: string;
|
|
24
|
+
type: string;
|
|
25
|
+
param: null;
|
|
26
|
+
code: string;
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
|
|
16
30
|
/**
|
|
17
31
|
* The structure of llama-server's /props?model=<id> endpoint
|
|
18
32
|
*/
|
|
@@ -132,4 +132,10 @@ export interface LlamaSettings {
|
|
|
132
132
|
* @default "asc"
|
|
133
133
|
*/
|
|
134
134
|
sortBy?: SortBy;
|
|
135
|
+
/**
|
|
136
|
+
* Whether to show the `[Server: <url>]` suffix in the /models model list.
|
|
137
|
+
* When `false`, only model names are shown without server annotations.
|
|
138
|
+
* @default true
|
|
139
|
+
*/
|
|
140
|
+
showServerUrls?: boolean;
|
|
135
141
|
}
|
|
@@ -10,6 +10,7 @@ import { BaseModel } from "../../models/baseModel";
|
|
|
10
10
|
import { errorMessage } from "../../utils/errors";
|
|
11
11
|
import { EventManager } from "../events";
|
|
12
12
|
import type { ServerManager } from "../server";
|
|
13
|
+
import type { LlamaSettingsManager } from "../settings";
|
|
13
14
|
|
|
14
15
|
/**
|
|
15
16
|
* Interactive model selection/action flow for the `/models` command.
|
|
@@ -24,7 +25,10 @@ import type { ServerManager } from "../server";
|
|
|
24
25
|
* server registry without coupling the menu to command routing.
|
|
25
26
|
*/
|
|
26
27
|
export class ModelsMenu {
|
|
27
|
-
constructor(
|
|
28
|
+
constructor(
|
|
29
|
+
private readonly serverManager: ServerManager,
|
|
30
|
+
private readonly settings: LlamaSettingsManager,
|
|
31
|
+
) {}
|
|
28
32
|
|
|
29
33
|
/**
|
|
30
34
|
* Runs the interactive model selection menu.
|
|
@@ -173,6 +177,8 @@ export class ModelsMenu {
|
|
|
173
177
|
ctx: ExtensionCommandContext,
|
|
174
178
|
models: BaseModel[],
|
|
175
179
|
): Promise<BaseModel | null> {
|
|
180
|
+
const showServerUrls = await this.settings.resolveShowServerUrls();
|
|
181
|
+
|
|
176
182
|
const labels = await Promise.all(
|
|
177
183
|
models.map(async (model) => ({
|
|
178
184
|
label: (await model.getLabel()).trim(),
|
|
@@ -191,7 +197,8 @@ export class ModelsMenu {
|
|
|
191
197
|
const choices = labels.map(({ label, serverUrl }) => {
|
|
192
198
|
const extraPadding = 2;
|
|
193
199
|
const padLen = maxLength - graphemeLength(label) + extraPadding;
|
|
194
|
-
|
|
200
|
+
const serverSuffix = showServerUrls ? ` [Server: ${serverUrl}]` : "";
|
|
201
|
+
return `${label}${" ".repeat(padLen)}${serverSuffix}`;
|
|
195
202
|
});
|
|
196
203
|
|
|
197
204
|
const choice = await ctx.ui.select(`${PROVIDER_NAME} models:`, choices);
|
|
@@ -210,10 +217,9 @@ export class ModelsMenu {
|
|
|
210
217
|
const base = [Action.INFO, Action.CANCEL];
|
|
211
218
|
|
|
212
219
|
const actions: Record<Status, Array<Action>> = {
|
|
213
|
-
[Status.LOADED]:
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
: [Action.SWITCH, ...base],
|
|
220
|
+
[Status.LOADED]: [Mode.ROUTER, Mode.LLAMASWAP].includes(model.mode)
|
|
221
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
222
|
+
: [Action.SWITCH, ...base],
|
|
217
223
|
[Status.LOADING]: [...base],
|
|
218
224
|
[Status.FAILED]: [Action.RETRY, ...base],
|
|
219
225
|
[Status.SLEEPING]:
|
package/src/managers/command.ts
CHANGED
|
@@ -51,7 +51,7 @@ export class CommandManager {
|
|
|
51
51
|
private readonly serverManager: ServerManager,
|
|
52
52
|
private readonly settings: LlamaSettingsManager,
|
|
53
53
|
) {
|
|
54
|
-
this.modelsMenu = new ModelsMenu(serverManager);
|
|
54
|
+
this.modelsMenu = new ModelsMenu(serverManager, settings);
|
|
55
55
|
}
|
|
56
56
|
|
|
57
57
|
/**
|
|
@@ -108,6 +108,11 @@ export class CommandManager {
|
|
|
108
108
|
this.notifyNotFound(ctx, url);
|
|
109
109
|
}
|
|
110
110
|
|
|
111
|
+
// Notify about other warnings (e.g. unauthorized servers)
|
|
112
|
+
for (const warning of this.serverManager.getWarnings()) {
|
|
113
|
+
ctx.ui.notify(warning, "warning");
|
|
114
|
+
}
|
|
115
|
+
|
|
111
116
|
if (args === "unload") {
|
|
112
117
|
const models = await this.serverManager.getAllModels();
|
|
113
118
|
await Promise.all(models.map((model) => model.unload()));
|
package/src/managers/server.ts
CHANGED
|
@@ -1,10 +1,14 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type {
|
|
2
|
+
ExtensionAPI,
|
|
3
|
+
ProviderModelConfig,
|
|
4
|
+
} from "@earendil-works/pi-coding-agent";
|
|
2
5
|
import { ApiError } from "../api/client";
|
|
3
6
|
import { API_TYPE, PROVIDER_NAME } from "../constants";
|
|
4
7
|
import { ServerStatus } from "../enums/serverStatus";
|
|
5
8
|
import type { SortBy } from "../interfaces/sortBy";
|
|
6
9
|
import { BaseModel } from "../models/baseModel";
|
|
7
10
|
import { Server } from "../server";
|
|
11
|
+
import { authRequiredMessage } from "../ui/strings";
|
|
8
12
|
import type { LlamaSettingsManager } from "./settings";
|
|
9
13
|
|
|
10
14
|
/** Model-list comparator: negative if a sorts first, positive if b does. */
|
|
@@ -45,6 +49,7 @@ export class ServerManager {
|
|
|
45
49
|
*/
|
|
46
50
|
async update(pi: ExtensionAPI, timeout?: number) {
|
|
47
51
|
this.failedUrls.length = 0;
|
|
52
|
+
this.warnings.length = 0;
|
|
48
53
|
|
|
49
54
|
// Surface warnings from strict URL parsing (dropped invalid entries)
|
|
50
55
|
this.warnings.push(...this.settings.takeWarnings());
|
|
@@ -60,9 +65,9 @@ export class ServerManager {
|
|
|
60
65
|
}
|
|
61
66
|
|
|
62
67
|
// Unregister providers that disappeared (removed or edited away);
|
|
63
|
-
//
|
|
68
|
+
// `seen` tracks all kept providerIds, so we skip those still present.
|
|
64
69
|
for (const old of this.servers) {
|
|
65
|
-
if (
|
|
70
|
+
if (seen.has(old.providerId)) continue;
|
|
66
71
|
pi.unregisterProvider(old.providerId);
|
|
67
72
|
// Optional chain is intentional despite the non-optional type: `sse`
|
|
68
73
|
// is undefined until initialize() runs (async-constructor hack — see Server)
|
|
@@ -79,34 +84,7 @@ export class ServerManager {
|
|
|
79
84
|
|
|
80
85
|
// Initialization and registration
|
|
81
86
|
for (const server of registrableServers) {
|
|
82
|
-
|
|
83
|
-
await server.initialize();
|
|
84
|
-
await this.registerProvider(server, pi);
|
|
85
|
-
} catch (err) {
|
|
86
|
-
if (err instanceof ApiError && err.type === "authentication") {
|
|
87
|
-
// Register the provider with an empty model list so the user can
|
|
88
|
-
// still configure the API key via `/login` or `auth.json`. On the
|
|
89
|
-
// next scan the provider will re-initialize and discover models.
|
|
90
|
-
// Don't add to `failedUrls` — the server IS reachable, auth just
|
|
91
|
-
// isn't configured yet, so the health indicator should stay green.
|
|
92
|
-
const message = [
|
|
93
|
-
"[pi-llama-cpp]",
|
|
94
|
-
`Server at '${server.baseUrl}' requires a valid API key.`,
|
|
95
|
-
"Configure the key via `/login` or in `~/.pi/agent/auth.json`.",
|
|
96
|
-
].join("\n");
|
|
97
|
-
this.warnings.push(message);
|
|
98
|
-
pi.registerProvider(server.providerId, {
|
|
99
|
-
name: server.providerName,
|
|
100
|
-
baseUrl: server.apiBaseUrl,
|
|
101
|
-
api: API_TYPE,
|
|
102
|
-
apiKey: server.getApiKey(),
|
|
103
|
-
models: [],
|
|
104
|
-
});
|
|
105
|
-
continue;
|
|
106
|
-
}
|
|
107
|
-
this.failedUrls.push(server.baseUrl);
|
|
108
|
-
continue;
|
|
109
|
-
}
|
|
87
|
+
await this.registerProvider(server, pi);
|
|
110
88
|
}
|
|
111
89
|
}
|
|
112
90
|
|
|
@@ -151,23 +129,37 @@ export class ServerManager {
|
|
|
151
129
|
}
|
|
152
130
|
|
|
153
131
|
/**
|
|
154
|
-
*
|
|
132
|
+
* Initializes the server and creates a Pi provider.
|
|
133
|
+
* Handles auth errors by registering with an empty model list so the user
|
|
134
|
+
* can still configure the API key via `/login` or `auth.json`.
|
|
155
135
|
*
|
|
156
136
|
* @param server The server
|
|
157
137
|
* @param pi The Pi API
|
|
158
138
|
*/
|
|
159
139
|
private async registerProvider(server: Server, pi: ExtensionAPI) {
|
|
160
|
-
const { apiBaseUrl, models, providerId, providerName } =
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
140
|
+
const { apiBaseUrl, apiKey, baseUrl, models, providerId, providerName } =
|
|
141
|
+
server;
|
|
142
|
+
let modelConfigs: ProviderModelConfig[] = [];
|
|
143
|
+
|
|
144
|
+
try {
|
|
145
|
+
await server.initialize();
|
|
146
|
+
modelConfigs = await Promise.all(models.map((m) => m.toProviderConfig()));
|
|
147
|
+
} catch (err) {
|
|
148
|
+
if (err instanceof ApiError && err.type === "authentication") {
|
|
149
|
+
// Don't add to `failedUrls` — the server IS reachable, auth just
|
|
150
|
+
// isn't configured yet, so the health indicator should stay green.
|
|
151
|
+
this.warnings.push(authRequiredMessage(baseUrl, providerId));
|
|
152
|
+
} else {
|
|
153
|
+
this.failedUrls.push(baseUrl);
|
|
154
|
+
return;
|
|
155
|
+
}
|
|
156
|
+
}
|
|
165
157
|
|
|
166
158
|
pi.registerProvider(providerId, {
|
|
167
159
|
name: providerName,
|
|
168
160
|
baseUrl: apiBaseUrl,
|
|
169
161
|
api: API_TYPE,
|
|
170
|
-
apiKey
|
|
162
|
+
apiKey,
|
|
171
163
|
models: modelConfigs,
|
|
172
164
|
});
|
|
173
165
|
}
|
package/src/managers/settings.ts
CHANGED
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
REACT_TO_MODEL_SELECT,
|
|
14
14
|
SERVER_TIMEOUT,
|
|
15
15
|
SETTINGS_KEY,
|
|
16
|
+
SHOW_SERVER_URLS,
|
|
16
17
|
SORT_BY,
|
|
17
18
|
THINKING_BUDGETS,
|
|
18
19
|
} from "../constants";
|
|
@@ -23,6 +24,7 @@ import {
|
|
|
23
24
|
} from "../interfaces/settings";
|
|
24
25
|
import type { SortBy } from "../interfaces/sortBy";
|
|
25
26
|
import { Server } from "../server";
|
|
27
|
+
import { CredentialResolver } from "../utils/credentialResolver";
|
|
26
28
|
import { SettingsStore } from "../utils/settingsStore";
|
|
27
29
|
import { UrlResolver } from "../utils/urlResolver";
|
|
28
30
|
|
|
@@ -46,6 +48,9 @@ export class LlamaSettingsManager {
|
|
|
46
48
|
}
|
|
47
49
|
}
|
|
48
50
|
|
|
51
|
+
/** Delegates credential key resolution (see `utils/credentialResolver`). */
|
|
52
|
+
private credentialResolver = new CredentialResolver();
|
|
53
|
+
|
|
49
54
|
/** Delegated multi-source URL resolution chain (see `utils/urlResolver`). */
|
|
50
55
|
private urlResolver = new UrlResolver({
|
|
51
56
|
getLlamaSettings: () => this.getLlamaSettings(),
|
|
@@ -148,12 +153,16 @@ export class LlamaSettingsManager {
|
|
|
148
153
|
|
|
149
154
|
/**
|
|
150
155
|
* Resolves API key for the provider ID using Pi's stored credentials.
|
|
156
|
+
* Delegates to `CredentialResolver` for key format handling.
|
|
151
157
|
*
|
|
158
|
+
* @param providerId The provider ID
|
|
152
159
|
* @returns The API key to use for the provider
|
|
153
160
|
*/
|
|
154
161
|
resolveApiKey(providerId: string): string {
|
|
155
162
|
const credential = readStoredCredential(providerId) as ApiKeyCredential;
|
|
156
|
-
|
|
163
|
+
if (!credential?.key) return API_KEY_PLACEHOLDER;
|
|
164
|
+
|
|
165
|
+
return this.credentialResolver.resolve(credential.key, credential.env);
|
|
157
166
|
}
|
|
158
167
|
|
|
159
168
|
/**
|
|
@@ -226,6 +235,15 @@ export class LlamaSettingsManager {
|
|
|
226
235
|
return (await this.getLlamaSettings()).sortBy ?? SORT_BY;
|
|
227
236
|
}
|
|
228
237
|
|
|
238
|
+
/**
|
|
239
|
+
* Resolves whether to show server URLs in the /models model list.
|
|
240
|
+
*
|
|
241
|
+
* @returns `true` if server URLs should be shown
|
|
242
|
+
*/
|
|
243
|
+
async resolveShowServerUrls(): Promise<boolean> {
|
|
244
|
+
return (await this.getLlamaSettings()).showServerUrls ?? SHOW_SERVER_URLS;
|
|
245
|
+
}
|
|
246
|
+
|
|
229
247
|
/**
|
|
230
248
|
* Persists one llamaSettings field to settings and reloads the in-memory
|
|
231
249
|
* settings so resolvers see the change immediately.
|
|
@@ -262,6 +280,8 @@ export class LlamaSettingsManager {
|
|
|
262
280
|
}
|
|
263
281
|
|
|
264
282
|
/**
|
|
265
|
-
*
|
|
283
|
+
* Creates a new LlamaSettingsManager instance.
|
|
266
284
|
*/
|
|
267
|
-
export
|
|
285
|
+
export function createSettingsManager(): LlamaSettingsManager {
|
|
286
|
+
return new LlamaSettingsManager();
|
|
287
|
+
}
|
|
@@ -3,7 +3,7 @@ import { Mode } from "../enums/mode";
|
|
|
3
3
|
import { SingleModel } from "./singleModel";
|
|
4
4
|
|
|
5
5
|
export class LegacyModel extends SingleModel {
|
|
6
|
-
get mode(): Mode {
|
|
6
|
+
override get mode(): Mode {
|
|
7
7
|
return Mode.LEGACY;
|
|
8
8
|
}
|
|
9
9
|
|
|
@@ -13,7 +13,7 @@ export class LegacyModel extends SingleModel {
|
|
|
13
13
|
*
|
|
14
14
|
* @returns The context size
|
|
15
15
|
*/
|
|
16
|
-
protected async getContextSize(): Promise<number> {
|
|
16
|
+
protected override async getContextSize(): Promise<number> {
|
|
17
17
|
const props = await this.server.fetchModelProps(this.id);
|
|
18
18
|
const models = await this.server.fetchModels();
|
|
19
19
|
|
|
@@ -34,7 +34,7 @@ export class LegacyModel extends SingleModel {
|
|
|
34
34
|
*
|
|
35
35
|
* @returns An array of capabilities, as expected by Pi
|
|
36
36
|
*/
|
|
37
|
-
protected async getCapabilities(): Promise<("text" | "image")[]> {
|
|
37
|
+
protected override async getCapabilities(): Promise<("text" | "image")[]> {
|
|
38
38
|
try {
|
|
39
39
|
return await super.getCapabilities();
|
|
40
40
|
} catch {
|