pi-llama-cpp 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +52 -13
  2. package/package.json +3 -3
  3. package/src/api/client.ts +59 -22
  4. package/src/constants.ts +5 -0
  5. package/src/enums/mode.ts +1 -0
  6. package/src/index.ts +2 -1
  7. package/src/interfaces/endpoints/models.ts +6 -0
  8. package/src/interfaces/endpoints/props.ts +14 -0
  9. package/src/interfaces/settings.ts +6 -0
  10. package/src/managers/command/models.ts +12 -6
  11. package/src/managers/command.ts +6 -1
  12. package/src/managers/server.ts +30 -38
  13. package/src/managers/settings.ts +23 -3
  14. package/src/models/legacyModel.ts +3 -3
  15. package/src/models/llamaSwapModel.ts +90 -0
  16. package/src/models/routerModel.ts +2 -2
  17. package/src/models/singleModel.ts +2 -2
  18. package/src/server.ts +42 -15
  19. package/src/sse/manager.ts +2 -2
  20. package/src/ui/dialog/confirm.ts +1 -1
  21. package/src/ui/dialog/input.ts +1 -1
  22. package/src/ui/editors/editorOptions.ts +8 -1
  23. package/src/ui/editors/itemBuilder.ts +2 -6
  24. package/src/ui/editors/listEditor.ts +24 -15
  25. package/src/ui/editors/override/entryEditor.ts +105 -23
  26. package/src/ui/editors/override/fields/base.ts +53 -0
  27. package/src/ui/editors/override/fields/capabilities.ts +36 -0
  28. package/src/ui/editors/override/fields/cost.ts +64 -0
  29. package/src/ui/editors/override/fields/index.ts +68 -0
  30. package/src/ui/editors/override/fields/numeric.ts +55 -0
  31. package/src/ui/editors/override/fields/pattern.ts +24 -0
  32. package/src/ui/editors/override/fields/reasoning.ts +33 -0
  33. package/src/ui/editors/override/itemBuilder.ts +4 -4
  34. package/src/ui/editors/server/fields.ts +2 -2
  35. package/src/ui/editors/server/itemBuilder.ts +2 -3
  36. package/src/ui/editors/server/serverEditor.ts +58 -22
  37. package/src/ui/editors/server/utils.ts +50 -0
  38. package/src/ui/settings/index.ts +11 -0
  39. package/src/ui/strings.ts +17 -0
  40. package/src/utils/credentialResolver.ts +110 -0
  41. package/src/utils/health.ts +2 -2
  42. package/src/utils/urlResolver.ts +1 -1
  43. package/tests/{commandManager.test.ts → command/commandManager.test.ts} +80 -9
  44. package/tests/{events.test.ts → events/events.test.ts} +6 -6
  45. package/tests/mocks.ts +4 -0
  46. package/tests/{legacyModel.test.ts → models/legacyModel.test.ts} +5 -5
  47. package/tests/models/llamaSwapModel.test.ts +327 -0
  48. package/tests/{routerModel.test.ts → models/routerModel.test.ts} +5 -5
  49. package/tests/{singleModel.test.ts → models/singleModel.test.ts} +5 -5
  50. package/tests/{health.test.ts → server/health.test.ts} +4 -4
  51. package/tests/{server.test.ts → server/server.test.ts} +57 -7
  52. package/tests/{serverManager.test.ts → server/serverManager.test.ts} +4 -4
  53. package/tests/{settings.test.ts → settings/settings.test.ts} +198 -355
  54. package/tests/{settingsStore.test.ts → settings/settingsStore.test.ts} +1 -1
  55. package/tests/{sseManager.test.ts → sse/sseManager.test.ts} +2 -2
  56. package/tests/{dialog.test.ts → ui/dialog.test.ts} +5 -4
  57. package/tests/{overrides.test.ts → ui/overrides.test.ts} +9 -42
  58. package/tests/utils/credentialResolver.test.ts +66 -0
  59. package/tests/utils/urlResolver.test.ts +116 -0
  60. package/src/ui/editors/override/fields.ts +0 -294
  61. package/src/ui/editors/override/handlers.ts +0 -118
  62. package/src/ui/editors/server/handlers.ts +0 -32
package/README.md CHANGED
@@ -12,6 +12,7 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
12
12
  - **Flexible URL resolution** — configures the server via `llamaSettings` (project/global), environment variable, or legacy `llamaServerUrl`
13
13
  - **Auth support** — allows to login into a llama.cpp server that was secured with an API key
14
14
  - **Multiple server support** — connect to multiple llama.cpp servers simultaneously via `llamaSettings.servers` or semicolon-separated URLs
15
+ - **Basic llama-swap support** — auto-detects and provides basic integration with [llama-swap](https://github.com/mostlygeek/llama-swap) gateways
15
16
  - **Thinking budget support** — configurable token budgets for model reasoning/thinking, mapped to Pi's thinking levels
16
17
  - **Real-time progress tracking** — live loading progress via SSE (falls back to polling)
17
18
 
@@ -35,11 +36,12 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
35
36
 
36
37
  When browsing servers via `/models servers`, each server URL is prefixed with a health indicator:
37
38
 
38
- | Icon | Status | Description |
39
- | ---- | ----------- | --------------------------------------------- |
40
- | 🟢 | Healthy | Server responded successfully to health check |
41
- | 🟡 | Timeout | Server health check timed out |
42
- | 🔴 | Unreachable | Server could not be reached |
39
+ | Icon | Status | Description |
40
+ | ---- | ------------ | --------------------------------------------- |
41
+ | 🟢 | Healthy | Server responded successfully to health check |
42
+ | 🟡 | Timeout | Server health check timed out |
43
+ | 🔴 | Unreachable | Server could not be reached |
44
+ | ⛔ | Unauthorized | Server requires an API key |
43
45
 
44
46
  ## Installation
45
47
 
@@ -127,6 +129,7 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
127
129
  | `sortBy` | string | `"asc"` | Sort order for models (see below) |
128
130
  | `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
129
131
  | `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
132
+ | `showServerUrls` | boolean | `true` | Show `[Server: <url>]` suffix in the /models model list |
130
133
 
131
134
  > **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
132
135
 
@@ -137,11 +140,11 @@ Run `/models settings` to edit the scalar settings above without hand-editing JS
137
140
  #### Server list editor
138
141
 
139
142
  Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
140
- without hand-editing JSON. Each server URL is prefixed with a health indicator
141
- (🟢 healthy, 🟡 timeout, 🔴 unreachable) that reflects the result of a
142
- health check against the server. Each change is written immediately to the **project**
143
- `.pi/settings.json` if it exists, otherwise to **global**
144
- `~/.pi/agent/settings.json`.
143
+ without hand-editing JSON. Each server URL is prefixed with a status indicator
144
+ (🟢 healthy, 🟡 timeout, 🔴 unreachable, ⛔ unauthorized) that reflects the
145
+ result of a health check and an auth probe against the server. Each change is
146
+ written immediately to the **project** `.pi/settings.json` if it exists,
147
+ otherwise to **global** `~/.pi/agent/settings.json`.
145
148
 
146
149
  Changes take effect immediately after closing the editor: new servers
147
150
  register their providers, removed ones leave pi's registry right away,
@@ -205,17 +208,48 @@ Each server gets its own provider (e.g., **Llama.cpp (http://127.0.0.1:8080)**)
205
208
  If your llama.cpp server requires authentication, use `/login` in Pi, select the "API key" option, and choose the provider from the list that correlates with the server needing the API key.
206
209
 
207
210
  Alternatively, configure the API key in `~/.pi/agent/auth.json`:
208
- Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`):
211
+ Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`).
212
+
213
+ The `key` field supports several formats:
214
+
215
+ | Format | Example | Description |
216
+ | ----------------- | -------------------------------------------- | ------------------------------------------ |
217
+ | **Literal** | `"sk-abc123"` | API key stored directly |
218
+ | **Env ref** | `"$OPENAI_API_KEY"` or `"${OPENAI_API_KEY}"` | Resolved from `process.env` or `env` field |
219
+ | **Shell command** | `"!cat ~/.secrets/llama-key"` | Stdout of the command is used |
220
+ | **Escape** | `"$$literal"` | `$$` → literal `$`, `$!` → literal `!` |
209
221
 
210
222
  ```json
211
223
  {
212
224
  "llama-server=http://127.0.0.1:8080": {
213
225
  "type": "api_key",
214
- "key": "<key-for-server-1>"
226
+ "key": "sk-abc123"
215
227
  },
216
228
  "llama-server=https://some-url-for-llama-cpp": {
217
229
  "type": "api_key",
218
- "key": "<key-for-server-2>"
230
+ "key": "$LLAMA_API_KEY"
231
+ },
232
+ "llama-server=https://secure-server": {
233
+ "type": "api_key",
234
+ "key": "!cat ~/.secrets/llama-key"
235
+ },
236
+ "llama-server=https://braced-ref": {
237
+ "type": "api_key",
238
+ "key": "${API_KEY}"
239
+ }
240
+ }
241
+ ```
242
+
243
+ For env ref formats, you can also store the variable value alongside the key using the `env` field:
244
+
245
+ ```json
246
+ {
247
+ "llama-server=http://127.0.0.1:8080": {
248
+ "type": "api_key",
249
+ "key": "$MY_KEY",
250
+ "env": {
251
+ "MY_KEY": "sk-abc123"
252
+ }
219
253
  }
220
254
  }
221
255
  ```
@@ -244,6 +278,10 @@ llama-server --model path/to/model.gguf ...
244
278
 
245
279
  > **Note:** The ik_llama.cpp fork is not legacy at all, but it uses an old way of describing models compared to llama.cpp.
246
280
 
281
+ - For llama-swap mode, point the extension at a running [llama-swap](https://github.com/mostlygeek/llama-swap) instance instead of a raw llama.cpp server. The extension auto-detects this mode via the `src` field in the server props response.
282
+
283
+ > **Note:** llama-swap support is basic — only model listing, status, load/unload, and capability detection are implemented.
284
+
247
285
  The extension determines the context size as follows:
248
286
 
249
287
  - A per-model `contextSize` override (see [Model Overrides](#model-overrides)) takes precedence over everything below
@@ -252,6 +290,7 @@ The extension determines the context size as follows:
252
290
  - When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
253
291
  - **Single mode** — reads `meta.n_ctx` from the `/v1/models` endpoint
254
292
  - **Legacy mode** — reads `max_model_len` from `/v1/models`, falling back to `n_ctx` from `/props`
293
+ - **Llama-swap mode** — reads `meta.n_ctx` from the llama-swap server via `/v1/models`
255
294
  - Falls back to `128000` if not available
256
295
 
257
296
  ### Commands
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.14.0",
3
+ "version": "0.16.0",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -38,8 +38,8 @@
38
38
  },
39
39
  "type": "module",
40
40
  "devDependencies": {
41
- "@types/node": "^26.6.2",
41
+ "@types/node": "^26.6.3",
42
42
  "prettier-plugin-organize-imports": "^4.3.0",
43
- "vitest": "^5.0.1"
43
+ "vitest": "^5.0.2"
44
44
  }
45
45
  }
package/src/api/client.ts CHANGED
@@ -81,6 +81,30 @@ export class ApiClient {
81
81
  );
82
82
  }
83
83
 
84
+ /**
85
+ * Makes a raw GET request that does not parse JSON. Useful for endpoints
86
+ * that return empty or non-JSON responses.
87
+ *
88
+ * @param endpoint The endpoint path to fetch
89
+ */
90
+ async rawGet(endpoint: string): Promise<void> {
91
+ return this.do_request<void>(endpoint, "GET", undefined, false);
92
+ }
93
+
94
+ /**
95
+ * Makes a raw POST request that does not parse JSON. Useful for endpoints
96
+ * that return empty or non-JSON responses.
97
+ *
98
+ * @param endpoint The endpoint path to post to
99
+ * @param body The optional request body
100
+ */
101
+ async rawPost(
102
+ endpoint: string,
103
+ body?: Record<string, unknown>,
104
+ ): Promise<void> {
105
+ return this.do_request<void>(endpoint, "POST", body, false);
106
+ }
107
+
84
108
  /**
85
109
  * Clears the entire cache.
86
110
  */
@@ -134,26 +158,54 @@ export class ApiClient {
134
158
  }
135
159
 
136
160
  /**
137
- * Makes a raw GET request to the llama-server.
161
+ * Makes a raw request to the llama-server.
138
162
  * This bypasses caching and deduplication.
139
163
  *
140
- * @param endpoint The endpoint path to fetch (e.g. "/health")
141
- * @returns The parsed JSON response from the server
164
+ * @param endpoint The endpoint path
165
+ * @param method The HTTP method
166
+ * @param body The optional request body
167
+ * @param parseJson Whether to parse the response as JSON (default: true)
168
+ * @returns The parsed JSON response, or void if parseJson is false
142
169
  * @throws ApiError with status 401 when the server rejects the request
143
170
  * due to authentication (invalid/missing API key).
144
171
  */
145
- private async do_get<T>(endpoint: string): Promise<T> {
172
+ private async do_request<T>(
173
+ endpoint: string,
174
+ method: "GET" | "POST",
175
+ body?: Record<string, unknown>,
176
+ parseJson = true,
177
+ ): Promise<T | void> {
146
178
  const url = `${this.baseUrl}${endpoint}`;
147
179
 
148
180
  const res = await fetch(url, {
149
- headers: { Authorization: `Bearer ${this.apiKey}` },
181
+ method,
182
+ headers: {
183
+ ...(method === "POST"
184
+ ? { "Content-Type": "application/json" }
185
+ : undefined),
186
+ Authorization: `Bearer ${this.apiKey}`,
187
+ },
188
+ ...(body ? { body: JSON.stringify(body) } : undefined),
150
189
  });
151
190
 
152
191
  if (res.status === 401) {
153
192
  throw new ApiError("authentication", res.status);
154
193
  }
155
194
 
156
- return res.json();
195
+ return parseJson ? ((await res.json()) as T) : (undefined as T);
196
+ }
197
+
198
+ /**
199
+ * Makes a raw GET request to the llama-server.
200
+ * This bypasses caching and deduplication.
201
+ *
202
+ * @param endpoint The endpoint path to fetch (e.g. "/health")
203
+ * @returns The parsed JSON response from the server
204
+ * @throws ApiError with status 401 when the server rejects the request
205
+ * due to authentication (invalid/missing API key).
206
+ */
207
+ private async do_get<T>(endpoint: string): Promise<T> {
208
+ return this.do_request<T>(endpoint, "GET") as Promise<T>;
157
209
  }
158
210
 
159
211
  /**
@@ -170,21 +222,6 @@ export class ApiClient {
170
222
  endpoint: string,
171
223
  body?: Record<string, unknown>,
172
224
  ): Promise<T> {
173
- const url = `${this.baseUrl}${endpoint}`;
174
-
175
- const res = await fetch(url, {
176
- method: "POST",
177
- headers: {
178
- "Content-Type": "application/json",
179
- Authorization: `Bearer ${this.apiKey}`,
180
- },
181
- body: body ? JSON.stringify(body) : undefined,
182
- });
183
-
184
- if (res.status === 401) {
185
- throw new ApiError("authentication", res.status);
186
- }
187
-
188
- return res.json();
225
+ return this.do_request<T>(endpoint, "POST", body) as Promise<T>;
189
226
  }
190
227
  }
package/src/constants.ts CHANGED
@@ -76,6 +76,11 @@ export const AUTOLOAD_ON_MESSAGE = false;
76
76
  */
77
77
  export const SORT_BY = "asc";
78
78
 
79
+ /**
80
+ * Default for showServerUrls — show server URLs in /models by default.
81
+ */
82
+ export const SHOW_SERVER_URLS = true;
83
+
79
84
  /**
80
85
  * Thinking budgets to send to the server, depending on user-selected level in Pi.
81
86
  */
package/src/enums/mode.ts CHANGED
@@ -3,4 +3,5 @@ export enum Mode {
3
3
  ROUTER = "router",
4
4
  SINGLE = "single",
5
5
  LEGACY = "legacy",
6
+ LLAMASWAP = "llama-swap",
6
7
  }
package/src/index.ts CHANGED
@@ -11,9 +11,10 @@ import { ModelSelectEvent } from "./interfaces/events";
11
11
  import { CommandManager } from "./managers/command";
12
12
  import { EventManager } from "./managers/events";
13
13
  import { ServerManager } from "./managers/server";
14
- import { settings } from "./managers/settings";
14
+ import { createSettingsManager } from "./managers/settings";
15
15
 
16
16
  export default async function (pi: ExtensionAPI) {
17
+ const settings = createSettingsManager();
17
18
  const serverManager = new ServerManager(settings);
18
19
  const eventManager = new EventManager(serverManager, settings);
19
20
  const commandManager = new CommandManager(serverManager, settings);
@@ -67,4 +67,10 @@ interface MetaProperty {
67
67
  n_embd: number;
68
68
  n_params: number;
69
69
  size: number;
70
+
71
+ // llama-swap extension
72
+ llamaswap?: {
73
+ aliases: string[];
74
+ type: string;
75
+ };
70
76
  }
@@ -13,6 +13,20 @@ export interface PropsEndpoint {
13
13
  cors_proxy_enabled: boolean;
14
14
  }
15
15
 
16
+ /**
17
+ * llama-swap returns a different shape for /props (without a model ID):
18
+ * an error response with a `src` that we can use as a trick to detect it.
19
+ */
20
+ export interface LlamaSwapPropsError {
21
+ src: "llama-swap";
22
+ error: {
23
+ message: string;
24
+ type: string;
25
+ param: null;
26
+ code: string;
27
+ };
28
+ }
29
+
16
30
  /**
17
31
  * The structure of llama-server's /props?model=<id> endpoint
18
32
  */
@@ -132,4 +132,10 @@ export interface LlamaSettings {
132
132
  * @default "asc"
133
133
  */
134
134
  sortBy?: SortBy;
135
+ /**
136
+ * Whether to show the `[Server: <url>]` suffix in the /models model list.
137
+ * When `false`, only model names are shown without server annotations.
138
+ * @default true
139
+ */
140
+ showServerUrls?: boolean;
135
141
  }
@@ -10,6 +10,7 @@ import { BaseModel } from "../../models/baseModel";
10
10
  import { errorMessage } from "../../utils/errors";
11
11
  import { EventManager } from "../events";
12
12
  import type { ServerManager } from "../server";
13
+ import type { LlamaSettingsManager } from "../settings";
13
14
 
14
15
  /**
15
16
  * Interactive model selection/action flow for the `/models` command.
@@ -24,7 +25,10 @@ import type { ServerManager } from "../server";
24
25
  * server registry without coupling the menu to command routing.
25
26
  */
26
27
  export class ModelsMenu {
27
- constructor(private readonly serverManager: ServerManager) {}
28
+ constructor(
29
+ private readonly serverManager: ServerManager,
30
+ private readonly settings: LlamaSettingsManager,
31
+ ) {}
28
32
 
29
33
  /**
30
34
  * Runs the interactive model selection menu.
@@ -173,6 +177,8 @@ export class ModelsMenu {
173
177
  ctx: ExtensionCommandContext,
174
178
  models: BaseModel[],
175
179
  ): Promise<BaseModel | null> {
180
+ const showServerUrls = await this.settings.resolveShowServerUrls();
181
+
176
182
  const labels = await Promise.all(
177
183
  models.map(async (model) => ({
178
184
  label: (await model.getLabel()).trim(),
@@ -191,7 +197,8 @@ export class ModelsMenu {
191
197
  const choices = labels.map(({ label, serverUrl }) => {
192
198
  const extraPadding = 2;
193
199
  const padLen = maxLength - graphemeLength(label) + extraPadding;
194
- return `${label}${" ".repeat(padLen)} [Server: ${serverUrl}]`;
200
+ const serverSuffix = showServerUrls ? ` [Server: ${serverUrl}]` : "";
201
+ return `${label}${" ".repeat(padLen)}${serverSuffix}`;
195
202
  });
196
203
 
197
204
  const choice = await ctx.ui.select(`${PROVIDER_NAME} models:`, choices);
@@ -210,10 +217,9 @@ export class ModelsMenu {
210
217
  const base = [Action.INFO, Action.CANCEL];
211
218
 
212
219
  const actions: Record<Status, Array<Action>> = {
213
- [Status.LOADED]:
214
- model.mode === Mode.ROUTER
215
- ? [Action.SWITCH, Action.UNLOAD, ...base]
216
- : [Action.SWITCH, ...base],
220
+ [Status.LOADED]: [Mode.ROUTER, Mode.LLAMASWAP].includes(model.mode)
221
+ ? [Action.SWITCH, Action.UNLOAD, ...base]
222
+ : [Action.SWITCH, ...base],
217
223
  [Status.LOADING]: [...base],
218
224
  [Status.FAILED]: [Action.RETRY, ...base],
219
225
  [Status.SLEEPING]:
@@ -51,7 +51,7 @@ export class CommandManager {
51
51
  private readonly serverManager: ServerManager,
52
52
  private readonly settings: LlamaSettingsManager,
53
53
  ) {
54
- this.modelsMenu = new ModelsMenu(serverManager);
54
+ this.modelsMenu = new ModelsMenu(serverManager, settings);
55
55
  }
56
56
 
57
57
  /**
@@ -108,6 +108,11 @@ export class CommandManager {
108
108
  this.notifyNotFound(ctx, url);
109
109
  }
110
110
 
111
+ // Notify about other warnings (e.g. unauthorized servers)
112
+ for (const warning of this.serverManager.getWarnings()) {
113
+ ctx.ui.notify(warning, "warning");
114
+ }
115
+
111
116
  if (args === "unload") {
112
117
  const models = await this.serverManager.getAllModels();
113
118
  await Promise.all(models.map((model) => model.unload()));
@@ -1,10 +1,14 @@
1
- import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
1
+ import type {
2
+ ExtensionAPI,
3
+ ProviderModelConfig,
4
+ } from "@earendil-works/pi-coding-agent";
2
5
  import { ApiError } from "../api/client";
3
6
  import { API_TYPE, PROVIDER_NAME } from "../constants";
4
7
  import { ServerStatus } from "../enums/serverStatus";
5
8
  import type { SortBy } from "../interfaces/sortBy";
6
9
  import { BaseModel } from "../models/baseModel";
7
10
  import { Server } from "../server";
11
+ import { authRequiredMessage } from "../ui/strings";
8
12
  import type { LlamaSettingsManager } from "./settings";
9
13
 
10
14
  /** Model-list comparator: negative if a sorts first, positive if b does. */
@@ -45,6 +49,7 @@ export class ServerManager {
45
49
  */
46
50
  async update(pi: ExtensionAPI, timeout?: number) {
47
51
  this.failedUrls.length = 0;
52
+ this.warnings.length = 0;
48
53
 
49
54
  // Surface warnings from strict URL parsing (dropped invalid entries)
50
55
  this.warnings.push(...this.settings.takeWarnings());
@@ -60,9 +65,9 @@ export class ServerManager {
60
65
  }
61
66
 
62
67
  // Unregister providers that disappeared (removed or edited away);
63
- // no-op for providers that were never registered
68
+ // `seen` tracks all kept providerIds, so we skip those still present.
64
69
  for (const old of this.servers) {
65
- if (fresh.some((f) => f.providerId === old.providerId)) continue;
70
+ if (seen.has(old.providerId)) continue;
66
71
  pi.unregisterProvider(old.providerId);
67
72
  // Optional chain is intentional despite the non-optional type: `sse`
68
73
  // is undefined until initialize() runs (async-constructor hack — see Server)
@@ -79,34 +84,7 @@ export class ServerManager {
79
84
 
80
85
  // Initialization and registration
81
86
  for (const server of registrableServers) {
82
- try {
83
- await server.initialize();
84
- await this.registerProvider(server, pi);
85
- } catch (err) {
86
- if (err instanceof ApiError && err.type === "authentication") {
87
- // Register the provider with an empty model list so the user can
88
- // still configure the API key via `/login` or `auth.json`. On the
89
- // next scan the provider will re-initialize and discover models.
90
- // Don't add to `failedUrls` — the server IS reachable, auth just
91
- // isn't configured yet, so the health indicator should stay green.
92
- const message = [
93
- "[pi-llama-cpp]",
94
- `Server at '${server.baseUrl}' requires a valid API key.`,
95
- "Configure the key via `/login` or in `~/.pi/agent/auth.json`.",
96
- ].join("\n");
97
- this.warnings.push(message);
98
- pi.registerProvider(server.providerId, {
99
- name: server.providerName,
100
- baseUrl: server.apiBaseUrl,
101
- api: API_TYPE,
102
- apiKey: server.getApiKey(),
103
- models: [],
104
- });
105
- continue;
106
- }
107
- this.failedUrls.push(server.baseUrl);
108
- continue;
109
- }
87
+ await this.registerProvider(server, pi);
110
88
  }
111
89
  }
112
90
 
@@ -151,23 +129,37 @@ export class ServerManager {
151
129
  }
152
130
 
153
131
  /**
154
- * Creates a Pi provider for the given server.
132
+ * Initializes the server and creates a Pi provider.
133
+ * Handles auth errors by registering with an empty model list so the user
134
+ * can still configure the API key via `/login` or `auth.json`.
155
135
  *
156
136
  * @param server The server
157
137
  * @param pi The Pi API
158
138
  */
159
139
  private async registerProvider(server: Server, pi: ExtensionAPI) {
160
- const { apiBaseUrl, models, providerId, providerName } = server;
161
- const apiKey = server.getApiKey();
162
- const modelConfigs = await Promise.all(
163
- models.map((m) => m.toProviderConfig()),
164
- );
140
+ const { apiBaseUrl, apiKey, baseUrl, models, providerId, providerName } =
141
+ server;
142
+ let modelConfigs: ProviderModelConfig[] = [];
143
+
144
+ try {
145
+ await server.initialize();
146
+ modelConfigs = await Promise.all(models.map((m) => m.toProviderConfig()));
147
+ } catch (err) {
148
+ if (err instanceof ApiError && err.type === "authentication") {
149
+ // Don't add to `failedUrls` — the server IS reachable, auth just
150
+ // isn't configured yet, so the health indicator should stay green.
151
+ this.warnings.push(authRequiredMessage(baseUrl, providerId));
152
+ } else {
153
+ this.failedUrls.push(baseUrl);
154
+ return;
155
+ }
156
+ }
165
157
 
166
158
  pi.registerProvider(providerId, {
167
159
  name: providerName,
168
160
  baseUrl: apiBaseUrl,
169
161
  api: API_TYPE,
170
- apiKey: apiKey,
162
+ apiKey,
171
163
  models: modelConfigs,
172
164
  });
173
165
  }
@@ -13,6 +13,7 @@ import {
13
13
  REACT_TO_MODEL_SELECT,
14
14
  SERVER_TIMEOUT,
15
15
  SETTINGS_KEY,
16
+ SHOW_SERVER_URLS,
16
17
  SORT_BY,
17
18
  THINKING_BUDGETS,
18
19
  } from "../constants";
@@ -23,6 +24,7 @@ import {
23
24
  } from "../interfaces/settings";
24
25
  import type { SortBy } from "../interfaces/sortBy";
25
26
  import { Server } from "../server";
27
+ import { CredentialResolver } from "../utils/credentialResolver";
26
28
  import { SettingsStore } from "../utils/settingsStore";
27
29
  import { UrlResolver } from "../utils/urlResolver";
28
30
 
@@ -46,6 +48,9 @@ export class LlamaSettingsManager {
46
48
  }
47
49
  }
48
50
 
51
+ /** Delegates credential key resolution (see `utils/credentialResolver`). */
52
+ private credentialResolver = new CredentialResolver();
53
+
49
54
  /** Delegated multi-source URL resolution chain (see `utils/urlResolver`). */
50
55
  private urlResolver = new UrlResolver({
51
56
  getLlamaSettings: () => this.getLlamaSettings(),
@@ -148,12 +153,16 @@ export class LlamaSettingsManager {
148
153
 
149
154
  /**
150
155
  * Resolves API key for the provider ID using Pi's stored credentials.
156
+ * Delegates to `CredentialResolver` for key format handling.
151
157
  *
158
+ * @param providerId The provider ID
152
159
  * @returns The API key to use for the provider
153
160
  */
154
161
  resolveApiKey(providerId: string): string {
155
162
  const credential = readStoredCredential(providerId) as ApiKeyCredential;
156
- return credential?.key ?? API_KEY_PLACEHOLDER;
163
+ if (!credential?.key) return API_KEY_PLACEHOLDER;
164
+
165
+ return this.credentialResolver.resolve(credential.key, credential.env);
157
166
  }
158
167
 
159
168
  /**
@@ -226,6 +235,15 @@ export class LlamaSettingsManager {
226
235
  return (await this.getLlamaSettings()).sortBy ?? SORT_BY;
227
236
  }
228
237
 
238
+ /**
239
+ * Resolves whether to show server URLs in the /models model list.
240
+ *
241
+ * @returns `true` if server URLs should be shown
242
+ */
243
+ async resolveShowServerUrls(): Promise<boolean> {
244
+ return (await this.getLlamaSettings()).showServerUrls ?? SHOW_SERVER_URLS;
245
+ }
246
+
229
247
  /**
230
248
  * Persists one llamaSettings field to settings and reloads the in-memory
231
249
  * settings so resolvers see the change immediately.
@@ -262,6 +280,8 @@ export class LlamaSettingsManager {
262
280
  }
263
281
 
264
282
  /**
265
- * Shared singleton instance used across the extension.
283
+ * Creates a new LlamaSettingsManager instance.
266
284
  */
267
- export const settings = new LlamaSettingsManager();
285
+ export function createSettingsManager(): LlamaSettingsManager {
286
+ return new LlamaSettingsManager();
287
+ }
@@ -3,7 +3,7 @@ import { Mode } from "../enums/mode";
3
3
  import { SingleModel } from "./singleModel";
4
4
 
5
5
  export class LegacyModel extends SingleModel {
6
- get mode(): Mode {
6
+ override get mode(): Mode {
7
7
  return Mode.LEGACY;
8
8
  }
9
9
 
@@ -13,7 +13,7 @@ export class LegacyModel extends SingleModel {
13
13
  *
14
14
  * @returns The context size
15
15
  */
16
- protected async getContextSize(): Promise<number> {
16
+ protected override async getContextSize(): Promise<number> {
17
17
  const props = await this.server.fetchModelProps(this.id);
18
18
  const models = await this.server.fetchModels();
19
19
 
@@ -34,7 +34,7 @@ export class LegacyModel extends SingleModel {
34
34
  *
35
35
  * @returns An array of capabilities, as expected by Pi
36
36
  */
37
- protected async getCapabilities(): Promise<("text" | "image")[]> {
37
+ protected override async getCapabilities(): Promise<("text" | "image")[]> {
38
38
  try {
39
39
  return await super.getCapabilities();
40
40
  } catch {