pi-llama-cpp 0.13.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.md +19 -18
  2. package/package.json +3 -3
  3. package/src/api/client.ts +25 -0
  4. package/src/constants.ts +2 -2
  5. package/src/enums/status.ts +0 -1
  6. package/src/index.ts +2 -1
  7. package/src/interfaces/endpoints/models.ts +1 -1
  8. package/src/interfaces/settings.ts +7 -1
  9. package/src/interfaces/sortBy.ts +4 -0
  10. package/src/managers/command/models.ts +254 -0
  11. package/src/managers/command.ts +34 -461
  12. package/src/managers/events.ts +2 -2
  13. package/src/managers/server.ts +34 -18
  14. package/src/managers/settings.ts +26 -90
  15. package/src/models/baseModel.ts +27 -10
  16. package/src/models/legacyModel.ts +2 -2
  17. package/src/models/routerModel.ts +2 -2
  18. package/src/models/singleModel.ts +1 -1
  19. package/src/server.ts +17 -38
  20. package/src/sse/client.ts +113 -59
  21. package/src/sse/fetch.ts +43 -0
  22. package/src/sse/manager.ts +9 -27
  23. package/src/sse/types.ts +0 -4
  24. package/src/ui/dialog/base.ts +118 -0
  25. package/src/ui/dialog/confirm.ts +45 -0
  26. package/src/ui/dialog/factory.ts +111 -0
  27. package/src/ui/dialog/input.ts +63 -0
  28. package/src/ui/dialog/options.ts +23 -0
  29. package/src/ui/editors/editorOptions.ts +50 -0
  30. package/src/ui/editors/itemBuilder.ts +43 -0
  31. package/src/ui/editors/listEditor.ts +300 -0
  32. package/src/ui/editors/override/entry.ts +24 -0
  33. package/src/ui/editors/override/entryEditor.ts +248 -0
  34. package/src/ui/editors/override/fields/base.ts +53 -0
  35. package/src/ui/editors/override/fields/capabilities.ts +36 -0
  36. package/src/ui/editors/override/fields/cost.ts +64 -0
  37. package/src/ui/editors/override/fields/index.ts +68 -0
  38. package/src/ui/editors/override/fields/numeric.ts +55 -0
  39. package/src/ui/editors/override/fields/pattern.ts +24 -0
  40. package/src/ui/editors/override/fields/reasoning.ts +33 -0
  41. package/src/ui/editors/override/itemBuilder.ts +39 -0
  42. package/src/ui/editors/override/overrideList.ts +127 -0
  43. package/src/ui/editors/server/builder.ts +60 -0
  44. package/src/ui/editors/server/fields.ts +84 -0
  45. package/src/ui/editors/server/itemBuilder.ts +64 -0
  46. package/src/ui/editors/server/serverEditor.ts +197 -0
  47. package/src/ui/editors/server/utils.ts +95 -0
  48. package/src/ui/editors/server/wizard.ts +110 -0
  49. package/src/ui/editors/settingField.ts +37 -0
  50. package/src/ui/editors/settingsListFactory.ts +33 -0
  51. package/src/ui/settings/index.ts +248 -0
  52. package/src/ui/strings.ts +25 -4
  53. package/src/utils/health.ts +2 -1
  54. package/src/utils/serverIds.ts +21 -0
  55. package/src/utils/settingsStore.ts +1 -1
  56. package/src/utils/urlResolver.ts +129 -0
  57. package/src/utils/urls.ts +33 -13
  58. package/tests/{commandManager.test.ts → command/commandManager.test.ts} +88 -12
  59. package/tests/{events.test.ts → events/events.test.ts} +6 -6
  60. package/tests/mocks.ts +2 -0
  61. package/tests/models/legacyModel.test.ts +103 -0
  62. package/tests/{routerModel.test.ts → models/routerModel.test.ts} +71 -72
  63. package/tests/{singleModel.test.ts → models/singleModel.test.ts} +17 -17
  64. package/tests/{health.test.ts → server/health.test.ts} +2 -2
  65. package/tests/{server.test.ts → server/server.test.ts} +5 -17
  66. package/tests/{serverManager.test.ts → server/serverManager.test.ts} +4 -4
  67. package/tests/{settings.test.ts → settings/settings.test.ts} +184 -141
  68. package/tests/{settingsStore.test.ts → settings/settingsStore.test.ts} +1 -1
  69. package/tests/{sseManager.test.ts → sse/sseManager.test.ts} +8 -26
  70. package/tests/{dialog.test.ts → ui/dialog.test.ts} +56 -4
  71. package/tests/{overrides.test.ts → ui/overrides.test.ts} +101 -111
  72. package/src/ui/dialog.ts +0 -290
  73. package/src/ui/overrideEntryEditor.ts +0 -119
  74. package/src/ui/overrideSettingsList.ts +0 -710
  75. package/src/ui/serverListEditor.ts +0 -59
  76. package/src/ui/serverSettingsList.ts +0 -513
  77. package/tests/legacyModel.test.ts +0 -97
package/README.md CHANGED
@@ -17,14 +17,13 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
17
17
 
18
18
  ### Status Indicators
19
19
 
20
- | Icon | Status | Description |
21
- | ---- | ------------ | -------------------------------------- |
22
- | 🟢 | Loaded | Model is active and ready to use |
23
- | 🟡 | Loading | Model is currently being loaded |
24
- | 🔴 | Failed | Model failed to load |
25
- | 🔵 | Sleeping | Model is available, but inactive |
26
- | ⚪ | Unloaded | Model is not loaded on the server |
27
- | ⛔ | Unauthorized | Model can't be used (API key required) |
20
+ | Icon | Status | Description |
21
+ | ---- | -------- | --------------------------------- |
22
+ | 🟢 | Loaded | Model is active and ready to use |
23
+ | 🟡 | Loading | Model is currently being loaded |
24
+ | 🔴 | Failed | Model failed to load |
25
+ | 🔵 | Sleeping | Model is available, but inactive |
26
+ | ⚪ | Unloaded | Model is not loaded on the server |
28
27
 
29
28
  > **Note**: The `Sleeping` status only shows when you start your server with `llama-server --sleep-idle-seconds <n> ...`.
30
29
  > This is a **llama.cpp server flag** that tells the server to put idle models to sleep after `n` seconds.
@@ -36,11 +35,12 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
36
35
 
37
36
  When browsing servers via `/models servers`, each server URL is prefixed with a health indicator:
38
37
 
39
- | Icon | Status | Description |
40
- | ---- | ----------- | --------------------------------------------- |
41
- | 🟢 | Healthy | Server responded successfully to health check |
42
- | 🟡 | Timeout | Server health check timed out |
43
- | 🔴 | Unreachable | Server could not be reached |
38
+ | Icon | Status | Description |
39
+ | ---- | ------------ | --------------------------------------------- |
40
+ | 🟢 | Healthy | Server responded successfully to health check |
41
+ | 🟡 | Timeout | Server health check timed out |
42
+ | 🔴 | Unreachable | Server could not be reached |
43
+ | ⛔ | Unauthorized | Server requires an API key |
44
44
 
45
45
  ## Installation
46
46
 
@@ -128,6 +128,7 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
128
128
  | `sortBy` | string | `"asc"` | Sort order for models (see below) |
129
129
  | `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
130
130
  | `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
131
+ | `showServerUrls` | boolean | `true` | Show `[Server: <url>]` suffix in the /models model list |
131
132
 
132
133
  > **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
133
134
 
@@ -138,11 +139,11 @@ Run `/models settings` to edit the scalar settings above without hand-editing JS
138
139
  #### Server list editor
139
140
 
140
141
  Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
141
- without hand-editing JSON. Each server URL is prefixed with a health indicator
142
- (🟢 healthy, 🟡 timeout, 🔴 unreachable) that reflects the result of a
143
- health check against the server. Each change is written immediately to the **project**
144
- `.pi/settings.json` if it exists, otherwise to **global**
145
- `~/.pi/agent/settings.json`.
142
+ without hand-editing JSON. Each server URL is prefixed with a status indicator
143
+ (🟢 healthy, 🟡 timeout, 🔴 unreachable, ⛔ unauthorized) that reflects the
144
+ result of a health check and an auth probe against the server. Each change is
145
+ written immediately to the **project** `.pi/settings.json` if it exists,
146
+ otherwise to **global** `~/.pi/agent/settings.json`.
146
147
 
147
148
  Changes take effect immediately after closing the editor: new servers
148
149
  register their providers, removed ones leave pi's registry right away,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.13.0",
3
+ "version": "0.15.0",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -38,8 +38,8 @@
38
38
  },
39
39
  "type": "module",
40
40
  "devDependencies": {
41
- "@types/node": "^26.4.1",
41
+ "@types/node": "^26.6.3",
42
42
  "prettier-plugin-organize-imports": "^4.3.0",
43
- "vitest": "^5.0.1"
43
+ "vitest": "^5.0.2"
44
44
  }
45
45
  }
package/src/api/client.ts CHANGED
@@ -1,5 +1,18 @@
1
1
  import { POLLING_INTERVAL } from "../constants";
2
2
 
3
+ /**
4
+ * Error thrown by the API client when the server rejects a request.
5
+ */
6
+ export class ApiError extends Error {
7
+ constructor(
8
+ public readonly type: "authentication" | "server" | "unknown",
9
+ public readonly statusCode: number,
10
+ ) {
11
+ super(`API error (${statusCode})`);
12
+ this.name = "ApiError";
13
+ }
14
+ }
15
+
3
16
  /**
4
17
  * How long GET responses stay cached: half the polling interval, so a poll
5
18
  * tick always reaches the server while multiple reads within one tick
@@ -126,6 +139,8 @@ export class ApiClient {
126
139
  *
127
140
  * @param endpoint The endpoint path to fetch (e.g. "/health")
128
141
  * @returns The parsed JSON response from the server
142
+ * @throws ApiError with status 401 when the server rejects the request
143
+ * due to authentication (invalid/missing API key).
129
144
  */
130
145
  private async do_get<T>(endpoint: string): Promise<T> {
131
146
  const url = `${this.baseUrl}${endpoint}`;
@@ -134,6 +149,10 @@ export class ApiClient {
134
149
  headers: { Authorization: `Bearer ${this.apiKey}` },
135
150
  });
136
151
 
152
+ if (res.status === 401) {
153
+ throw new ApiError("authentication", res.status);
154
+ }
155
+
137
156
  return res.json();
138
157
  }
139
158
 
@@ -144,6 +163,8 @@ export class ApiClient {
144
163
  * @param endpoint The endpoint path to post to
145
164
  * @param body The optional request body
146
165
  * @returns The parsed JSON response from the server
166
+ * @throws ApiError with status 401 when the server rejects the request
167
+ * due to authentication (invalid/missing API key).
147
168
  */
148
169
  private async do_post<T>(
149
170
  endpoint: string,
@@ -160,6 +181,10 @@ export class ApiClient {
160
181
  body: body ? JSON.stringify(body) : undefined,
161
182
  });
162
183
 
184
+ if (res.status === 401) {
185
+ throw new ApiError("authentication", res.status);
186
+ }
187
+
163
188
  return res.json();
164
189
  }
165
190
  }
package/src/constants.ts CHANGED
@@ -77,9 +77,9 @@ export const AUTOLOAD_ON_MESSAGE = false;
77
77
  export const SORT_BY = "asc";
78
78
 
79
79
  /**
80
- * Sort order options for model lists.
80
+ * Default for showServerUrls — show server URLs in /models by default.
81
81
  */
82
- export type SortBy = "asc" | "desc" | "asc-name" | "desc-name" | "api";
82
+ export const SHOW_SERVER_URLS = true;
83
83
 
84
84
  /**
85
85
  * Thinking budgets to send to the server, depending on user-selected level in Pi.
@@ -5,5 +5,4 @@ export enum Status {
5
5
  FAILED = "failed",
6
6
  SLEEPING = "sleeping",
7
7
  UNLOADED = "unloaded",
8
- UNAUTHORIZED = "unauthorized",
9
8
  }
package/src/index.ts CHANGED
@@ -11,9 +11,10 @@ import { ModelSelectEvent } from "./interfaces/events";
11
11
  import { CommandManager } from "./managers/command";
12
12
  import { EventManager } from "./managers/events";
13
13
  import { ServerManager } from "./managers/server";
14
- import { settings } from "./managers/settings";
14
+ import { createSettingsManager } from "./managers/settings";
15
15
 
16
16
  export default async function (pi: ExtensionAPI) {
17
+ const settings = createSettingsManager();
17
18
  const serverManager = new ServerManager(settings);
18
19
  const eventManager = new EventManager(serverManager, settings);
19
20
  const commandManager = new CommandManager(serverManager, settings);
@@ -10,7 +10,7 @@ export interface ModelsEndpoint {
10
10
  data: DataProperty[];
11
11
  }
12
12
 
13
- export interface ModelProperty {
13
+ interface ModelProperty {
14
14
  name: string;
15
15
  model: string;
16
16
  modified_at: string;
@@ -1,5 +1,5 @@
1
1
  import type { ModelCost, OpenAICompletionsCompat } from "@earendil-works/pi-ai";
2
- import type { SortBy } from "../constants";
2
+ import type { SortBy } from "./sortBy";
3
3
 
4
4
  /**
5
5
  * Per-model overrides applied on top of what llama-server reports.
@@ -132,4 +132,10 @@ export interface LlamaSettings {
132
132
  * @default "asc"
133
133
  */
134
134
  sortBy?: SortBy;
135
+ /**
136
+ * Whether to show the `[Server: <url>]` suffix in the /models model list.
137
+ * When `false`, only model names are shown without server annotations.
138
+ * @default true
139
+ */
140
+ showServerUrls?: boolean;
135
141
  }
@@ -0,0 +1,4 @@
1
+ /**
2
+ * Sort order options for model lists.
3
+ */
4
+ export type SortBy = "asc" | "desc" | "asc-name" | "desc-name" | "api";
@@ -0,0 +1,254 @@
1
+ import type {
2
+ ExtensionAPI,
3
+ ExtensionCommandContext,
4
+ } from "@earendil-works/pi-coding-agent";
5
+ import { PROVIDER_NAME } from "../../constants";
6
+ import { Action } from "../../enums/action";
7
+ import { Mode } from "../../enums/mode";
8
+ import { Status } from "../../enums/status";
9
+ import { BaseModel } from "../../models/baseModel";
10
+ import { errorMessage } from "../../utils/errors";
11
+ import { EventManager } from "../events";
12
+ import type { ServerManager } from "../server";
13
+ import type { LlamaSettingsManager } from "../settings";
14
+
15
+ /**
16
+ * Interactive model selection/action flow for the `/models` command.
17
+ *
18
+ * Owns the select-model → select-action loop and the per-status action
19
+ * tables, plus the non-blocking model load pipeline (progress events,
20
+ * inflight tracking, success/failure handling).
21
+ *
22
+ * Split out of `CommandManager`: command routing stays there;
23
+ * this class is the self-contained interactive concern.
24
+ * `ServerManager` is injected so load/refresh paths can reach the
25
+ * server registry without coupling the menu to command routing.
26
+ */
27
+ export class ModelsMenu {
28
+ constructor(
29
+ private readonly serverManager: ServerManager,
30
+ private readonly settings: LlamaSettingsManager,
31
+ ) {}
32
+
33
+ /**
34
+ * Runs the interactive model selection menu.
35
+ */
36
+ async show(ctx: ExtensionCommandContext, pi: ExtensionAPI): Promise<void> {
37
+ const event = await this.selectionLoop(
38
+ ctx,
39
+ await this.serverManager.getAllModels(),
40
+ );
41
+
42
+ if (!event) return;
43
+ const { action, model } = event;
44
+
45
+ // Action: Cancel
46
+ if (!action || action === Action.CANCEL) return;
47
+
48
+ // Action: Info
49
+ if (action === Action.INFO) {
50
+ const info = await model.getInfo();
51
+ ctx.ui.notify(`${info}`, "info");
52
+ return;
53
+ }
54
+
55
+ // Action: Unload
56
+ if (action === Action.UNLOAD) {
57
+ await model.unload();
58
+ ctx.ui.notify(`Unloaded ${model.name}`, "info");
59
+ return;
60
+ }
61
+
62
+ // Action: Switch
63
+ if (action === Action.SWITCH) {
64
+ const { serverId } = model;
65
+ const piModel = ctx.modelRegistry.find(serverId, model.id);
66
+ if (!piModel)
67
+ throw new Error(`Cannot find model ${model.name} in pi registry`);
68
+
69
+ await pi.setModel(piModel);
70
+ ctx.ui.notify(`Model ${model.name} ready`, "info");
71
+ return;
72
+ }
73
+
74
+ // Actions: Load / Load & Switch / Retry
75
+ const loadActions = [Action.LOAD, Action.LOAD_AND_SWITCH, Action.RETRY];
76
+ if (loadActions.includes(action)) {
77
+ ctx.ui.notify(`Loading ${model.name}...`, "info");
78
+ // Mark the load as in-flight so session_before_switch can warn about
79
+ // it (see EventManager.inflightModel for the coupling rationale)
80
+ EventManager.inflightModel = model;
81
+
82
+ // Subscribe to progress events; skip when the server is gone
83
+ // (removed/edited away mid-load → getServer returns undefined)
84
+ const server = this.serverManager.getServer(model);
85
+ const cleanupProgress =
86
+ server?.sseManager.subscribeToProgress(
87
+ model.id,
88
+ (percentage, stage) => {
89
+ const stageText = stage ? ` (${stage})` : "";
90
+ ctx.ui.notify(
91
+ `Loading ${model.name}... [${percentage}%${stageText}]`,
92
+ "info",
93
+ );
94
+ },
95
+ ) ?? (() => {});
96
+
97
+ const onSuccess = async () => {
98
+ const { serverId } = model;
99
+ const piModel = ctx.modelRegistry.find(serverId, model.id);
100
+ if (!piModel)
101
+ throw new Error(`Cannot find model ${model.name} in pi registry`);
102
+
103
+ // Verify failure
104
+ if ((await model.getStatus()) === Status.FAILED)
105
+ throw new Error(`Failed to load model ${model.name}`);
106
+
107
+ // Select the model if asked
108
+ if (action === Action.LOAD_AND_SWITCH) await pi.setModel(piModel);
109
+
110
+ ctx.ui.notify(`Model ${model.name} ready`, "info");
111
+ };
112
+
113
+ const onFailure = (err: any) => {
114
+ const message = errorMessage(err);
115
+
116
+ try {
117
+ ctx.ui.notify(message, "error");
118
+ } catch {
119
+ // ctx went stale between error and notification
120
+ }
121
+ };
122
+
123
+ const onFinished = async () => {
124
+ cleanupProgress();
125
+ EventManager.resetInflightModel();
126
+
127
+ // Re-scan providers to ensure accuracy of loaded models
128
+ await this.serverManager.update(pi);
129
+
130
+ // Force TUI refresh so Pi picks up the updated model states
131
+ ctx.ui.setStatus(PROVIDER_NAME, " ");
132
+ ctx.ui.setStatus(PROVIDER_NAME, undefined);
133
+ };
134
+
135
+ // Load the model without blocking the UI
136
+ model.load().then(onSuccess).catch(onFailure).finally(onFinished);
137
+ }
138
+ }
139
+
140
+ /**
141
+ * Handles the menu for model selection.
142
+ * Loops: select model → select action → handle action.
143
+ *
144
+ * Escape on actions menu goes back to model selection.
145
+ * Escape on model selection exits.
146
+ *
147
+ * @returns The selected action and model
148
+ */
149
+ private async selectionLoop(
150
+ ctx: ExtensionCommandContext,
151
+ models: BaseModel[],
152
+ ): Promise<{ action: Action; model: BaseModel } | null> {
153
+ while (true) {
154
+ // Select the model
155
+ const model = await this.selectModel(ctx, models);
156
+ if (!model) return null;
157
+
158
+ // Select the action
159
+ const actions = await this.getActionsForModel(model);
160
+ const action = await this.selectAction(ctx, model, actions);
161
+ if (action === null) {
162
+ // Escape key pressed => back to model selection
163
+ continue;
164
+ }
165
+
166
+ // Return the selected action and model
167
+ return { action, model };
168
+ }
169
+ }
170
+
171
+ /**
172
+ * Select a model from the list. Returns null if user cancels.
173
+ *
174
+ * @returns The model selected by the user
175
+ */
176
+ private async selectModel(
177
+ ctx: ExtensionCommandContext,
178
+ models: BaseModel[],
179
+ ): Promise<BaseModel | null> {
180
+ const showServerUrls = await this.settings.resolveShowServerUrls();
181
+
182
+ const labels = await Promise.all(
183
+ models.map(async (model) => ({
184
+ label: (await model.getLabel()).trim(),
185
+ serverUrl: model.serverUrl,
186
+ })),
187
+ );
188
+
189
+ // Count grapheme clusters (not UTF-16 code units) so emoji padding aligns visually
190
+ const graphemeLength = (str: string) =>
191
+ [...new Intl.Segmenter().segment(str)].length;
192
+
193
+ // Decorate the label so the spacing makes it seem more like a table
194
+ const maxLength = Math.max(
195
+ ...labels.map(({ label }) => graphemeLength(label)),
196
+ );
197
+ const choices = labels.map(({ label, serverUrl }) => {
198
+ const extraPadding = 2;
199
+ const padLen = maxLength - graphemeLength(label) + extraPadding;
200
+ const serverSuffix = showServerUrls ? ` [Server: ${serverUrl}]` : "";
201
+ return `${label}${" ".repeat(padLen)}${serverSuffix}`;
202
+ });
203
+
204
+ const choice = await ctx.ui.select(`${PROVIDER_NAME} models:`, choices);
205
+ if (!choice) return null;
206
+ const idx = choices.indexOf(choice);
207
+
208
+ return models[idx];
209
+ }
210
+
211
+ /**
212
+ * Get available actions for a model based on its mode and status.
213
+ *
214
+ * @returns A mapping of actions for each status
215
+ */
216
+ private async getActionsForModel(model: BaseModel): Promise<Array<Action>> {
217
+ const base = [Action.INFO, Action.CANCEL];
218
+
219
+ const actions: Record<Status, Array<Action>> = {
220
+ [Status.LOADED]:
221
+ model.mode === Mode.ROUTER
222
+ ? [Action.SWITCH, Action.UNLOAD, ...base]
223
+ : [Action.SWITCH, ...base],
224
+ [Status.LOADING]: [...base],
225
+ [Status.FAILED]: [Action.RETRY, ...base],
226
+ [Status.SLEEPING]:
227
+ model.mode === Mode.ROUTER
228
+ ? [Action.SWITCH, Action.UNLOAD, ...base]
229
+ : [Action.SWITCH, ...base],
230
+ [Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, ...base],
231
+ };
232
+
233
+ const status = await model.getStatus();
234
+ return actions[status];
235
+ }
236
+
237
+ /**
238
+ * Selects an action for a model.
239
+ *
240
+ * @returns The selected action
241
+ */
242
+ private async selectAction(
243
+ ctx: ExtensionCommandContext,
244
+ model: BaseModel,
245
+ actions: Array<Action>,
246
+ ): Promise<Action | null> {
247
+ const labels = actions.map((a) => String(a));
248
+ const choice = await ctx.ui.select(`${model.name}`, labels);
249
+ if (!choice) return null;
250
+
251
+ const idx = labels.indexOf(choice);
252
+ return actions[idx];
253
+ }
254
+ }