pi-llama-cpp 0.9.2 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +163 -27
- package/package.json +4 -3
- package/src/api/client.ts +76 -32
- package/src/constants.ts +28 -3
- package/src/index.ts +4 -12
- package/src/interfaces/events.ts +14 -3
- package/src/interfaces/server.ts +20 -0
- package/src/interfaces/settings.ts +69 -0
- package/src/managers/command.ts +285 -43
- package/src/managers/events.ts +50 -12
- package/src/managers/server.ts +90 -17
- package/src/managers/settings.ts +268 -0
- package/src/models/baseModel.ts +32 -17
- package/src/models/legacyModel.ts +2 -2
- package/src/models/routerModel.ts +3 -3
- package/src/server.ts +86 -21
- package/src/sse/client.ts +28 -16
- package/src/sse/manager.ts +23 -10
- package/src/ui/serverListEditor.ts +481 -0
- package/src/utils/errors.ts +5 -0
- package/src/utils/settingsStore.ts +60 -0
- package/src/utils/urls.ts +16 -0
- package/tests/commandManager.test.ts +256 -11
- package/tests/events.test.ts +229 -64
- package/tests/mocks.ts +145 -32
- package/tests/server.test.ts +54 -6
- package/tests/serverListEditor.test.ts +637 -0
- package/tests/serverManager.test.ts +282 -39
- package/tests/settings.test.ts +793 -0
- package/tests/settingsStore.test.ts +209 -0
- package/tests/sseManager.test.ts +98 -0
- package/src/interfaces/auth.ts +0 -6
- package/src/resolver.ts +0 -149
- package/src/utils/cache.ts +0 -39
- package/src/utils/mutex.ts +0 -24
- package/tests/resolver.test.ts +0 -184
package/src/managers/command.ts
CHANGED
|
@@ -1,18 +1,180 @@
|
|
|
1
|
-
import
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
import {
|
|
2
|
+
getSettingsListTheme,
|
|
3
|
+
type ExtensionAPI,
|
|
4
|
+
type ExtensionCommandContext,
|
|
4
5
|
} from "@earendil-works/pi-coding-agent";
|
|
5
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
AutocompleteItem,
|
|
8
|
+
SettingsList,
|
|
9
|
+
type SettingItem,
|
|
10
|
+
} from "@earendil-works/pi-tui";
|
|
6
11
|
import { PROVIDER_NAME } from "../constants";
|
|
7
12
|
import { Action } from "../enums/action";
|
|
8
13
|
import { Mode } from "../enums/mode";
|
|
9
14
|
import { Status } from "../enums/status";
|
|
15
|
+
import { LlamaSettings } from "../interfaces/settings";
|
|
10
16
|
import { BaseModel } from "../models/baseModel";
|
|
17
|
+
import { ServerListEditor } from "../ui/serverListEditor";
|
|
18
|
+
import { errorMessage } from "../utils/errors";
|
|
11
19
|
import { EventManager } from "./events";
|
|
12
20
|
import { ServerManager } from "./server";
|
|
21
|
+
import type { LlamaSettingsManager } from "./settings";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Identifiers of the editable fields shown in `/models settings`.
|
|
25
|
+
* Values match the scalar `LlamaSettings` keys.
|
|
26
|
+
*/
|
|
27
|
+
export enum Options {
|
|
28
|
+
REACT_TO_MODEL_SELECT = "reactToModelSelect",
|
|
29
|
+
AUTOLOAD_ON_MESSAGE = "autoloadOnMessage",
|
|
30
|
+
SORT_BY = "sortBy",
|
|
31
|
+
POLLING_TIMEOUT = "pollingTimeout",
|
|
32
|
+
SERVER_TIMEOUT = "serverTimeout",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
type SortByValue = NonNullable<LlamaSettings["sortBy"]>;
|
|
36
|
+
|
|
37
|
+
const SORT_VALUES: SortByValue[] = [
|
|
38
|
+
"asc",
|
|
39
|
+
"desc",
|
|
40
|
+
"asc-name",
|
|
41
|
+
"desc-name",
|
|
42
|
+
"api",
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
/** Presets (ms) for `pollingTimeout` */
|
|
46
|
+
const POLLING_PRESETS = [15000, 30000, 60000, 120000, 300000];
|
|
47
|
+
|
|
48
|
+
/** Presets (ms) for `serverTimeout` */
|
|
49
|
+
const SERVER_PRESETS = [500, 1000, 2000, 5000, 10000];
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* `/models` subcommand completions. Module-level so
|
|
53
|
+
* {@link CommandManager.getArgumentCompletions} doesn't rebuild the
|
|
54
|
+
* table on every keystroke.
|
|
55
|
+
*/
|
|
56
|
+
const ARGUMENT_COMPLETIONS: AutocompleteItem[] = [
|
|
57
|
+
{
|
|
58
|
+
value: "info",
|
|
59
|
+
label: "info",
|
|
60
|
+
description: "Show information of all models",
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
value: "unload",
|
|
64
|
+
label: "unload",
|
|
65
|
+
description: "Unload all models",
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
value: "servers",
|
|
69
|
+
label: "servers",
|
|
70
|
+
description: "Manage llama.cpp server URLs",
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
value: "settings",
|
|
74
|
+
label: "settings",
|
|
75
|
+
description: "Configure llamaSettings",
|
|
76
|
+
},
|
|
77
|
+
];
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Formats milliseconds compactly for display (e.g. `500 -> "500ms"`,
|
|
81
|
+
* `60000 -> "60s"`).
|
|
82
|
+
*/
|
|
83
|
+
export const formatMs = (ms: number): string =>
|
|
84
|
+
ms % 1000 === 0 ? `${ms / 1000}s` : `${ms}ms`;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Parses a value produced by `formatMs()` back to milliseconds.
|
|
88
|
+
* Only ever called with values from the preset lists.
|
|
89
|
+
*/
|
|
90
|
+
const parseMs = (value: string): number =>
|
|
91
|
+
value.endsWith("ms")
|
|
92
|
+
? Number(value.slice(0, -2))
|
|
93
|
+
: Number(value.slice(0, -1)) * 1000;
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Builds the `SettingsList` items for `/models settings` from the current
|
|
97
|
+
* (merged) values of the scalar `llamaSettings` fields.
|
|
98
|
+
*/
|
|
99
|
+
export const buildSettingsItems = (
|
|
100
|
+
settings: LlamaSettingsManager,
|
|
101
|
+
): SettingItem[] => {
|
|
102
|
+
const { pollingTimeout, serverTimeout } = settings.resolveTimeouts();
|
|
103
|
+
|
|
104
|
+
return [
|
|
105
|
+
{
|
|
106
|
+
id: Options.REACT_TO_MODEL_SELECT,
|
|
107
|
+
label: "React to model selection",
|
|
108
|
+
description: "Load the model when you pick it in Pi (immediate)",
|
|
109
|
+
currentValue: settings.resolveReactToModelSelect() ? "on" : "off",
|
|
110
|
+
values: ["on", "off"],
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
id: Options.AUTOLOAD_ON_MESSAGE,
|
|
114
|
+
label: "Autoload on message",
|
|
115
|
+
description:
|
|
116
|
+
"Auto-load the selected model when you send a message (immediate)",
|
|
117
|
+
currentValue: settings.resolveAutoloadOnMessage() ? "on" : "off",
|
|
118
|
+
values: ["on", "off"],
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
id: Options.SORT_BY,
|
|
122
|
+
label: "Sort models by",
|
|
123
|
+
description: "Order of models in /models (next open)",
|
|
124
|
+
currentValue: settings.resolveSortBy(),
|
|
125
|
+
values: [...SORT_VALUES],
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
id: Options.POLLING_TIMEOUT,
|
|
129
|
+
label: "Polling timeout",
|
|
130
|
+
description: "Max model-load wait (next model load)",
|
|
131
|
+
currentValue: formatMs(pollingTimeout),
|
|
132
|
+
values: POLLING_PRESETS.map(formatMs),
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
id: Options.SERVER_TIMEOUT,
|
|
136
|
+
label: "Server timeout",
|
|
137
|
+
description: "Health check / SSE probe timeout (next model load)",
|
|
138
|
+
currentValue: formatMs(serverTimeout),
|
|
139
|
+
values: SERVER_PRESETS.map(formatMs),
|
|
140
|
+
},
|
|
141
|
+
];
|
|
142
|
+
};
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Persists a change made in the settings menu.
|
|
146
|
+
* Maps the `SettingsList` id/value pair to the matching `llamaSettings`
|
|
147
|
+
* key and writes it via `LlamaSettingsManager.setLlamaSetting()`.
|
|
148
|
+
*/
|
|
149
|
+
export const applySettingChange = async (
|
|
150
|
+
id: string,
|
|
151
|
+
newValue: string,
|
|
152
|
+
settings: LlamaSettingsManager,
|
|
153
|
+
): Promise<void> => {
|
|
154
|
+
switch (id) {
|
|
155
|
+
case Options.REACT_TO_MODEL_SELECT:
|
|
156
|
+
await settings.setLlamaSetting("reactToModelSelect", newValue === "on");
|
|
157
|
+
return;
|
|
158
|
+
case Options.AUTOLOAD_ON_MESSAGE:
|
|
159
|
+
await settings.setLlamaSetting("autoloadOnMessage", newValue === "on");
|
|
160
|
+
return;
|
|
161
|
+
case Options.SORT_BY:
|
|
162
|
+
await settings.setLlamaSetting("sortBy", newValue as SortByValue);
|
|
163
|
+
return;
|
|
164
|
+
case Options.POLLING_TIMEOUT:
|
|
165
|
+
await settings.setLlamaSetting("pollingTimeout", parseMs(newValue));
|
|
166
|
+
return;
|
|
167
|
+
case Options.SERVER_TIMEOUT:
|
|
168
|
+
await settings.setLlamaSetting("serverTimeout", parseMs(newValue));
|
|
169
|
+
return;
|
|
170
|
+
}
|
|
171
|
+
};
|
|
13
172
|
|
|
14
173
|
export class CommandManager {
|
|
15
|
-
constructor(
|
|
174
|
+
constructor(
|
|
175
|
+
private readonly serverManager: ServerManager,
|
|
176
|
+
private readonly settings: LlamaSettingsManager,
|
|
177
|
+
) {}
|
|
16
178
|
|
|
17
179
|
/**
|
|
18
180
|
* Sets up the argument completions for the `/models` command
|
|
@@ -21,19 +183,9 @@ export class CommandManager {
|
|
|
21
183
|
* @returns Completions with that prefix
|
|
22
184
|
*/
|
|
23
185
|
getArgumentCompletions(prefix: string): AutocompleteItem[] | null {
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
label: "info",
|
|
28
|
-
description: "Show information of all models",
|
|
29
|
-
},
|
|
30
|
-
{
|
|
31
|
-
value: "unload",
|
|
32
|
-
label: "unload",
|
|
33
|
-
description: "Unload all models",
|
|
34
|
-
},
|
|
35
|
-
];
|
|
36
|
-
const filtered = available.filter((a) => a.value.startsWith(prefix));
|
|
186
|
+
const filtered = ARGUMENT_COMPLETIONS.filter((a) =>
|
|
187
|
+
a.value.startsWith(prefix),
|
|
188
|
+
);
|
|
37
189
|
return filtered.length > 0 ? filtered : null;
|
|
38
190
|
}
|
|
39
191
|
|
|
@@ -49,6 +201,19 @@ export class CommandManager {
|
|
|
49
201
|
ctx: ExtensionCommandContext,
|
|
50
202
|
pi: ExtensionAPI,
|
|
51
203
|
) {
|
|
204
|
+
// Settings menu: no network round-trip needed, handle before any
|
|
205
|
+
// server updates / unreachable-server notifications
|
|
206
|
+
if (args === "settings") {
|
|
207
|
+
await this.runSettingsMenu(ctx);
|
|
208
|
+
return;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// Servers editor: same — edits are passive until the next provider scan
|
|
212
|
+
if (args === "servers") {
|
|
213
|
+
await this.runServersEditor(ctx);
|
|
214
|
+
return;
|
|
215
|
+
}
|
|
216
|
+
|
|
52
217
|
// Re-register providers so Pi sees updated model states
|
|
53
218
|
await this.serverManager.update(pi);
|
|
54
219
|
|
|
@@ -77,6 +242,80 @@ export class CommandManager {
|
|
|
77
242
|
await this.runModelsMenu(ctx, pi);
|
|
78
243
|
}
|
|
79
244
|
|
|
245
|
+
/**
|
|
246
|
+
* Runs the interactive settings menu for the scalar `llamaSettings`
|
|
247
|
+
* fields. Enter/Space cycles the value under the cursor; Esc closes.
|
|
248
|
+
*
|
|
249
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
250
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified
|
|
251
|
+
* and leave the dialog open with values unchanged.
|
|
252
|
+
*/
|
|
253
|
+
private async runSettingsMenu(ctx: ExtensionCommandContext): Promise<void> {
|
|
254
|
+
if (ctx.mode !== "tui") {
|
|
255
|
+
ctx.ui.notify(
|
|
256
|
+
"/models settings requires an interactive session (TUI)",
|
|
257
|
+
"warning",
|
|
258
|
+
);
|
|
259
|
+
return;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const items = buildSettingsItems(this.settings);
|
|
263
|
+
|
|
264
|
+
await ctx.ui.custom<void>(
|
|
265
|
+
(_tui, _theme, _kb, done) =>
|
|
266
|
+
new SettingsList(
|
|
267
|
+
items,
|
|
268
|
+
Math.min(items.length + 2, 15),
|
|
269
|
+
getSettingsListTheme(),
|
|
270
|
+
(id, newValue) => {
|
|
271
|
+
applySettingChange(id, newValue, this.settings).catch(
|
|
272
|
+
(err: unknown) => {
|
|
273
|
+
const message = errorMessage(err);
|
|
274
|
+
ctx.ui.notify(message, "error");
|
|
275
|
+
},
|
|
276
|
+
);
|
|
277
|
+
},
|
|
278
|
+
() => done(undefined),
|
|
279
|
+
),
|
|
280
|
+
);
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Runs the interactive servers editor for `llamaSettings.servers`.
|
|
285
|
+
* Enter/e edits the selected URL, i its id, n its name, a adds,
|
|
286
|
+
* d deletes (after a confirmation prompt); Esc closes.
|
|
287
|
+
*
|
|
288
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
289
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified and
|
|
290
|
+
* the editor stays open with the pre-mutation list. List changes (add,
|
|
291
|
+
* remove, URL/`id`/`name` edits) apply the next time providers are
|
|
292
|
+
* scanned — run `/models` to see them.
|
|
293
|
+
*/
|
|
294
|
+
private async runServersEditor(ctx: ExtensionCommandContext): Promise<void> {
|
|
295
|
+
if (ctx.mode !== "tui") {
|
|
296
|
+
ctx.ui.notify(
|
|
297
|
+
"/models servers requires an interactive session (TUI)",
|
|
298
|
+
"warning",
|
|
299
|
+
);
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
const servers = this.settings.llamaServers;
|
|
304
|
+
|
|
305
|
+
await ctx.ui.custom<void>(
|
|
306
|
+
(tui, theme, keybindings, done) =>
|
|
307
|
+
new ServerListEditor({
|
|
308
|
+
tui,
|
|
309
|
+
theme,
|
|
310
|
+
keybindings,
|
|
311
|
+
servers,
|
|
312
|
+
persist: (next) => this.settings.setLlamaSetting("servers", next),
|
|
313
|
+
done: () => done(undefined),
|
|
314
|
+
onError: (message) => ctx.ui.notify(message, "error"),
|
|
315
|
+
}),
|
|
316
|
+
);
|
|
317
|
+
}
|
|
318
|
+
|
|
80
319
|
/**
|
|
81
320
|
* Notifies the user that a server is unreachable.
|
|
82
321
|
*/
|
|
@@ -132,18 +371,24 @@ export class CommandManager {
|
|
|
132
371
|
const loadActions = [Action.LOAD, Action.LOAD_AND_SWITCH, Action.RETRY];
|
|
133
372
|
if (loadActions.includes(action)) {
|
|
134
373
|
ctx.ui.notify(`Loading ${model.name}...`, "info");
|
|
374
|
+
// Mark the load as in-flight so session_before_switch can warn about
|
|
375
|
+
// it (see EventManager.inflightModel for the coupling rationale)
|
|
135
376
|
EventManager.inflightModel = model;
|
|
136
377
|
|
|
137
|
-
// Subscribe to progress events
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
"
|
|
145
|
-
|
|
146
|
-
|
|
378
|
+
// Subscribe to progress events; skip when the server is gone
|
|
379
|
+
// (removed/edited away mid-load → getServer returns undefined)
|
|
380
|
+
const server = this.serverManager.getServer(model);
|
|
381
|
+
const cleanupProgress =
|
|
382
|
+
server?.sseManager.subscribeToProgress(
|
|
383
|
+
model.id,
|
|
384
|
+
(percentage, stage) => {
|
|
385
|
+
const stageText = stage ? ` (${stage})` : "";
|
|
386
|
+
ctx.ui.notify(
|
|
387
|
+
`Loading ${model.name}... [${percentage}%${stageText}]`,
|
|
388
|
+
"info",
|
|
389
|
+
);
|
|
390
|
+
},
|
|
391
|
+
) ?? (() => {});
|
|
147
392
|
|
|
148
393
|
const onSuccess = async () => {
|
|
149
394
|
const { serverId } = model;
|
|
@@ -168,7 +413,7 @@ export class CommandManager {
|
|
|
168
413
|
};
|
|
169
414
|
|
|
170
415
|
const onFailure = (err: any) => {
|
|
171
|
-
const message =
|
|
416
|
+
const message = errorMessage(err);
|
|
172
417
|
|
|
173
418
|
try {
|
|
174
419
|
ctx.ui.notify(message, "error");
|
|
@@ -268,25 +513,22 @@ export class CommandManager {
|
|
|
268
513
|
* @returns A mapping of actions for each status
|
|
269
514
|
*/
|
|
270
515
|
private async getActionsForModel(model: BaseModel): Promise<Array<Action>> {
|
|
271
|
-
const
|
|
516
|
+
const base = [Action.INFO, Action.CANCEL];
|
|
517
|
+
|
|
518
|
+
const actions: Record<Status, Array<Action>> = {
|
|
272
519
|
[Status.LOADED]:
|
|
273
520
|
model.mode === Mode.ROUTER
|
|
274
|
-
? [Action.SWITCH, Action.UNLOAD,
|
|
275
|
-
: [Action.SWITCH,
|
|
276
|
-
[Status.LOADING]: [
|
|
277
|
-
[Status.FAILED]: [Action.RETRY,
|
|
278
|
-
[Status.SLEEPING]: [
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
Action.INFO,
|
|
282
|
-
Action.CANCEL,
|
|
283
|
-
],
|
|
284
|
-
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, Action.CANCEL],
|
|
285
|
-
[Status.UNAUTHORIZED]: [Action.INFO, Action.CANCEL],
|
|
521
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
522
|
+
: [Action.SWITCH, ...base],
|
|
523
|
+
[Status.LOADING]: [...base],
|
|
524
|
+
[Status.FAILED]: [Action.RETRY, ...base],
|
|
525
|
+
[Status.SLEEPING]: [Action.SWITCH, Action.UNLOAD, ...base],
|
|
526
|
+
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, ...base],
|
|
527
|
+
[Status.UNAUTHORIZED]: [...base],
|
|
286
528
|
};
|
|
287
529
|
|
|
288
530
|
const status = await model.getStatus();
|
|
289
|
-
return
|
|
531
|
+
return actions[status];
|
|
290
532
|
}
|
|
291
533
|
|
|
292
534
|
/**
|
package/src/managers/events.ts
CHANGED
|
@@ -3,18 +3,37 @@ import {
|
|
|
3
3
|
type ExtensionContext,
|
|
4
4
|
} from "@earendil-works/pi-coding-agent";
|
|
5
5
|
import { READABLE_TIMEOUT } from "../constants";
|
|
6
|
+
import { Status } from "../enums/status";
|
|
6
7
|
import { ModelSelectEvent } from "../interfaces/events";
|
|
8
|
+
import type { LlamaSettingsManager } from "../managers/settings";
|
|
7
9
|
import { BaseModel } from "../models/baseModel";
|
|
8
|
-
import {
|
|
9
|
-
import { Server } from "../server";
|
|
10
|
+
import { ServerManager } from "./server";
|
|
10
11
|
|
|
11
12
|
export class EventManager {
|
|
13
|
+
/**
|
|
14
|
+
* Model with a load currently in flight. Deliberately a class static
|
|
15
|
+
* (REFACTOR.md §3.3): the load is started by CommandManager
|
|
16
|
+
* (fire-and-forget from the /models editor) while the "session switched
|
|
17
|
+
* mid-load" warning must be emitted here, from the session_before_switch
|
|
18
|
+
* hook — a shared static is the least-plumbing bridge between the two
|
|
19
|
+
* event-owning managers.
|
|
20
|
+
*
|
|
21
|
+
* Known limits, accepted: (1) single slot — a second overlapping load
|
|
22
|
+
* overwrites the first, and the first load's onFinished reset can clear
|
|
23
|
+
* the flag while the second is still loading, so the session-switch
|
|
24
|
+
* warning may be missed; (2) loads initiated from this class
|
|
25
|
+
* (onModelSelect / autoLoadIfNeeded) never set the flag.
|
|
26
|
+
*/
|
|
12
27
|
static inflightModel: BaseModel | null = null;
|
|
13
28
|
|
|
14
|
-
constructor(
|
|
29
|
+
constructor(
|
|
30
|
+
private readonly serverManager: ServerManager,
|
|
31
|
+
private readonly settings: LlamaSettingsManager,
|
|
32
|
+
) {}
|
|
15
33
|
|
|
16
34
|
/**
|
|
17
|
-
* Resets the in-flight model reference.
|
|
35
|
+
* Resets the in-flight model reference. Called by CommandManager when
|
|
36
|
+
* its load settles (see `inflightModel` for why this lives on a static).
|
|
18
37
|
*/
|
|
19
38
|
static resetInflightModel() {
|
|
20
39
|
EventManager.inflightModel = null;
|
|
@@ -27,7 +46,10 @@ export class EventManager {
|
|
|
27
46
|
* @param ctx Pi context
|
|
28
47
|
*/
|
|
29
48
|
async onModelSelect(event: ModelSelectEvent, ctx: ExtensionContext) {
|
|
30
|
-
|
|
49
|
+
// Check if the model_select event should be used
|
|
50
|
+
if (!this.settings.resolveReactToModelSelect()) return;
|
|
51
|
+
|
|
52
|
+
for (const { providerId, models } of this.serverManager.servers) {
|
|
31
53
|
if (event.model.provider !== providerId) continue;
|
|
32
54
|
|
|
33
55
|
const model = models.find((m) => m.id === event.model.id);
|
|
@@ -44,6 +66,20 @@ export class EventManager {
|
|
|
44
66
|
}
|
|
45
67
|
}
|
|
46
68
|
|
|
69
|
+
/**
|
|
70
|
+
* Loads the model if auto-loading is enabled and the model is unloaded.
|
|
71
|
+
*
|
|
72
|
+
* @param model The model to potentially auto-load
|
|
73
|
+
*/
|
|
74
|
+
private async autoLoadIfNeeded(model: BaseModel): Promise<void> {
|
|
75
|
+
if (!this.settings.resolveAutoloadOnMessage()) return;
|
|
76
|
+
|
|
77
|
+
const status = await model.getStatus();
|
|
78
|
+
if (status !== Status.UNLOADED) return;
|
|
79
|
+
|
|
80
|
+
await model.load();
|
|
81
|
+
}
|
|
82
|
+
|
|
47
83
|
/**
|
|
48
84
|
* Session-switch handler. Registered once at extension init.
|
|
49
85
|
* Only notifies if a model load is actually in-flight.
|
|
@@ -81,17 +117,19 @@ export class EventManager {
|
|
|
81
117
|
if (!model) return payload;
|
|
82
118
|
|
|
83
119
|
// Check if this model belongs to one of our servers
|
|
84
|
-
const
|
|
85
|
-
|
|
86
|
-
|
|
120
|
+
const serverModel = this.serverManager.servers
|
|
121
|
+
.flatMap((s) => s.models)
|
|
122
|
+
.find((m) => m.id === model);
|
|
123
|
+
|
|
124
|
+
if (!serverModel) return payload;
|
|
87
125
|
|
|
88
|
-
if
|
|
126
|
+
// Auto-load if enabled and model is unloaded
|
|
127
|
+
await this.autoLoadIfNeeded(serverModel);
|
|
89
128
|
|
|
90
129
|
// Retrieve pi's current thinking level, so we can setup a budget
|
|
91
|
-
const resolver = new ConfigResolver();
|
|
92
130
|
const level =
|
|
93
|
-
ctx.thinkingLevel ??
|
|
94
|
-
const budgets =
|
|
131
|
+
ctx.thinkingLevel ?? this.settings.resolveThinkingLevel() ?? "medium";
|
|
132
|
+
const budgets = this.settings.resolveThinkingBudgets();
|
|
95
133
|
const thinking_budget_tokens = budgets[level];
|
|
96
134
|
|
|
97
135
|
// Setup payload
|
package/src/managers/server.ts
CHANGED
|
@@ -1,14 +1,27 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { API_TYPE, PROVIDER_NAME,
|
|
2
|
+
import { API_TYPE, PROVIDER_NAME, type SortBy } from "../constants";
|
|
3
3
|
import { ServerStatus } from "../enums/serverStatus";
|
|
4
4
|
import { BaseModel } from "../models/baseModel";
|
|
5
5
|
import { Server } from "../server";
|
|
6
|
+
import type { LlamaSettingsManager } from "./settings";
|
|
7
|
+
|
|
8
|
+
/** Model-list comparator: negative if a sorts first, positive if b does. */
|
|
9
|
+
type ModelComparator = (a: BaseModel, b: BaseModel) => number;
|
|
6
10
|
|
|
7
11
|
export class ServerManager {
|
|
12
|
+
constructor(private readonly settings: LlamaSettingsManager) {}
|
|
8
13
|
readonly failedUrls: string[] = [];
|
|
9
14
|
private readonly warnings: string[] = [];
|
|
15
|
+
private readonly serverList: Server[] = [];
|
|
10
16
|
|
|
11
|
-
|
|
17
|
+
/**
|
|
18
|
+
* Live view of the server list. `update()` re-derives the list from
|
|
19
|
+
* settings on every scan (in place), so `/models servers` edits apply
|
|
20
|
+
* without a restart.
|
|
21
|
+
*/
|
|
22
|
+
get servers(): readonly Server[] {
|
|
23
|
+
return this.serverList;
|
|
24
|
+
}
|
|
12
25
|
|
|
13
26
|
/**
|
|
14
27
|
* Verifies reachability of servers and registers the providers
|
|
@@ -16,8 +29,9 @@ export class ServerManager {
|
|
|
16
29
|
* @param pi The Pi extension API
|
|
17
30
|
*/
|
|
18
31
|
async initialize(pi: ExtensionAPI) {
|
|
19
|
-
// Register the providers with
|
|
20
|
-
|
|
32
|
+
// Register the providers with the configured server timeout
|
|
33
|
+
const { serverTimeout } = this.settings.resolveTimeouts();
|
|
34
|
+
await this.update(pi, serverTimeout);
|
|
21
35
|
}
|
|
22
36
|
|
|
23
37
|
/**
|
|
@@ -30,6 +44,33 @@ export class ServerManager {
|
|
|
30
44
|
async update(pi: ExtensionAPI, timeout?: number) {
|
|
31
45
|
this.failedUrls.length = 0;
|
|
32
46
|
|
|
47
|
+
// Surface warnings from strict URL parsing (dropped invalid entries)
|
|
48
|
+
this.warnings.push(...this.settings.takeWarnings());
|
|
49
|
+
|
|
50
|
+
// Re-derive the server list from settings so `/models servers` edits
|
|
51
|
+
// (add / remove / URL / id / name) apply on the next scan
|
|
52
|
+
const fresh: Server[] = [];
|
|
53
|
+
const seen = new Set<string>(); // dedupe repeated URLs (same providerId)
|
|
54
|
+
for (const server of this.settings.resolveServers()) {
|
|
55
|
+
if (seen.has(server.providerId)) continue;
|
|
56
|
+
seen.add(server.providerId);
|
|
57
|
+
fresh.push(server);
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Unregister providers that disappeared (removed or edited away);
|
|
61
|
+
// no-op for providers that were never registered
|
|
62
|
+
for (const old of this.servers) {
|
|
63
|
+
if (fresh.some((f) => f.providerId === old.providerId)) continue;
|
|
64
|
+
pi.unregisterProvider(old.providerId);
|
|
65
|
+
// Optional chain is intentional despite the non-optional type: `sse`
|
|
66
|
+
// is undefined until initialize() runs (async-constructor hack — see Server)
|
|
67
|
+
old.sseManager?.disconnect();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Replace in place so the live `servers` view stays valid (D1)
|
|
71
|
+
this.serverList.length = 0;
|
|
72
|
+
this.serverList.push(...fresh);
|
|
73
|
+
|
|
33
74
|
const registrableServers = timeout
|
|
34
75
|
? await this.findRegistrableServers(timeout)
|
|
35
76
|
: this.servers;
|
|
@@ -67,7 +108,7 @@ export class ServerManager {
|
|
|
67
108
|
} else if (status === ServerStatus.TIMEOUT) {
|
|
68
109
|
const message = [
|
|
69
110
|
"[pi-llama-cpp]",
|
|
70
|
-
`${PROVIDER_NAME} server initialization for '${server.baseUrl}' took more than ${
|
|
111
|
+
`${PROVIDER_NAME} server initialization for '${server.baseUrl}' took more than ${timeout} ms, so it has been skipped.`,
|
|
71
112
|
"Run `/models` to retry without timeout and see all models.",
|
|
72
113
|
].join("\n");
|
|
73
114
|
this.warnings.push(message);
|
|
@@ -93,7 +134,7 @@ export class ServerManager {
|
|
|
93
134
|
*/
|
|
94
135
|
private async registerProvider(server: Server, pi: ExtensionAPI) {
|
|
95
136
|
const { baseUrl, models, providerId, providerName } = server;
|
|
96
|
-
const apiKey =
|
|
137
|
+
const apiKey = server.getApiKey();
|
|
97
138
|
const modelConfigs = await Promise.all(
|
|
98
139
|
models.map((m) => m.toProviderConfig()),
|
|
99
140
|
);
|
|
@@ -121,26 +162,58 @@ export class ServerManager {
|
|
|
121
162
|
* Returns the server for a given model.
|
|
122
163
|
*
|
|
123
164
|
* @param model - The model to find the server for
|
|
124
|
-
* @returns The server containing the model
|
|
165
|
+
* @returns The server containing the model, or `undefined` when no
|
|
166
|
+
* current server matches (e.g. removed while a model was loading)
|
|
125
167
|
*/
|
|
126
|
-
getServer(model: BaseModel): Server {
|
|
127
|
-
return this.servers.find((s) => s.baseUrl === model.serverUrl)
|
|
168
|
+
getServer(model: BaseModel): Server | undefined {
|
|
169
|
+
return this.servers.find((s) => s.baseUrl === model.serverUrl);
|
|
128
170
|
}
|
|
129
171
|
|
|
130
172
|
/**
|
|
131
|
-
* Returns all models from all servers.
|
|
173
|
+
* Returns all models from all servers, sorted by the configured sort mode.
|
|
132
174
|
*
|
|
133
175
|
* @returns Flat array of all models across all servers
|
|
134
176
|
*/
|
|
135
177
|
getAllModels(): BaseModel[] {
|
|
136
|
-
const
|
|
178
|
+
const sortBy = this.settings.resolveSortBy();
|
|
179
|
+
const allModels = this.servers.flatMap((s) => s.models);
|
|
137
180
|
|
|
138
|
-
|
|
139
|
-
for (const model of models) {
|
|
140
|
-
response.push(model);
|
|
141
|
-
}
|
|
142
|
-
}
|
|
181
|
+
if (sortBy === "api") return allModels;
|
|
143
182
|
|
|
144
|
-
return
|
|
183
|
+
return allModels.sort(ServerManager.SORTERS[sortBy]);
|
|
145
184
|
}
|
|
185
|
+
|
|
186
|
+
private static sortByIdAsc(a: BaseModel, b: BaseModel): number {
|
|
187
|
+
return a.id.localeCompare(b.id);
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
private static sortByIdDesc(a: BaseModel, b: BaseModel): number {
|
|
191
|
+
return b.id.localeCompare(a.id);
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Name ascending, with ID as tiebreaker. */
|
|
195
|
+
private static sortByNameAsc(a: BaseModel, b: BaseModel): number {
|
|
196
|
+
const cmp = a.name.localeCompare(b.name);
|
|
197
|
+
return cmp !== 0 ? cmp : a.id.localeCompare(b.id);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** Name descending, with ID as tiebreaker. */
|
|
201
|
+
private static sortByNameDesc(a: BaseModel, b: BaseModel): number {
|
|
202
|
+
const cmp = b.name.localeCompare(a.name);
|
|
203
|
+
return cmp !== 0 ? cmp : a.id.localeCompare(b.id);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Comparators for each sort mode except "api", which preserves server
|
|
208
|
+
* order (short-circuited in {@link ServerManager.getAllModels}).
|
|
209
|
+
*/
|
|
210
|
+
private static readonly SORTERS: Record<
|
|
211
|
+
Exclude<SortBy, "api">,
|
|
212
|
+
ModelComparator
|
|
213
|
+
> = {
|
|
214
|
+
asc: ServerManager.sortByIdAsc,
|
|
215
|
+
desc: ServerManager.sortByIdDesc,
|
|
216
|
+
"asc-name": ServerManager.sortByNameAsc,
|
|
217
|
+
"desc-name": ServerManager.sortByNameDesc,
|
|
218
|
+
};
|
|
146
219
|
}
|