pi-llama-cpp 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -5
- package/package.json +3 -3
- package/src/api/client.ts +76 -32
- package/src/constants.ts +15 -0
- package/src/index.ts +3 -5
- package/src/interfaces/events.ts +14 -3
- package/src/interfaces/server.ts +20 -0
- package/src/interfaces/settings.ts +11 -2
- package/src/managers/command.ts +285 -43
- package/src/managers/events.ts +28 -10
- package/src/managers/server.ts +87 -16
- package/src/managers/settings.ts +78 -16
- package/src/models/baseModel.ts +0 -8
- package/src/server.ts +69 -28
- package/src/sse/client.ts +28 -16
- package/src/sse/manager.ts +22 -10
- package/src/ui/serverListEditor.ts +481 -0
- package/src/utils/errors.ts +5 -0
- package/src/utils/settingsStore.ts +60 -0
- package/src/utils/urls.ts +16 -0
- package/tests/commandManager.test.ts +256 -11
- package/tests/events.test.ts +117 -88
- package/tests/mocks.ts +145 -32
- package/tests/server.test.ts +37 -35
- package/tests/serverListEditor.test.ts +637 -0
- package/tests/serverManager.test.ts +282 -39
- package/tests/settings.test.ts +173 -16
- package/tests/settingsStore.test.ts +209 -0
- package/tests/sseManager.test.ts +88 -11
- package/src/interfaces/auth.ts +0 -6
- package/src/utils/cache.ts +0 -39
- package/src/utils/mutex.ts +0 -24
package/src/managers/command.ts
CHANGED
|
@@ -1,18 +1,180 @@
|
|
|
1
|
-
import
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
import {
|
|
2
|
+
getSettingsListTheme,
|
|
3
|
+
type ExtensionAPI,
|
|
4
|
+
type ExtensionCommandContext,
|
|
4
5
|
} from "@earendil-works/pi-coding-agent";
|
|
5
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
AutocompleteItem,
|
|
8
|
+
SettingsList,
|
|
9
|
+
type SettingItem,
|
|
10
|
+
} from "@earendil-works/pi-tui";
|
|
6
11
|
import { PROVIDER_NAME } from "../constants";
|
|
7
12
|
import { Action } from "../enums/action";
|
|
8
13
|
import { Mode } from "../enums/mode";
|
|
9
14
|
import { Status } from "../enums/status";
|
|
15
|
+
import { LlamaSettings } from "../interfaces/settings";
|
|
10
16
|
import { BaseModel } from "../models/baseModel";
|
|
17
|
+
import { ServerListEditor } from "../ui/serverListEditor";
|
|
18
|
+
import { errorMessage } from "../utils/errors";
|
|
11
19
|
import { EventManager } from "./events";
|
|
12
20
|
import { ServerManager } from "./server";
|
|
21
|
+
import type { LlamaSettingsManager } from "./settings";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Identifiers of the editable fields shown in `/models settings`.
|
|
25
|
+
* Values match the scalar `LlamaSettings` keys.
|
|
26
|
+
*/
|
|
27
|
+
export enum Options {
|
|
28
|
+
REACT_TO_MODEL_SELECT = "reactToModelSelect",
|
|
29
|
+
AUTOLOAD_ON_MESSAGE = "autoloadOnMessage",
|
|
30
|
+
SORT_BY = "sortBy",
|
|
31
|
+
POLLING_TIMEOUT = "pollingTimeout",
|
|
32
|
+
SERVER_TIMEOUT = "serverTimeout",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
type SortByValue = NonNullable<LlamaSettings["sortBy"]>;
|
|
36
|
+
|
|
37
|
+
const SORT_VALUES: SortByValue[] = [
|
|
38
|
+
"asc",
|
|
39
|
+
"desc",
|
|
40
|
+
"asc-name",
|
|
41
|
+
"desc-name",
|
|
42
|
+
"api",
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
/** Presets (ms) for `pollingTimeout` */
|
|
46
|
+
const POLLING_PRESETS = [15000, 30000, 60000, 120000, 300000];
|
|
47
|
+
|
|
48
|
+
/** Presets (ms) for `serverTimeout` */
|
|
49
|
+
const SERVER_PRESETS = [500, 1000, 2000, 5000, 10000];
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* `/models` subcommand completions. Module-level so
|
|
53
|
+
* {@link CommandManager.getArgumentCompletions} doesn't rebuild the
|
|
54
|
+
* table on every keystroke.
|
|
55
|
+
*/
|
|
56
|
+
const ARGUMENT_COMPLETIONS: AutocompleteItem[] = [
|
|
57
|
+
{
|
|
58
|
+
value: "info",
|
|
59
|
+
label: "info",
|
|
60
|
+
description: "Show information of all models",
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
value: "unload",
|
|
64
|
+
label: "unload",
|
|
65
|
+
description: "Unload all models",
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
value: "servers",
|
|
69
|
+
label: "servers",
|
|
70
|
+
description: "Manage llama.cpp server URLs",
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
value: "settings",
|
|
74
|
+
label: "settings",
|
|
75
|
+
description: "Configure llamaSettings",
|
|
76
|
+
},
|
|
77
|
+
];
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Formats milliseconds compactly for display (e.g. `500 -> "500ms"`,
|
|
81
|
+
* `60000 -> "60s"`).
|
|
82
|
+
*/
|
|
83
|
+
export const formatMs = (ms: number): string =>
|
|
84
|
+
ms % 1000 === 0 ? `${ms / 1000}s` : `${ms}ms`;
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Parses a value produced by `formatMs()` back to milliseconds.
|
|
88
|
+
* Only ever called with values from the preset lists.
|
|
89
|
+
*/
|
|
90
|
+
const parseMs = (value: string): number =>
|
|
91
|
+
value.endsWith("ms")
|
|
92
|
+
? Number(value.slice(0, -2))
|
|
93
|
+
: Number(value.slice(0, -1)) * 1000;
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Builds the `SettingsList` items for `/models settings` from the current
|
|
97
|
+
* (merged) values of the scalar `llamaSettings` fields.
|
|
98
|
+
*/
|
|
99
|
+
export const buildSettingsItems = (
|
|
100
|
+
settings: LlamaSettingsManager,
|
|
101
|
+
): SettingItem[] => {
|
|
102
|
+
const { pollingTimeout, serverTimeout } = settings.resolveTimeouts();
|
|
103
|
+
|
|
104
|
+
return [
|
|
105
|
+
{
|
|
106
|
+
id: Options.REACT_TO_MODEL_SELECT,
|
|
107
|
+
label: "React to model selection",
|
|
108
|
+
description: "Load the model when you pick it in Pi (immediate)",
|
|
109
|
+
currentValue: settings.resolveReactToModelSelect() ? "on" : "off",
|
|
110
|
+
values: ["on", "off"],
|
|
111
|
+
},
|
|
112
|
+
{
|
|
113
|
+
id: Options.AUTOLOAD_ON_MESSAGE,
|
|
114
|
+
label: "Autoload on message",
|
|
115
|
+
description:
|
|
116
|
+
"Auto-load the selected model when you send a message (immediate)",
|
|
117
|
+
currentValue: settings.resolveAutoloadOnMessage() ? "on" : "off",
|
|
118
|
+
values: ["on", "off"],
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
id: Options.SORT_BY,
|
|
122
|
+
label: "Sort models by",
|
|
123
|
+
description: "Order of models in /models (next open)",
|
|
124
|
+
currentValue: settings.resolveSortBy(),
|
|
125
|
+
values: [...SORT_VALUES],
|
|
126
|
+
},
|
|
127
|
+
{
|
|
128
|
+
id: Options.POLLING_TIMEOUT,
|
|
129
|
+
label: "Polling timeout",
|
|
130
|
+
description: "Max model-load wait (next model load)",
|
|
131
|
+
currentValue: formatMs(pollingTimeout),
|
|
132
|
+
values: POLLING_PRESETS.map(formatMs),
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
id: Options.SERVER_TIMEOUT,
|
|
136
|
+
label: "Server timeout",
|
|
137
|
+
description: "Health check / SSE probe timeout (next model load)",
|
|
138
|
+
currentValue: formatMs(serverTimeout),
|
|
139
|
+
values: SERVER_PRESETS.map(formatMs),
|
|
140
|
+
},
|
|
141
|
+
];
|
|
142
|
+
};
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Persists a change made in the settings menu.
|
|
146
|
+
* Maps the `SettingsList` id/value pair to the matching `llamaSettings`
|
|
147
|
+
* key and writes it via `LlamaSettingsManager.setLlamaSetting()`.
|
|
148
|
+
*/
|
|
149
|
+
export const applySettingChange = async (
|
|
150
|
+
id: string,
|
|
151
|
+
newValue: string,
|
|
152
|
+
settings: LlamaSettingsManager,
|
|
153
|
+
): Promise<void> => {
|
|
154
|
+
switch (id) {
|
|
155
|
+
case Options.REACT_TO_MODEL_SELECT:
|
|
156
|
+
await settings.setLlamaSetting("reactToModelSelect", newValue === "on");
|
|
157
|
+
return;
|
|
158
|
+
case Options.AUTOLOAD_ON_MESSAGE:
|
|
159
|
+
await settings.setLlamaSetting("autoloadOnMessage", newValue === "on");
|
|
160
|
+
return;
|
|
161
|
+
case Options.SORT_BY:
|
|
162
|
+
await settings.setLlamaSetting("sortBy", newValue as SortByValue);
|
|
163
|
+
return;
|
|
164
|
+
case Options.POLLING_TIMEOUT:
|
|
165
|
+
await settings.setLlamaSetting("pollingTimeout", parseMs(newValue));
|
|
166
|
+
return;
|
|
167
|
+
case Options.SERVER_TIMEOUT:
|
|
168
|
+
await settings.setLlamaSetting("serverTimeout", parseMs(newValue));
|
|
169
|
+
return;
|
|
170
|
+
}
|
|
171
|
+
};
|
|
13
172
|
|
|
14
173
|
export class CommandManager {
|
|
15
|
-
constructor(
|
|
174
|
+
constructor(
|
|
175
|
+
private readonly serverManager: ServerManager,
|
|
176
|
+
private readonly settings: LlamaSettingsManager,
|
|
177
|
+
) {}
|
|
16
178
|
|
|
17
179
|
/**
|
|
18
180
|
* Sets up the argument completions for the `/models` command
|
|
@@ -21,19 +183,9 @@ export class CommandManager {
|
|
|
21
183
|
* @returns Completions with that prefix
|
|
22
184
|
*/
|
|
23
185
|
getArgumentCompletions(prefix: string): AutocompleteItem[] | null {
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
label: "info",
|
|
28
|
-
description: "Show information of all models",
|
|
29
|
-
},
|
|
30
|
-
{
|
|
31
|
-
value: "unload",
|
|
32
|
-
label: "unload",
|
|
33
|
-
description: "Unload all models",
|
|
34
|
-
},
|
|
35
|
-
];
|
|
36
|
-
const filtered = available.filter((a) => a.value.startsWith(prefix));
|
|
186
|
+
const filtered = ARGUMENT_COMPLETIONS.filter((a) =>
|
|
187
|
+
a.value.startsWith(prefix),
|
|
188
|
+
);
|
|
37
189
|
return filtered.length > 0 ? filtered : null;
|
|
38
190
|
}
|
|
39
191
|
|
|
@@ -49,6 +201,19 @@ export class CommandManager {
|
|
|
49
201
|
ctx: ExtensionCommandContext,
|
|
50
202
|
pi: ExtensionAPI,
|
|
51
203
|
) {
|
|
204
|
+
// Settings menu: no network round-trip needed, handle before any
|
|
205
|
+
// server updates / unreachable-server notifications
|
|
206
|
+
if (args === "settings") {
|
|
207
|
+
await this.runSettingsMenu(ctx);
|
|
208
|
+
return;
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// Servers editor: same — edits are passive until the next provider scan
|
|
212
|
+
if (args === "servers") {
|
|
213
|
+
await this.runServersEditor(ctx);
|
|
214
|
+
return;
|
|
215
|
+
}
|
|
216
|
+
|
|
52
217
|
// Re-register providers so Pi sees updated model states
|
|
53
218
|
await this.serverManager.update(pi);
|
|
54
219
|
|
|
@@ -77,6 +242,80 @@ export class CommandManager {
|
|
|
77
242
|
await this.runModelsMenu(ctx, pi);
|
|
78
243
|
}
|
|
79
244
|
|
|
245
|
+
/**
|
|
246
|
+
* Runs the interactive settings menu for the scalar `llamaSettings`
|
|
247
|
+
* fields. Enter/Space cycles the value under the cursor; Esc closes.
|
|
248
|
+
*
|
|
249
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
250
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified
|
|
251
|
+
* and leave the dialog open with values unchanged.
|
|
252
|
+
*/
|
|
253
|
+
private async runSettingsMenu(ctx: ExtensionCommandContext): Promise<void> {
|
|
254
|
+
if (ctx.mode !== "tui") {
|
|
255
|
+
ctx.ui.notify(
|
|
256
|
+
"/models settings requires an interactive session (TUI)",
|
|
257
|
+
"warning",
|
|
258
|
+
);
|
|
259
|
+
return;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const items = buildSettingsItems(this.settings);
|
|
263
|
+
|
|
264
|
+
await ctx.ui.custom<void>(
|
|
265
|
+
(_tui, _theme, _kb, done) =>
|
|
266
|
+
new SettingsList(
|
|
267
|
+
items,
|
|
268
|
+
Math.min(items.length + 2, 15),
|
|
269
|
+
getSettingsListTheme(),
|
|
270
|
+
(id, newValue) => {
|
|
271
|
+
applySettingChange(id, newValue, this.settings).catch(
|
|
272
|
+
(err: unknown) => {
|
|
273
|
+
const message = errorMessage(err);
|
|
274
|
+
ctx.ui.notify(message, "error");
|
|
275
|
+
},
|
|
276
|
+
);
|
|
277
|
+
},
|
|
278
|
+
() => done(undefined),
|
|
279
|
+
),
|
|
280
|
+
);
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Runs the interactive servers editor for `llamaSettings.servers`.
|
|
285
|
+
* Enter/e edits the selected URL, i its id, n its name, a adds,
|
|
286
|
+
* d deletes (after a confirmation prompt); Esc closes.
|
|
287
|
+
*
|
|
288
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
289
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified and
|
|
290
|
+
* the editor stays open with the pre-mutation list. List changes (add,
|
|
291
|
+
* remove, URL/`id`/`name` edits) apply the next time providers are
|
|
292
|
+
* scanned — run `/models` to see them.
|
|
293
|
+
*/
|
|
294
|
+
private async runServersEditor(ctx: ExtensionCommandContext): Promise<void> {
|
|
295
|
+
if (ctx.mode !== "tui") {
|
|
296
|
+
ctx.ui.notify(
|
|
297
|
+
"/models servers requires an interactive session (TUI)",
|
|
298
|
+
"warning",
|
|
299
|
+
);
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
const servers = this.settings.llamaServers;
|
|
304
|
+
|
|
305
|
+
await ctx.ui.custom<void>(
|
|
306
|
+
(tui, theme, keybindings, done) =>
|
|
307
|
+
new ServerListEditor({
|
|
308
|
+
tui,
|
|
309
|
+
theme,
|
|
310
|
+
keybindings,
|
|
311
|
+
servers,
|
|
312
|
+
persist: (next) => this.settings.setLlamaSetting("servers", next),
|
|
313
|
+
done: () => done(undefined),
|
|
314
|
+
onError: (message) => ctx.ui.notify(message, "error"),
|
|
315
|
+
}),
|
|
316
|
+
);
|
|
317
|
+
}
|
|
318
|
+
|
|
80
319
|
/**
|
|
81
320
|
* Notifies the user that a server is unreachable.
|
|
82
321
|
*/
|
|
@@ -132,18 +371,24 @@ export class CommandManager {
|
|
|
132
371
|
const loadActions = [Action.LOAD, Action.LOAD_AND_SWITCH, Action.RETRY];
|
|
133
372
|
if (loadActions.includes(action)) {
|
|
134
373
|
ctx.ui.notify(`Loading ${model.name}...`, "info");
|
|
374
|
+
// Mark the load as in-flight so session_before_switch can warn about
|
|
375
|
+
// it (see EventManager.inflightModel for the coupling rationale)
|
|
135
376
|
EventManager.inflightModel = model;
|
|
136
377
|
|
|
137
|
-
// Subscribe to progress events
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
"
|
|
145
|
-
|
|
146
|
-
|
|
378
|
+
// Subscribe to progress events; skip when the server is gone
|
|
379
|
+
// (removed/edited away mid-load → getServer returns undefined)
|
|
380
|
+
const server = this.serverManager.getServer(model);
|
|
381
|
+
const cleanupProgress =
|
|
382
|
+
server?.sseManager.subscribeToProgress(
|
|
383
|
+
model.id,
|
|
384
|
+
(percentage, stage) => {
|
|
385
|
+
const stageText = stage ? ` (${stage})` : "";
|
|
386
|
+
ctx.ui.notify(
|
|
387
|
+
`Loading ${model.name}... [${percentage}%${stageText}]`,
|
|
388
|
+
"info",
|
|
389
|
+
);
|
|
390
|
+
},
|
|
391
|
+
) ?? (() => {});
|
|
147
392
|
|
|
148
393
|
const onSuccess = async () => {
|
|
149
394
|
const { serverId } = model;
|
|
@@ -168,7 +413,7 @@ export class CommandManager {
|
|
|
168
413
|
};
|
|
169
414
|
|
|
170
415
|
const onFailure = (err: any) => {
|
|
171
|
-
const message =
|
|
416
|
+
const message = errorMessage(err);
|
|
172
417
|
|
|
173
418
|
try {
|
|
174
419
|
ctx.ui.notify(message, "error");
|
|
@@ -268,25 +513,22 @@ export class CommandManager {
|
|
|
268
513
|
* @returns A mapping of actions for each status
|
|
269
514
|
*/
|
|
270
515
|
private async getActionsForModel(model: BaseModel): Promise<Array<Action>> {
|
|
271
|
-
const
|
|
516
|
+
const base = [Action.INFO, Action.CANCEL];
|
|
517
|
+
|
|
518
|
+
const actions: Record<Status, Array<Action>> = {
|
|
272
519
|
[Status.LOADED]:
|
|
273
520
|
model.mode === Mode.ROUTER
|
|
274
|
-
? [Action.SWITCH, Action.UNLOAD,
|
|
275
|
-
: [Action.SWITCH,
|
|
276
|
-
[Status.LOADING]: [
|
|
277
|
-
[Status.FAILED]: [Action.RETRY,
|
|
278
|
-
[Status.SLEEPING]: [
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
Action.INFO,
|
|
282
|
-
Action.CANCEL,
|
|
283
|
-
],
|
|
284
|
-
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, Action.CANCEL],
|
|
285
|
-
[Status.UNAUTHORIZED]: [Action.INFO, Action.CANCEL],
|
|
521
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
522
|
+
: [Action.SWITCH, ...base],
|
|
523
|
+
[Status.LOADING]: [...base],
|
|
524
|
+
[Status.FAILED]: [Action.RETRY, ...base],
|
|
525
|
+
[Status.SLEEPING]: [Action.SWITCH, Action.UNLOAD, ...base],
|
|
526
|
+
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, ...base],
|
|
527
|
+
[Status.UNAUTHORIZED]: [...base],
|
|
286
528
|
};
|
|
287
529
|
|
|
288
530
|
const status = await model.getStatus();
|
|
289
|
-
return
|
|
531
|
+
return actions[status];
|
|
290
532
|
}
|
|
291
533
|
|
|
292
534
|
/**
|
package/src/managers/events.ts
CHANGED
|
@@ -5,17 +5,35 @@ import {
|
|
|
5
5
|
import { READABLE_TIMEOUT } from "../constants";
|
|
6
6
|
import { Status } from "../enums/status";
|
|
7
7
|
import { ModelSelectEvent } from "../interfaces/events";
|
|
8
|
-
import {
|
|
8
|
+
import type { LlamaSettingsManager } from "../managers/settings";
|
|
9
9
|
import { BaseModel } from "../models/baseModel";
|
|
10
|
-
import {
|
|
10
|
+
import { ServerManager } from "./server";
|
|
11
11
|
|
|
12
12
|
export class EventManager {
|
|
13
|
+
/**
|
|
14
|
+
* Model with a load currently in flight. Deliberately a class static
|
|
15
|
+
* (REFACTOR.md §3.3): the load is started by CommandManager
|
|
16
|
+
* (fire-and-forget from the /models editor) while the "session switched
|
|
17
|
+
* mid-load" warning must be emitted here, from the session_before_switch
|
|
18
|
+
* hook — a shared static is the least-plumbing bridge between the two
|
|
19
|
+
* event-owning managers.
|
|
20
|
+
*
|
|
21
|
+
* Known limits, accepted: (1) single slot — a second overlapping load
|
|
22
|
+
* overwrites the first, and the first load's onFinished reset can clear
|
|
23
|
+
* the flag while the second is still loading, so the session-switch
|
|
24
|
+
* warning may be missed; (2) loads initiated from this class
|
|
25
|
+
* (onModelSelect / autoLoadIfNeeded) never set the flag.
|
|
26
|
+
*/
|
|
13
27
|
static inflightModel: BaseModel | null = null;
|
|
14
28
|
|
|
15
|
-
constructor(
|
|
29
|
+
constructor(
|
|
30
|
+
private readonly serverManager: ServerManager,
|
|
31
|
+
private readonly settings: LlamaSettingsManager,
|
|
32
|
+
) {}
|
|
16
33
|
|
|
17
34
|
/**
|
|
18
|
-
* Resets the in-flight model reference.
|
|
35
|
+
* Resets the in-flight model reference. Called by CommandManager when
|
|
36
|
+
* its load settles (see `inflightModel` for why this lives on a static).
|
|
19
37
|
*/
|
|
20
38
|
static resetInflightModel() {
|
|
21
39
|
EventManager.inflightModel = null;
|
|
@@ -29,9 +47,9 @@ export class EventManager {
|
|
|
29
47
|
*/
|
|
30
48
|
async onModelSelect(event: ModelSelectEvent, ctx: ExtensionContext) {
|
|
31
49
|
// Check if the model_select event should be used
|
|
32
|
-
if (!settings.resolveReactToModelSelect()) return;
|
|
50
|
+
if (!this.settings.resolveReactToModelSelect()) return;
|
|
33
51
|
|
|
34
|
-
for (const { providerId, models } of this.servers) {
|
|
52
|
+
for (const { providerId, models } of this.serverManager.servers) {
|
|
35
53
|
if (event.model.provider !== providerId) continue;
|
|
36
54
|
|
|
37
55
|
const model = models.find((m) => m.id === event.model.id);
|
|
@@ -54,7 +72,7 @@ export class EventManager {
|
|
|
54
72
|
* @param model The model to potentially auto-load
|
|
55
73
|
*/
|
|
56
74
|
private async autoLoadIfNeeded(model: BaseModel): Promise<void> {
|
|
57
|
-
if (!settings.resolveAutoloadOnMessage()) return;
|
|
75
|
+
if (!this.settings.resolveAutoloadOnMessage()) return;
|
|
58
76
|
|
|
59
77
|
const status = await model.getStatus();
|
|
60
78
|
if (status !== Status.UNLOADED) return;
|
|
@@ -99,7 +117,7 @@ export class EventManager {
|
|
|
99
117
|
if (!model) return payload;
|
|
100
118
|
|
|
101
119
|
// Check if this model belongs to one of our servers
|
|
102
|
-
const serverModel = this.servers
|
|
120
|
+
const serverModel = this.serverManager.servers
|
|
103
121
|
.flatMap((s) => s.models)
|
|
104
122
|
.find((m) => m.id === model);
|
|
105
123
|
|
|
@@ -110,8 +128,8 @@ export class EventManager {
|
|
|
110
128
|
|
|
111
129
|
// Retrieve pi's current thinking level, so we can setup a budget
|
|
112
130
|
const level =
|
|
113
|
-
ctx.thinkingLevel ?? settings.resolveThinkingLevel() ?? "medium";
|
|
114
|
-
const budgets = settings.resolveThinkingBudgets();
|
|
131
|
+
ctx.thinkingLevel ?? this.settings.resolveThinkingLevel() ?? "medium";
|
|
132
|
+
const budgets = this.settings.resolveThinkingBudgets();
|
|
115
133
|
const thinking_budget_tokens = budgets[level];
|
|
116
134
|
|
|
117
135
|
// Setup payload
|
package/src/managers/server.ts
CHANGED
|
@@ -1,15 +1,27 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { API_TYPE, PROVIDER_NAME } from "../constants";
|
|
2
|
+
import { API_TYPE, PROVIDER_NAME, type SortBy } from "../constants";
|
|
3
3
|
import { ServerStatus } from "../enums/serverStatus";
|
|
4
4
|
import { BaseModel } from "../models/baseModel";
|
|
5
5
|
import { Server } from "../server";
|
|
6
|
-
import {
|
|
6
|
+
import type { LlamaSettingsManager } from "./settings";
|
|
7
|
+
|
|
8
|
+
/** Model-list comparator: negative if a sorts first, positive if b does. */
|
|
9
|
+
type ModelComparator = (a: BaseModel, b: BaseModel) => number;
|
|
7
10
|
|
|
8
11
|
export class ServerManager {
|
|
12
|
+
constructor(private readonly settings: LlamaSettingsManager) {}
|
|
9
13
|
readonly failedUrls: string[] = [];
|
|
10
14
|
private readonly warnings: string[] = [];
|
|
15
|
+
private readonly serverList: Server[] = [];
|
|
11
16
|
|
|
12
|
-
|
|
17
|
+
/**
|
|
18
|
+
* Live view of the server list. `update()` re-derives the list from
|
|
19
|
+
* settings on every scan (in place), so `/models servers` edits apply
|
|
20
|
+
* without a restart.
|
|
21
|
+
*/
|
|
22
|
+
get servers(): readonly Server[] {
|
|
23
|
+
return this.serverList;
|
|
24
|
+
}
|
|
13
25
|
|
|
14
26
|
/**
|
|
15
27
|
* Verifies reachability of servers and registers the providers
|
|
@@ -18,7 +30,7 @@ export class ServerManager {
|
|
|
18
30
|
*/
|
|
19
31
|
async initialize(pi: ExtensionAPI) {
|
|
20
32
|
// Register the providers with the configured server timeout
|
|
21
|
-
const { serverTimeout } = settings.resolveTimeouts();
|
|
33
|
+
const { serverTimeout } = this.settings.resolveTimeouts();
|
|
22
34
|
await this.update(pi, serverTimeout);
|
|
23
35
|
}
|
|
24
36
|
|
|
@@ -32,6 +44,33 @@ export class ServerManager {
|
|
|
32
44
|
async update(pi: ExtensionAPI, timeout?: number) {
|
|
33
45
|
this.failedUrls.length = 0;
|
|
34
46
|
|
|
47
|
+
// Surface warnings from strict URL parsing (dropped invalid entries)
|
|
48
|
+
this.warnings.push(...this.settings.takeWarnings());
|
|
49
|
+
|
|
50
|
+
// Re-derive the server list from settings so `/models servers` edits
|
|
51
|
+
// (add / remove / URL / id / name) apply on the next scan
|
|
52
|
+
const fresh: Server[] = [];
|
|
53
|
+
const seen = new Set<string>(); // dedupe repeated URLs (same providerId)
|
|
54
|
+
for (const server of this.settings.resolveServers()) {
|
|
55
|
+
if (seen.has(server.providerId)) continue;
|
|
56
|
+
seen.add(server.providerId);
|
|
57
|
+
fresh.push(server);
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// Unregister providers that disappeared (removed or edited away);
|
|
61
|
+
// no-op for providers that were never registered
|
|
62
|
+
for (const old of this.servers) {
|
|
63
|
+
if (fresh.some((f) => f.providerId === old.providerId)) continue;
|
|
64
|
+
pi.unregisterProvider(old.providerId);
|
|
65
|
+
// Optional chain is intentional despite the non-optional type: `sse`
|
|
66
|
+
// is undefined until initialize() runs (async-constructor hack — see Server)
|
|
67
|
+
old.sseManager?.disconnect();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// Replace in place so the live `servers` view stays valid (D1)
|
|
71
|
+
this.serverList.length = 0;
|
|
72
|
+
this.serverList.push(...fresh);
|
|
73
|
+
|
|
35
74
|
const registrableServers = timeout
|
|
36
75
|
? await this.findRegistrableServers(timeout)
|
|
37
76
|
: this.servers;
|
|
@@ -95,7 +134,7 @@ export class ServerManager {
|
|
|
95
134
|
*/
|
|
96
135
|
private async registerProvider(server: Server, pi: ExtensionAPI) {
|
|
97
136
|
const { baseUrl, models, providerId, providerName } = server;
|
|
98
|
-
const apiKey =
|
|
137
|
+
const apiKey = server.getApiKey();
|
|
99
138
|
const modelConfigs = await Promise.all(
|
|
100
139
|
models.map((m) => m.toProviderConfig()),
|
|
101
140
|
);
|
|
@@ -123,26 +162,58 @@ export class ServerManager {
|
|
|
123
162
|
* Returns the server for a given model.
|
|
124
163
|
*
|
|
125
164
|
* @param model - The model to find the server for
|
|
126
|
-
* @returns The server containing the model
|
|
165
|
+
* @returns The server containing the model, or `undefined` when no
|
|
166
|
+
* current server matches (e.g. removed while a model was loading)
|
|
127
167
|
*/
|
|
128
|
-
getServer(model: BaseModel): Server {
|
|
129
|
-
return this.servers.find((s) => s.baseUrl === model.serverUrl)
|
|
168
|
+
getServer(model: BaseModel): Server | undefined {
|
|
169
|
+
return this.servers.find((s) => s.baseUrl === model.serverUrl);
|
|
130
170
|
}
|
|
131
171
|
|
|
132
172
|
/**
|
|
133
|
-
* Returns all models from all servers.
|
|
173
|
+
* Returns all models from all servers, sorted by the configured sort mode.
|
|
134
174
|
*
|
|
135
175
|
* @returns Flat array of all models across all servers
|
|
136
176
|
*/
|
|
137
177
|
getAllModels(): BaseModel[] {
|
|
138
|
-
const
|
|
178
|
+
const sortBy = this.settings.resolveSortBy();
|
|
179
|
+
const allModels = this.servers.flatMap((s) => s.models);
|
|
139
180
|
|
|
140
|
-
|
|
141
|
-
for (const model of models) {
|
|
142
|
-
response.push(model);
|
|
143
|
-
}
|
|
144
|
-
}
|
|
181
|
+
if (sortBy === "api") return allModels;
|
|
145
182
|
|
|
146
|
-
return
|
|
183
|
+
return allModels.sort(ServerManager.SORTERS[sortBy]);
|
|
147
184
|
}
|
|
185
|
+
|
|
186
|
+
private static sortByIdAsc(a: BaseModel, b: BaseModel): number {
|
|
187
|
+
return a.id.localeCompare(b.id);
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
private static sortByIdDesc(a: BaseModel, b: BaseModel): number {
|
|
191
|
+
return b.id.localeCompare(a.id);
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Name ascending, with ID as tiebreaker. */
|
|
195
|
+
private static sortByNameAsc(a: BaseModel, b: BaseModel): number {
|
|
196
|
+
const cmp = a.name.localeCompare(b.name);
|
|
197
|
+
return cmp !== 0 ? cmp : a.id.localeCompare(b.id);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** Name descending, with ID as tiebreaker. */
|
|
201
|
+
private static sortByNameDesc(a: BaseModel, b: BaseModel): number {
|
|
202
|
+
const cmp = b.name.localeCompare(a.name);
|
|
203
|
+
return cmp !== 0 ? cmp : a.id.localeCompare(b.id);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Comparators for each sort mode except "api", which preserves server
|
|
208
|
+
* order (short-circuited in {@link ServerManager.getAllModels}).
|
|
209
|
+
*/
|
|
210
|
+
private static readonly SORTERS: Record<
|
|
211
|
+
Exclude<SortBy, "api">,
|
|
212
|
+
ModelComparator
|
|
213
|
+
> = {
|
|
214
|
+
asc: ServerManager.sortByIdAsc,
|
|
215
|
+
desc: ServerManager.sortByIdDesc,
|
|
216
|
+
"asc-name": ServerManager.sortByNameAsc,
|
|
217
|
+
"desc-name": ServerManager.sortByNameDesc,
|
|
218
|
+
};
|
|
148
219
|
}
|