pi-llama-cpp 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +187 -10
- package/package.json +6 -6
- package/src/api/client.ts +76 -32
- package/src/constants.ts +15 -0
- package/src/index.ts +3 -5
- package/src/interfaces/events.ts +14 -3
- package/src/interfaces/server.ts +27 -0
- package/src/interfaces/settings.ts +76 -2
- package/src/managers/command.ts +369 -50
- package/src/managers/events.ts +28 -10
- package/src/managers/server.ts +90 -16
- package/src/managers/settings.ts +181 -44
- package/src/models/baseModel.ts +37 -15
- package/src/models/routerModel.ts +2 -1
- package/src/server.ts +103 -28
- package/src/sse/client.ts +28 -16
- package/src/sse/manager.ts +26 -13
- package/src/ui/dialog.ts +287 -0
- package/src/ui/overrideEntryEditor.ts +119 -0
- package/src/ui/overrideSettingsList.ts +682 -0
- package/src/ui/serverListEditor.ts +32 -0
- package/src/ui/serverSettingsList.ts +466 -0
- package/src/ui/strings.ts +127 -0
- package/src/utils/errors.ts +5 -0
- package/src/utils/settingsStore.ts +56 -0
- package/src/utils/urls.ts +16 -0
- package/tests/commandManager.test.ts +346 -11
- package/tests/dialog.test.ts +186 -0
- package/tests/events.test.ts +120 -88
- package/tests/legacyModel.test.ts +4 -19
- package/tests/mocks.ts +149 -32
- package/tests/overrides.test.ts +352 -0
- package/tests/server.test.ts +42 -40
- package/tests/serverManager.test.ts +264 -55
- package/tests/settings.test.ts +654 -51
- package/tests/settingsStore.test.ts +190 -0
- package/tests/singleModel.test.ts +32 -0
- package/tests/sseManager.test.ts +88 -11
- package/src/interfaces/auth.ts +0 -6
- package/src/utils/cache.ts +0 -39
- package/src/utils/mutex.ts +0 -24
package/src/managers/command.ts
CHANGED
|
@@ -1,18 +1,186 @@
|
|
|
1
|
-
import
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
import {
|
|
2
|
+
getSettingsListTheme,
|
|
3
|
+
type ExtensionAPI,
|
|
4
|
+
type ExtensionCommandContext,
|
|
4
5
|
} from "@earendil-works/pi-coding-agent";
|
|
5
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
AutocompleteItem,
|
|
8
|
+
SettingsList,
|
|
9
|
+
type SettingItem,
|
|
10
|
+
} from "@earendil-works/pi-tui";
|
|
6
11
|
import { PROVIDER_NAME } from "../constants";
|
|
7
12
|
import { Action } from "../enums/action";
|
|
8
13
|
import { Mode } from "../enums/mode";
|
|
9
14
|
import { Status } from "../enums/status";
|
|
15
|
+
import { LlamaSettings } from "../interfaces/settings";
|
|
10
16
|
import { BaseModel } from "../models/baseModel";
|
|
17
|
+
import { createOverrideSettingsList } from "../ui/overrideSettingsList";
|
|
18
|
+
import { ServerSettingsList } from "../ui/serverSettingsList";
|
|
19
|
+
import { errorMessage } from "../utils/errors";
|
|
11
20
|
import { EventManager } from "./events";
|
|
12
21
|
import { ServerManager } from "./server";
|
|
22
|
+
import type { LlamaSettingsManager } from "./settings";
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Identifiers of the editable fields shown in `/models settings`.
|
|
26
|
+
* Values match the scalar `LlamaSettings` keys.
|
|
27
|
+
*/
|
|
28
|
+
export enum Options {
|
|
29
|
+
REACT_TO_MODEL_SELECT = "reactToModelSelect",
|
|
30
|
+
AUTOLOAD_ON_MESSAGE = "autoloadOnMessage",
|
|
31
|
+
SORT_BY = "sortBy",
|
|
32
|
+
POLLING_TIMEOUT = "pollingTimeout",
|
|
33
|
+
SERVER_TIMEOUT = "serverTimeout",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
type SortByValue = NonNullable<LlamaSettings["sortBy"]>;
|
|
37
|
+
|
|
38
|
+
const SORT_VALUES: SortByValue[] = [
|
|
39
|
+
"asc",
|
|
40
|
+
"desc",
|
|
41
|
+
"asc-name",
|
|
42
|
+
"desc-name",
|
|
43
|
+
"api",
|
|
44
|
+
];
|
|
45
|
+
|
|
46
|
+
/** Presets (ms) for `pollingTimeout` */
|
|
47
|
+
const POLLING_PRESETS = [15000, 30000, 60000, 120000, 300000];
|
|
48
|
+
|
|
49
|
+
/** Presets (ms) for `serverTimeout` */
|
|
50
|
+
const SERVER_PRESETS = [500, 1000, 2000, 5000, 10000];
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* `/models` subcommand completions. Module-level so
|
|
54
|
+
* {@link CommandManager.getArgumentCompletions} doesn't rebuild the
|
|
55
|
+
* table on every keystroke.
|
|
56
|
+
*/
|
|
57
|
+
const ARGUMENT_COMPLETIONS: AutocompleteItem[] = [
|
|
58
|
+
{
|
|
59
|
+
value: "info",
|
|
60
|
+
label: "info",
|
|
61
|
+
description: "Show information of all models",
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
value: "unload",
|
|
65
|
+
label: "unload",
|
|
66
|
+
description: "Unload all models",
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
value: "settings",
|
|
70
|
+
label: "settings",
|
|
71
|
+
description: "Configure llamaSettings",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
value: "servers",
|
|
75
|
+
label: "servers",
|
|
76
|
+
description: "Manage llama.cpp server URLs",
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
value: "overrides",
|
|
80
|
+
label: "overrides",
|
|
81
|
+
description: "Manage llama.cpp model overrides",
|
|
82
|
+
},
|
|
83
|
+
];
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Formats milliseconds compactly for display (e.g. `500 -> "500ms"`,
|
|
87
|
+
* `60000 -> "60s"`).
|
|
88
|
+
*/
|
|
89
|
+
export const formatMs = (ms: number): string =>
|
|
90
|
+
ms % 1000 === 0 ? `${ms / 1000}s` : `${ms}ms`;
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Parses a value produced by `formatMs()` back to milliseconds.
|
|
94
|
+
* Only ever called with values from the preset lists.
|
|
95
|
+
*/
|
|
96
|
+
const parseMs = (value: string): number =>
|
|
97
|
+
value.endsWith("ms")
|
|
98
|
+
? Number(value.slice(0, -2))
|
|
99
|
+
: Number(value.slice(0, -1)) * 1000;
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Builds the `SettingsList` items for `/models settings` from the current
|
|
103
|
+
* (merged) values of the scalar `llamaSettings` fields.
|
|
104
|
+
*/
|
|
105
|
+
export const buildSettingsItems = async (
|
|
106
|
+
settings: LlamaSettingsManager,
|
|
107
|
+
): Promise<SettingItem[]> => {
|
|
108
|
+
const { pollingTimeout, serverTimeout } = await settings.resolveTimeouts();
|
|
109
|
+
|
|
110
|
+
return [
|
|
111
|
+
{
|
|
112
|
+
id: Options.REACT_TO_MODEL_SELECT,
|
|
113
|
+
label: "React to model selection",
|
|
114
|
+
description: "Load the model when you pick it in Pi (immediate)",
|
|
115
|
+
currentValue: (await settings.resolveReactToModelSelect()) ? "on" : "off",
|
|
116
|
+
values: ["on", "off"],
|
|
117
|
+
},
|
|
118
|
+
{
|
|
119
|
+
id: Options.AUTOLOAD_ON_MESSAGE,
|
|
120
|
+
label: "Autoload on message",
|
|
121
|
+
description:
|
|
122
|
+
"Auto-load the selected model when you send a message (immediate)",
|
|
123
|
+
currentValue: (await settings.resolveAutoloadOnMessage()) ? "on" : "off",
|
|
124
|
+
values: ["on", "off"],
|
|
125
|
+
},
|
|
126
|
+
{
|
|
127
|
+
id: Options.SORT_BY,
|
|
128
|
+
label: "Sort models by",
|
|
129
|
+
description: "Order of models in /models (next open)",
|
|
130
|
+
currentValue: await settings.resolveSortBy(),
|
|
131
|
+
values: [...SORT_VALUES],
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
id: Options.POLLING_TIMEOUT,
|
|
135
|
+
label: "Polling timeout",
|
|
136
|
+
description: "Max model-load wait (next model load)",
|
|
137
|
+
currentValue: formatMs(pollingTimeout),
|
|
138
|
+
values: POLLING_PRESETS.map(formatMs),
|
|
139
|
+
},
|
|
140
|
+
{
|
|
141
|
+
id: Options.SERVER_TIMEOUT,
|
|
142
|
+
label: "Server timeout",
|
|
143
|
+
description: "Health check / SSE probe timeout (next model load)",
|
|
144
|
+
currentValue: formatMs(serverTimeout),
|
|
145
|
+
values: SERVER_PRESETS.map(formatMs),
|
|
146
|
+
},
|
|
147
|
+
];
|
|
148
|
+
};
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Persists a change made in the settings menu.
|
|
152
|
+
* Maps the `SettingsList` id/value pair to the matching `llamaSettings`
|
|
153
|
+
* key and writes it via `LlamaSettingsManager.setLlamaSetting()`.
|
|
154
|
+
*/
|
|
155
|
+
export const applySettingChange = async (
|
|
156
|
+
id: string,
|
|
157
|
+
newValue: string,
|
|
158
|
+
settings: LlamaSettingsManager,
|
|
159
|
+
): Promise<void> => {
|
|
160
|
+
switch (id) {
|
|
161
|
+
case Options.REACT_TO_MODEL_SELECT:
|
|
162
|
+
await settings.setLlamaSetting("reactToModelSelect", newValue === "on");
|
|
163
|
+
return;
|
|
164
|
+
case Options.AUTOLOAD_ON_MESSAGE:
|
|
165
|
+
await settings.setLlamaSetting("autoloadOnMessage", newValue === "on");
|
|
166
|
+
return;
|
|
167
|
+
case Options.SORT_BY:
|
|
168
|
+
await settings.setLlamaSetting("sortBy", newValue as SortByValue);
|
|
169
|
+
return;
|
|
170
|
+
case Options.POLLING_TIMEOUT:
|
|
171
|
+
await settings.setLlamaSetting("pollingTimeout", parseMs(newValue));
|
|
172
|
+
return;
|
|
173
|
+
case Options.SERVER_TIMEOUT:
|
|
174
|
+
await settings.setLlamaSetting("serverTimeout", parseMs(newValue));
|
|
175
|
+
return;
|
|
176
|
+
}
|
|
177
|
+
};
|
|
13
178
|
|
|
14
179
|
export class CommandManager {
|
|
15
|
-
constructor(
|
|
180
|
+
constructor(
|
|
181
|
+
private readonly serverManager: ServerManager,
|
|
182
|
+
private readonly settings: LlamaSettingsManager,
|
|
183
|
+
) {}
|
|
16
184
|
|
|
17
185
|
/**
|
|
18
186
|
* Sets up the argument completions for the `/models` command
|
|
@@ -21,19 +189,9 @@ export class CommandManager {
|
|
|
21
189
|
* @returns Completions with that prefix
|
|
22
190
|
*/
|
|
23
191
|
getArgumentCompletions(prefix: string): AutocompleteItem[] | null {
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
label: "info",
|
|
28
|
-
description: "Show information of all models",
|
|
29
|
-
},
|
|
30
|
-
{
|
|
31
|
-
value: "unload",
|
|
32
|
-
label: "unload",
|
|
33
|
-
description: "Unload all models",
|
|
34
|
-
},
|
|
35
|
-
];
|
|
36
|
-
const filtered = available.filter((a) => a.value.startsWith(prefix));
|
|
192
|
+
const filtered = ARGUMENT_COMPLETIONS.filter((a) =>
|
|
193
|
+
a.value.startsWith(prefix),
|
|
194
|
+
);
|
|
37
195
|
return filtered.length > 0 ? filtered : null;
|
|
38
196
|
}
|
|
39
197
|
|
|
@@ -49,6 +207,27 @@ export class CommandManager {
|
|
|
49
207
|
ctx: ExtensionCommandContext,
|
|
50
208
|
pi: ExtensionAPI,
|
|
51
209
|
) {
|
|
210
|
+
// Settings menu: no network round-trip needed, handle before any
|
|
211
|
+
// server updates / unreachable-server notifications
|
|
212
|
+
if (args === "settings") {
|
|
213
|
+
await this.runSettingsMenu(ctx);
|
|
214
|
+
return;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// Servers editor: re-registers providers after editing so changes
|
|
218
|
+
// (add / remove / URL / id / name) apply immediately
|
|
219
|
+
if (args === "servers") {
|
|
220
|
+
await this.runServersEditor(ctx, pi);
|
|
221
|
+
return;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// Overrides editor: re-registers providers after editing so new
|
|
225
|
+
// overrides take effect on the next request
|
|
226
|
+
if (args === "overrides") {
|
|
227
|
+
await this.runOverridesEditor(ctx, pi);
|
|
228
|
+
return;
|
|
229
|
+
}
|
|
230
|
+
|
|
52
231
|
// Re-register providers so Pi sees updated model states
|
|
53
232
|
await this.serverManager.update(pi);
|
|
54
233
|
|
|
@@ -58,17 +237,15 @@ export class CommandManager {
|
|
|
58
237
|
}
|
|
59
238
|
|
|
60
239
|
if (args === "unload") {
|
|
61
|
-
await
|
|
62
|
-
|
|
63
|
-
);
|
|
240
|
+
const models = await this.serverManager.getAllModels();
|
|
241
|
+
await Promise.all(models.map((model) => model.unload()));
|
|
64
242
|
ctx.ui.notify(`Unloaded all ${PROVIDER_NAME} models`, "info");
|
|
65
243
|
return;
|
|
66
244
|
}
|
|
67
245
|
|
|
68
246
|
if (args === "info") {
|
|
69
|
-
const
|
|
70
|
-
|
|
71
|
-
);
|
|
247
|
+
const models = await this.serverManager.getAllModels();
|
|
248
|
+
const infos = await Promise.all(models.map((model) => model.getInfo()));
|
|
72
249
|
ctx.ui.notify(ctx.ui.theme.fg("accent", infos.join("\n")), "info");
|
|
73
250
|
return;
|
|
74
251
|
}
|
|
@@ -77,6 +254,142 @@ export class CommandManager {
|
|
|
77
254
|
await this.runModelsMenu(ctx, pi);
|
|
78
255
|
}
|
|
79
256
|
|
|
257
|
+
/**
|
|
258
|
+
* Runs the interactive settings menu for the scalar `llamaSettings`
|
|
259
|
+
* fields. Enter/Space cycles the value under the cursor; Esc closes.
|
|
260
|
+
*
|
|
261
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
262
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified
|
|
263
|
+
* and leave the dialog open with values unchanged. These settings
|
|
264
|
+
* (reactToModelSelect, autoloadOnMessage, sortBy, timeouts) do not
|
|
265
|
+
* require provider re-registration.
|
|
266
|
+
*/
|
|
267
|
+
private async runSettingsMenu(ctx: ExtensionCommandContext): Promise<void> {
|
|
268
|
+
if (ctx.mode !== "tui") {
|
|
269
|
+
ctx.ui.notify(
|
|
270
|
+
"/models settings requires an interactive session (TUI)",
|
|
271
|
+
"warning",
|
|
272
|
+
);
|
|
273
|
+
return;
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const items = await buildSettingsItems(this.settings);
|
|
277
|
+
|
|
278
|
+
await ctx.ui.custom<void>(
|
|
279
|
+
(_tui, _theme, _kb, done) =>
|
|
280
|
+
new SettingsList(
|
|
281
|
+
items,
|
|
282
|
+
Math.min(items.length + 2, 15),
|
|
283
|
+
getSettingsListTheme(),
|
|
284
|
+
(id, newValue) => {
|
|
285
|
+
applySettingChange(id, newValue, this.settings).catch(
|
|
286
|
+
(err: unknown) => {
|
|
287
|
+
const message = errorMessage(err);
|
|
288
|
+
ctx.ui.notify(message, "error");
|
|
289
|
+
},
|
|
290
|
+
);
|
|
291
|
+
},
|
|
292
|
+
() => done(undefined),
|
|
293
|
+
),
|
|
294
|
+
);
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
/**
|
|
298
|
+
* Runs the interactive servers editor for `llamaSettings.servers`.
|
|
299
|
+
* Enter on a server row drills into its field-edit submenu (URL/id/name);
|
|
300
|
+
* a adds a new server (inline Input), d deletes (after confirmation);
|
|
301
|
+
* Esc closes.
|
|
302
|
+
*
|
|
303
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
304
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified and
|
|
305
|
+
* the editor stays open with the pre-mutation list. After closing,
|
|
306
|
+
* providers are re-registered so server changes apply immediately.
|
|
307
|
+
*/
|
|
308
|
+
private async runServersEditor(
|
|
309
|
+
ctx: ExtensionCommandContext,
|
|
310
|
+
pi: ExtensionAPI,
|
|
311
|
+
): Promise<void> {
|
|
312
|
+
if (ctx.mode !== "tui") {
|
|
313
|
+
ctx.ui.notify(
|
|
314
|
+
"/models servers requires an interactive session (TUI)",
|
|
315
|
+
"warning",
|
|
316
|
+
);
|
|
317
|
+
return;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
const servers = await this.settings.getLlamaServers();
|
|
321
|
+
|
|
322
|
+
await ctx.ui.custom<void>(
|
|
323
|
+
(tui, theme, keybindings, done) =>
|
|
324
|
+
new ServerSettingsList({
|
|
325
|
+
tui,
|
|
326
|
+
theme,
|
|
327
|
+
keybindings,
|
|
328
|
+
servers,
|
|
329
|
+
persist: (next) => this.settings.setLlamaSetting("servers", next),
|
|
330
|
+
done: () => {
|
|
331
|
+
done(undefined);
|
|
332
|
+
// Re-register providers so the updated server list takes effect
|
|
333
|
+
this.serverManager.update(pi);
|
|
334
|
+
},
|
|
335
|
+
onError: (message) => ctx.ui.notify(message, "error"),
|
|
336
|
+
}),
|
|
337
|
+
);
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* Runs the interactive overrides editor for
|
|
342
|
+
* `llamaSettings.servers[].overrides`: a SettingsList of servers drilling
|
|
343
|
+
* down into each server's override entries (one row per pattern, with
|
|
344
|
+
* add/delete support). Within a server's entry list: Enter drills into
|
|
345
|
+
* the field-edit submenu; a adds, d deletes (after confirmation).
|
|
346
|
+
*
|
|
347
|
+
* Fields use a mix of finite (Enter to cycle) and infinite (Enter to
|
|
348
|
+
* type) editing:
|
|
349
|
+
*
|
|
350
|
+
* - Pattern / costs (input, output, cacheRead, cacheWrite): infinite —
|
|
351
|
+
* Enter opens an Input for typing.
|
|
352
|
+
* - Capabilities: finite — Enter cycles between `text` and `text | image`.
|
|
353
|
+
* - Reasoning: finite — Enter cycles between `true` and `false`.
|
|
354
|
+
*
|
|
355
|
+
* Servers themselves are not managed here — use `/models servers`.
|
|
356
|
+
*
|
|
357
|
+
* Writes go to the global `~/.pi/agent/settings.json` via
|
|
358
|
+
* `LlamaSettingsManager.setLlamaSetting()`; write errors are notified and
|
|
359
|
+
* leave the values unchanged. After closing, providers are
|
|
360
|
+
* re-registered so new overrides take effect on the next request.
|
|
361
|
+
*/
|
|
362
|
+
private async runOverridesEditor(
|
|
363
|
+
ctx: ExtensionCommandContext,
|
|
364
|
+
pi: ExtensionAPI,
|
|
365
|
+
): Promise<void> {
|
|
366
|
+
if (ctx.mode !== "tui") {
|
|
367
|
+
ctx.ui.notify(
|
|
368
|
+
"/models overrides requires an interactive session (TUI)",
|
|
369
|
+
"warning",
|
|
370
|
+
);
|
|
371
|
+
return;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
const servers = await this.settings.getLlamaServers();
|
|
375
|
+
await ctx.ui.custom<void>((tui, theme, keybindings, done) =>
|
|
376
|
+
createOverrideSettingsList({
|
|
377
|
+
tui,
|
|
378
|
+
theme,
|
|
379
|
+
keybindings,
|
|
380
|
+
servers,
|
|
381
|
+
persist: (next) => this.settings.setLlamaSetting("servers", next),
|
|
382
|
+
done: () => {
|
|
383
|
+
done(undefined);
|
|
384
|
+
// Re-register providers so the updated overrides take effect
|
|
385
|
+
this.serverManager.update(pi);
|
|
386
|
+
},
|
|
387
|
+
onError: (message) => ctx.ui.notify(message, "error"),
|
|
388
|
+
onChanged: () => {}, // no per-change notification needed
|
|
389
|
+
}),
|
|
390
|
+
);
|
|
391
|
+
}
|
|
392
|
+
|
|
80
393
|
/**
|
|
81
394
|
* Notifies the user that a server is unreachable.
|
|
82
395
|
*/
|
|
@@ -93,7 +406,7 @@ export class CommandManager {
|
|
|
93
406
|
): Promise<void> {
|
|
94
407
|
const event = await this.modelSelectionHandler(
|
|
95
408
|
ctx,
|
|
96
|
-
this.serverManager.getAllModels(),
|
|
409
|
+
await this.serverManager.getAllModels(),
|
|
97
410
|
);
|
|
98
411
|
|
|
99
412
|
if (!event) return;
|
|
@@ -132,18 +445,24 @@ export class CommandManager {
|
|
|
132
445
|
const loadActions = [Action.LOAD, Action.LOAD_AND_SWITCH, Action.RETRY];
|
|
133
446
|
if (loadActions.includes(action)) {
|
|
134
447
|
ctx.ui.notify(`Loading ${model.name}...`, "info");
|
|
448
|
+
// Mark the load as in-flight so session_before_switch can warn about
|
|
449
|
+
// it (see EventManager.inflightModel for the coupling rationale)
|
|
135
450
|
EventManager.inflightModel = model;
|
|
136
451
|
|
|
137
|
-
// Subscribe to progress events
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
"
|
|
145
|
-
|
|
146
|
-
|
|
452
|
+
// Subscribe to progress events; skip when the server is gone
|
|
453
|
+
// (removed/edited away mid-load → getServer returns undefined)
|
|
454
|
+
const server = this.serverManager.getServer(model);
|
|
455
|
+
const cleanupProgress =
|
|
456
|
+
server?.sseManager.subscribeToProgress(
|
|
457
|
+
model.id,
|
|
458
|
+
(percentage, stage) => {
|
|
459
|
+
const stageText = stage ? ` (${stage})` : "";
|
|
460
|
+
ctx.ui.notify(
|
|
461
|
+
`Loading ${model.name}... [${percentage}%${stageText}]`,
|
|
462
|
+
"info",
|
|
463
|
+
);
|
|
464
|
+
},
|
|
465
|
+
) ?? (() => {});
|
|
147
466
|
|
|
148
467
|
const onSuccess = async () => {
|
|
149
468
|
const { serverId } = model;
|
|
@@ -168,7 +487,7 @@ export class CommandManager {
|
|
|
168
487
|
};
|
|
169
488
|
|
|
170
489
|
const onFailure = (err: any) => {
|
|
171
|
-
const message =
|
|
490
|
+
const message = errorMessage(err);
|
|
172
491
|
|
|
173
492
|
try {
|
|
174
493
|
ctx.ui.notify(message, "error");
|
|
@@ -268,25 +587,25 @@ export class CommandManager {
|
|
|
268
587
|
* @returns A mapping of actions for each status
|
|
269
588
|
*/
|
|
270
589
|
private async getActionsForModel(model: BaseModel): Promise<Array<Action>> {
|
|
271
|
-
const
|
|
590
|
+
const base = [Action.INFO, Action.CANCEL];
|
|
591
|
+
|
|
592
|
+
const actions: Record<Status, Array<Action>> = {
|
|
272
593
|
[Status.LOADED]:
|
|
273
594
|
model.mode === Mode.ROUTER
|
|
274
|
-
? [Action.SWITCH, Action.UNLOAD,
|
|
275
|
-
: [Action.SWITCH,
|
|
276
|
-
[Status.LOADING]: [
|
|
277
|
-
[Status.FAILED]: [Action.RETRY,
|
|
278
|
-
[Status.SLEEPING]:
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
],
|
|
284
|
-
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, Action.CANCEL],
|
|
285
|
-
[Status.UNAUTHORIZED]: [Action.INFO, Action.CANCEL],
|
|
595
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
596
|
+
: [Action.SWITCH, ...base],
|
|
597
|
+
[Status.LOADING]: [...base],
|
|
598
|
+
[Status.FAILED]: [Action.RETRY, ...base],
|
|
599
|
+
[Status.SLEEPING]:
|
|
600
|
+
model.mode === Mode.ROUTER
|
|
601
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
602
|
+
: [Action.SWITCH, ...base],
|
|
603
|
+
[Status.UNLOADED]: [Action.LOAD_AND_SWITCH, Action.LOAD, ...base],
|
|
604
|
+
[Status.UNAUTHORIZED]: [...base],
|
|
286
605
|
};
|
|
287
606
|
|
|
288
607
|
const status = await model.getStatus();
|
|
289
|
-
return
|
|
608
|
+
return actions[status];
|
|
290
609
|
}
|
|
291
610
|
|
|
292
611
|
/**
|
package/src/managers/events.ts
CHANGED
|
@@ -5,17 +5,35 @@ import {
|
|
|
5
5
|
import { READABLE_TIMEOUT } from "../constants";
|
|
6
6
|
import { Status } from "../enums/status";
|
|
7
7
|
import { ModelSelectEvent } from "../interfaces/events";
|
|
8
|
-
import {
|
|
8
|
+
import type { LlamaSettingsManager } from "../managers/settings";
|
|
9
9
|
import { BaseModel } from "../models/baseModel";
|
|
10
|
-
import {
|
|
10
|
+
import { ServerManager } from "./server";
|
|
11
11
|
|
|
12
12
|
export class EventManager {
|
|
13
|
+
/**
|
|
14
|
+
* Model with a load currently in flight. Deliberately a class static
|
|
15
|
+
* (REFACTOR.md §3.3): the load is started by CommandManager
|
|
16
|
+
* (fire-and-forget from the /models editor) while the "session switched
|
|
17
|
+
* mid-load" warning must be emitted here, from the session_before_switch
|
|
18
|
+
* hook — a shared static is the least-plumbing bridge between the two
|
|
19
|
+
* event-owning managers.
|
|
20
|
+
*
|
|
21
|
+
* Known limits, accepted: (1) single slot — a second overlapping load
|
|
22
|
+
* overwrites the first, and the first load's onFinished reset can clear
|
|
23
|
+
* the flag while the second is still loading, so the session-switch
|
|
24
|
+
* warning may be missed; (2) loads initiated from this class
|
|
25
|
+
* (onModelSelect / autoLoadIfNeeded) never set the flag.
|
|
26
|
+
*/
|
|
13
27
|
static inflightModel: BaseModel | null = null;
|
|
14
28
|
|
|
15
|
-
constructor(
|
|
29
|
+
constructor(
|
|
30
|
+
private readonly serverManager: ServerManager,
|
|
31
|
+
private readonly settings: LlamaSettingsManager,
|
|
32
|
+
) {}
|
|
16
33
|
|
|
17
34
|
/**
|
|
18
|
-
* Resets the in-flight model reference.
|
|
35
|
+
* Resets the in-flight model reference. Called by CommandManager when
|
|
36
|
+
* its load settles (see `inflightModel` for why this lives on a static).
|
|
19
37
|
*/
|
|
20
38
|
static resetInflightModel() {
|
|
21
39
|
EventManager.inflightModel = null;
|
|
@@ -29,9 +47,9 @@ export class EventManager {
|
|
|
29
47
|
*/
|
|
30
48
|
async onModelSelect(event: ModelSelectEvent, ctx: ExtensionContext) {
|
|
31
49
|
// Check if the model_select event should be used
|
|
32
|
-
if (!settings.resolveReactToModelSelect()) return;
|
|
50
|
+
if (!(await this.settings.resolveReactToModelSelect())) return;
|
|
33
51
|
|
|
34
|
-
for (const { providerId, models } of this.servers) {
|
|
52
|
+
for (const { providerId, models } of this.serverManager.servers) {
|
|
35
53
|
if (event.model.provider !== providerId) continue;
|
|
36
54
|
|
|
37
55
|
const model = models.find((m) => m.id === event.model.id);
|
|
@@ -54,7 +72,7 @@ export class EventManager {
|
|
|
54
72
|
* @param model The model to potentially auto-load
|
|
55
73
|
*/
|
|
56
74
|
private async autoLoadIfNeeded(model: BaseModel): Promise<void> {
|
|
57
|
-
if (!settings.resolveAutoloadOnMessage()) return;
|
|
75
|
+
if (!(await this.settings.resolveAutoloadOnMessage())) return;
|
|
58
76
|
|
|
59
77
|
const status = await model.getStatus();
|
|
60
78
|
if (status !== Status.UNLOADED) return;
|
|
@@ -99,7 +117,7 @@ export class EventManager {
|
|
|
99
117
|
if (!model) return payload;
|
|
100
118
|
|
|
101
119
|
// Check if this model belongs to one of our servers
|
|
102
|
-
const serverModel = this.servers
|
|
120
|
+
const serverModel = this.serverManager.servers
|
|
103
121
|
.flatMap((s) => s.models)
|
|
104
122
|
.find((m) => m.id === model);
|
|
105
123
|
|
|
@@ -110,8 +128,8 @@ export class EventManager {
|
|
|
110
128
|
|
|
111
129
|
// Retrieve pi's current thinking level, so we can setup a budget
|
|
112
130
|
const level =
|
|
113
|
-
ctx.thinkingLevel ?? settings.resolveThinkingLevel() ?? "medium";
|
|
114
|
-
const budgets = settings.resolveThinkingBudgets();
|
|
131
|
+
ctx.thinkingLevel ?? this.settings.resolveThinkingLevel() ?? "medium";
|
|
132
|
+
const budgets = this.settings.resolveThinkingBudgets();
|
|
115
133
|
const thinking_budget_tokens = budgets[level];
|
|
116
134
|
|
|
117
135
|
// Setup payload
|