pi-llama-cpp 0.15.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -3
- package/package.json +1 -1
- package/src/api/client.ts +59 -22
- package/src/enums/mode.ts +1 -0
- package/src/interfaces/endpoints/models.ts +6 -0
- package/src/interfaces/endpoints/props.ts +14 -0
- package/src/managers/command/models.ts +3 -4
- package/src/managers/settings.ts +9 -1
- package/src/models/legacyModel.ts +3 -3
- package/src/models/llamaSwapModel.ts +90 -0
- package/src/models/routerModel.ts +2 -2
- package/src/models/singleModel.ts +2 -2
- package/src/server.ts +36 -6
- package/src/ui/dialog/confirm.ts +1 -1
- package/src/ui/dialog/input.ts +1 -1
- package/src/ui/editors/override/entryEditor.ts +9 -9
- package/src/ui/editors/override/itemBuilder.ts +5 -2
- package/src/ui/editors/server/fields.ts +2 -2
- package/src/ui/editors/server/itemBuilder.ts +5 -2
- package/src/ui/editors/server/serverEditor.ts +7 -7
- package/src/utils/credentialResolver.ts +110 -0
- package/src/utils/health.ts +2 -3
- package/tests/mocks.ts +2 -0
- package/tests/models/llamaSwapModel.test.ts +327 -0
- package/tests/server/health.test.ts +2 -2
- package/tests/server/server.test.ts +52 -2
- package/tests/settings/settings.test.ts +54 -252
- package/tests/utils/credentialResolver.test.ts +66 -0
- package/tests/utils/urlResolver.test.ts +116 -0
package/README.md
CHANGED
|
@@ -12,6 +12,7 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
|
|
|
12
12
|
- **Flexible URL resolution** — configures the server via `llamaSettings` (project/global), environment variable, or legacy `llamaServerUrl`
|
|
13
13
|
- **Auth support** — allows to login into a llama.cpp server that was secured with an API key
|
|
14
14
|
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously via `llamaSettings.servers` or semicolon-separated URLs
|
|
15
|
+
- **Basic llama-swap support** — auto-detects and provides basic integration with [llama-swap](https://github.com/mostlygeek/llama-swap) gateways
|
|
15
16
|
- **Thinking budget support** — configurable token budgets for model reasoning/thinking, mapped to Pi's thinking levels
|
|
16
17
|
- **Real-time progress tracking** — live loading progress via SSE (falls back to polling)
|
|
17
18
|
|
|
@@ -207,17 +208,48 @@ Each server gets its own provider (e.g., **Llama.cpp (http://127.0.0.1:8080)**)
|
|
|
207
208
|
If your llama.cpp server requires authentication, use `/login` in Pi, select the "API key" option, and choose the provider from the list that correlates with the server needing the API key.
|
|
208
209
|
|
|
209
210
|
Alternatively, configure the API key in `~/.pi/agent/auth.json`:
|
|
210
|
-
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`)
|
|
211
|
+
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`).
|
|
212
|
+
|
|
213
|
+
The `key` field supports several formats:
|
|
214
|
+
|
|
215
|
+
| Format | Example | Description |
|
|
216
|
+
| ----------------- | -------------------------------------------- | ------------------------------------------ |
|
|
217
|
+
| **Literal** | `"sk-abc123"` | API key stored directly |
|
|
218
|
+
| **Env ref** | `"$OPENAI_API_KEY"` or `"${OPENAI_API_KEY}"` | Resolved from `process.env` or `env` field |
|
|
219
|
+
| **Shell command** | `"!cat ~/.secrets/llama-key"` | Stdout of the command is used |
|
|
220
|
+
| **Escape** | `"$$literal"` | `$$` → literal `$`, `$!` → literal `!` |
|
|
211
221
|
|
|
212
222
|
```json
|
|
213
223
|
{
|
|
214
224
|
"llama-server=http://127.0.0.1:8080": {
|
|
215
225
|
"type": "api_key",
|
|
216
|
-
"key": "
|
|
226
|
+
"key": "sk-abc123"
|
|
217
227
|
},
|
|
218
228
|
"llama-server=https://some-url-for-llama-cpp": {
|
|
219
229
|
"type": "api_key",
|
|
220
|
-
"key": "
|
|
230
|
+
"key": "$LLAMA_API_KEY"
|
|
231
|
+
},
|
|
232
|
+
"llama-server=https://secure-server": {
|
|
233
|
+
"type": "api_key",
|
|
234
|
+
"key": "!cat ~/.secrets/llama-key"
|
|
235
|
+
},
|
|
236
|
+
"llama-server=https://braced-ref": {
|
|
237
|
+
"type": "api_key",
|
|
238
|
+
"key": "${API_KEY}"
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
For env ref formats, you can also store the variable value alongside the key using the `env` field:
|
|
244
|
+
|
|
245
|
+
```json
|
|
246
|
+
{
|
|
247
|
+
"llama-server=http://127.0.0.1:8080": {
|
|
248
|
+
"type": "api_key",
|
|
249
|
+
"key": "$MY_KEY",
|
|
250
|
+
"env": {
|
|
251
|
+
"MY_KEY": "sk-abc123"
|
|
252
|
+
}
|
|
221
253
|
}
|
|
222
254
|
}
|
|
223
255
|
```
|
|
@@ -246,6 +278,10 @@ llama-server --model path/to/model.gguf ...
|
|
|
246
278
|
|
|
247
279
|
> **Note:** The ik_llama.cpp fork is not legacy at all, but it uses an old way of describing models compared to llama.cpp.
|
|
248
280
|
|
|
281
|
+
- For llama-swap mode, point the extension at a running [llama-swap](https://github.com/mostlygeek/llama-swap) instance instead of a raw llama.cpp server. The extension auto-detects this mode via the `src` field in the server props response.
|
|
282
|
+
|
|
283
|
+
> **Note:** llama-swap support is basic — only model listing, status, load/unload, and capability detection are implemented.
|
|
284
|
+
|
|
249
285
|
The extension determines the context size as follows:
|
|
250
286
|
|
|
251
287
|
- A per-model `contextSize` override (see [Model Overrides](#model-overrides)) takes precedence over everything below
|
|
@@ -254,6 +290,7 @@ The extension determines the context size as follows:
|
|
|
254
290
|
- When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
|
|
255
291
|
- **Single mode** — reads `meta.n_ctx` from the `/v1/models` endpoint
|
|
256
292
|
- **Legacy mode** — reads `max_model_len` from `/v1/models`, falling back to `n_ctx` from `/props`
|
|
293
|
+
- **Llama-swap mode** — reads `meta.n_ctx` from the llama-swap server via `/v1/models`
|
|
257
294
|
- Falls back to `128000` if not available
|
|
258
295
|
|
|
259
296
|
### Commands
|
package/package.json
CHANGED
package/src/api/client.ts
CHANGED
|
@@ -81,6 +81,30 @@ export class ApiClient {
|
|
|
81
81
|
);
|
|
82
82
|
}
|
|
83
83
|
|
|
84
|
+
/**
|
|
85
|
+
* Makes a raw GET request that does not parse JSON. Useful for endpoints
|
|
86
|
+
* that return empty or non-JSON responses.
|
|
87
|
+
*
|
|
88
|
+
* @param endpoint The endpoint path to fetch
|
|
89
|
+
*/
|
|
90
|
+
async rawGet(endpoint: string): Promise<void> {
|
|
91
|
+
return this.do_request<void>(endpoint, "GET", undefined, false);
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Makes a raw POST request that does not parse JSON. Useful for endpoints
|
|
96
|
+
* that return empty or non-JSON responses.
|
|
97
|
+
*
|
|
98
|
+
* @param endpoint The endpoint path to post to
|
|
99
|
+
* @param body The optional request body
|
|
100
|
+
*/
|
|
101
|
+
async rawPost(
|
|
102
|
+
endpoint: string,
|
|
103
|
+
body?: Record<string, unknown>,
|
|
104
|
+
): Promise<void> {
|
|
105
|
+
return this.do_request<void>(endpoint, "POST", body, false);
|
|
106
|
+
}
|
|
107
|
+
|
|
84
108
|
/**
|
|
85
109
|
* Clears the entire cache.
|
|
86
110
|
*/
|
|
@@ -134,26 +158,54 @@ export class ApiClient {
|
|
|
134
158
|
}
|
|
135
159
|
|
|
136
160
|
/**
|
|
137
|
-
* Makes a raw
|
|
161
|
+
* Makes a raw request to the llama-server.
|
|
138
162
|
* This bypasses caching and deduplication.
|
|
139
163
|
*
|
|
140
|
-
* @param endpoint The endpoint path
|
|
141
|
-
* @
|
|
164
|
+
* @param endpoint The endpoint path
|
|
165
|
+
* @param method The HTTP method
|
|
166
|
+
* @param body The optional request body
|
|
167
|
+
* @param parseJson Whether to parse the response as JSON (default: true)
|
|
168
|
+
* @returns The parsed JSON response, or void if parseJson is false
|
|
142
169
|
* @throws ApiError with status 401 when the server rejects the request
|
|
143
170
|
* due to authentication (invalid/missing API key).
|
|
144
171
|
*/
|
|
145
|
-
private async
|
|
172
|
+
private async do_request<T>(
|
|
173
|
+
endpoint: string,
|
|
174
|
+
method: "GET" | "POST",
|
|
175
|
+
body?: Record<string, unknown>,
|
|
176
|
+
parseJson = true,
|
|
177
|
+
): Promise<T | void> {
|
|
146
178
|
const url = `${this.baseUrl}${endpoint}`;
|
|
147
179
|
|
|
148
180
|
const res = await fetch(url, {
|
|
149
|
-
|
|
181
|
+
method,
|
|
182
|
+
headers: {
|
|
183
|
+
...(method === "POST"
|
|
184
|
+
? { "Content-Type": "application/json" }
|
|
185
|
+
: undefined),
|
|
186
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
187
|
+
},
|
|
188
|
+
...(body ? { body: JSON.stringify(body) } : undefined),
|
|
150
189
|
});
|
|
151
190
|
|
|
152
191
|
if (res.status === 401) {
|
|
153
192
|
throw new ApiError("authentication", res.status);
|
|
154
193
|
}
|
|
155
194
|
|
|
156
|
-
return res.json();
|
|
195
|
+
return parseJson ? ((await res.json()) as T) : (undefined as T);
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Makes a raw GET request to the llama-server.
|
|
200
|
+
* This bypasses caching and deduplication.
|
|
201
|
+
*
|
|
202
|
+
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
203
|
+
* @returns The parsed JSON response from the server
|
|
204
|
+
* @throws ApiError with status 401 when the server rejects the request
|
|
205
|
+
* due to authentication (invalid/missing API key).
|
|
206
|
+
*/
|
|
207
|
+
private async do_get<T>(endpoint: string): Promise<T> {
|
|
208
|
+
return this.do_request<T>(endpoint, "GET") as Promise<T>;
|
|
157
209
|
}
|
|
158
210
|
|
|
159
211
|
/**
|
|
@@ -170,21 +222,6 @@ export class ApiClient {
|
|
|
170
222
|
endpoint: string,
|
|
171
223
|
body?: Record<string, unknown>,
|
|
172
224
|
): Promise<T> {
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
const res = await fetch(url, {
|
|
176
|
-
method: "POST",
|
|
177
|
-
headers: {
|
|
178
|
-
"Content-Type": "application/json",
|
|
179
|
-
Authorization: `Bearer ${this.apiKey}`,
|
|
180
|
-
},
|
|
181
|
-
body: body ? JSON.stringify(body) : undefined,
|
|
182
|
-
});
|
|
183
|
-
|
|
184
|
-
if (res.status === 401) {
|
|
185
|
-
throw new ApiError("authentication", res.status);
|
|
186
|
-
}
|
|
187
|
-
|
|
188
|
-
return res.json();
|
|
225
|
+
return this.do_request<T>(endpoint, "POST", body) as Promise<T>;
|
|
189
226
|
}
|
|
190
227
|
}
|
package/src/enums/mode.ts
CHANGED
|
@@ -13,6 +13,20 @@ export interface PropsEndpoint {
|
|
|
13
13
|
cors_proxy_enabled: boolean;
|
|
14
14
|
}
|
|
15
15
|
|
|
16
|
+
/**
|
|
17
|
+
* llama-swap returns a different shape for /props (without a model ID):
|
|
18
|
+
* an error response with a `src` that we can use as a trick to detect it.
|
|
19
|
+
*/
|
|
20
|
+
export interface LlamaSwapPropsError {
|
|
21
|
+
src: "llama-swap";
|
|
22
|
+
error: {
|
|
23
|
+
message: string;
|
|
24
|
+
type: string;
|
|
25
|
+
param: null;
|
|
26
|
+
code: string;
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
|
|
16
30
|
/**
|
|
17
31
|
* The structure of llama-server's /props?model=<id> endpoint
|
|
18
32
|
*/
|
|
@@ -217,10 +217,9 @@ export class ModelsMenu {
|
|
|
217
217
|
const base = [Action.INFO, Action.CANCEL];
|
|
218
218
|
|
|
219
219
|
const actions: Record<Status, Array<Action>> = {
|
|
220
|
-
[Status.LOADED]:
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
: [Action.SWITCH, ...base],
|
|
220
|
+
[Status.LOADED]: [Mode.ROUTER, Mode.LLAMASWAP].includes(model.mode)
|
|
221
|
+
? [Action.SWITCH, Action.UNLOAD, ...base]
|
|
222
|
+
: [Action.SWITCH, ...base],
|
|
224
223
|
[Status.LOADING]: [...base],
|
|
225
224
|
[Status.FAILED]: [Action.RETRY, ...base],
|
|
226
225
|
[Status.SLEEPING]:
|
package/src/managers/settings.ts
CHANGED
|
@@ -24,6 +24,7 @@ import {
|
|
|
24
24
|
} from "../interfaces/settings";
|
|
25
25
|
import type { SortBy } from "../interfaces/sortBy";
|
|
26
26
|
import { Server } from "../server";
|
|
27
|
+
import { CredentialResolver } from "../utils/credentialResolver";
|
|
27
28
|
import { SettingsStore } from "../utils/settingsStore";
|
|
28
29
|
import { UrlResolver } from "../utils/urlResolver";
|
|
29
30
|
|
|
@@ -47,6 +48,9 @@ export class LlamaSettingsManager {
|
|
|
47
48
|
}
|
|
48
49
|
}
|
|
49
50
|
|
|
51
|
+
/** Delegates credential key resolution (see `utils/credentialResolver`). */
|
|
52
|
+
private credentialResolver = new CredentialResolver();
|
|
53
|
+
|
|
50
54
|
/** Delegated multi-source URL resolution chain (see `utils/urlResolver`). */
|
|
51
55
|
private urlResolver = new UrlResolver({
|
|
52
56
|
getLlamaSettings: () => this.getLlamaSettings(),
|
|
@@ -149,12 +153,16 @@ export class LlamaSettingsManager {
|
|
|
149
153
|
|
|
150
154
|
/**
|
|
151
155
|
* Resolves API key for the provider ID using Pi's stored credentials.
|
|
156
|
+
* Delegates to `CredentialResolver` for key format handling.
|
|
152
157
|
*
|
|
158
|
+
* @param providerId The provider ID
|
|
153
159
|
* @returns The API key to use for the provider
|
|
154
160
|
*/
|
|
155
161
|
resolveApiKey(providerId: string): string {
|
|
156
162
|
const credential = readStoredCredential(providerId) as ApiKeyCredential;
|
|
157
|
-
|
|
163
|
+
if (!credential?.key) return API_KEY_PLACEHOLDER;
|
|
164
|
+
|
|
165
|
+
return this.credentialResolver.resolve(credential.key, credential.env);
|
|
158
166
|
}
|
|
159
167
|
|
|
160
168
|
/**
|
|
@@ -3,7 +3,7 @@ import { Mode } from "../enums/mode";
|
|
|
3
3
|
import { SingleModel } from "./singleModel";
|
|
4
4
|
|
|
5
5
|
export class LegacyModel extends SingleModel {
|
|
6
|
-
get mode(): Mode {
|
|
6
|
+
override get mode(): Mode {
|
|
7
7
|
return Mode.LEGACY;
|
|
8
8
|
}
|
|
9
9
|
|
|
@@ -13,7 +13,7 @@ export class LegacyModel extends SingleModel {
|
|
|
13
13
|
*
|
|
14
14
|
* @returns The context size
|
|
15
15
|
*/
|
|
16
|
-
protected async getContextSize(): Promise<number> {
|
|
16
|
+
protected override async getContextSize(): Promise<number> {
|
|
17
17
|
const props = await this.server.fetchModelProps(this.id);
|
|
18
18
|
const models = await this.server.fetchModels();
|
|
19
19
|
|
|
@@ -34,7 +34,7 @@ export class LegacyModel extends SingleModel {
|
|
|
34
34
|
*
|
|
35
35
|
* @returns An array of capabilities, as expected by Pi
|
|
36
36
|
*/
|
|
37
|
-
protected async getCapabilities(): Promise<("text" | "image")[]> {
|
|
37
|
+
protected override async getCapabilities(): Promise<("text" | "image")[]> {
|
|
38
38
|
try {
|
|
39
39
|
return await super.getCapabilities();
|
|
40
40
|
} catch {
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import { FALLBACK_CTX } from "../constants";
|
|
2
|
+
import { Mode } from "../enums/mode";
|
|
3
|
+
import { Status } from "../enums/status";
|
|
4
|
+
import { BaseModel } from "./baseModel";
|
|
5
|
+
|
|
6
|
+
export class LlamaSwapModel extends BaseModel {
|
|
7
|
+
get mode(): Mode {
|
|
8
|
+
return Mode.LLAMASWAP;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
override get name(): string {
|
|
12
|
+
// llama-swap adds a `llamaswap` key inside `meta` with aliases;
|
|
13
|
+
// fall back to the standard aliases array, then to id
|
|
14
|
+
return (
|
|
15
|
+
this.model.meta?.llamaswap?.aliases?.[0] ??
|
|
16
|
+
this.model.aliases?.[0] ??
|
|
17
|
+
this.model.id
|
|
18
|
+
);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Retrieves the context size of the model.
|
|
23
|
+
* An override's `contextSize` takes precedence; otherwise reads
|
|
24
|
+
* `meta.n_ctx` from the llama-swap `/v1/models` response,
|
|
25
|
+
* falling back to `FALLBACK_CTX`.
|
|
26
|
+
*
|
|
27
|
+
* @returns The context size
|
|
28
|
+
*/
|
|
29
|
+
protected override async getContextSize(): Promise<number> {
|
|
30
|
+
const overridden = this.server.findOverrideForModel(this.id)?.contextSize;
|
|
31
|
+
if (overridden && overridden > 0) return overridden;
|
|
32
|
+
|
|
33
|
+
const { data } = await this.server.fetchModels();
|
|
34
|
+
const model = data.find((m) => m.id === this.id);
|
|
35
|
+
|
|
36
|
+
return model?.meta?.n_ctx ?? FALLBACK_CTX;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Detects the capabilities of the model.
|
|
41
|
+
* An override's `capabilities` fully replaces detection.
|
|
42
|
+
*
|
|
43
|
+
* @returns An array of capabilities, as expected by Pi
|
|
44
|
+
*/
|
|
45
|
+
override async getCapabilities(): Promise<("text" | "image")[]> {
|
|
46
|
+
const overridden = this.server.findOverrideForModel(this.id)?.capabilities;
|
|
47
|
+
if (overridden) return overridden;
|
|
48
|
+
|
|
49
|
+
const { data } = await this.server.fetchModels();
|
|
50
|
+
const model = data.find((m) => m.id === this.id);
|
|
51
|
+
|
|
52
|
+
const inputModalities = model?.architecture?.input_modalities ?? [];
|
|
53
|
+
return inputModalities.includes("image") ? ["text", "image"] : ["text"];
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Detects the load status of a model.
|
|
58
|
+
* For simplicity, we'll only handle loaded/unloaded
|
|
59
|
+
*
|
|
60
|
+
* @returns The current {@link Status}
|
|
61
|
+
*/
|
|
62
|
+
override async getStatus(): Promise<Status> {
|
|
63
|
+
const { data } = await this.server.fetchModels();
|
|
64
|
+
const model = data.find((m) => m.id === this.id);
|
|
65
|
+
|
|
66
|
+
return model?.status?.value === "loaded" ? Status.LOADED : Status.UNLOADED;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Loads the model in the llama-swap server
|
|
71
|
+
*/
|
|
72
|
+
override async load(): Promise<void> {
|
|
73
|
+
const status = await this.getStatus();
|
|
74
|
+
if (status === Status.LOADED) return;
|
|
75
|
+
|
|
76
|
+
try {
|
|
77
|
+
await this.server.llamaSwapLoad(this.id);
|
|
78
|
+
} catch (err) {
|
|
79
|
+
console.warn({ err });
|
|
80
|
+
throw new Error(`Model loading failed: ${this.id}`);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Unloads the model in the llama-swap server
|
|
86
|
+
*/
|
|
87
|
+
override async unload(): Promise<void> {
|
|
88
|
+
await this.server.llamaSwapUnload(this.id);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
@@ -26,7 +26,7 @@ export class RouterModel extends BaseModel {
|
|
|
26
26
|
*
|
|
27
27
|
* In exchange, it will allow unloaded models to be correctly shown as "unloaded".
|
|
28
28
|
*/
|
|
29
|
-
protected async pollStatus(startTime = Date.now()): Promise<void> {
|
|
29
|
+
protected override async pollStatus(startTime = Date.now()): Promise<void> {
|
|
30
30
|
let elapsed = 0;
|
|
31
31
|
const limit = 5000;
|
|
32
32
|
|
|
@@ -52,7 +52,7 @@ export class RouterModel extends BaseModel {
|
|
|
52
52
|
*
|
|
53
53
|
* @returns The context size in tokens
|
|
54
54
|
*/
|
|
55
|
-
protected async getContextSize(): Promise<number> {
|
|
55
|
+
protected override async getContextSize(): Promise<number> {
|
|
56
56
|
// We can get a more accurate context size if the model is already loaded
|
|
57
57
|
if ((await this.getStatus()) === Status.LOADED) {
|
|
58
58
|
return super.getContextSize();
|
|
@@ -2,11 +2,11 @@ import { Mode } from "../enums/mode";
|
|
|
2
2
|
import { BaseModel } from "./baseModel";
|
|
3
3
|
|
|
4
4
|
export class SingleModel extends BaseModel {
|
|
5
|
-
get mode(): Mode {
|
|
5
|
+
override get mode(): Mode {
|
|
6
6
|
return Mode.SINGLE;
|
|
7
7
|
}
|
|
8
8
|
|
|
9
|
-
protected async getCapabilities(): Promise<("text" | "image")[]> {
|
|
9
|
+
protected override async getCapabilities(): Promise<("text" | "image")[]> {
|
|
10
10
|
try {
|
|
11
11
|
return await super.getCapabilities();
|
|
12
12
|
} catch {
|
package/src/server.ts
CHANGED
|
@@ -8,6 +8,7 @@ import { Mode } from "./enums/mode";
|
|
|
8
8
|
import { ServerStatus } from "./enums/serverStatus";
|
|
9
9
|
import { ModelsEndpoint } from "./interfaces/endpoints/models";
|
|
10
10
|
import {
|
|
11
|
+
LlamaSwapPropsError,
|
|
11
12
|
PropsEndpoint,
|
|
12
13
|
PropsModelEndpoint,
|
|
13
14
|
} from "./interfaces/endpoints/props";
|
|
@@ -16,6 +17,7 @@ import type { ModelOverride } from "./interfaces/settings";
|
|
|
16
17
|
import type { LlamaSettingsManager } from "./managers/settings";
|
|
17
18
|
import { BaseModel } from "./models/baseModel";
|
|
18
19
|
import { LegacyModel } from "./models/legacyModel";
|
|
20
|
+
import { LlamaSwapModel } from "./models/llamaSwapModel";
|
|
19
21
|
import { RouterModel } from "./models/routerModel";
|
|
20
22
|
import { SingleModel } from "./models/singleModel";
|
|
21
23
|
import { SSEManager } from "./sse/manager";
|
|
@@ -147,6 +149,7 @@ export class Server {
|
|
|
147
149
|
[Mode.ROUTER]: RouterModel,
|
|
148
150
|
[Mode.LEGACY]: LegacyModel,
|
|
149
151
|
[Mode.SINGLE]: SingleModel,
|
|
152
|
+
[Mode.LLAMASWAP]: LlamaSwapModel,
|
|
150
153
|
}[mode];
|
|
151
154
|
|
|
152
155
|
const models: BaseModel[] = data.map((m) => new modelCtor(m, this));
|
|
@@ -163,9 +166,12 @@ export class Server {
|
|
|
163
166
|
* @returns The detected mode
|
|
164
167
|
*/
|
|
165
168
|
private async detectServerMode(data: ModelsEndpoint["data"]): Promise<Mode> {
|
|
166
|
-
const
|
|
169
|
+
const serverProps = await this.fetchServerProps();
|
|
167
170
|
|
|
168
|
-
if (
|
|
171
|
+
if ("src" in serverProps && serverProps.src === "llama-swap")
|
|
172
|
+
return Mode.LLAMASWAP;
|
|
173
|
+
if ("role" in serverProps && serverProps.role === "router")
|
|
174
|
+
return Mode.ROUTER;
|
|
169
175
|
if ("max_model_len" in data[0]) return Mode.LEGACY;
|
|
170
176
|
return Mode.SINGLE;
|
|
171
177
|
}
|
|
@@ -197,12 +203,16 @@ export class Server {
|
|
|
197
203
|
}
|
|
198
204
|
|
|
199
205
|
/**
|
|
200
|
-
* Fetches general properties of the server
|
|
206
|
+
* Fetches general properties of the server.
|
|
207
|
+
* llama-server returns {@link PropsEndpoint}; llama-swap returns
|
|
208
|
+
* {@link LlamaSwapPropsError} (an error response with a `src` discriminator).
|
|
201
209
|
*
|
|
202
|
-
* @return The properties
|
|
210
|
+
* @return The server properties
|
|
203
211
|
*/
|
|
204
|
-
async fetchServerProps(): Promise<PropsEndpoint> {
|
|
205
|
-
return await this.apiClient.get<PropsEndpoint>(
|
|
212
|
+
async fetchServerProps(): Promise<PropsEndpoint | LlamaSwapPropsError> {
|
|
213
|
+
return await this.apiClient.get<PropsEndpoint | LlamaSwapPropsError>(
|
|
214
|
+
"/props?autoload=false",
|
|
215
|
+
);
|
|
206
216
|
}
|
|
207
217
|
|
|
208
218
|
/**
|
|
@@ -258,4 +268,24 @@ export class Server {
|
|
|
258
268
|
model,
|
|
259
269
|
});
|
|
260
270
|
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Loads a model on a llama-swap server (GET /upstream/{id}).
|
|
274
|
+
*
|
|
275
|
+
* @param model The model ID to load
|
|
276
|
+
*/
|
|
277
|
+
async llamaSwapLoad(model: string): Promise<void> {
|
|
278
|
+
this.apiClient.clearCache();
|
|
279
|
+
await this.apiClient.rawGet(`/upstream/${model}`);
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/**
|
|
283
|
+
* Unloads a model on a llama-swap server (POST /api/models/unload/{id}).
|
|
284
|
+
*
|
|
285
|
+
* @param model The model ID to unload
|
|
286
|
+
*/
|
|
287
|
+
async llamaSwapUnload(model: string): Promise<void> {
|
|
288
|
+
this.apiClient.clearCache();
|
|
289
|
+
await this.apiClient.rawPost(`/api/models/unload/${model}`);
|
|
290
|
+
}
|
|
261
291
|
}
|
package/src/ui/dialog/confirm.ts
CHANGED
package/src/ui/dialog/input.ts
CHANGED
|
@@ -31,13 +31,13 @@ export class OverrideEntryListEditor extends ListEditor<OverrideSettingsListOpti
|
|
|
31
31
|
|
|
32
32
|
/** Esc steps back to the server list, not out of the whole dialog —
|
|
33
33
|
* `options.done` is the top-level close (shared options object). */
|
|
34
|
-
protected close(): void {
|
|
34
|
+
protected override close(): void {
|
|
35
35
|
this.done();
|
|
36
36
|
}
|
|
37
37
|
|
|
38
38
|
// -- abstract hooks -------------------------------------------------------
|
|
39
39
|
|
|
40
|
-
protected buildSettingsList(): SettingsList {
|
|
40
|
+
protected override buildSettingsList(): SettingsList {
|
|
41
41
|
const builder = new OverrideItemBuilder(this.dialogs);
|
|
42
42
|
const items: SettingItem[] = this.entries().map((entry, i) => ({
|
|
43
43
|
id: `entry-${i}`,
|
|
@@ -103,7 +103,7 @@ export class OverrideEntryListEditor extends ListEditor<OverrideSettingsListOpti
|
|
|
103
103
|
);
|
|
104
104
|
}
|
|
105
105
|
|
|
106
|
-
protected beginAdd(): void {
|
|
106
|
+
protected override beginAdd(): void {
|
|
107
107
|
this.openDialog(
|
|
108
108
|
this.dialogs.input({
|
|
109
109
|
title: TITLES.addOverride,
|
|
@@ -119,7 +119,7 @@ export class OverrideEntryListEditor extends ListEditor<OverrideSettingsListOpti
|
|
|
119
119
|
);
|
|
120
120
|
}
|
|
121
121
|
|
|
122
|
-
protected deleteSelected(): void {
|
|
122
|
+
protected override deleteSelected(): void {
|
|
123
123
|
const idx = this.selectedIndex;
|
|
124
124
|
const next = this.removeEntry(idx);
|
|
125
125
|
void this.persistSnapshot(next, () => {
|
|
@@ -129,21 +129,21 @@ export class OverrideEntryListEditor extends ListEditor<OverrideSettingsListOpti
|
|
|
129
129
|
});
|
|
130
130
|
}
|
|
131
131
|
|
|
132
|
-
protected readonly emptyHintKey = "emptyOverrideEntries" as const;
|
|
132
|
+
protected override readonly emptyHintKey = "emptyOverrideEntries" as const;
|
|
133
133
|
|
|
134
|
-
protected getCount(): number {
|
|
134
|
+
protected override getCount(): number {
|
|
135
135
|
return this.entries().length;
|
|
136
136
|
}
|
|
137
137
|
|
|
138
|
-
protected getRowId(index: number): string {
|
|
138
|
+
protected override getRowId(index: number): string {
|
|
139
139
|
return `entry-${index}`;
|
|
140
140
|
}
|
|
141
141
|
|
|
142
|
-
protected getRowLabel(index: number): string {
|
|
142
|
+
protected override getRowLabel(index: number): string {
|
|
143
143
|
return this.entries()[index]?.pattern ?? "";
|
|
144
144
|
}
|
|
145
145
|
|
|
146
|
-
protected get deleteTitle(): string {
|
|
146
|
+
protected override get deleteTitle(): string {
|
|
147
147
|
return TITLES.deleteOverride;
|
|
148
148
|
}
|
|
149
149
|
|
|
@@ -12,7 +12,7 @@ export class OverrideItemBuilder extends ItemBuilder<
|
|
|
12
12
|
OverrideEntry,
|
|
13
13
|
OverrideField
|
|
14
14
|
> {
|
|
15
|
-
protected get fields(): readonly OverrideField[] {
|
|
15
|
+
protected override get fields(): readonly OverrideField[] {
|
|
16
16
|
return OverrideFields.all;
|
|
17
17
|
}
|
|
18
18
|
|
|
@@ -21,7 +21,10 @@ export class OverrideItemBuilder extends ItemBuilder<
|
|
|
21
21
|
* an `InputDialog` submenu (validating against the field definition)
|
|
22
22
|
* for "input" fields.
|
|
23
23
|
*/
|
|
24
|
-
protected decorate(
|
|
24
|
+
protected override decorate(
|
|
25
|
+
def: OverrideField,
|
|
26
|
+
base: SettingItem,
|
|
27
|
+
): SettingItem {
|
|
25
28
|
if (def.type === "finite") {
|
|
26
29
|
return { ...base, values: [...(def.options ?? [])] };
|
|
27
30
|
}
|
|
@@ -26,7 +26,7 @@ class UrlField extends ServerField {
|
|
|
26
26
|
readonly field = FIELDS.serverUrl;
|
|
27
27
|
readonly validate = ServerUrl.parse;
|
|
28
28
|
|
|
29
|
-
currentValue(server: LlamaServer): string {
|
|
29
|
+
override currentValue(server: LlamaServer): string {
|
|
30
30
|
return server.url;
|
|
31
31
|
}
|
|
32
32
|
|
|
@@ -51,7 +51,7 @@ class OptionalTextField extends ServerField {
|
|
|
51
51
|
this.field = field;
|
|
52
52
|
}
|
|
53
53
|
|
|
54
|
-
currentValue(server: LlamaServer): string {
|
|
54
|
+
override currentValue(server: LlamaServer): string {
|
|
55
55
|
return server[this.id] ?? "";
|
|
56
56
|
}
|
|
57
57
|
|
|
@@ -10,7 +10,7 @@ import { ServerDisplay } from "./utils";
|
|
|
10
10
|
* Builds `SettingItem` objects for server rows and their field-edit submenus.
|
|
11
11
|
*/
|
|
12
12
|
export class ServerItemBuilder extends ItemBuilder<LlamaServer, ServerField> {
|
|
13
|
-
protected get fields(): readonly ServerField[] {
|
|
13
|
+
protected override get fields(): readonly ServerField[] {
|
|
14
14
|
return ServerFields.all;
|
|
15
15
|
}
|
|
16
16
|
|
|
@@ -18,7 +18,10 @@ export class ServerItemBuilder extends ItemBuilder<LlamaServer, ServerField> {
|
|
|
18
18
|
* Adds the field-edit submenu: an `InputDialog` validating against the
|
|
19
19
|
* field definition.
|
|
20
20
|
*/
|
|
21
|
-
protected decorate(
|
|
21
|
+
protected override decorate(
|
|
22
|
+
def: ServerField,
|
|
23
|
+
base: SettingItem,
|
|
24
|
+
): SettingItem {
|
|
22
25
|
return {
|
|
23
26
|
...base,
|
|
24
27
|
submenu: this.dialogs.inputSubmenu(
|