pi-llama-cpp 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -18
- package/package.json +3 -3
- package/src/api/client.ts +25 -0
- package/src/constants.ts +2 -2
- package/src/enums/status.ts +0 -1
- package/src/index.ts +2 -1
- package/src/interfaces/endpoints/models.ts +1 -1
- package/src/interfaces/settings.ts +7 -1
- package/src/interfaces/sortBy.ts +4 -0
- package/src/managers/command/models.ts +254 -0
- package/src/managers/command.ts +34 -461
- package/src/managers/events.ts +2 -2
- package/src/managers/server.ts +34 -18
- package/src/managers/settings.ts +26 -90
- package/src/models/baseModel.ts +27 -10
- package/src/models/legacyModel.ts +2 -2
- package/src/models/routerModel.ts +2 -2
- package/src/models/singleModel.ts +1 -1
- package/src/server.ts +17 -38
- package/src/sse/client.ts +113 -59
- package/src/sse/fetch.ts +43 -0
- package/src/sse/manager.ts +9 -27
- package/src/sse/types.ts +0 -4
- package/src/ui/dialog/base.ts +118 -0
- package/src/ui/dialog/confirm.ts +45 -0
- package/src/ui/dialog/factory.ts +111 -0
- package/src/ui/dialog/input.ts +63 -0
- package/src/ui/dialog/options.ts +23 -0
- package/src/ui/editors/editorOptions.ts +50 -0
- package/src/ui/editors/itemBuilder.ts +43 -0
- package/src/ui/editors/listEditor.ts +300 -0
- package/src/ui/editors/override/entry.ts +24 -0
- package/src/ui/editors/override/entryEditor.ts +248 -0
- package/src/ui/editors/override/fields/base.ts +53 -0
- package/src/ui/editors/override/fields/capabilities.ts +36 -0
- package/src/ui/editors/override/fields/cost.ts +64 -0
- package/src/ui/editors/override/fields/index.ts +68 -0
- package/src/ui/editors/override/fields/numeric.ts +55 -0
- package/src/ui/editors/override/fields/pattern.ts +24 -0
- package/src/ui/editors/override/fields/reasoning.ts +33 -0
- package/src/ui/editors/override/itemBuilder.ts +39 -0
- package/src/ui/editors/override/overrideList.ts +127 -0
- package/src/ui/editors/server/builder.ts +60 -0
- package/src/ui/editors/server/fields.ts +84 -0
- package/src/ui/editors/server/itemBuilder.ts +64 -0
- package/src/ui/editors/server/serverEditor.ts +197 -0
- package/src/ui/editors/server/utils.ts +95 -0
- package/src/ui/editors/server/wizard.ts +110 -0
- package/src/ui/editors/settingField.ts +37 -0
- package/src/ui/editors/settingsListFactory.ts +33 -0
- package/src/ui/settings/index.ts +248 -0
- package/src/ui/strings.ts +25 -4
- package/src/utils/health.ts +2 -1
- package/src/utils/serverIds.ts +21 -0
- package/src/utils/settingsStore.ts +1 -1
- package/src/utils/urlResolver.ts +129 -0
- package/src/utils/urls.ts +33 -13
- package/tests/{commandManager.test.ts → command/commandManager.test.ts} +88 -12
- package/tests/{events.test.ts → events/events.test.ts} +6 -6
- package/tests/mocks.ts +2 -0
- package/tests/models/legacyModel.test.ts +103 -0
- package/tests/{routerModel.test.ts → models/routerModel.test.ts} +71 -72
- package/tests/{singleModel.test.ts → models/singleModel.test.ts} +17 -17
- package/tests/{health.test.ts → server/health.test.ts} +2 -2
- package/tests/{server.test.ts → server/server.test.ts} +5 -17
- package/tests/{serverManager.test.ts → server/serverManager.test.ts} +4 -4
- package/tests/{settings.test.ts → settings/settings.test.ts} +184 -141
- package/tests/{settingsStore.test.ts → settings/settingsStore.test.ts} +1 -1
- package/tests/{sseManager.test.ts → sse/sseManager.test.ts} +8 -26
- package/tests/{dialog.test.ts → ui/dialog.test.ts} +56 -4
- package/tests/{overrides.test.ts → ui/overrides.test.ts} +101 -111
- package/src/ui/dialog.ts +0 -290
- package/src/ui/overrideEntryEditor.ts +0 -119
- package/src/ui/overrideSettingsList.ts +0 -710
- package/src/ui/serverListEditor.ts +0 -59
- package/src/ui/serverSettingsList.ts +0 -513
- package/tests/legacyModel.test.ts +0 -97
package/src/managers/settings.ts
CHANGED
|
@@ -9,23 +9,23 @@ import { join } from "node:path";
|
|
|
9
9
|
import {
|
|
10
10
|
API_KEY_PLACEHOLDER,
|
|
11
11
|
AUTOLOAD_ON_MESSAGE,
|
|
12
|
-
LLAMA_SERVER_URL,
|
|
13
12
|
POLLING_TIMEOUT,
|
|
14
13
|
REACT_TO_MODEL_SELECT,
|
|
15
14
|
SERVER_TIMEOUT,
|
|
16
15
|
SETTINGS_KEY,
|
|
16
|
+
SHOW_SERVER_URLS,
|
|
17
17
|
SORT_BY,
|
|
18
18
|
THINKING_BUDGETS,
|
|
19
|
-
type SortBy,
|
|
20
19
|
} from "../constants";
|
|
21
20
|
import {
|
|
22
21
|
LlamaServer,
|
|
23
22
|
LlamaSettings,
|
|
24
23
|
ModelOverride,
|
|
25
24
|
} from "../interfaces/settings";
|
|
25
|
+
import type { SortBy } from "../interfaces/sortBy";
|
|
26
26
|
import { Server } from "../server";
|
|
27
27
|
import { SettingsStore } from "../utils/settingsStore";
|
|
28
|
-
import {
|
|
28
|
+
import { UrlResolver } from "../utils/urlResolver";
|
|
29
29
|
|
|
30
30
|
export class LlamaSettingsManager {
|
|
31
31
|
private settingsManager = SettingsManager.create(process.cwd());
|
|
@@ -47,16 +47,17 @@ export class LlamaSettingsManager {
|
|
|
47
47
|
}
|
|
48
48
|
}
|
|
49
49
|
|
|
50
|
-
/**
|
|
51
|
-
private
|
|
50
|
+
/** Delegated multi-source URL resolution chain (see `utils/urlResolver`). */
|
|
51
|
+
private urlResolver = new UrlResolver({
|
|
52
|
+
getLlamaSettings: () => this.getLlamaSettings(),
|
|
53
|
+
getMergedSettings: () => this.getMergedSettings(),
|
|
54
|
+
});
|
|
52
55
|
|
|
53
56
|
/**
|
|
54
57
|
* Returns and clears warnings collected during URL resolution.
|
|
55
58
|
*/
|
|
56
59
|
takeWarnings(): string[] {
|
|
57
|
-
|
|
58
|
-
this.warnings.length = 0;
|
|
59
|
-
return warnings;
|
|
60
|
+
return this.urlResolver.takeWarnings();
|
|
60
61
|
}
|
|
61
62
|
|
|
62
63
|
/**
|
|
@@ -75,7 +76,7 @@ export class LlamaSettingsManager {
|
|
|
75
76
|
* Convenience method for the `llamaSettings` key.
|
|
76
77
|
* Reloads settings from disk before reading.
|
|
77
78
|
*/
|
|
78
|
-
async getLlamaSettings(): Promise<LlamaSettings> {
|
|
79
|
+
private async getLlamaSettings(): Promise<LlamaSettings> {
|
|
79
80
|
return (await this.getMergedSettings())[SETTINGS_KEY] ?? {};
|
|
80
81
|
}
|
|
81
82
|
|
|
@@ -88,89 +89,13 @@ export class LlamaSettingsManager {
|
|
|
88
89
|
}
|
|
89
90
|
|
|
90
91
|
/**
|
|
91
|
-
* Resolves the server URLs to use
|
|
92
|
-
*
|
|
93
|
-
* - `LLAMA_SERVER_URL` env variable
|
|
94
|
-
* - `llamaSettings` key (current - project, then global)
|
|
95
|
-
* - `llamaServerUrl` key (legacy - project, then global)
|
|
96
|
-
* - Default URL
|
|
92
|
+
* Resolves the server URLs to use. Delegates to the URL resolver chain
|
|
93
|
+
* (env → settings → legacy → default, see `utils/urlResolver`).
|
|
97
94
|
*
|
|
98
95
|
* @returns The list of URLs to use
|
|
99
96
|
*/
|
|
100
97
|
async resolveUrls(): Promise<string[]> {
|
|
101
|
-
|
|
102
|
-
if (response.length > 0) return response;
|
|
103
|
-
|
|
104
|
-
response = await this.resolveServerUrls();
|
|
105
|
-
if (response.length > 0) return response;
|
|
106
|
-
|
|
107
|
-
response = await this.resolveLegacyUrls();
|
|
108
|
-
if (response.length > 0) return response;
|
|
109
|
-
|
|
110
|
-
return [LLAMA_SERVER_URL];
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
/**
|
|
114
|
-
* Resolves the llama-server URLs from the environment variable.
|
|
115
|
-
*
|
|
116
|
-
* @returns A list of detected URLs
|
|
117
|
-
*/
|
|
118
|
-
private resolveEnvUrls(): string[] {
|
|
119
|
-
const raw = process.env.LLAMA_SERVER_URL;
|
|
120
|
-
if (!raw) return [];
|
|
121
|
-
|
|
122
|
-
return this.parseUrls(raw);
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
/**
|
|
126
|
-
* Resolves the llama-server URLs from `llamaSettings.servers`.
|
|
127
|
-
* Settings are merged, prioritizing project over global settings.
|
|
128
|
-
* Reloads settings from disk before reading.
|
|
129
|
-
*
|
|
130
|
-
* @returns A list of detected URLs
|
|
131
|
-
*/
|
|
132
|
-
private async resolveServerUrls(): Promise<string[]> {
|
|
133
|
-
const { servers = [] } = await this.getLlamaSettings();
|
|
134
|
-
return servers.map((s) => this.parseUrls(s.url)).flat();
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
/**
|
|
138
|
-
* Resolves the llama-server URLs from `llamaServerUrl` legacy key.
|
|
139
|
-
* Settings are merged, prioritizing project over global settings.
|
|
140
|
-
* Reloads settings from disk before reading.
|
|
141
|
-
*
|
|
142
|
-
* @returns A list of detected URLs
|
|
143
|
-
*/
|
|
144
|
-
private async resolveLegacyUrls(): Promise<string[]> {
|
|
145
|
-
const { llamaServerUrl = null } = await this.getMergedSettings();
|
|
146
|
-
if (!llamaServerUrl) return [];
|
|
147
|
-
|
|
148
|
-
return this.parseUrls(llamaServerUrl);
|
|
149
|
-
}
|
|
150
|
-
|
|
151
|
-
/**
|
|
152
|
-
* Parses a raw URL string into an array of cleaned URLs.
|
|
153
|
-
* Splits on semicolons, trims whitespace, filters empty strings, strips
|
|
154
|
-
* trailing slashes, and drops entries without an http(s) scheme —
|
|
155
|
-
* collecting a warning for each dropped entry (same validation the
|
|
156
|
-
* `/models servers` editor applies).
|
|
157
|
-
*
|
|
158
|
-
* @returns A sanitized URL
|
|
159
|
-
*/
|
|
160
|
-
private parseUrls(raw: string): string[] {
|
|
161
|
-
return raw
|
|
162
|
-
.split(";")
|
|
163
|
-
.map(normalizeUrl)
|
|
164
|
-
.filter((u) => {
|
|
165
|
-
if (u.length === 0) return false;
|
|
166
|
-
if (!isValidServerUrl(u)) {
|
|
167
|
-
this.warnings.push(
|
|
168
|
-
`Ignoring invalid server URL '${u}' (needs http(s)://)`,
|
|
169
|
-
);
|
|
170
|
-
return false;
|
|
171
|
-
}
|
|
172
|
-
return true;
|
|
173
|
-
});
|
|
98
|
+
return this.urlResolver.resolveUrls();
|
|
174
99
|
}
|
|
175
100
|
|
|
176
101
|
/**
|
|
@@ -302,6 +227,15 @@ export class LlamaSettingsManager {
|
|
|
302
227
|
return (await this.getLlamaSettings()).sortBy ?? SORT_BY;
|
|
303
228
|
}
|
|
304
229
|
|
|
230
|
+
/**
|
|
231
|
+
* Resolves whether to show server URLs in the /models model list.
|
|
232
|
+
*
|
|
233
|
+
* @returns `true` if server URLs should be shown
|
|
234
|
+
*/
|
|
235
|
+
async resolveShowServerUrls(): Promise<boolean> {
|
|
236
|
+
return (await this.getLlamaSettings()).showServerUrls ?? SHOW_SERVER_URLS;
|
|
237
|
+
}
|
|
238
|
+
|
|
305
239
|
/**
|
|
306
240
|
* Persists one llamaSettings field to settings and reloads the in-memory
|
|
307
241
|
* settings so resolvers see the change immediately.
|
|
@@ -338,6 +272,8 @@ export class LlamaSettingsManager {
|
|
|
338
272
|
}
|
|
339
273
|
|
|
340
274
|
/**
|
|
341
|
-
*
|
|
275
|
+
* Creates a new LlamaSettingsManager instance.
|
|
342
276
|
*/
|
|
343
|
-
export
|
|
277
|
+
export function createSettingsManager(): LlamaSettingsManager {
|
|
278
|
+
return new LlamaSettingsManager();
|
|
279
|
+
}
|
package/src/models/baseModel.ts
CHANGED
|
@@ -23,7 +23,6 @@ export abstract class BaseModel {
|
|
|
23
23
|
[Status.FAILED]: "🔴",
|
|
24
24
|
[Status.SLEEPING]: "🔵",
|
|
25
25
|
[Status.UNLOADED]: "⚪",
|
|
26
|
-
[Status.UNAUTHORIZED]: "⛔",
|
|
27
26
|
};
|
|
28
27
|
|
|
29
28
|
abstract get mode(): Mode;
|
|
@@ -72,7 +71,7 @@ export abstract class BaseModel {
|
|
|
72
71
|
*
|
|
73
72
|
* @returns An array of capabilities, as expected by Pi
|
|
74
73
|
*/
|
|
75
|
-
async getCapabilities(): Promise<("text" | "image")[]> {
|
|
74
|
+
protected async getCapabilities(): Promise<("text" | "image")[]> {
|
|
76
75
|
const overridden = this.server.findOverrideForModel(this.id)?.capabilities;
|
|
77
76
|
if (overridden) return overridden;
|
|
78
77
|
|
|
@@ -108,7 +107,6 @@ export abstract class BaseModel {
|
|
|
108
107
|
|
|
109
108
|
if (is_sleeping) return Status.SLEEPING;
|
|
110
109
|
if (!error) return Status.LOADED;
|
|
111
|
-
if (error.code === 401) return Status.UNAUTHORIZED;
|
|
112
110
|
if (error.code === 503) return Status.LOADING;
|
|
113
111
|
if (error.code === 400 && error.message === "model is not loaded")
|
|
114
112
|
return Status.UNLOADED;
|
|
@@ -128,14 +126,13 @@ export abstract class BaseModel {
|
|
|
128
126
|
*
|
|
129
127
|
* @returns The context size in tokens
|
|
130
128
|
*/
|
|
131
|
-
async getContextSize(): Promise<number> {
|
|
129
|
+
protected async getContextSize(): Promise<number> {
|
|
132
130
|
const overridden = this.server.findOverrideForModel(this.id)?.contextSize;
|
|
133
131
|
if (overridden && overridden > 0) return overridden;
|
|
134
132
|
|
|
135
133
|
try {
|
|
136
134
|
const { data } = await this.server.fetchModels();
|
|
137
|
-
const
|
|
138
|
-
|
|
135
|
+
const n_ctx = data.find((m) => m.id === this.id)?.meta?.n_ctx;
|
|
139
136
|
return n_ctx ?? FALLBACK_CTX;
|
|
140
137
|
} catch {
|
|
141
138
|
return FALLBACK_CTX;
|
|
@@ -187,6 +184,10 @@ export abstract class BaseModel {
|
|
|
187
184
|
cacheWrite: userCost.cacheWrite ?? 0,
|
|
188
185
|
};
|
|
189
186
|
|
|
187
|
+
const input = await this.getCapabilities();
|
|
188
|
+
const contextWindow = await this.getContextSize();
|
|
189
|
+
const maxTokens = this.getMaxTokens(contextWindow);
|
|
190
|
+
|
|
190
191
|
const response: ProviderModelConfig = {
|
|
191
192
|
id: this.id,
|
|
192
193
|
name: this.name,
|
|
@@ -199,10 +200,10 @@ export abstract class BaseModel {
|
|
|
199
200
|
xhigh: "xhigh",
|
|
200
201
|
max: "max",
|
|
201
202
|
},
|
|
202
|
-
input
|
|
203
|
-
contextWindow
|
|
203
|
+
input,
|
|
204
|
+
contextWindow,
|
|
204
205
|
cost,
|
|
205
|
-
maxTokens
|
|
206
|
+
maxTokens,
|
|
206
207
|
};
|
|
207
208
|
|
|
208
209
|
// Add compat if the override specifies it
|
|
@@ -213,6 +214,22 @@ export abstract class BaseModel {
|
|
|
213
214
|
return response;
|
|
214
215
|
}
|
|
215
216
|
|
|
217
|
+
/**
|
|
218
|
+
* Gets the maximum number of tokens the model can generate.
|
|
219
|
+
*
|
|
220
|
+
* An override's `maxTokens` (when set and `> 0`) replaces the value;
|
|
221
|
+
* otherwise it falls back to the context size — a model cannot generate
|
|
222
|
+
* more tokens than its context holds. A stored `0` behaves as if the
|
|
223
|
+
* key were absent.
|
|
224
|
+
*
|
|
225
|
+
* @param contextSize - The already-resolved context size, used as fallback
|
|
226
|
+
* @returns The maximum number of tokens
|
|
227
|
+
*/
|
|
228
|
+
protected getMaxTokens(contextSize: number): number {
|
|
229
|
+
const overridden = this.server.findOverrideForModel(this.id)?.maxTokens;
|
|
230
|
+
return overridden && overridden > 0 ? overridden : contextSize;
|
|
231
|
+
}
|
|
232
|
+
|
|
216
233
|
/**
|
|
217
234
|
* Loads the model in llama-server.
|
|
218
235
|
* Uses SSE status events when available, falling back to polling.
|
|
@@ -269,7 +286,7 @@ export abstract class BaseModel {
|
|
|
269
286
|
* @param timeout The maximum amount of ms before timeout. Defaults to server's pollingTimeout
|
|
270
287
|
* @param interval The polling interval. Defaults to POLLING_INTERVAL
|
|
271
288
|
*/
|
|
272
|
-
async pollStatus(
|
|
289
|
+
protected async pollStatus(
|
|
273
290
|
startTime: number = Date.now(),
|
|
274
291
|
timeout?: number,
|
|
275
292
|
interval: number = POLLING_INTERVAL,
|
|
@@ -13,7 +13,7 @@ export class LegacyModel extends SingleModel {
|
|
|
13
13
|
*
|
|
14
14
|
* @returns The context size
|
|
15
15
|
*/
|
|
16
|
-
async getContextSize(): Promise<number> {
|
|
16
|
+
protected async getContextSize(): Promise<number> {
|
|
17
17
|
const props = await this.server.fetchModelProps(this.id);
|
|
18
18
|
const models = await this.server.fetchModels();
|
|
19
19
|
|
|
@@ -34,7 +34,7 @@ export class LegacyModel extends SingleModel {
|
|
|
34
34
|
*
|
|
35
35
|
* @returns An array of capabilities, as expected by Pi
|
|
36
36
|
*/
|
|
37
|
-
async getCapabilities(): Promise<("text" | "image")[]> {
|
|
37
|
+
protected async getCapabilities(): Promise<("text" | "image")[]> {
|
|
38
38
|
try {
|
|
39
39
|
return await super.getCapabilities();
|
|
40
40
|
} catch {
|
|
@@ -26,7 +26,7 @@ export class RouterModel extends BaseModel {
|
|
|
26
26
|
*
|
|
27
27
|
* In exchange, it will allow unloaded models to be correctly shown as "unloaded".
|
|
28
28
|
*/
|
|
29
|
-
async pollStatus(startTime = Date.now()): Promise<void> {
|
|
29
|
+
protected async pollStatus(startTime = Date.now()): Promise<void> {
|
|
30
30
|
let elapsed = 0;
|
|
31
31
|
const limit = 5000;
|
|
32
32
|
|
|
@@ -52,7 +52,7 @@ export class RouterModel extends BaseModel {
|
|
|
52
52
|
*
|
|
53
53
|
* @returns The context size in tokens
|
|
54
54
|
*/
|
|
55
|
-
async getContextSize(): Promise<number> {
|
|
55
|
+
protected async getContextSize(): Promise<number> {
|
|
56
56
|
// We can get a more accurate context size if the model is already loaded
|
|
57
57
|
if ((await this.getStatus()) === Status.LOADED) {
|
|
58
58
|
return super.getContextSize();
|
package/src/server.ts
CHANGED
|
@@ -3,11 +3,9 @@ import {
|
|
|
3
3
|
API_KEY_PLACEHOLDER,
|
|
4
4
|
ENDPOINT_PREFIX,
|
|
5
5
|
PROVIDER_NAME,
|
|
6
|
-
PROVIDER_PREFIX,
|
|
7
6
|
} from "./constants";
|
|
8
7
|
import { Mode } from "./enums/mode";
|
|
9
8
|
import { ServerStatus } from "./enums/serverStatus";
|
|
10
|
-
import { HealthEndpoint } from "./interfaces/endpoints/health";
|
|
11
9
|
import { ModelsEndpoint } from "./interfaces/endpoints/models";
|
|
12
10
|
import {
|
|
13
11
|
PropsEndpoint,
|
|
@@ -22,20 +20,22 @@ import { RouterModel } from "./models/routerModel";
|
|
|
22
20
|
import { SingleModel } from "./models/singleModel";
|
|
23
21
|
import { SSEManager } from "./sse/manager";
|
|
24
22
|
import { checkServerHealth } from "./utils/health";
|
|
23
|
+
import { ServerIds } from "./utils/serverIds";
|
|
25
24
|
|
|
26
25
|
/**
|
|
27
26
|
* Optional constructor collaborators for {@link Server} — the seam tests use
|
|
28
27
|
* to run the real Server against fake clients.
|
|
29
28
|
*
|
|
30
29
|
* Both are factories because their arguments only exist around construction:
|
|
31
|
-
* the API key is (re-)resolved by the Server, and SSEManager needs its owner
|
|
32
|
-
*
|
|
33
|
-
* re-invokes both on every scan
|
|
34
|
-
* by design), so captured
|
|
30
|
+
* the API key is (re-)resolved by the Server, and SSEManager needs its owner
|
|
31
|
+
* (it reads the key and timeouts live through it). Factories must stay pure
|
|
32
|
+
* functions of their arguments — `initialize()` re-invokes both on every scan
|
|
33
|
+
* (the ApiClient rebuild picks up a fresh key, by design), so captured
|
|
34
|
+
* per-server state would leak across re-scans.
|
|
35
35
|
*/
|
|
36
36
|
export type ServerDeps = {
|
|
37
37
|
createApiClient?: (apiKey: string) => ApiClient;
|
|
38
|
-
createSSEManager?: (server: Server
|
|
38
|
+
createSSEManager?: (server: Server) => SSEManager;
|
|
39
39
|
};
|
|
40
40
|
|
|
41
41
|
export class Server {
|
|
@@ -52,8 +52,8 @@ export class Server {
|
|
|
52
52
|
// in ServerManager), so no lazy fallback is needed. initialize()
|
|
53
53
|
// rebuilds the client to re-resolve the API key.
|
|
54
54
|
this.apiClient =
|
|
55
|
-
deps.createApiClient?.(this.
|
|
56
|
-
new ApiClient(options.baseUrl, this.
|
|
55
|
+
deps.createApiClient?.(this.apiKey) ??
|
|
56
|
+
new ApiClient(options.baseUrl, this.apiKey);
|
|
57
57
|
}
|
|
58
58
|
|
|
59
59
|
/** Base URL of this server endpoint. */
|
|
@@ -102,7 +102,7 @@ export class Server {
|
|
|
102
102
|
* Uses custom ID if provided, otherwise falls back to URL-based ID.
|
|
103
103
|
*/
|
|
104
104
|
get providerId(): string {
|
|
105
|
-
return this.options.customId
|
|
105
|
+
return ServerIds.resolve(this.baseUrl, this.options.customId);
|
|
106
106
|
}
|
|
107
107
|
|
|
108
108
|
/**
|
|
@@ -119,17 +119,15 @@ export class Server {
|
|
|
119
119
|
/**
|
|
120
120
|
* Retrieves the API key from the config resolver.
|
|
121
121
|
* Tries custom ID first, then falls back to URL-based ID.
|
|
122
|
-
*
|
|
123
|
-
* @returns The API key
|
|
124
122
|
*/
|
|
125
|
-
|
|
123
|
+
get apiKey(): string {
|
|
126
124
|
// Try custom ID first
|
|
127
125
|
if (this.options.customId) {
|
|
128
126
|
const key = this.settings.resolveApiKey(this.options.customId);
|
|
129
127
|
if (key !== API_KEY_PLACEHOLDER) return key;
|
|
130
128
|
}
|
|
131
129
|
// Fall back to URL-based ID
|
|
132
|
-
return this.settings.resolveApiKey(
|
|
130
|
+
return this.settings.resolveApiKey(ServerIds.fromUrl(this.baseUrl));
|
|
133
131
|
}
|
|
134
132
|
|
|
135
133
|
/**
|
|
@@ -137,13 +135,10 @@ export class Server {
|
|
|
137
135
|
* Clears the cache first so we always fetch fresh data.
|
|
138
136
|
*/
|
|
139
137
|
async initialize() {
|
|
140
|
-
const apiKey = this.getApiKey();
|
|
141
138
|
this.apiClient =
|
|
142
|
-
this.deps.createApiClient?.(apiKey) ??
|
|
143
|
-
new ApiClient(this.baseUrl, apiKey);
|
|
144
|
-
this.sse =
|
|
145
|
-
this.deps.createSSEManager?.(this, apiKey) ??
|
|
146
|
-
new SSEManager(this, apiKey);
|
|
139
|
+
this.deps.createApiClient?.(this.apiKey) ??
|
|
140
|
+
new ApiClient(this.baseUrl, this.apiKey);
|
|
141
|
+
this.sse = this.deps.createSSEManager?.(this) ?? new SSEManager(this);
|
|
147
142
|
const { data } = await this.fetchModels();
|
|
148
143
|
const mode = await this.detectServerMode(data);
|
|
149
144
|
|
|
@@ -187,16 +182,7 @@ export class Server {
|
|
|
187
182
|
* @returns The server status
|
|
188
183
|
*/
|
|
189
184
|
async isReady(timeout: number): Promise<ServerStatus> {
|
|
190
|
-
return checkServerHealth(this.baseUrl, timeout, this.
|
|
191
|
-
}
|
|
192
|
-
|
|
193
|
-
/**
|
|
194
|
-
* Retrieves the health status of the server
|
|
195
|
-
*
|
|
196
|
-
* @returns The health status
|
|
197
|
-
*/
|
|
198
|
-
async fetchServerHealth(): Promise<HealthEndpoint> {
|
|
199
|
-
return await this.apiClient.get<HealthEndpoint>("/health");
|
|
185
|
+
return checkServerHealth(this.baseUrl, timeout, this.apiKey);
|
|
200
186
|
}
|
|
201
187
|
|
|
202
188
|
/**
|
|
@@ -231,13 +217,6 @@ export class Server {
|
|
|
231
217
|
);
|
|
232
218
|
}
|
|
233
219
|
|
|
234
|
-
/**
|
|
235
|
-
* Returns the per-model override configuration for this server.
|
|
236
|
-
*/
|
|
237
|
-
getOverrides(): Record<string, ModelOverride> {
|
|
238
|
-
return this.options.overrides ?? {};
|
|
239
|
-
}
|
|
240
|
-
|
|
241
220
|
/**
|
|
242
221
|
* Resolves the override for a given model ID using prefix matching.
|
|
243
222
|
*
|
|
@@ -249,7 +228,7 @@ export class Server {
|
|
|
249
228
|
* @returns The matching override, or `undefined` if no key matches.
|
|
250
229
|
*/
|
|
251
230
|
findOverrideForModel(modelId: string): ModelOverride | undefined {
|
|
252
|
-
const overrides = this.
|
|
231
|
+
const overrides = this.options.overrides ?? {};
|
|
253
232
|
let best: ModelOverride | undefined;
|
|
254
233
|
let bestLen = 0;
|
|
255
234
|
|
package/src/sse/client.ts
CHANGED
|
@@ -1,30 +1,22 @@
|
|
|
1
1
|
import { POLLING_INTERVAL } from "../constants";
|
|
2
|
+
import { openSSEStream } from "./fetch";
|
|
2
3
|
import type { SSECallback, SSECleanup, SSEEvent } from "./types";
|
|
3
4
|
|
|
4
|
-
/**
|
|
5
|
-
* Builds the full SSE endpoint URL, appending the API key as a query
|
|
6
|
-
* parameter when one is set. Shared by {@link SSEClient} and
|
|
7
|
-
* {@link SSEManager.probeSSE} so the two can't drift.
|
|
8
|
-
*/
|
|
9
|
-
export const buildSSEUrl = (endpoint: string, apiKey?: string): string => {
|
|
10
|
-
if (apiKey) {
|
|
11
|
-
return `${endpoint}?api_key=${encodeURIComponent(apiKey)}`;
|
|
12
|
-
}
|
|
13
|
-
return endpoint;
|
|
14
|
-
};
|
|
15
|
-
|
|
16
5
|
/**
|
|
17
6
|
* SSE client for llama-server's /models/sse endpoint.
|
|
18
7
|
*
|
|
19
|
-
* Uses a single shared
|
|
8
|
+
* Uses a single shared stream per server instance, consumed with `fetch`
|
|
9
|
+
* (EventSource cannot send the API key as a header, and llama-server no
|
|
10
|
+
* longer accepts it as a query parameter — see `fetch.ts`).
|
|
20
11
|
* Supports multiple model subscriptions with automatic event routing.
|
|
21
12
|
* Handles reconnection by re-subscribing all callbacks.
|
|
22
13
|
*/
|
|
23
14
|
export class SSEClient {
|
|
24
|
-
private
|
|
15
|
+
private abortController: AbortController | null = null;
|
|
16
|
+
private disposed: boolean = false;
|
|
25
17
|
private subscribers: Map<string, SSECallback> = new Map();
|
|
26
18
|
private connected: boolean = false;
|
|
27
|
-
private reconnecting: boolean = false; // tracks if
|
|
19
|
+
private reconnecting: boolean = false; // tracks if reconnect is in progress
|
|
28
20
|
/**
|
|
29
21
|
* Single shared slot — each setOnConnectFailed call overwrites the
|
|
30
22
|
* previous callback (see there for the constraint this imposes).
|
|
@@ -42,7 +34,15 @@ export class SSEClient {
|
|
|
42
34
|
) {}
|
|
43
35
|
|
|
44
36
|
/**
|
|
45
|
-
*
|
|
37
|
+
* Waits the given amount of time (used between reconnection attempts).
|
|
38
|
+
*/
|
|
39
|
+
private delay(ms: number): Promise<void> {
|
|
40
|
+
return new Promise<void>((resolve) => setTimeout(resolve, ms));
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Connects to the SSE endpoint and keeps it open, reconnecting with a
|
|
45
|
+
* fixed delay until `disconnect()` is called.
|
|
46
46
|
*
|
|
47
47
|
* No current caller consumes the result: `subscribe()` triggers the
|
|
48
48
|
* connection without awaiting it, and connection failures before the
|
|
@@ -50,56 +50,110 @@ export class SSEClient {
|
|
|
50
50
|
*
|
|
51
51
|
* @returns true if the connection was established successfully
|
|
52
52
|
*/
|
|
53
|
-
async connect(): Promise<boolean> {
|
|
53
|
+
private async connect(): Promise<boolean> {
|
|
54
54
|
if (this.connected) return true;
|
|
55
|
+
this.disposed = false;
|
|
55
56
|
|
|
56
|
-
|
|
57
|
+
// Loop-local so a stale, already-aborted controller from a previous
|
|
58
|
+
// connection can't be confused with the current one after a revival.
|
|
59
|
+
let abortController: AbortController | null = this.abortController;
|
|
57
60
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
this.
|
|
62
|
-
return false;
|
|
63
|
-
}
|
|
61
|
+
while (!this.disposed) {
|
|
62
|
+
if (abortController?.signal.aborted) return false;
|
|
63
|
+
abortController = new AbortController();
|
|
64
|
+
this.abortController = abortController;
|
|
64
65
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
66
|
+
let body: ReadableStream<Uint8Array>;
|
|
67
|
+
try {
|
|
68
|
+
body = await openSSEStream(
|
|
69
|
+
this.sseEndpoint,
|
|
70
|
+
this.apiKey,
|
|
71
|
+
abortController.signal,
|
|
72
|
+
);
|
|
73
|
+
} catch {
|
|
74
|
+
if (this.disposed || abortController.signal.aborted) return false;
|
|
75
|
+
this.notifyConnectFailed();
|
|
76
|
+
this.reconnecting = true;
|
|
77
|
+
await this.delay(POLLING_INTERVAL);
|
|
78
|
+
continue;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
this.connected = true;
|
|
82
|
+
this.reconnecting = false;
|
|
83
|
+
|
|
84
|
+
await this.consume(body);
|
|
85
|
+
|
|
86
|
+
if (this.disposed || abortController.signal.aborted) return false;
|
|
87
|
+
// Stream ended (server closed or network error): reconnect
|
|
68
88
|
this.reconnecting = true;
|
|
89
|
+
await this.delay(POLLING_INTERVAL);
|
|
90
|
+
}
|
|
69
91
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
this._onConnectFailed();
|
|
73
|
-
}
|
|
74
|
-
};
|
|
92
|
+
return false;
|
|
93
|
+
}
|
|
75
94
|
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
95
|
+
/**
|
|
96
|
+
* Reads the raw byte stream, parses the SSE framing and dispatches
|
|
97
|
+
* `data:` payloads. Resolves when the stream ends or errors.
|
|
98
|
+
*/
|
|
99
|
+
private async consume(body: ReadableStream<Uint8Array>): Promise<void> {
|
|
100
|
+
const reader = body.getReader();
|
|
101
|
+
const decoder = new TextDecoder();
|
|
102
|
+
let buffer = "";
|
|
103
|
+
|
|
104
|
+
try {
|
|
105
|
+
for (;;) {
|
|
106
|
+
const { done, value } = await reader.read();
|
|
107
|
+
if (done) break;
|
|
108
|
+
|
|
109
|
+
buffer += decoder.decode(value, { stream: true });
|
|
110
|
+
let newlineIndex: number;
|
|
111
|
+
while ((newlineIndex = buffer.indexOf("\n")) !== -1) {
|
|
112
|
+
const line = buffer.slice(0, newlineIndex);
|
|
113
|
+
buffer = buffer.slice(newlineIndex + 1);
|
|
114
|
+
this.handleLine(line);
|
|
115
|
+
}
|
|
88
116
|
}
|
|
89
|
-
|
|
117
|
+
// Flush a trailing data line that lacked its event-terminating newline
|
|
118
|
+
if (buffer) this.handleLine(buffer);
|
|
119
|
+
} catch {
|
|
120
|
+
// Stream aborted or network error — handled by the reconnect logic
|
|
121
|
+
} finally {
|
|
122
|
+
reader.releaseLock();
|
|
123
|
+
}
|
|
124
|
+
}
|
|
90
125
|
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
126
|
+
/**
|
|
127
|
+
* Handles a single SSE line, dispatching `data:` payloads as events.
|
|
128
|
+
*/
|
|
129
|
+
private handleLine(line: string): void {
|
|
130
|
+
if (!line.startsWith("data:")) return;
|
|
131
|
+
|
|
132
|
+
const payload = line.slice(5).trim();
|
|
133
|
+
if (!payload || payload === "[DONE]") return;
|
|
134
|
+
|
|
135
|
+
try {
|
|
136
|
+
const data = JSON.parse(payload);
|
|
137
|
+
const sseEvent: SSEEvent = {
|
|
138
|
+
event: data.event ?? "unknown",
|
|
139
|
+
model: data.model ?? "*",
|
|
140
|
+
data: data.data,
|
|
99
141
|
};
|
|
100
|
-
|
|
142
|
+
this._hasReceivedEvents = true;
|
|
143
|
+
this.dispatch(sseEvent);
|
|
144
|
+
} catch {
|
|
145
|
+
// Invalid JSON, ignore
|
|
146
|
+
}
|
|
147
|
+
}
|
|
101
148
|
|
|
102
|
-
|
|
149
|
+
/**
|
|
150
|
+
* Notifies the connect-failed callback if the connection failed before
|
|
151
|
+
* any event was received.
|
|
152
|
+
*/
|
|
153
|
+
private notifyConnectFailed(): void {
|
|
154
|
+
if (!this._hasReceivedEvents && this._onConnectFailed) {
|
|
155
|
+
this._onConnectFailed();
|
|
156
|
+
}
|
|
103
157
|
}
|
|
104
158
|
|
|
105
159
|
/**
|
|
@@ -143,11 +197,11 @@ export class SSEClient {
|
|
|
143
197
|
* Disconnects from the SSE endpoint and clears all subscriptions.
|
|
144
198
|
*/
|
|
145
199
|
disconnect(): void {
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
}
|
|
200
|
+
this.disposed = true;
|
|
201
|
+
this.abortController?.abort();
|
|
202
|
+
this.abortController = null;
|
|
150
203
|
this.connected = false;
|
|
204
|
+
this.reconnecting = false;
|
|
151
205
|
this.subscribers.clear();
|
|
152
206
|
}
|
|
153
207
|
|