pi-llama-cpp 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,16 +10,33 @@ import {
10
10
  POLLING_TIMEOUT,
11
11
  REACT_TO_MODEL_SELECT,
12
12
  SERVER_TIMEOUT,
13
+ SETTINGS_KEY,
14
+ SORT_BY,
13
15
  THINKING_BUDGETS,
16
+ type SortBy,
14
17
  } from "../constants";
15
- import { LlamaSettings } from "../interfaces/settings";
18
+ import { LlamaServer, LlamaSettings } from "../interfaces/settings";
16
19
  import { Server } from "../server";
17
-
18
- const SETTINGS_KEY = "llamaSettings";
20
+ import { SettingsStore } from "../utils/settingsStore";
21
+ import { isValidServerUrl, normalizeUrl } from "../utils/urls";
19
22
 
20
23
  export class LlamaSettingsManager {
21
24
  private settingsManager = SettingsManager.create(process.cwd());
22
25
 
26
+ constructor(private readonly store: SettingsStore = new SettingsStore()) {}
27
+
28
+ /** Warnings collected during URL resolution (dropped invalid entries). */
29
+ private warnings: string[] = [];
30
+
31
+ /**
32
+ * Returns and clears warnings collected during URL resolution.
33
+ */
34
+ takeWarnings(): string[] {
35
+ const warnings = [...this.warnings];
36
+ this.warnings.length = 0;
37
+ return warnings;
38
+ }
39
+
23
40
  /**
24
41
  * Convenience getter for merged project/global settings
25
42
  */
@@ -38,6 +55,14 @@ export class LlamaSettingsManager {
38
55
  return this.mergedSettings[SETTINGS_KEY] ?? {};
39
56
  }
40
57
 
58
+ /**
59
+ * Convenience getter for the merged `servers` list (project overrides
60
+ * global, per-key merge)
61
+ */
62
+ get llamaServers(): LlamaServer[] {
63
+ return this.llamaSettings.servers ?? [];
64
+ }
65
+
41
66
  /**
42
67
  * Resolves the server URLs to use in the following order:
43
68
  *
@@ -99,17 +124,27 @@ export class LlamaSettingsManager {
99
124
 
100
125
  /**
101
126
  * Parses a raw URL string into an array of cleaned URLs.
102
- * Splits on semicolons, trims whitespace, filters empty strings,
103
- * and strips trailing slashes.
127
+ * Splits on semicolons, trims whitespace, filters empty strings, strips
128
+ * trailing slashes, and drops entries without an http(s) scheme —
129
+ * collecting a warning for each dropped entry (same validation the
130
+ * `/models servers` editor applies).
104
131
  *
105
132
  * @returns A sanitized URL
106
133
  */
107
134
  private parseUrls(raw: string): string[] {
108
135
  return raw
109
136
  .split(";")
110
- .map((u) => u.trim())
111
- .filter((u) => u.length > 0)
112
- .map((u) => u.replace(/\/+$/, ""));
137
+ .map(normalizeUrl)
138
+ .filter((u) => {
139
+ if (u.length === 0) return false;
140
+ if (!isValidServerUrl(u)) {
141
+ this.warnings.push(
142
+ `Ignoring invalid server URL '${u}' (needs http(s)://)`,
143
+ );
144
+ return false;
145
+ }
146
+ return true;
147
+ });
113
148
  }
114
149
 
115
150
  /**
@@ -121,19 +156,16 @@ export class LlamaSettingsManager {
121
156
  * @returns A list of Server objects
122
157
  */
123
158
  resolveServers(): Server[] {
124
- const { pollingTimeout, serverTimeout } = this.resolveTimeouts();
125
159
  const urls = this.resolveUrls();
126
160
  const serverConfigs = this.llamaSettings.servers ?? [];
127
161
 
128
162
  return urls.map((url) => {
129
163
  const config = serverConfigs.find((s) => s.url === url);
130
- return new Server(
131
- url,
132
- config?.id,
133
- config?.name,
134
- serverTimeout,
135
- pollingTimeout,
136
- );
164
+ return new Server(this, {
165
+ baseUrl: url,
166
+ customId: config?.id,
167
+ customName: config?.name,
168
+ });
137
169
  });
138
170
  }
139
171
 
@@ -198,6 +230,36 @@ export class LlamaSettingsManager {
198
230
  serverTimeout: this.llamaSettings.serverTimeout ?? SERVER_TIMEOUT,
199
231
  };
200
232
  }
233
+
234
+ /**
235
+ * Resolves the sort order for model lists.
236
+ *
237
+ * @returns The sort order: "asc", "desc", "asc-name", "desc-name", or "api"
238
+ */
239
+ resolveSortBy(): SortBy {
240
+ return this.llamaSettings.sortBy ?? SORT_BY;
241
+ }
242
+
243
+ /**
244
+ * Persists one llamaSettings field to the global settings file and
245
+ * reloads the in-memory settings so resolvers see the change immediately.
246
+ *
247
+ * Rejects if the file can't be read (e.g. invalid JSON) or written —
248
+ * in-memory state stays consistent (reload only on success).
249
+ */
250
+ async setLlamaSetting<K extends keyof LlamaSettings>(
251
+ key: K,
252
+ value: LlamaSettings[K],
253
+ ): Promise<void> {
254
+ await this.store.updateKey(SETTINGS_KEY, (current) => {
255
+ const merged =
256
+ typeof current === "object" && current !== null
257
+ ? (current as Record<string, unknown>)
258
+ : {};
259
+ return { ...merged, [key]: value };
260
+ });
261
+ await this.settingsManager.reload();
262
+ }
201
263
  }
202
264
 
203
265
  /**
@@ -16,14 +16,6 @@ export abstract class BaseModel {
16
16
  protected readonly server: Server,
17
17
  ) {}
18
18
 
19
- protected readonly statusMapper: Record<string, Status> = {
20
- loaded: Status.LOADED,
21
- loading: Status.LOADING,
22
- failed: Status.FAILED,
23
- sleeping: Status.SLEEPING,
24
- unloaded: Status.UNLOADED,
25
- };
26
-
27
19
  protected readonly labelIcons: Record<Status, string> = {
28
20
  [Status.LOADED]: "🟢",
29
21
  [Status.LOADING]: "🟡",
package/src/server.ts CHANGED
@@ -1,10 +1,8 @@
1
1
  import { ApiClient } from "./api/client";
2
2
  import {
3
3
  API_KEY_PLACEHOLDER,
4
- POLLING_TIMEOUT,
5
4
  PROVIDER_NAME,
6
5
  PROVIDER_PREFIX,
7
- SERVER_TIMEOUT,
8
6
  } from "./constants";
9
7
  import { Mode } from "./enums/mode";
10
8
  import { ServerStatus } from "./enums/serverStatus";
@@ -14,25 +12,67 @@ import {
14
12
  PropsEndpoint,
15
13
  PropsModelEndpoint,
16
14
  } from "./interfaces/endpoints/props";
17
- import { settings } from "./managers/settings";
15
+ import type { ServerOptions } from "./interfaces/server";
16
+ import type { LlamaSettingsManager } from "./managers/settings";
18
17
  import { BaseModel } from "./models/baseModel";
19
18
  import { LegacyModel } from "./models/legacyModel";
20
19
  import { RouterModel } from "./models/routerModel";
21
20
  import { SingleModel } from "./models/singleModel";
22
21
  import { SSEManager } from "./sse/manager";
23
22
 
23
+ /**
24
+ * Optional constructor collaborators for {@link Server} — the seam tests use
25
+ * to run the real Server against fake clients.
26
+ *
27
+ * Both are factories because their arguments only exist around construction:
28
+ * the API key is (re-)resolved by the Server, and SSEManager needs its owner.
29
+ * Factories must stay pure functions of their arguments — `initialize()`
30
+ * re-invokes both on every scan (the ApiClient rebuild picks up a fresh key,
31
+ * by design), so captured per-server state would leak across re-scans.
32
+ */
33
+ export type ServerDeps = {
34
+ createApiClient?: (apiKey: string) => ApiClient;
35
+ createSSEManager?: (server: Server, apiKey: string) => SSEManager;
36
+ };
37
+
24
38
  export class Server {
25
39
  public readonly models: BaseModel[] = [];
26
- private apiClient!: ApiClient;
40
+ private apiClient: ApiClient;
27
41
  private sse!: SSEManager;
28
42
 
29
43
  constructor(
30
- readonly baseUrl: string,
31
- private readonly customId?: string,
32
- private readonly customName?: string,
33
- readonly serverTimeout: number = SERVER_TIMEOUT,
34
- readonly pollingTimeout: number = POLLING_TIMEOUT,
35
- ) {}
44
+ private readonly settings: LlamaSettingsManager,
45
+ private readonly options: ServerOptions,
46
+ private readonly deps: ServerDeps = {},
47
+ ) {
48
+ // Eager client: `isReady` may run before `initialize()` (health probing
49
+ // in ServerManager), so no lazy fallback is needed. initialize()
50
+ // rebuilds the client to re-resolve the API key.
51
+ this.apiClient =
52
+ deps.createApiClient?.(this.getApiKey()) ??
53
+ new ApiClient(options.baseUrl, this.getApiKey());
54
+ }
55
+
56
+ /** Base URL of this server endpoint. */
57
+ get baseUrl(): string {
58
+ return this.options.baseUrl;
59
+ }
60
+
61
+ /**
62
+ * Maximum time (ms) for server verification and SSE support probe.
63
+ * Resolved live from the injected settings manager.
64
+ */
65
+ get serverTimeout(): number {
66
+ return this.settings.resolveTimeouts().serverTimeout;
67
+ }
68
+
69
+ /**
70
+ * Maximum time (ms) to wait for model loading before giving up.
71
+ * Resolved live from the injected settings manager.
72
+ */
73
+ get pollingTimeout(): number {
74
+ return this.settings.resolveTimeouts().pollingTimeout;
75
+ }
36
76
 
37
77
  /**
38
78
  * Provides access to the SSE manager for direct subscriptions.
@@ -46,7 +86,7 @@ export class Server {
46
86
  * Uses custom ID if provided, otherwise falls back to URL-based ID.
47
87
  */
48
88
  get providerId(): string {
49
- return this.customId ?? `${PROVIDER_PREFIX}=${this.baseUrl}`;
89
+ return this.options.customId ?? `${PROVIDER_PREFIX}=${this.baseUrl}`;
50
90
  }
51
91
 
52
92
  /**
@@ -54,8 +94,8 @@ export class Server {
54
94
  * Uses custom name as suffix if provided.
55
95
  */
56
96
  get providerName(): string {
57
- if (this.customName) {
58
- return `${PROVIDER_NAME} (${this.customName})`;
97
+ if (this.options.customName) {
98
+ return `${PROVIDER_NAME} (${this.options.customName})`;
59
99
  }
60
100
  return `${PROVIDER_NAME} (${this.baseUrl})`;
61
101
  }
@@ -68,12 +108,12 @@ export class Server {
68
108
  */
69
109
  getApiKey(): string {
70
110
  // Try custom ID first
71
- if (this.customId) {
72
- const key = settings.resolveApiKey(this.customId);
111
+ if (this.options.customId) {
112
+ const key = this.settings.resolveApiKey(this.options.customId);
73
113
  if (key !== API_KEY_PLACEHOLDER) return key;
74
114
  }
75
115
  // Fall back to URL-based ID
76
- return settings.resolveApiKey(`${PROVIDER_PREFIX}=${this.baseUrl}`);
116
+ return this.settings.resolveApiKey(`${PROVIDER_PREFIX}=${this.baseUrl}`);
77
117
  }
78
118
 
79
119
  /**
@@ -81,11 +121,15 @@ export class Server {
81
121
  * Clears the cache first so we always fetch fresh data.
82
122
  */
83
123
  async initialize() {
84
- const apiKey = await this.getApiKey();
85
- this.apiClient = new ApiClient(this.baseUrl, apiKey);
86
- this.sse = new SSEManager(this.baseUrl, apiKey, this.serverTimeout);
124
+ const apiKey = this.getApiKey();
125
+ this.apiClient =
126
+ this.deps.createApiClient?.(apiKey) ??
127
+ new ApiClient(this.baseUrl, apiKey);
128
+ this.sse =
129
+ this.deps.createSSEManager?.(this, apiKey) ??
130
+ new SSEManager(this, apiKey);
87
131
  const { data } = await this.fetchModels();
88
- const mode = await this.detectServerMode();
132
+ const mode = await this.detectServerMode(data);
89
133
 
90
134
  // Setup models
91
135
  const modelCtor = {
@@ -94,22 +138,21 @@ export class Server {
94
138
  [Mode.SINGLE]: SingleModel,
95
139
  }[mode];
96
140
 
97
- const models: BaseModel[] = data
98
- .map((m) => new modelCtor(m, this))
99
- .sort((a, b) => (a.id > b.id ? 1 : a.id === b.id ? 0 : -1));
141
+ const models: BaseModel[] = data.map((m) => new modelCtor(m, this));
100
142
 
101
143
  this.models.length = 0;
102
144
  this.models.push(...models);
103
145
  }
104
146
 
105
147
  /**
106
- * Detects the mode of the server
148
+ * Detects the mode of the server from the models data already fetched by
149
+ * {@link initialize} — no second /v1/models round-trip.
107
150
  *
151
+ * @param data Models endpoint data fetched by initialize()
108
152
  * @returns The detected mode
109
153
  */
110
- private async detectServerMode(): Promise<Mode> {
154
+ private async detectServerMode(data: ModelsEndpoint["data"]): Promise<Mode> {
111
155
  const { role } = await this.fetchServerProps();
112
- const { data } = await this.fetchModels();
113
156
 
114
157
  if (role === "router") return Mode.ROUTER;
115
158
  if ("max_model_len" in data[0]) return Mode.LEGACY;
@@ -123,8 +166,6 @@ export class Server {
123
166
  * @returns The server status
124
167
  */
125
168
  async isReady(timeout: number): Promise<ServerStatus> {
126
- this.apiClient ??= new ApiClient(this.baseUrl, await this.getApiKey());
127
-
128
169
  try {
129
170
  const timeoutPromise = new Promise<never>((_, reject) =>
130
171
  setTimeout(() => reject(new Error("timeout")), timeout),
package/src/sse/client.ts CHANGED
@@ -1,6 +1,18 @@
1
1
  import { POLLING_INTERVAL } from "../constants";
2
2
  import type { SSECallback, SSECleanup, SSEEvent } from "./types";
3
3
 
4
+ /**
5
+ * Builds the full SSE endpoint URL, appending the API key as a query
6
+ * parameter when one is set. Shared by {@link SSEClient} and
7
+ * {@link SSEManager.probeSSE} so the two can't drift.
8
+ */
9
+ export const buildSSEUrl = (endpoint: string, apiKey?: string): string => {
10
+ if (apiKey) {
11
+ return `${endpoint}?api_key=${encodeURIComponent(apiKey)}`;
12
+ }
13
+ return endpoint;
14
+ };
15
+
4
16
  /**
5
17
  * SSE client for llama-server's /models/sse endpoint.
6
18
  *
@@ -13,6 +25,10 @@ export class SSEClient {
13
25
  private subscribers: Map<string, SSECallback> = new Map();
14
26
  private connected: boolean = false;
15
27
  private reconnecting: boolean = false; // tracks if EventSource auto-reconnect is in progress
28
+ /**
29
+ * Single shared slot — each setOnConnectFailed call overwrites the
30
+ * previous callback (see there for the constraint this imposes).
31
+ */
16
32
  private _onConnectFailed: (() => void) | null = null;
17
33
  private _hasReceivedEvents: boolean = false;
18
34
 
@@ -28,12 +44,16 @@ export class SSEClient {
28
44
  /**
29
45
  * Connects to the SSE endpoint.
30
46
  *
47
+ * No current caller consumes the result: `subscribe()` triggers the
48
+ * connection without awaiting it, and connection failures before the
49
+ * first event are surfaced through the `setOnConnectFailed` callback.
50
+ *
31
51
  * @returns true if the connection was established successfully
32
52
  */
33
53
  async connect(): Promise<boolean> {
34
54
  if (this.connected) return true;
35
55
 
36
- const url = this.buildUrl();
56
+ const url = buildSSEUrl(this.sseEndpoint, this.apiKey);
37
57
 
38
58
  try {
39
59
  this.eventSource = new EventSource(url);
@@ -42,11 +62,6 @@ export class SSEClient {
42
62
  return false;
43
63
  }
44
64
 
45
- this.eventSource.onopen = () => {
46
- this.connected = true;
47
- this.reconnecting = false;
48
- };
49
-
50
65
  this.eventSource.onerror = () => {
51
66
  // EventSource will auto-reconnect; we just track state
52
67
  this.connected = false;
@@ -91,6 +106,13 @@ export class SSEClient {
91
106
  * Sets a callback to be called when the connection fails before
92
107
  * any event is received. Useful for rejecting promises early.
93
108
  *
109
+ * Single shared slot: each call overwrites the previous callback, so at
110
+ * most one caller may depend on it at a time. The only caller today is
111
+ * `SSEManager.subscribeToStatus`, which must therefore not be invoked
112
+ * twice concurrently on the same client — the second registration would
113
+ * take over the failure signal and the first promise would only reject
114
+ * via its own timeout.
115
+ *
94
116
  * @param callback - Called once when connection fails
95
117
  */
96
118
  setOnConnectFailed(callback: () => void): void {
@@ -129,16 +151,6 @@ export class SSEClient {
129
151
  this.subscribers.clear();
130
152
  }
131
153
 
132
- /**
133
- * Builds the full URL with optional API key query param.
134
- */
135
- private buildUrl(): string {
136
- if (this.apiKey) {
137
- return `${this.sseEndpoint}?api_key=${encodeURIComponent(this.apiKey)}`;
138
- }
139
- return this.sseEndpoint;
140
- }
141
-
142
154
  /**
143
155
  * Dispatches an SSE event to all matching subscribers.
144
156
  */
@@ -1,5 +1,5 @@
1
- import { POLLING_TIMEOUT, SERVER_TIMEOUT } from "../constants";
2
- import { SSEClient } from "./client";
1
+ import type { Server } from "../server";
2
+ import { SSEClient, buildSSEUrl } from "./client";
3
3
  import {
4
4
  DownloadProgressData,
5
5
  ProgressData,
@@ -25,16 +25,31 @@ export class SSEManager {
25
25
  private sseSupported: boolean | null = null;
26
26
 
27
27
  constructor(
28
- private readonly baseUrl: string,
28
+ private readonly server: Server,
29
29
  private readonly apiKey: string,
30
- readonly serverTimeout: number = SERVER_TIMEOUT,
31
30
  ) {}
32
31
 
32
+ /**
33
+ * Maximum time (ms) for server verification and SSE support probe.
34
+ * Delegates to the owning {@link Server}.
35
+ */
36
+ get serverTimeout(): number {
37
+ return this.server.serverTimeout;
38
+ }
39
+
40
+ /**
41
+ * Maximum time (ms) to wait for model loading before giving up.
42
+ * Delegates to the owning {@link Server}.
43
+ */
44
+ get pollingTimeout(): number {
45
+ return this.server.pollingTimeout;
46
+ }
47
+
33
48
  /**
34
49
  * The SSE endpoint URL.
35
50
  */
36
51
  private get sseEndpoint(): string {
37
- return `${this.baseUrl}/models/sse`;
52
+ return `${this.server.baseUrl}/models/sse`;
38
53
  }
39
54
 
40
55
  /**
@@ -47,10 +62,7 @@ export class SSEManager {
47
62
  if (this.sseSupported !== null) return this.sseSupported;
48
63
 
49
64
  try {
50
- let url = this.sseEndpoint;
51
- if (this.apiKey) {
52
- url = `${url}?api_key=${encodeURIComponent(this.apiKey)}`;
53
- }
65
+ const url = buildSSEUrl(this.sseEndpoint, this.apiKey);
54
66
  const response = await fetch(url, {
55
67
  method: "GET",
56
68
  signal: AbortSignal.timeout(this.serverTimeout),
@@ -171,7 +183,7 @@ export class SSEManager {
171
183
  return new Promise((resolve, reject) => {
172
184
  const timeout = setTimeout(
173
185
  () => reject(new Error(`SSE status timeout for model: ${modelId}`)),
174
- POLLING_TIMEOUT,
186
+ this.pollingTimeout,
175
187
  );
176
188
 
177
189
  this.subscribeToSSE(modelId, (event: SSEEvent) => {