pi-llama-cpp 0.7.0 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -127,10 +127,10 @@ llama-server --model path/to/model.gguf ...
127
127
  The extension determines the context size as follows:
128
128
 
129
129
  - **Router mode**
130
- - When loaded, reads `meta.n_ctx` from the `/models` endpoint
130
+ - When loaded, reads `meta.n_ctx` from the `/v1/models` endpoint
131
131
  - When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
132
- - **Single mode** — reads `meta.n_ctx` from the `/models` endpoint
133
- - **Legacy mode** — reads `max_model_len` from `/models`, falling back to `n_ctx` from `/props`
132
+ - **Single mode** — reads `meta.n_ctx` from the `/v1/models` endpoint
133
+ - **Legacy mode** — reads `max_model_len` from `/v1/models`, falling back to `n_ctx` from `/props`
134
134
  - Falls back to `128000` if not available
135
135
 
136
136
  ### Commands
@@ -209,7 +209,7 @@ When you trigger a load, switch, or retry action, the extension polls the server
209
209
  Each model exposed to Pi includes the following defaults:
210
210
 
211
211
  - **`maxTokens`** — dynamically set to the model's context window (detected from llama-server)
212
- - **`reasoning`** — `true` (assumed, as llama.cpp's `/models` endpoint does not expose it)
212
+ - **`reasoning`** — `true` (assumed, as llama.cpp's `/v1/models` endpoint does not expose it)
213
213
  - **`cost`** — all zero (local models)
214
214
 
215
215
  ## Dependencies
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.7.0",
3
+ "version": "0.7.2",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -36,8 +36,8 @@
36
36
  "@earendil-works/pi-tui": "*"
37
37
  },
38
38
  "devDependencies": {
39
- "@types/node": "^25.9.3",
39
+ "@types/node": "^26.0.0",
40
40
  "prettier-plugin-organize-imports": "^4.3.0",
41
- "vitest": "^4.1.8"
41
+ "vitest": "^4.1.9"
42
42
  }
43
43
  }
@@ -10,7 +10,6 @@ import { Server } from "../server";
10
10
 
11
11
  export class EventManager {
12
12
  static inflightModel: BaseModel | null = null;
13
- private readonly resolver = new ConfigResolver();
14
13
 
15
14
  constructor(private readonly servers: Server[]) {}
16
15
 
@@ -86,8 +85,9 @@ export class EventManager {
86
85
  if (!isLlamaCpp) return payload;
87
86
 
88
87
  // Retrieve pi's current thinking level, so we can setup a budget
89
- const level = this.resolver.resolveThinkingLevel() ?? "medium";
90
- const budgets = this.resolver.resolveThinkingBudgets();
88
+ const resolver = new ConfigResolver();
89
+ const level = resolver.resolveThinkingLevel() ?? "medium";
90
+ const budgets = resolver.resolveThinkingBudgets();
91
91
  const thinking_budget_tokens = budgets[level];
92
92
 
93
93
  // Setup payload
@@ -14,8 +14,7 @@ export class RouterModel extends BaseModel {
14
14
  }
15
15
 
16
16
  /**
17
- * Workaround for the currently-bugged /models status detection
18
- * (I suspect it was introduced in PR #22683 of llama.cpp)
17
+ * Workaround for /models status detection
19
18
  *
20
19
  * When a model is loaded for the very first time,
21
20
  * this workaround will try to poll to /props instead of /models
package/src/server.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { PROVIDER_NAME, PROVIDER_PREFIX } from "./constants";
1
+ import { POLLING_INTERVAL, PROVIDER_NAME, PROVIDER_PREFIX } from "./constants";
2
2
  import { Mode } from "./enums/mode";
3
3
  import { ServerStatus } from "./enums/serverStatus";
4
4
  import { HealthEndpoint } from "./interfaces/endpoints/health";
@@ -9,10 +9,14 @@ import { LegacyModel } from "./models/legacyModel";
9
9
  import { RouterModel } from "./models/routerModel";
10
10
  import { SingleModel } from "./models/singleModel";
11
11
  import { ConfigResolver } from "./resolver";
12
+ import { Cache } from "./utils/cache";
13
+ import { Mutex } from "./utils/mutex";
12
14
 
13
15
  export class Server {
14
16
  public readonly models: BaseModel[] = [];
15
17
  private configResolver = new ConfigResolver();
18
+ private cache = new Cache(POLLING_INTERVAL / 2);
19
+ private mutex = new Mutex();
16
20
 
17
21
  constructor(readonly baseUrl: string) {}
18
22
 
@@ -39,9 +43,11 @@ export class Server {
39
43
  }
40
44
 
41
45
  /**
42
- * Fetches models from the server and populates {@link models}
46
+ * Fetches models from the server and populates {@link models}.
47
+ * Clears the cache first so we always fetch fresh data.
43
48
  */
44
49
  async initialize() {
50
+ this.cache.clear();
45
51
  const { data } = await this.fetchModels();
46
52
  const mode = await this.detectServerMode();
47
53
 
@@ -150,11 +156,13 @@ export class Server {
150
156
  resource: "load" | "unload",
151
157
  model: string,
152
158
  ): Promise<ModelsEndpoint> {
159
+ this.cache.clear();
153
160
  return await this.rpc<ModelsEndpoint>(`/models/${resource}`, { model });
154
161
  }
155
162
 
156
163
  /**
157
- * Makes an HTTP request to the llama-server and returns the parsed JSON response
164
+ * Makes a cached, deduplicated request to the llama-server.
165
+ * Results are cached for half the {@link POLLING_INTERVAL} and in-flight requests are deduplicated.
158
166
  *
159
167
  * @param endpoint The endpoint path to fetch (e.g. "/health")
160
168
  * @param body The optional request body for POST requests
@@ -163,6 +171,33 @@ export class Server {
163
171
  private async rpc<T>(
164
172
  endpoint: string,
165
173
  body?: Record<string, unknown>,
174
+ ): Promise<T> {
175
+ const key = this.cacheKey(endpoint, body);
176
+
177
+ // Check cache
178
+ const cached = this.cache.get<T>(key);
179
+ if (cached !== undefined) {
180
+ return cached;
181
+ }
182
+
183
+ // Deduplicate in-flight requests
184
+ return this.mutex.getOrCreate(key, async () => {
185
+ const data = await this.fetch<T>(endpoint, body);
186
+ this.cache.set(key, data);
187
+ return data;
188
+ });
189
+ }
190
+
191
+ /**
192
+ * Makes an HTTP request to the llama-server and returns the parsed JSON response.
193
+ *
194
+ * @param endpoint The endpoint path to fetch (e.g. "/health")
195
+ * @param body The optional request body for POST requests
196
+ * @returns The parsed JSON response from the server
197
+ */
198
+ private async fetch<T>(
199
+ endpoint: string,
200
+ body?: Record<string, unknown>,
166
201
  ): Promise<T> {
167
202
  const url = `${this.baseUrl}${endpoint}`;
168
203
  const apiKey = await this.getApiKey();
@@ -184,4 +219,15 @@ export class Server {
184
219
  const response: T = await res.json();
185
220
  return response;
186
221
  }
222
+
223
+ /**
224
+ * Generates a cache key from the endpoint and body.
225
+ *
226
+ * @param endpoint The endpoint path (e.g. "/health")
227
+ * @param body The optional request body for POST requests
228
+ * @returns A key used for caching
229
+ */
230
+ private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
231
+ return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
232
+ }
187
233
  }
@@ -0,0 +1,39 @@
1
+ /**
2
+ * Generic TTL cache.
3
+ * Entries expire after `ttl` milliseconds from the time they were set.
4
+ */
5
+ export class Cache {
6
+ private entries = new Map<string, { data: unknown; timestamp: number }>();
7
+
8
+ /**
9
+ * @param ttl Time-to-live in milliseconds
10
+ */
11
+ constructor(private readonly ttl: number) {}
12
+
13
+ /**
14
+ * Gets a cached value by key. Returns `undefined` if missing or expired.
15
+ */
16
+ get<T>(key: string): T | undefined {
17
+ const entry = this.entries.get(key);
18
+ if (!entry) return undefined;
19
+ if (Date.now() - entry.timestamp > this.ttl) {
20
+ this.entries.delete(key);
21
+ return undefined;
22
+ }
23
+ return entry.data as T;
24
+ }
25
+
26
+ /**
27
+ * Stores a value in the cache with the current timestamp.
28
+ */
29
+ set(key: string, data: unknown): void {
30
+ this.entries.set(key, { data, timestamp: Date.now() });
31
+ }
32
+
33
+ /**
34
+ * Clears all cached entries.
35
+ */
36
+ clear(): void {
37
+ this.entries.clear();
38
+ }
39
+ }
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Ensures only one in-flight operation exists per key.
3
+ * Concurrent callers for the same key share the same promise.
4
+ */
5
+ export class Mutex {
6
+ private promises = new Map<string, Promise<unknown>>();
7
+
8
+ /**
9
+ * Runs `fn` for the given key, or returns an existing in-flight promise.
10
+ * Concurrent callers for the same key share the same promise.
11
+ */
12
+ getOrCreate<T>(key: string, fn: () => Promise<T>): Promise<T> {
13
+ const existing = this.promises.get(key);
14
+ if (existing) {
15
+ return existing as Promise<T>;
16
+ }
17
+
18
+ const promise = fn().finally(() => {
19
+ this.promises.delete(key);
20
+ });
21
+ this.promises.set(key, promise);
22
+ return promise;
23
+ }
24
+ }