pi-llama-cpp 0.7.0 → 0.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/package.json +3 -3
- package/src/managers/events.ts +3 -3
- package/src/models/routerModel.ts +1 -2
- package/src/server.ts +49 -3
- package/src/utils/cache.ts +39 -0
- package/src/utils/mutex.ts +24 -0
package/README.md
CHANGED
|
@@ -127,10 +127,10 @@ llama-server --model path/to/model.gguf ...
|
|
|
127
127
|
The extension determines the context size as follows:
|
|
128
128
|
|
|
129
129
|
- **Router mode**
|
|
130
|
-
- When loaded, reads `meta.n_ctx` from the `/models` endpoint
|
|
130
|
+
- When loaded, reads `meta.n_ctx` from the `/v1/models` endpoint
|
|
131
131
|
- When not loaded, reads `--ctx-size` and/or `--fit-ctx` from the server arguments (which can also originate from the **presets.ini** file the llama.cpp server uses to load its models).
|
|
132
|
-
- **Single mode** — reads `meta.n_ctx` from the `/models` endpoint
|
|
133
|
-
- **Legacy mode** — reads `max_model_len` from `/models`, falling back to `n_ctx` from `/props`
|
|
132
|
+
- **Single mode** — reads `meta.n_ctx` from the `/v1/models` endpoint
|
|
133
|
+
- **Legacy mode** — reads `max_model_len` from `/v1/models`, falling back to `n_ctx` from `/props`
|
|
134
134
|
- Falls back to `128000` if not available
|
|
135
135
|
|
|
136
136
|
### Commands
|
|
@@ -209,7 +209,7 @@ When you trigger a load, switch, or retry action, the extension polls the server
|
|
|
209
209
|
Each model exposed to Pi includes the following defaults:
|
|
210
210
|
|
|
211
211
|
- **`maxTokens`** — dynamically set to the model's context window (detected from llama-server)
|
|
212
|
-
- **`reasoning`** — `true` (assumed, as llama.cpp's `/models` endpoint does not expose it)
|
|
212
|
+
- **`reasoning`** — `true` (assumed, as llama.cpp's `/v1/models` endpoint does not expose it)
|
|
213
213
|
- **`cost`** — all zero (local models)
|
|
214
214
|
|
|
215
215
|
## Dependencies
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.2",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -36,8 +36,8 @@
|
|
|
36
36
|
"@earendil-works/pi-tui": "*"
|
|
37
37
|
},
|
|
38
38
|
"devDependencies": {
|
|
39
|
-
"@types/node": "^
|
|
39
|
+
"@types/node": "^26.0.0",
|
|
40
40
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
41
|
-
"vitest": "^4.1.
|
|
41
|
+
"vitest": "^4.1.9"
|
|
42
42
|
}
|
|
43
43
|
}
|
package/src/managers/events.ts
CHANGED
|
@@ -10,7 +10,6 @@ import { Server } from "../server";
|
|
|
10
10
|
|
|
11
11
|
export class EventManager {
|
|
12
12
|
static inflightModel: BaseModel | null = null;
|
|
13
|
-
private readonly resolver = new ConfigResolver();
|
|
14
13
|
|
|
15
14
|
constructor(private readonly servers: Server[]) {}
|
|
16
15
|
|
|
@@ -86,8 +85,9 @@ export class EventManager {
|
|
|
86
85
|
if (!isLlamaCpp) return payload;
|
|
87
86
|
|
|
88
87
|
// Retrieve pi's current thinking level, so we can setup a budget
|
|
89
|
-
const
|
|
90
|
-
const
|
|
88
|
+
const resolver = new ConfigResolver();
|
|
89
|
+
const level = resolver.resolveThinkingLevel() ?? "medium";
|
|
90
|
+
const budgets = resolver.resolveThinkingBudgets();
|
|
91
91
|
const thinking_budget_tokens = budgets[level];
|
|
92
92
|
|
|
93
93
|
// Setup payload
|
|
@@ -14,8 +14,7 @@ export class RouterModel extends BaseModel {
|
|
|
14
14
|
}
|
|
15
15
|
|
|
16
16
|
/**
|
|
17
|
-
* Workaround for
|
|
18
|
-
* (I suspect it was introduced in PR #22683 of llama.cpp)
|
|
17
|
+
* Workaround for /models status detection
|
|
19
18
|
*
|
|
20
19
|
* When a model is loaded for the very first time,
|
|
21
20
|
* this workaround will try to poll to /props instead of /models
|
package/src/server.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { PROVIDER_NAME, PROVIDER_PREFIX } from "./constants";
|
|
1
|
+
import { POLLING_INTERVAL, PROVIDER_NAME, PROVIDER_PREFIX } from "./constants";
|
|
2
2
|
import { Mode } from "./enums/mode";
|
|
3
3
|
import { ServerStatus } from "./enums/serverStatus";
|
|
4
4
|
import { HealthEndpoint } from "./interfaces/endpoints/health";
|
|
@@ -9,10 +9,14 @@ import { LegacyModel } from "./models/legacyModel";
|
|
|
9
9
|
import { RouterModel } from "./models/routerModel";
|
|
10
10
|
import { SingleModel } from "./models/singleModel";
|
|
11
11
|
import { ConfigResolver } from "./resolver";
|
|
12
|
+
import { Cache } from "./utils/cache";
|
|
13
|
+
import { Mutex } from "./utils/mutex";
|
|
12
14
|
|
|
13
15
|
export class Server {
|
|
14
16
|
public readonly models: BaseModel[] = [];
|
|
15
17
|
private configResolver = new ConfigResolver();
|
|
18
|
+
private cache = new Cache(POLLING_INTERVAL / 2);
|
|
19
|
+
private mutex = new Mutex();
|
|
16
20
|
|
|
17
21
|
constructor(readonly baseUrl: string) {}
|
|
18
22
|
|
|
@@ -39,9 +43,11 @@ export class Server {
|
|
|
39
43
|
}
|
|
40
44
|
|
|
41
45
|
/**
|
|
42
|
-
* Fetches models from the server and populates {@link models}
|
|
46
|
+
* Fetches models from the server and populates {@link models}.
|
|
47
|
+
* Clears the cache first so we always fetch fresh data.
|
|
43
48
|
*/
|
|
44
49
|
async initialize() {
|
|
50
|
+
this.cache.clear();
|
|
45
51
|
const { data } = await this.fetchModels();
|
|
46
52
|
const mode = await this.detectServerMode();
|
|
47
53
|
|
|
@@ -150,11 +156,13 @@ export class Server {
|
|
|
150
156
|
resource: "load" | "unload",
|
|
151
157
|
model: string,
|
|
152
158
|
): Promise<ModelsEndpoint> {
|
|
159
|
+
this.cache.clear();
|
|
153
160
|
return await this.rpc<ModelsEndpoint>(`/models/${resource}`, { model });
|
|
154
161
|
}
|
|
155
162
|
|
|
156
163
|
/**
|
|
157
|
-
* Makes
|
|
164
|
+
* Makes a cached, deduplicated request to the llama-server.
|
|
165
|
+
* Results are cached for half the {@link POLLING_INTERVAL} and in-flight requests are deduplicated.
|
|
158
166
|
*
|
|
159
167
|
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
160
168
|
* @param body The optional request body for POST requests
|
|
@@ -163,6 +171,33 @@ export class Server {
|
|
|
163
171
|
private async rpc<T>(
|
|
164
172
|
endpoint: string,
|
|
165
173
|
body?: Record<string, unknown>,
|
|
174
|
+
): Promise<T> {
|
|
175
|
+
const key = this.cacheKey(endpoint, body);
|
|
176
|
+
|
|
177
|
+
// Check cache
|
|
178
|
+
const cached = this.cache.get<T>(key);
|
|
179
|
+
if (cached !== undefined) {
|
|
180
|
+
return cached;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
// Deduplicate in-flight requests
|
|
184
|
+
return this.mutex.getOrCreate(key, async () => {
|
|
185
|
+
const data = await this.fetch<T>(endpoint, body);
|
|
186
|
+
this.cache.set(key, data);
|
|
187
|
+
return data;
|
|
188
|
+
});
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Makes an HTTP request to the llama-server and returns the parsed JSON response.
|
|
193
|
+
*
|
|
194
|
+
* @param endpoint The endpoint path to fetch (e.g. "/health")
|
|
195
|
+
* @param body The optional request body for POST requests
|
|
196
|
+
* @returns The parsed JSON response from the server
|
|
197
|
+
*/
|
|
198
|
+
private async fetch<T>(
|
|
199
|
+
endpoint: string,
|
|
200
|
+
body?: Record<string, unknown>,
|
|
166
201
|
): Promise<T> {
|
|
167
202
|
const url = `${this.baseUrl}${endpoint}`;
|
|
168
203
|
const apiKey = await this.getApiKey();
|
|
@@ -184,4 +219,15 @@ export class Server {
|
|
|
184
219
|
const response: T = await res.json();
|
|
185
220
|
return response;
|
|
186
221
|
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* Generates a cache key from the endpoint and body.
|
|
225
|
+
*
|
|
226
|
+
* @param endpoint The endpoint path (e.g. "/health")
|
|
227
|
+
* @param body The optional request body for POST requests
|
|
228
|
+
* @returns A key used for caching
|
|
229
|
+
*/
|
|
230
|
+
private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
|
|
231
|
+
return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
|
|
232
|
+
}
|
|
187
233
|
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Generic TTL cache.
|
|
3
|
+
* Entries expire after `ttl` milliseconds from the time they were set.
|
|
4
|
+
*/
|
|
5
|
+
export class Cache {
|
|
6
|
+
private entries = new Map<string, { data: unknown; timestamp: number }>();
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* @param ttl Time-to-live in milliseconds
|
|
10
|
+
*/
|
|
11
|
+
constructor(private readonly ttl: number) {}
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Gets a cached value by key. Returns `undefined` if missing or expired.
|
|
15
|
+
*/
|
|
16
|
+
get<T>(key: string): T | undefined {
|
|
17
|
+
const entry = this.entries.get(key);
|
|
18
|
+
if (!entry) return undefined;
|
|
19
|
+
if (Date.now() - entry.timestamp > this.ttl) {
|
|
20
|
+
this.entries.delete(key);
|
|
21
|
+
return undefined;
|
|
22
|
+
}
|
|
23
|
+
return entry.data as T;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Stores a value in the cache with the current timestamp.
|
|
28
|
+
*/
|
|
29
|
+
set(key: string, data: unknown): void {
|
|
30
|
+
this.entries.set(key, { data, timestamp: Date.now() });
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Clears all cached entries.
|
|
35
|
+
*/
|
|
36
|
+
clear(): void {
|
|
37
|
+
this.entries.clear();
|
|
38
|
+
}
|
|
39
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ensures only one in-flight operation exists per key.
|
|
3
|
+
* Concurrent callers for the same key share the same promise.
|
|
4
|
+
*/
|
|
5
|
+
export class Mutex {
|
|
6
|
+
private promises = new Map<string, Promise<unknown>>();
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Runs `fn` for the given key, or returns an existing in-flight promise.
|
|
10
|
+
* Concurrent callers for the same key share the same promise.
|
|
11
|
+
*/
|
|
12
|
+
getOrCreate<T>(key: string, fn: () => Promise<T>): Promise<T> {
|
|
13
|
+
const existing = this.promises.get(key);
|
|
14
|
+
if (existing) {
|
|
15
|
+
return existing as Promise<T>;
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
const promise = fn().finally(() => {
|
|
19
|
+
this.promises.delete(key);
|
|
20
|
+
});
|
|
21
|
+
this.promises.set(key, promise);
|
|
22
|
+
return promise;
|
|
23
|
+
}
|
|
24
|
+
}
|