pi-llama-cpp 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -77,6 +77,7 @@ Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (g
77
77
  ],
78
78
  "reactToModelSelect": true,
79
79
  "autoloadOnMessage": false,
80
+ "sortBy": "asc",
80
81
  "pollingTimeout": 60000,
81
82
  "serverTimeout": 1000
82
83
  }
@@ -101,11 +102,57 @@ With this config, the servers will appear in Pi as **Llama.cpp (Local Server)**
101
102
  | -------------------- | ------- | ------- | ------------------------------------------------------------- |
102
103
  | `reactToModelSelect` | boolean | `true` | Load the model when you switch via Pi's model picker. |
103
104
  | `autoloadOnMessage` | boolean | `false` | Automatically load an unloaded model before sending a message |
105
+ | `sortBy` | string | `"asc"` | Sort order for models (see below) |
104
106
  | `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
105
107
  | `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
106
108
 
107
109
  > **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
108
110
 
111
+ #### In-session settings menu
112
+
113
+ Run `/models settings` to edit the scalar settings above without hand-editing JSON:
114
+
115
+ - **Enter/Space** cycles the value under the cursor; **Esc** closes the menu.
116
+ - Booleans toggle `on`/`off`, `sortBy` cycles through the sort orders, and the
117
+ timeouts cycle through presets (`pollingTimeout`: 15s/30s/60s/120s/300s,
118
+ `serverTimeout`: 500ms/1s/2s/5s/10s).
119
+ - Changes are written to the **global** `~/.pi/agent/settings.json` only. If a
120
+ project `.pi/settings.json` defines the same key, its value keeps winning in
121
+ the merged view until you remove it there.
122
+ - Boolean and sort changes apply immediately; timeout changes apply on the next
123
+ model load.
124
+ - The `servers` list is edited with `/models servers` (see below).
125
+
126
+ #### Server list editor
127
+
128
+ Run `/models servers` to add, edit or remove entries of `llamaSettings.servers`
129
+ without hand-editing JSON:
130
+
131
+ - **↑/↓** moves the cursor, **Enter/e** edits the selected URL, **i** edits
132
+ its `id`, **n** its `name`, **a** adds a new entry, **d** deletes it
133
+ (after an "Are you sure?" confirmation — only **y** confirms;
134
+ **Enter** is ignored, **Esc/n** cancels), **Esc** closes the editor.
135
+ - One URL per entry (`http://host:port`). Trailing slashes are stripped on
136
+ save; `;`-separated values are rejected — use separate entries instead.
137
+ - Each change is written immediately to the **global**
138
+ `~/.pi/agent/settings.json`. If a project `.pi/settings.json` defines
139
+ `servers`, its list keeps winning in the merged view until you remove it
140
+ there.
141
+ - Changes apply the next time providers are scanned — run `/models` to see
142
+ them. Additions, removals, and URL/`id`/`name` edits all take effect on
143
+ the next `/models`: new servers register their providers, removed ones
144
+ leave pi's registry immediately, and edited ones are re-registered with
145
+ the fresh config — no restart needed.
146
+ - Limitation: a model already loading in the background on a removed or
147
+ edited server finishes loading, but its progress notifications stop;
148
+ re-select it from the (new) provider afterwards.
149
+ - The editor shows a warning when the `LLAMA_SERVER_URL` environment variable
150
+ is set, since it overrides the configured servers.
151
+ - Per-server `id`/`name` overrides can be edited with **i**/**n**; saving an
152
+ empty value clears the override. The list shows them as a
153
+ `(<id> - <name>)` suffix, falling back to the auto-detected
154
+ `llama-server=<url>` id when no custom `id` is set.
155
+
109
156
  #### Environment variable
110
157
 
111
158
  For a quick setup, you can use the `LLAMA_SERVER_URL` environment variable instead of the JSON config:
@@ -209,11 +256,13 @@ The extension determines the context size as follows:
209
256
 
210
257
  ### Commands
211
258
 
212
- | Command | Description |
213
- | ---------------- | ---------------------------------------------------------------------------------- |
214
- | `/models` | Browse your models with live status. Select a model to load, switch, or unload it. |
215
- | `/models info` | Show detailed information for all available models at once. |
216
- | `/models unload` | Unload all loaded models at once. |
259
+ | Command | Description |
260
+ | ------------------ | ---------------------------------------------------------------------------------- |
261
+ | `/models` | Browse your models with live status. Select a model to load, switch, or unload it. |
262
+ | `/models info` | Show detailed information for all available models at once. |
263
+ | `/models unload` | Unload all loaded models at once. |
264
+ | `/models servers` | Add, edit or remove llama.cpp server URLs via a TUI editor. |
265
+ | `/models settings` | Open a menu to edit the scalar `llamaSettings` fields. |
217
266
 
218
267
  > **Note:** When a llama.cpp server is slow to respond, it will be skipped at startup with a warning. Run `/models` to retry without timeout and see all models.
219
268
 
@@ -221,6 +270,18 @@ The extension determines the context size as follows:
221
270
 
222
271
  > **Note:** The `/models unload` command only makes sense in router mode.
223
272
 
273
+ #### Model sorting
274
+
275
+ The order of models in the `/models` menu is controlled by the `sortBy` setting:
276
+
277
+ | Value | Description |
278
+ | ------------- | --------------------------------------------------------------------------------------- |
279
+ | `"asc"` | Sort by model ID ascending (default) |
280
+ | `"desc"` | Sort by model ID descending |
281
+ | `"asc-name"` | Sort by model name ascending (ties broken by ID) |
282
+ | `"desc-name"` | Sort by model name descending (ties broken by ID) |
283
+ | `"api"` | No sorting — models appear in the order returned by each server's `/v1/models` endpoint |
284
+
224
285
  ### Model Actions
225
286
 
226
287
  When browsing models via the `/models` command, you can:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.10.0",
3
+ "version": "0.11.0",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -38,8 +38,8 @@
38
38
  },
39
39
  "type": "module",
40
40
  "devDependencies": {
41
- "@types/node": "^26.4.0",
41
+ "@types/node": "^26.4.1",
42
42
  "prettier-plugin-organize-imports": "^4.3.0",
43
- "vitest": "^4.1.11"
43
+ "vitest": "^5.0.0"
44
44
  }
45
45
  }
package/src/api/client.ts CHANGED
@@ -1,13 +1,29 @@
1
1
  import { POLLING_INTERVAL } from "../constants";
2
- import { Cache } from "../utils/cache";
3
- import { Mutex } from "../utils/mutex";
4
2
 
5
3
  /**
6
- * HTTP client for llama-server with caching and deduplication.
4
+ * How long GET responses stay cached: half the polling interval, so a poll
5
+ * tick always reaches the server while multiple reads within one tick
6
+ * (fan-outs like `toProviderConfig`, concurrent model polls) do not.
7
+ */
8
+ const CACHE_TTL = POLLING_INTERVAL / 2;
9
+
10
+ /**
11
+ * HTTP client for llama-server with GET caching and request deduplication.
12
+ *
13
+ * Two complementary mechanisms keep repeated reads from reaching the server:
14
+ * - the TTL cache absorbs time-spaced repeats (poll loops, sequential reads);
15
+ * - the in-flight map absorbs simultaneous bursts: concurrent callers for the
16
+ * same key share one request's promise.
17
+ *
18
+ * POST requests are deduplicated but never cached — they are not idempotent
19
+ * reads, and the only caller (`Server.postRequest()`) clears the cache around
20
+ * them. The dedup matters: independent load triggers (command, auto-load,
21
+ * model-select) can race, and merging their duplicate `POST /models/load`
22
+ * into one server call avoids load-state glitches.
7
23
  */
8
24
  export class ApiClient {
9
- private cache = new Cache(POLLING_INTERVAL / 2);
10
- private mutex = new Mutex();
25
+ private cache = new Map<string, { data: unknown; timestamp: number }>();
26
+ private inflight = new Map<string, Promise<unknown>>();
11
27
 
12
28
  /**
13
29
  * Creates a new ApiClient.
@@ -22,40 +38,34 @@ export class ApiClient {
22
38
 
23
39
  /**
24
40
  * Makes a cached, deduplicated GET request to the llama-server.
25
- * Results are cached for half the polling interval and in-flight requests are deduplicated.
26
41
  *
27
42
  * @param endpoint The endpoint path to fetch (e.g. "/health")
28
43
  * @returns The parsed JSON response from the server
29
44
  */
30
45
  async get<T>(endpoint: string): Promise<T> {
31
- const cached = this.cache.get<T>(endpoint);
46
+ const cached = this.cacheGet<T>(endpoint);
32
47
  if (cached !== undefined) return cached;
33
48
 
34
- return this.mutex.getOrCreate(endpoint, async () => {
35
- const data = (await this.do_get<T>(endpoint)) as T;
36
- this.cache.set(endpoint, data);
49
+ return this.dedupe(endpoint, async () => {
50
+ const data = await this.do_get<T>(endpoint);
51
+ this.cacheSet(endpoint, data);
37
52
  return data;
38
53
  });
39
54
  }
40
55
 
41
56
  /**
42
- * Makes a cached, deduplicated POST request to the llama-server.
43
- * Results are cached for half the polling interval and in-flight requests are deduplicated.
57
+ * Makes a deduplicated POST request to the llama-server.
58
+ * Concurrent duplicate requests share one server call; responses are
59
+ * never cached.
44
60
  *
45
61
  * @param endpoint The endpoint path to post to
46
62
  * @param body The optional request body
47
63
  * @returns The parsed JSON response from the server
48
64
  */
49
65
  async post<T>(endpoint: string, body?: Record<string, unknown>): Promise<T> {
50
- const key = this.cacheKey(endpoint, body);
51
- const cached = this.cache.get<T>(key);
52
- if (cached !== undefined) return cached;
53
-
54
- return this.mutex.getOrCreate(key, async () => {
55
- const data = (await this.do_post<T>(endpoint, body)) as T;
56
- this.cache.set(key, data);
57
- return data;
58
- });
66
+ return this.dedupe(this.cacheKey(endpoint, body), async () =>
67
+ this.do_post<T>(endpoint, body),
68
+ );
59
69
  }
60
70
 
61
71
  /**
@@ -65,6 +75,51 @@ export class ApiClient {
65
75
  this.cache.clear();
66
76
  }
67
77
 
78
+ /**
79
+ * Runs `fn` for the given key, or returns an existing in-flight promise.
80
+ * Concurrent callers for the same key share the same promise.
81
+ */
82
+ private dedupe<T>(key: string, fn: () => Promise<T>): Promise<T> {
83
+ const existing = this.inflight.get(key);
84
+ if (existing) return existing as Promise<T>;
85
+
86
+ const promise = fn().finally(() => {
87
+ this.inflight.delete(key);
88
+ });
89
+ this.inflight.set(key, promise);
90
+ return promise;
91
+ }
92
+
93
+ /**
94
+ * Gets a cached GET response. Returns `undefined` if missing or expired.
95
+ */
96
+ private cacheGet<T>(key: string): T | undefined {
97
+ const entry = this.cache.get(key);
98
+ if (!entry) return undefined;
99
+ if (Date.now() - entry.timestamp > CACHE_TTL) {
100
+ this.cache.delete(key);
101
+ return undefined;
102
+ }
103
+ return entry.data as T;
104
+ }
105
+
106
+ /**
107
+ * Stores a GET response in the cache with the current timestamp.
108
+ */
109
+ private cacheSet(key: string, data: unknown): void {
110
+ this.cache.set(key, { data, timestamp: Date.now() });
111
+ }
112
+
113
+ /**
114
+ * Builds the dedup key for a POST request.
115
+ *
116
+ * @param endpoint The endpoint path to post to
117
+ * @param body The optional request body
118
+ */
119
+ private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
120
+ return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
121
+ }
122
+
68
123
  /**
69
124
  * Makes a raw GET request to the llama-server.
70
125
  * This bypasses caching and deduplication.
@@ -107,15 +162,4 @@ export class ApiClient {
107
162
 
108
163
  return res.json();
109
164
  }
110
-
111
- /**
112
- * Sets a cache key
113
- *
114
- * @param endpoint The endpoint path to post to
115
- * @param body The optional request body
116
- * @returns The cache key
117
- */
118
- private cacheKey(endpoint: string, body?: Record<string, unknown>): string {
119
- return body ? `${endpoint}:${JSON.stringify(body)}` : endpoint;
120
- }
121
165
  }
package/src/constants.ts CHANGED
@@ -3,6 +3,11 @@
3
3
  */
4
4
  export const PROVIDER_PREFIX = "llama-server";
5
5
 
6
+ /**
7
+ * The settings key used in project/global settings.
8
+ */
9
+ export const SETTINGS_KEY = "llamaSettings";
10
+
6
11
  /**
7
12
  * This provider's name
8
13
  */
@@ -58,6 +63,16 @@ export const REACT_TO_MODEL_SELECT = true;
58
63
  */
59
64
  export const AUTOLOAD_ON_MESSAGE = false;
60
65
 
66
+ /**
67
+ * Default sort order for model lists.
68
+ */
69
+ export const SORT_BY = "asc";
70
+
71
+ /**
72
+ * Sort order options for model lists.
73
+ */
74
+ export type SortBy = "asc" | "desc" | "asc-name" | "desc-name" | "api";
75
+
61
76
  /**
62
77
  * Thinking budgets to send to the server, depending on user-selected level in Pi.
63
78
  */
package/src/index.ts CHANGED
@@ -14,11 +14,9 @@ import { ServerManager } from "./managers/server";
14
14
  import { settings } from "./managers/settings";
15
15
 
16
16
  export default async function (pi: ExtensionAPI) {
17
- const servers = await settings.resolveServers();
18
-
19
- const eventManager = new EventManager(servers);
20
- const serverManager = new ServerManager(servers);
21
- const commandManager = new CommandManager(serverManager);
17
+ const serverManager = new ServerManager(settings);
18
+ const eventManager = new EventManager(serverManager, settings);
19
+ const commandManager = new CommandManager(serverManager, settings);
22
20
 
23
21
  // Register providers once at startup
24
22
  await serverManager.initialize(pi);
@@ -1,3 +1,14 @@
1
- export interface ModelSelectEvent {
2
- model: { id: string; provider: string };
3
- }
1
+ import type { ExtensionEvent } from "@earendil-works/pi-coding-agent";
2
+
3
+ /**
4
+ * pi-coding-agent does not re-export its `ModelSelectEvent` from the public
5
+ * API (it exists in the internal extensions/index but is omitted from the
6
+ * root re-export list, and the package's `exports` map blocks deep imports).
7
+ * It is therefore derived from the exported `ExtensionEvent` union, which
8
+ * yields pi's real event shape (`model: Model<any>`, `previousModel`,
9
+ * `source`) and tracks the API automatically.
10
+ */
11
+ export type ModelSelectEvent = Extract<
12
+ ExtensionEvent,
13
+ { type: "model_select" }
14
+ >;
@@ -0,0 +1,20 @@
1
+ /**
2
+ * Identity of a llama.cpp server endpoint.
3
+ *
4
+ * Persisted counterpart: {@link LlamaServer} (`llamaSettings.servers`),
5
+ * whose `url`/`id`/`name` map onto these fields.
6
+ */
7
+ export interface ServerOptions {
8
+ /**
9
+ * Base URL of the llama.cpp server (e.g. "http://127.0.0.1:8080").
10
+ */
11
+ baseUrl: string;
12
+ /**
13
+ * Custom provider ID; falls back to a URL-based one.
14
+ */
15
+ customId?: string;
16
+ /**
17
+ * Custom provider name suffix; falls back to the base URL.
18
+ */
19
+ customName?: string;
20
+ }
@@ -1,7 +1,9 @@
1
+ import type { SortBy } from "../constants";
2
+
1
3
  /**
2
4
  * A description of a server in the "llamaSettings" key
3
5
  */
4
- interface LlamaServer {
6
+ export interface LlamaServer {
5
7
  /**
6
8
  * The URL of the llama.cpp server.
7
9
  */
@@ -29,14 +31,16 @@ interface LlamaServer {
29
31
  * }],
30
32
  * "reactToModelSelect": true
31
33
  * "autoloadOnMessage": false
34
+ * "sortBy": "asc"
32
35
  * }
33
36
  }
34
37
  */
35
38
  export interface LlamaSettings {
36
39
  /**
37
40
  * List of servers to connect to.
41
+ * @default []
38
42
  */
39
- servers: LlamaServer[];
43
+ servers?: LlamaServer[];
40
44
  /**
41
45
  * Whether to react to model selection events by loading the model.
42
46
  * @default true
@@ -57,4 +61,9 @@ export interface LlamaSettings {
57
61
  * @default 1000
58
62
  */
59
63
  serverTimeout?: number;
64
+ /**
65
+ * How to sort models in the /models command.
66
+ * @default "asc"
67
+ */
68
+ sortBy?: SortBy;
60
69
  }