pi-llamacpp-infra 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +45 -17
- package/package.json +5 -3
- package/src/index.ts +291 -47
- package/src/prompt-warmup.ts +15 -2
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
# llamacpp-infra — Discovery, Metrics & Control for llama.cpp Servers
|
|
1
|
+
# llamacpp-infra — Discovery, Metrics & Control for llama.cpp-family Servers
|
|
2
2
|
|
|
3
|
-
**llamacpp-infra** turns pi into a first-class citizen of local llama.cpp infrastructure. It probes any number of machines — localhost, LAN or Tailscale — discovers every model served by llama.cpp and its variants, registers them into pi's native `/model` list, and gives you live Prometheus metrics, per-model thinking budgets, vision detection and a full configuration UI — all without leaving the pi prompt.
|
|
3
|
+
**llamacpp-infra** turns pi into a first-class citizen of local llama.cpp infrastructure. It probes any number of machines — localhost, LAN or Tailscale — discovers every model served by llama.cpp and its variants (including LM Studio), registers them into pi's native `/model` list, and gives you live Prometheus metrics, per-model thinking budgets, vision detection and a full configuration UI — all without leaving the pi prompt.
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
@@ -14,17 +14,20 @@ Every endpoint llamacpp-infra talks to runs llama.cpp or a direct variant:
|
|
|
14
14
|
| **ZINC** | `owned_by: "zinc"` | Payload workaround: empty model field + tool normalization |
|
|
15
15
|
| **DwarfStar / ds4** | Opt-in ping to `/v1/chat/completions` | antirez's ds4-server for DeepSeek V4 (per-server `probeDs4` flag) |
|
|
16
16
|
| **lucebox** | `GET /props` with `server.name: "luce-*"` | DeepSeek dflash server with rich metadata |
|
|
17
|
+
| **LM Studio** | `GET /v1/models` + optional `/api/v1/models` metadata | Local OpenAI-compatible server backed by llama.cpp; default port `1234` |
|
|
17
18
|
|
|
18
|
-
Anything else (
|
|
19
|
+
Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-in providers for those.
|
|
19
20
|
|
|
20
21
|
## Features
|
|
21
22
|
|
|
22
23
|
- **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
|
|
24
|
+
- **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
|
|
23
25
|
- **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
|
|
24
26
|
- **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
|
|
25
27
|
- **Live Prometheus metrics** — polls `/metrics` (or `/stats`) and renders a compact widget with instantaneous prompt/gen throughput; auto-activates for llamacpp-infra models only
|
|
26
28
|
- **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
|
|
27
29
|
- **Header warmup** — pre-caches the system prompt KV on llama.cpp-family servers so the first real request is faster
|
|
30
|
+
- **LM Studio support** — uses LM Studio's OpenAI-compatible `/v1` API, enriches names/context/quant/vision from `/api/v1/models` (or legacy `/api/v0/models`), and avoids llama.cpp-only request fields
|
|
28
31
|
- **ZINC workaround** — ZINC rejects non-empty model IDs; the payload hook rewrites the request and normalizes tool definitions automatically
|
|
29
32
|
- **Vision detection** — scans `/proc` for local llama-server processes launched with `--mmproj` and marks those models as image-capable; also reads server-reported `modalities` / `input_modalities`
|
|
30
33
|
- **Native configuration UI** — everything configurable through `/llamacpp-infra config` with pi's native menus; no config file editing required
|
|
@@ -39,7 +42,7 @@ llamacpp-infra is a [pi package](https://pi.dev/packages): one extension (`src/i
|
|
|
39
42
|
pi install git:github.com/noguerol/llamacpp-infra
|
|
40
43
|
|
|
41
44
|
# Pin a tag/commit
|
|
42
|
-
pi install git:github.com/noguerol/llamacpp-infra@v1.
|
|
45
|
+
pi install git:github.com/noguerol/llamacpp-infra@v1.2.0
|
|
43
46
|
|
|
44
47
|
# From npm
|
|
45
48
|
pi install npm:pi-llamacpp-infra
|
|
@@ -58,18 +61,33 @@ pi remove npm:pi-llamacpp-infra
|
|
|
58
61
|
|
|
59
62
|
> **Security:** pi packages run with full system access. Install only packages you trust and review the source.
|
|
60
63
|
|
|
61
|
-
**Requirements:** a working pi installation and at least one llama.cpp-family server running somewhere accessible (localhost, LAN or Tailscale).
|
|
64
|
+
**Requirements:** a working pi installation and at least one llama.cpp-family server running somewhere accessible (localhost, LAN or Tailscale). LM Studio works when its local server is started (Developer tab or `lms server start`, usually on `http://localhost:1234/v1`).
|
|
62
65
|
|
|
63
66
|
## Quick Start
|
|
64
67
|
|
|
65
68
|
```
|
|
66
|
-
/llamacpp-infra config # open the config menu → add your first server
|
|
69
|
+
/llamacpp-infra config # open the config menu → add your first server (LM Studio: port 1234)
|
|
67
70
|
/llamacpp-infra scan # discover models now
|
|
68
71
|
/llamacpp-infra list # see what was found
|
|
69
72
|
```
|
|
70
73
|
|
|
71
74
|
That's it. On the next pi startup, llamacpp-infra probes your servers automatically and registers every model into `/model`. Switch models with `/model` as usual.
|
|
72
75
|
|
|
76
|
+
### LM Studio quick setup
|
|
77
|
+
|
|
78
|
+
LM Studio exposes an OpenAI-compatible API on `/v1` (default `http://localhost:1234/v1`) and a richer local REST API for model metadata on `/api/v1/models` (older LM Studio versions used `/api/v0/models`). llamacpp-infra probes `/v1/models` as the source of usable model IDs and, when available, enriches them with LM Studio's context length, display name, quantization and vision capability.
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
# Start LM Studio's local server (or use the Developer tab in the GUI)
|
|
82
|
+
lms server start
|
|
83
|
+
|
|
84
|
+
# In pi: add/select host 127.0.0.1 with port 1234, then scan
|
|
85
|
+
/llamacpp-infra config
|
|
86
|
+
/llamacpp-infra scan
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
No special payload workaround is required: LM Studio accepts standard OpenAI chat-completions requests. llamacpp-infra deliberately does **not** send llama.cpp-only fields such as `cache_prompt` or `thinking_budget_tokens` to LM Studio.
|
|
90
|
+
|
|
73
91
|
## Commands
|
|
74
92
|
|
|
75
93
|
| Command | Description |
|
|
@@ -102,12 +120,14 @@ Shows every discovered model with metadata badges:
|
|
|
102
120
|
```
|
|
103
121
|
📋 Discovered models (8)
|
|
104
122
|
|
|
105
|
-
1.
|
|
106
|
-
2.
|
|
107
|
-
3.
|
|
108
|
-
4.
|
|
123
|
+
1. Qwen3.6-27B-UD-Q3_K_XL (local:8080) 👁️ 🗜️ UD-Q3_K_XL
|
|
124
|
+
2. DeepSeek-V4-Flash-ROCMFP2 (local:8081) 🗜️ ROCMFP2
|
|
125
|
+
3. Meta-Llama-3.1-8B (myserver:8080) 🚀 draft-model 🗜️ Q4_K_M
|
|
126
|
+
4. gemma-3-4b-it (myserver:8081) 👁️ 🗜️ Q4_K_M
|
|
109
127
|
```
|
|
110
128
|
|
|
129
|
+
The same compact id is what pi's `/model` picker shows, with the serving machine in parentheses.
|
|
130
|
+
|
|
111
131
|
### `/llamacpp-infra status`
|
|
112
132
|
|
|
113
133
|
Detailed per-endpoint report:
|
|
@@ -135,7 +155,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
135
155
|
"id": "local",
|
|
136
156
|
"host": "127.0.0.1",
|
|
137
157
|
"label": "Local",
|
|
138
|
-
"ports": [8000, 8001, 8002, 8080, 8081, 8082],
|
|
158
|
+
"ports": [8000, 8001, 8002, 8080, 8081, 8082, 1234],
|
|
139
159
|
"enabled": true,
|
|
140
160
|
"probeDs4": false
|
|
141
161
|
},
|
|
@@ -164,7 +184,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
164
184
|
"metricsPollMs": 5000
|
|
165
185
|
},
|
|
166
186
|
"modelOptions": {
|
|
167
|
-
"
|
|
187
|
+
"Qwen3.6-27B (myserver:8080)": {
|
|
168
188
|
"thinkingBudgets": {
|
|
169
189
|
"minimal": 256,
|
|
170
190
|
"low": 1024,
|
|
@@ -183,7 +203,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
183
203
|
| `id` | required | Unique short id (used in model IDs and logs) |
|
|
184
204
|
| `host` | required | Hostname, tailnet name or IP |
|
|
185
205
|
| `label` | `host` | Friendly name shown in menus |
|
|
186
|
-
| `ports` | required | Array of ports to probe |
|
|
206
|
+
| `ports` | required | Array of ports to probe (`1234` is LM Studio's usual local server port) |
|
|
187
207
|
| `enabled` | `true` | Whether to probe this server |
|
|
188
208
|
| `probeDs4` | `false` | Opt-in: ping `/v1/chat/completions` for DwarfStar/ds4 servers |
|
|
189
209
|
| `apiKey` | — | Optional bearer token sent on discovery and per-model requests |
|
|
@@ -198,7 +218,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
198
218
|
| `startupGraceMs` | `40000` | Keep trying at startup while nothing has answered |
|
|
199
219
|
| `knownGoodFailLimit` | `3` | Consecutive failures before a live endpoint is dropped |
|
|
200
220
|
| `detectVision` | `true` | Scan `/proc` for `--mmproj` + read server-reported modalities |
|
|
201
|
-
| `prefixModelIds` | `true` | `host:port
|
|
221
|
+
| `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
|
|
202
222
|
| `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
|
|
203
223
|
| `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
|
|
204
224
|
| `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
|
|
@@ -213,7 +233,15 @@ Levels: `minimal`, `low`, `medium`, `high`, `xhigh`, `max`.
|
|
|
213
233
|
|
|
214
234
|
## Model ID Format
|
|
215
235
|
|
|
216
|
-
|
|
236
|
+
Models are registered with compact display ids: `ModelName (host:port)`, e.g. `Qwen3.6-27B-UD-Q3_K_XL (myserver:8080)` — the compact model name plus the machine serving it in parentheses, matching how pi shows native provider models. Localhost servers (`127.0.0.1`, `localhost`) use `local:port` in the tag.
|
|
237
|
+
|
|
238
|
+
pi sends the compact id to the extension's request hook, which transparently rewrites it to the raw server-side id (the GGUF path, alias or router id the server advertised in `/v1/models`) before the request leaves pi. Config keys under `modelOptions` use the compact id; legacy `host:port/model` keys are migrated automatically on the first scan.
|
|
239
|
+
|
|
240
|
+
With `prefixModelIds: false` the machine tag is omitted (`ModelName`); it is re-added automatically only when two models would otherwise collide.
|
|
241
|
+
|
|
242
|
+
### Thinking budgets in the UI
|
|
243
|
+
|
|
244
|
+
llama.cpp-family models are registered as reasoning models, exactly like a native pi provider: the footer shows `ModelName (host:port) • <level>`, the thinking selector offers levels with token estimates, and pi sends the configured `thinking_budget_tokens` budget on each request. Per-model budgets configured in the extension override pi's global per-level budgets.
|
|
217
245
|
|
|
218
246
|
## Live Metrics Widget
|
|
219
247
|
|
|
@@ -233,13 +261,13 @@ llamacpp-infra/
|
|
|
233
261
|
├── LICENSE # MIT
|
|
234
262
|
├── README.md
|
|
235
263
|
└── src/
|
|
236
|
-
├── index.ts # Extension entry point (~
|
|
264
|
+
├── index.ts # Extension entry point (~2600 lines)
|
|
237
265
|
└── prompt-warmup.ts # Header warmup module (inlined, ~600 lines)
|
|
238
266
|
```
|
|
239
267
|
|
|
240
268
|
Two-file extension with zero external dependencies (only pi's bundled `@earendil-works/pi-coding-agent` + Node built-ins):
|
|
241
269
|
|
|
242
|
-
- **Discovery engine** — multi-server probing with timeouts, retry budgets, and per-server kind detection (llama.cpp, ZINC, DwarfStar, lucebox)
|
|
270
|
+
- **Discovery engine** — multi-server probing with timeouts, retry budgets, and per-server kind detection (llama.cpp, ZINC, DwarfStar, lucebox, LM Studio)
|
|
243
271
|
- **Router support** — single-model and multi-model llama.cpp modes with per-model status, args parsing and metadata extraction
|
|
244
272
|
- **Metrics subsystem** — Prometheus endpoint discovery, polling, and compact widget rendering
|
|
245
273
|
- **Thinking budgets** — per-model per-level configuration with automatic `reasoning` registration
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "Discovery, metrics and control of llama.cpp-family servers for pi: probes any number of machines (localhost, LAN, Tailscale), registers every model into pi's native /model list, and provides live Prometheus metrics, per-model thinking budgets, vision detection and a native config UI. Supports llama.cpp, ZINC, DwarfStar/ds4 and
|
|
3
|
+
"version": "1.2.0",
|
|
4
|
+
"description": "Discovery, metrics and control of llama.cpp-family servers for pi: probes any number of machines (localhost, LAN, Tailscale), registers every model into pi's native /model list, and provides live Prometheus metrics, per-model thinking budgets, vision detection and a native config UI. Supports llama.cpp, ZINC, DwarfStar/ds4, lucebox and LM Studio.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
|
7
7
|
"llamacpp",
|
|
@@ -11,7 +11,9 @@
|
|
|
11
11
|
"metrics",
|
|
12
12
|
"tailscale",
|
|
13
13
|
"vision",
|
|
14
|
-
"thinking"
|
|
14
|
+
"thinking",
|
|
15
|
+
"lm-studio",
|
|
16
|
+
"lmstudio"
|
|
15
17
|
],
|
|
16
18
|
"author": "Javier Noguerol <https://github.com/noguerol>",
|
|
17
19
|
"license": "MIT",
|
package/src/index.ts
CHANGED
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
* extension: llamacpp-infra
|
|
3
3
|
* =========================
|
|
4
4
|
* Discovery, metrics and control of models served by llama.cpp and its
|
|
5
|
-
* variants (ZINC, DwarfStar/ds4, lucebox,
|
|
6
|
-
*
|
|
5
|
+
* variants (ZINC, DwarfStar/ds4, lucebox, LM Studio) on any number of
|
|
6
|
+
* machines — localhost, LAN or Tailscale.
|
|
7
7
|
*
|
|
8
8
|
* SCOPE
|
|
9
9
|
* -----
|
|
@@ -15,8 +15,9 @@
|
|
|
15
15
|
* ZINC — llama.cpp-compatible runtime (owned_by: "zinc")
|
|
16
16
|
* DwarfStar — antirez's ds4-server for DeepSeek V4 (chat ping probe)
|
|
17
17
|
* lucebox — DeepSeek dflash server (rich /props metadata)
|
|
18
|
+
* LM Studio — local OpenAI-compatible server backed by llama.cpp
|
|
18
19
|
*
|
|
19
|
-
* Anything else (
|
|
20
|
+
* Anything else (vLLM, Ollama, cloud APIs…) is out of scope.
|
|
20
21
|
*
|
|
21
22
|
* FEATURES
|
|
22
23
|
* --------
|
|
@@ -47,6 +48,12 @@
|
|
|
47
48
|
* 7. Header warmup (pre-cache system prompt KV on llama.cpp-family servers).
|
|
48
49
|
* 8. ZINC workaround: ZINC rejects non-empty model ids; the payload hook
|
|
49
50
|
* rewrites the request accordingly and normalizes tool definitions.
|
|
51
|
+
* 9. LM Studio support: discovers `/v1/models`, enriches metadata from
|
|
52
|
+
* `/api/v1/models` or legacy `/api/v0/models`, and avoids llama.cpp-only
|
|
53
|
+
* request fields such as `cache_prompt` / `thinking_budget_tokens`.
|
|
54
|
+
* 10. Compact model ids: models are registered as "Name (host:port)" — the
|
|
55
|
+
* raw server-side model id (GGUF path / alias) is restored automatically
|
|
56
|
+
* in before_provider_request before the request leaves pi.
|
|
50
57
|
*
|
|
51
58
|
* HISTORY
|
|
52
59
|
* -------
|
|
@@ -149,6 +156,37 @@ interface LlamaCppMeta {
|
|
|
149
156
|
n_ctx_train?: number;
|
|
150
157
|
}
|
|
151
158
|
|
|
159
|
+
interface LmStudioModelInfo {
|
|
160
|
+
/** v1 REST API unique key; v0 uses `id` instead. */
|
|
161
|
+
key?: string;
|
|
162
|
+
id?: string;
|
|
163
|
+
display_name?: string;
|
|
164
|
+
/** v1: "llm" | "embedding"; v0: "llm" | "vlm" | "embeddings". */
|
|
165
|
+
type?: string;
|
|
166
|
+
publisher?: string;
|
|
167
|
+
architecture?: string | null;
|
|
168
|
+
arch?: string | null;
|
|
169
|
+
format?: string | null;
|
|
170
|
+
compatibility_type?: string | null;
|
|
171
|
+
quantization?: string | { name?: string | null; bits_per_weight?: number | null } | null;
|
|
172
|
+
state?: string;
|
|
173
|
+
max_context_length?: number;
|
|
174
|
+
loaded_instances?: Array<{ id?: string; config?: { context_length?: number; parallel?: number } }>;
|
|
175
|
+
capabilities?: {
|
|
176
|
+
vision?: boolean;
|
|
177
|
+
trained_for_tool_use?: boolean;
|
|
178
|
+
reasoning?: boolean | { allowed_options?: string[]; default?: string };
|
|
179
|
+
};
|
|
180
|
+
selected_variant?: string;
|
|
181
|
+
variants?: string[];
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
interface LmStudioModelsResponse {
|
|
185
|
+
models?: LmStudioModelInfo[];
|
|
186
|
+
data?: LmStudioModelInfo[];
|
|
187
|
+
object?: string;
|
|
188
|
+
}
|
|
189
|
+
|
|
152
190
|
interface LlamaCppModel {
|
|
153
191
|
id: string;
|
|
154
192
|
name?: string;
|
|
@@ -170,6 +208,8 @@ interface LlamaCppModel {
|
|
|
170
208
|
input_modalities?: string[];
|
|
171
209
|
output_modalities?: string[];
|
|
172
210
|
};
|
|
211
|
+
/** LM Studio REST metadata, when the local server exposes it. */
|
|
212
|
+
lmStudio?: LmStudioModelInfo;
|
|
173
213
|
meta?: LlamaCppMeta;
|
|
174
214
|
context_window?: number;
|
|
175
215
|
context_length?: number;
|
|
@@ -210,7 +250,7 @@ interface ServerProps {
|
|
|
210
250
|
cache_type_v?: string;
|
|
211
251
|
}
|
|
212
252
|
|
|
213
|
-
type ServerKind = "llamacpp" | "zinc" | "lucebox" | "dwarfstar";
|
|
253
|
+
type ServerKind = "llamacpp" | "zinc" | "lucebox" | "dwarfstar" | "lmstudio";
|
|
214
254
|
type ServerMode = "single" | "router" | "unknown";
|
|
215
255
|
|
|
216
256
|
/** Per-model metadata extracted during discovery. */
|
|
@@ -295,7 +335,7 @@ const DEFAULT_SERVERS: ServerConfig[] = [
|
|
|
295
335
|
id: "local",
|
|
296
336
|
host: "127.0.0.1",
|
|
297
337
|
label: "Local",
|
|
298
|
-
ports: [8000, 8001, 8002, 8080, 8081, 8082],
|
|
338
|
+
ports: [8000, 8001, 8002, 8080, 8081, 8082, 1234],
|
|
299
339
|
enabled: true,
|
|
300
340
|
probeDs4: false,
|
|
301
341
|
},
|
|
@@ -681,22 +721,109 @@ async function fetchServerProps(
|
|
|
681
721
|
return undefined;
|
|
682
722
|
}
|
|
683
723
|
|
|
724
|
+
function lmStudioKey(info: LmStudioModelInfo): string | undefined {
|
|
725
|
+
return info.key ?? info.id;
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
function lmStudioQuantName(info: LmStudioModelInfo | undefined): string | undefined {
|
|
729
|
+
if (!info) return undefined;
|
|
730
|
+
const q = info.quantization;
|
|
731
|
+
if (typeof q === "string") return q || undefined;
|
|
732
|
+
return q?.name ?? undefined;
|
|
733
|
+
}
|
|
734
|
+
|
|
735
|
+
function lmStudioContextLength(info: LmStudioModelInfo | undefined): number | undefined {
|
|
736
|
+
if (!info) return undefined;
|
|
737
|
+
const loadedCtx = info.loaded_instances?.find((inst) => typeof inst.config?.context_length === "number")?.config?.context_length;
|
|
738
|
+
return loadedCtx ?? info.max_context_length;
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
function isLmStudioModelInfo(value: unknown): value is LmStudioModelInfo {
|
|
742
|
+
const m = value as LmStudioModelInfo | undefined;
|
|
743
|
+
if (!m || typeof m !== "object") return false;
|
|
744
|
+
return Boolean(
|
|
745
|
+
m.key ||
|
|
746
|
+
m.id ||
|
|
747
|
+
m.display_name ||
|
|
748
|
+
m.compatibility_type ||
|
|
749
|
+
m.format ||
|
|
750
|
+
m.loaded_instances ||
|
|
751
|
+
typeof m.max_context_length === "number" ||
|
|
752
|
+
m.capabilities,
|
|
753
|
+
);
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
async function fetchLmStudioCatalog(
|
|
757
|
+
baseUrl: string,
|
|
758
|
+
timeoutMs: number,
|
|
759
|
+
apiKey?: string,
|
|
760
|
+
): Promise<LmStudioModelInfo[] | undefined> {
|
|
761
|
+
const rootUrl = baseUrl.replace(/\/v1\/?$/, "");
|
|
762
|
+
// LM Studio v1 is current; v0 is still common in older installations.
|
|
763
|
+
for (const path of ["/api/v1/models", "/api/v0/models"]) {
|
|
764
|
+
try {
|
|
765
|
+
const { status, body } = await httpGet(`${rootUrl}${path}`, timeoutMs, apiKey);
|
|
766
|
+
if (status < 200 || status >= 300) continue;
|
|
767
|
+
const payload = JSON.parse(body) as LmStudioModelsResponse;
|
|
768
|
+
const models = Array.isArray(payload.models) ? payload.models : Array.isArray(payload.data) ? payload.data : undefined;
|
|
769
|
+
if (models && models.length > 0 && models.some(isLmStudioModelInfo)) return models;
|
|
770
|
+
} catch {
|
|
771
|
+
// Not LM Studio, or an older server without the REST metadata endpoint.
|
|
772
|
+
}
|
|
773
|
+
}
|
|
774
|
+
return undefined;
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
function buildLmStudioCatalogMap(catalog: LmStudioModelInfo[]): Map<string, LmStudioModelInfo> {
|
|
778
|
+
const map = new Map<string, LmStudioModelInfo>();
|
|
779
|
+
const add = (key: string | undefined, info: LmStudioModelInfo) => {
|
|
780
|
+
const k = key?.trim().toLowerCase();
|
|
781
|
+
if (k && !map.has(k)) map.set(k, info);
|
|
782
|
+
};
|
|
783
|
+
for (const info of catalog) {
|
|
784
|
+
add(lmStudioKey(info), info);
|
|
785
|
+
add(info.key, info);
|
|
786
|
+
add(info.id, info);
|
|
787
|
+
add(info.selected_variant, info);
|
|
788
|
+
add(info.display_name, info);
|
|
789
|
+
for (const variant of info.variants ?? []) add(variant, info);
|
|
790
|
+
for (const inst of info.loaded_instances ?? []) add(inst.id, info);
|
|
791
|
+
}
|
|
792
|
+
return map;
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
function enrichWithLmStudioCatalog(models: LlamaCppModel[], catalog: LmStudioModelInfo[]): LlamaCppModel[] {
|
|
796
|
+
const byKey = buildLmStudioCatalogMap(catalog);
|
|
797
|
+
return models.map((model) => {
|
|
798
|
+
const rawId = String(model.id ?? "");
|
|
799
|
+
const info = byKey.get(rawId.toLowerCase()) ?? byKey.get(baseName(rawId).toLowerCase());
|
|
800
|
+
if (!info) return model;
|
|
801
|
+
return { ...model, display_name: info.display_name ?? model.display_name, lmStudio: info };
|
|
802
|
+
});
|
|
803
|
+
}
|
|
804
|
+
|
|
684
805
|
/**
|
|
685
806
|
* Detect the server kind for an endpoint:
|
|
686
807
|
* 1. lucebox: /props with server.name "luce-*"
|
|
687
808
|
* 2. DwarfStar: /props build_info "dwarf*"/"ds4*" (else via chat probe)
|
|
688
809
|
* 3. ZINC: /v1/models with owned_by === "zinc"
|
|
689
|
-
* 4.
|
|
810
|
+
* 4. LM Studio: OpenAI-compatible /v1/models + REST /api/v1|v0/models
|
|
811
|
+
* 5. llama.cpp: anything else serving models
|
|
690
812
|
*/
|
|
691
813
|
async function detectServerKind(
|
|
692
814
|
baseUrl: string,
|
|
693
815
|
models: LlamaCppModel[],
|
|
694
816
|
props: ServerProps | undefined,
|
|
817
|
+
lmStudioCatalog?: LmStudioModelInfo[],
|
|
695
818
|
): Promise<ServerKind | "unknown"> {
|
|
819
|
+
void baseUrl;
|
|
696
820
|
const serverName = String(props?.server?.name ?? props?.build_info ?? "").toLowerCase();
|
|
697
821
|
if (serverName.startsWith("luce")) return "lucebox";
|
|
698
822
|
if (serverName.startsWith("dwarf") || serverName.startsWith("ds4")) return "dwarfstar";
|
|
699
823
|
if (models.length > 0 && models[0].owned_by === "zinc") return "zinc";
|
|
824
|
+
if (models.some((m) => String(m.owned_by ?? "").toLowerCase().includes("lmstudio"))) return "lmstudio";
|
|
825
|
+
if (lmStudioCatalog && lmStudioCatalog.length > 0) return "lmstudio";
|
|
826
|
+
if (models.some((m) => m.lmStudio)) return "lmstudio";
|
|
700
827
|
if (models.length > 0) return "llamacpp";
|
|
701
828
|
return "unknown";
|
|
702
829
|
}
|
|
@@ -766,13 +893,14 @@ function buildModelMetadata(
|
|
|
766
893
|
): ModelMetadata {
|
|
767
894
|
const meta: ModelMetadata = {};
|
|
768
895
|
|
|
769
|
-
// ── Model quant (GGUF filename / router id) ──
|
|
896
|
+
// ── Model quant (GGUF filename / router id / LM Studio metadata) ──
|
|
770
897
|
const sourcePath = entry.path ?? props?.model_path ?? rawId;
|
|
771
|
-
meta.quant = extractQuantTag(sourcePath);
|
|
898
|
+
meta.quant = extractQuantTag(sourcePath) ?? lmStudioQuantName(entry.lmStudio)?.toUpperCase();
|
|
772
899
|
|
|
773
900
|
// ── Vision ──
|
|
774
901
|
if (entry.architecture?.input_modalities?.includes("image")) meta.vision = true;
|
|
775
902
|
if (meta.vision === undefined && props?.modalities?.vision === true) meta.vision = true;
|
|
903
|
+
if (!meta.vision && (entry.lmStudio?.capabilities?.vision === true || entry.lmStudio?.type === "vlm")) meta.vision = true;
|
|
776
904
|
if (!meta.vision && local?.hasMmproj) meta.vision = true;
|
|
777
905
|
|
|
778
906
|
// ── Drafter (speculative decoding) ──
|
|
@@ -791,8 +919,9 @@ function buildModelMetadata(
|
|
|
791
919
|
meta.cacheK = argsInfo?.cacheK ?? local?.cacheK ?? props?.cache_type_k?.toLowerCase();
|
|
792
920
|
meta.cacheV = argsInfo?.cacheV ?? local?.cacheV ?? props?.cache_type_v?.toLowerCase();
|
|
793
921
|
|
|
794
|
-
// ── Router load status ──
|
|
922
|
+
// ── Router / LM Studio load status ──
|
|
795
923
|
if (entry.status?.value) meta.routerStatus = entry.status.value;
|
|
924
|
+
else if (entry.lmStudio?.state && entry.lmStudio.state !== "loaded") meta.routerStatus = entry.lmStudio.state;
|
|
796
925
|
|
|
797
926
|
return meta;
|
|
798
927
|
}
|
|
@@ -827,7 +956,7 @@ async function fetchModelsFromEndpoint(
|
|
|
827
956
|
return ep;
|
|
828
957
|
}
|
|
829
958
|
const payload = JSON.parse(body) as LlamaCppModelsResponse;
|
|
830
|
-
|
|
959
|
+
let models = payload.data ?? [];
|
|
831
960
|
|
|
832
961
|
// llama.cpp provides a parallel models[] array with friendly names
|
|
833
962
|
const nameMap = new Map<string, string>();
|
|
@@ -837,18 +966,24 @@ async function fetchModelsFromEndpoint(
|
|
|
837
966
|
}
|
|
838
967
|
}
|
|
839
968
|
|
|
840
|
-
// ── Mode detection ──
|
|
969
|
+
// ── Mode + server metadata detection ──
|
|
841
970
|
// Router/multi-model entries carry path/status/architecture; the router's
|
|
842
|
-
// own /props answers with role: "router".
|
|
843
|
-
|
|
971
|
+
// own /props answers with role: "router". LM Studio answers OpenAI's
|
|
972
|
+
// /v1/models and exposes richer local metadata under /api/v1/models
|
|
973
|
+
// (legacy: /api/v0/models), without llama.cpp's /props endpoint.
|
|
844
974
|
let props = await fetchServerProps(baseUrl, settings.discoveryTimeoutMs, srv.apiKey);
|
|
975
|
+
const lmStudioCatalog = props ? undefined : await fetchLmStudioCatalog(baseUrl, settings.discoveryTimeoutMs, srv.apiKey);
|
|
976
|
+
if (lmStudioCatalog) models = enrichWithLmStudioCatalog(models, lmStudioCatalog);
|
|
977
|
+
const isRouterShape = models.some((m) => m.path !== undefined || m.status !== undefined);
|
|
845
978
|
const routerProps = props?.role === "router";
|
|
846
|
-
const mode: ServerMode =
|
|
979
|
+
const mode: ServerMode = lmStudioCatalog
|
|
980
|
+
? models.length > 1 ? "router" : "single"
|
|
981
|
+
: isRouterShape || routerProps ? "router" : "single";
|
|
847
982
|
// In single-model mode props describes THE model; in router mode the root
|
|
848
983
|
// props is the router itself (useless for per-model metadata).
|
|
849
984
|
if (mode === "router") props = undefined;
|
|
850
985
|
|
|
851
|
-
const kind = await detectServerKind(baseUrl, models, props);
|
|
986
|
+
const kind = await detectServerKind(baseUrl, models, props, lmStudioCatalog);
|
|
852
987
|
// lucebox always enriches via /props (its own schema, even without router).
|
|
853
988
|
if (kind === "lucebox") {
|
|
854
989
|
props = (await fetchServerProps(baseUrl, settings.discoveryTimeoutMs, srv.apiKey)) ?? props;
|
|
@@ -891,6 +1026,10 @@ async function fetchModelsFromEndpoint(
|
|
|
891
1026
|
// =============================================================================
|
|
892
1027
|
|
|
893
1028
|
/** Compat profile per server kind. */
|
|
1029
|
+
function supportsThinkingBudget(kind: ServerKind | "unknown" | "auto" | undefined): boolean {
|
|
1030
|
+
return kind === "llamacpp" || kind === "lucebox";
|
|
1031
|
+
}
|
|
1032
|
+
|
|
894
1033
|
function makeCompat(kind: ServerKind | "unknown" | "auto") {
|
|
895
1034
|
const usageInStreaming = kind !== "zinc";
|
|
896
1035
|
return {
|
|
@@ -900,8 +1039,9 @@ function makeCompat(kind: ServerKind | "unknown" | "auto") {
|
|
|
900
1039
|
supportsUsageInStreaming: usageInStreaming,
|
|
901
1040
|
supportsStrictMode: false,
|
|
902
1041
|
// llama.cpp accepts a per-request `thinking_budget_tokens` cap.
|
|
1042
|
+
// LM Studio is OpenAI-compatible but does not accept llama.cpp-only fields.
|
|
903
1043
|
// (Only honored by pi when the model is registered with reasoning: true.)
|
|
904
|
-
...(kind
|
|
1044
|
+
...(supportsThinkingBudget(kind) ? { thinkingTokenBudgetField: THINKING_BUDGET_FIELD } : {}),
|
|
905
1045
|
};
|
|
906
1046
|
}
|
|
907
1047
|
|
|
@@ -917,6 +1057,8 @@ interface PiModel {
|
|
|
917
1057
|
compat?: ReturnType<typeof makeCompat>;
|
|
918
1058
|
headers?: Record<string, string>;
|
|
919
1059
|
thinkingBudgets?: ThinkingBudgets;
|
|
1060
|
+
/** Raw model id expected by the server (compact ids are display-only). */
|
|
1061
|
+
serverModelId: string;
|
|
920
1062
|
/** Metadata badges are applied to the name at build time. */
|
|
921
1063
|
endpoint: { serverId: string; host: string; port: number; kind: ServerKind | "unknown" | "auto"; mode: ServerMode };
|
|
922
1064
|
quant?: string;
|
|
@@ -949,8 +1091,9 @@ function toPiModel(
|
|
|
949
1091
|
): PiModel {
|
|
950
1092
|
const rawId = String(model.id ?? "");
|
|
951
1093
|
const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
|
|
952
|
-
|
|
953
|
-
|
|
1094
|
+
// Compact display id: "ModelName (host:port)". The raw server-side id is
|
|
1095
|
+
// kept separately (serverModelId) and restored in before_provider_request.
|
|
1096
|
+
const machineTag = settings.prefixModelIds ? ` (${hostPort})` : "";
|
|
954
1097
|
const kind = ep.server;
|
|
955
1098
|
|
|
956
1099
|
const common = {
|
|
@@ -966,22 +1109,41 @@ function toPiModel(
|
|
|
966
1109
|
...(srv.apiKey ? { headers: { Authorization: `Bearer ${srv.apiKey}` } } : {}),
|
|
967
1110
|
};
|
|
968
1111
|
|
|
1112
|
+
// ── LM Studio: OpenAI-compatible runtime with rich REST model metadata ──
|
|
1113
|
+
if (kind === "lmstudio") {
|
|
1114
|
+
const info = model.lmStudio;
|
|
1115
|
+
const contextWindow = lmStudioContextLength(info) ?? model.context_window ?? model.context_length ?? 32768;
|
|
1116
|
+
const displayName = info?.display_name ? cleanModelName(info.display_name) : cleanModelName(rawId);
|
|
1117
|
+
return {
|
|
1118
|
+
...common,
|
|
1119
|
+
id: `${displayName}${machineTag}`,
|
|
1120
|
+
serverModelId: rawId,
|
|
1121
|
+
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
1122
|
+
// LM Studio exposes reasoning-capable models, but its OpenAI-compatible
|
|
1123
|
+
// endpoint does not use llama.cpp's thinking_budget_tokens field.
|
|
1124
|
+
reasoning: false,
|
|
1125
|
+
contextWindow,
|
|
1126
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
|
|
1127
|
+
compat: makeCompat(kind),
|
|
1128
|
+
};
|
|
1129
|
+
}
|
|
1130
|
+
|
|
969
1131
|
// ── lucebox: alias as source of truth, rich metadata from /props ──
|
|
970
1132
|
if (kind === "lucebox") {
|
|
971
1133
|
const modelPath = ep.props?.model_path || rawId;
|
|
972
1134
|
const alias = ep.props?.model_alias || rawId;
|
|
973
1135
|
const nCtx = ep.props?.default_generation_settings?.n_ctx;
|
|
974
1136
|
const contextWindow = model.context_length ?? nCtx ?? 8192;
|
|
975
|
-
const
|
|
1137
|
+
const displayName = cleanModelName(baseName(alias !== rawId ? alias : modelPath));
|
|
976
1138
|
return {
|
|
977
1139
|
...common,
|
|
978
|
-
id:
|
|
979
|
-
|
|
1140
|
+
id: `${displayName}${machineTag}`,
|
|
1141
|
+
serverModelId: alias,
|
|
1142
|
+
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
980
1143
|
reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
|
|
981
1144
|
contextWindow,
|
|
982
1145
|
maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
|
|
983
1146
|
compat: makeCompat(kind),
|
|
984
|
-
...(budgets ? { thinkingBudgets: budgets } : {}),
|
|
985
1147
|
};
|
|
986
1148
|
}
|
|
987
1149
|
|
|
@@ -997,21 +1159,19 @@ function toPiModel(
|
|
|
997
1159
|
const contextWindow =
|
|
998
1160
|
model.meta?.n_ctx ?? model.meta?.n_ctx_train ?? model.context_window ?? model.context_length ?? 32768;
|
|
999
1161
|
|
|
1000
|
-
// Models with configured thinking budgets must be registered as reasoning
|
|
1001
|
-
// models so pi's thinking-level machinery (and budget field) engages.
|
|
1002
|
-
const modelId = usePrefix ? prefixId(rawId) : rawId;
|
|
1003
|
-
const budgets = config_ModelOptions()?.[modelId]?.thinkingBudgets;
|
|
1004
1162
|
const isLlamaFamily = kind === "llamacpp";
|
|
1005
1163
|
|
|
1006
1164
|
return {
|
|
1007
1165
|
...common,
|
|
1008
|
-
id:
|
|
1166
|
+
id: `${displayName}${machineTag}`,
|
|
1167
|
+
serverModelId: rawId,
|
|
1009
1168
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
1010
|
-
reasoning:
|
|
1169
|
+
// llama.cpp-family models behave like native pi reasoning models: the
|
|
1170
|
+
// footer shows the thinking level and pi sends the configured budget.
|
|
1171
|
+
reasoning: isLlamaFamily,
|
|
1011
1172
|
contextWindow,
|
|
1012
1173
|
maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
|
|
1013
1174
|
compat: makeCompat(kind),
|
|
1014
|
-
...(budgets ? { thinkingBudgets: budgets } : {}),
|
|
1015
1175
|
};
|
|
1016
1176
|
}
|
|
1017
1177
|
|
|
@@ -1125,6 +1285,7 @@ function modelsSignature(endpoints: EndpointResult[]): string {
|
|
|
1125
1285
|
.map((r) => {
|
|
1126
1286
|
const models = r.models
|
|
1127
1287
|
.filter((m) => {
|
|
1288
|
+
if (r.server === "lmstudio") return true;
|
|
1128
1289
|
const st = r.meta.get(String(m.id ?? ""))?.routerStatus;
|
|
1129
1290
|
return !st || st === "loaded" || rIncludeUnloaded;
|
|
1130
1291
|
})
|
|
@@ -1160,6 +1321,22 @@ const zincModelIds = new Set<string>();
|
|
|
1160
1321
|
const modelBaseUrls = new Map<string, string>();
|
|
1161
1322
|
/** baseUrl → server kind (for the warmup request profile). */
|
|
1162
1323
|
const endpointKinds = new Map<string, string>();
|
|
1324
|
+
/** Registered (compact) model id → raw server-side model id (payload rewrite). */
|
|
1325
|
+
const serverModelIds = new Map<string, string>();
|
|
1326
|
+
/** Raw server-side model id → registered (compact) model id. */
|
|
1327
|
+
const compactModelIds = new Map<string, string>();
|
|
1328
|
+
|
|
1329
|
+
/** Compact id registered in pi for a raw server model id (or the input). */
|
|
1330
|
+
function compactIdFor(modelId: string | undefined): string | undefined {
|
|
1331
|
+
if (!modelId) return undefined;
|
|
1332
|
+
return compactModelIds.get(modelId) ?? modelId;
|
|
1333
|
+
}
|
|
1334
|
+
|
|
1335
|
+
/** Raw server-side id to send in requests for a registered model id (or the input). */
|
|
1336
|
+
function rawIdFor(modelId: string | undefined): string | undefined {
|
|
1337
|
+
if (!modelId) return undefined;
|
|
1338
|
+
return serverModelIds.get(modelId) ?? modelId;
|
|
1339
|
+
}
|
|
1163
1340
|
|
|
1164
1341
|
/**
|
|
1165
1342
|
* Build pi models from a scan and (re)register the provider.
|
|
@@ -1167,7 +1344,10 @@ const endpointKinds = new Map<string, string>();
|
|
|
1167
1344
|
*/
|
|
1168
1345
|
function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: InfraConfig): PiModel[] {
|
|
1169
1346
|
zincModelIds.clear();
|
|
1347
|
+
serverModelIds.clear();
|
|
1348
|
+
compactModelIds.clear();
|
|
1170
1349
|
const settings = config.settings;
|
|
1350
|
+
let configDirty = false;
|
|
1171
1351
|
|
|
1172
1352
|
const piModels: PiModel[] = [];
|
|
1173
1353
|
const seenIds = new Set<string>();
|
|
@@ -1177,7 +1357,11 @@ function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: In
|
|
|
1177
1357
|
if (!srv) continue;
|
|
1178
1358
|
|
|
1179
1359
|
// Router mode: skip unloaded models unless explicitly included.
|
|
1360
|
+
// LM Studio's /v1/models can expose embedding models too; this provider
|
|
1361
|
+
// registers chat/completions models only.
|
|
1180
1362
|
const visibleModels = ep.models.filter((m) => {
|
|
1363
|
+
const lmType = m.lmStudio?.type;
|
|
1364
|
+
if (ep.server === "lmstudio") return lmType !== "embedding" && lmType !== "embeddings";
|
|
1181
1365
|
const st = ep.meta.get(String(m.id ?? ""))?.routerStatus;
|
|
1182
1366
|
return !st || st === "loaded" || settings.includeUnloadedRouterModels;
|
|
1183
1367
|
});
|
|
@@ -1193,25 +1377,56 @@ function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: In
|
|
|
1193
1377
|
const rawId = String(model.id ?? "");
|
|
1194
1378
|
const modelMeta = ep.meta.get(rawId) ?? {};
|
|
1195
1379
|
const pm = toPiModel(model, ep, srv, settings, modelMeta);
|
|
1380
|
+
const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
|
|
1196
1381
|
|
|
1197
|
-
// ID collision guard:
|
|
1198
|
-
// then a numeric suffix if even that collides.
|
|
1382
|
+
// ID collision guard: add the machine tag, then a numeric suffix.
|
|
1199
1383
|
if (seenIds.has(pm.id)) {
|
|
1200
|
-
pm.id = `${
|
|
1384
|
+
if (!pm.id.includes(`(${hostPort})`)) pm.id = `${pm.id} (${hostPort})`;
|
|
1201
1385
|
let n = 2;
|
|
1202
1386
|
while (seenIds.has(pm.id)) pm.id = `${pm.id}-${n++}`;
|
|
1203
1387
|
}
|
|
1204
1388
|
seenIds.add(pm.id);
|
|
1205
1389
|
|
|
1390
|
+
// Thinking budgets: resolve by the registered (compact) id or legacy
|
|
1391
|
+
// "host:port/model" keys, then migrate legacy keys forward.
|
|
1392
|
+
const opts = config_ModelOptions();
|
|
1393
|
+
let budgets = opts[pm.id]?.thinkingBudgets;
|
|
1394
|
+
if (!budgets) {
|
|
1395
|
+
for (const legacyKey of [`${hostPort}/${pm.serverModelId.replace(/^\/+/, "")}`, pm.serverModelId]) {
|
|
1396
|
+
const entry = opts[legacyKey];
|
|
1397
|
+
if (entry?.thinkingBudgets) {
|
|
1398
|
+
if (!opts[pm.id]) opts[pm.id] = entry;
|
|
1399
|
+
delete opts[legacyKey]; // migrate: compact id replaces host:port/model
|
|
1400
|
+
configDirty = true;
|
|
1401
|
+
budgets = entry.thinkingBudgets;
|
|
1402
|
+
break;
|
|
1403
|
+
}
|
|
1404
|
+
}
|
|
1405
|
+
}
|
|
1406
|
+
if (budgets && ep.server !== "lmstudio") {
|
|
1407
|
+
pm.thinkingBudgets = budgets;
|
|
1408
|
+
// Models with configured thinking budgets must be registered as
|
|
1409
|
+
// reasoning models so pi's thinking-level machinery engages.
|
|
1410
|
+
pm.reasoning = true;
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
// Duplicate display names within one endpoint get the raw id appended.
|
|
1206
1414
|
const candidateName = cleanModelName(String(model.name ?? model.id));
|
|
1207
1415
|
if (nameCount.get(candidateName)! > 1) {
|
|
1208
|
-
pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${
|
|
1416
|
+
pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${pm.serverModelId})`;
|
|
1209
1417
|
}
|
|
1210
1418
|
piModels.push(pm);
|
|
1211
|
-
if (ep.server === "zinc")
|
|
1419
|
+
if (ep.server === "zinc") {
|
|
1420
|
+
zincModelIds.add(pm.id);
|
|
1421
|
+
zincModelIds.add(pm.serverModelId);
|
|
1422
|
+
}
|
|
1423
|
+
serverModelIds.set(pm.id, pm.serverModelId);
|
|
1424
|
+
if (!compactModelIds.has(pm.serverModelId)) compactModelIds.set(pm.serverModelId, pm.id);
|
|
1212
1425
|
}
|
|
1213
1426
|
}
|
|
1214
1427
|
|
|
1428
|
+
if (configDirty) saveConfig(config);
|
|
1429
|
+
|
|
1215
1430
|
// Maps for hooks
|
|
1216
1431
|
endpointKinds.clear();
|
|
1217
1432
|
for (const ep of scan.endpoints) {
|
|
@@ -1281,6 +1496,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1281
1496
|
provider: PROVIDER_NAME,
|
|
1282
1497
|
cacheFile: join(homedir(), ".pi", "agent", "warmup-llamacpp-infra.json"),
|
|
1283
1498
|
kindFor: (baseUrl) => endpointKinds.get(baseUrl),
|
|
1499
|
+
requestModelFor: (modelId) => rawIdFor(modelId) ?? modelId,
|
|
1284
1500
|
onEvent: (ev) => warmupStatus.handle(ev),
|
|
1285
1501
|
});
|
|
1286
1502
|
|
|
@@ -1587,6 +1803,21 @@ export default function (pi: ExtensionAPI) {
|
|
|
1587
1803
|
registerEmptyProvider();
|
|
1588
1804
|
void discoverAndRegister().then((r) => schedulePolling(r.shouldPoll));
|
|
1589
1805
|
|
|
1806
|
+
// ── Hook 0: compact registered id → raw server model id ─────────
|
|
1807
|
+
// Model ids registered in pi are compact display ids ("Name (host:port)");
|
|
1808
|
+
// llama.cpp-family servers expect the raw model path/alias they advertised
|
|
1809
|
+
// in /v1/models, so the payload model field is rewritten here. Registered
|
|
1810
|
+
// first so every later hook (and the server) sees the raw id.
|
|
1811
|
+
pi.on("before_provider_request", (event, _ctx) => {
|
|
1812
|
+
const payload = event.payload as Record<string, unknown>;
|
|
1813
|
+
const modelInPayload = typeof payload.model === "string" ? payload.model : undefined;
|
|
1814
|
+
if (!modelInPayload) return undefined;
|
|
1815
|
+
const raw = serverModelIds.get(modelInPayload);
|
|
1816
|
+
if (raw === undefined || raw === modelInPayload) return undefined;
|
|
1817
|
+
debugLog(`model id "${modelInPayload}" → "${raw}"`);
|
|
1818
|
+
return { ...payload, model: raw };
|
|
1819
|
+
});
|
|
1820
|
+
|
|
1590
1821
|
// ── Hook 1: ZINC payload workaround ────────────────────────────
|
|
1591
1822
|
// ZINC rejects non-empty model ids and is picky about tool formats.
|
|
1592
1823
|
function isZincModel(modelInPayload: unknown): boolean {
|
|
@@ -1638,18 +1869,23 @@ export default function (pi: ExtensionAPI) {
|
|
|
1638
1869
|
// ── Hook 2: per-model thinking budget (llama.cpp thinking_budget_tokens) ──
|
|
1639
1870
|
// pi injects thinking_token budgets from its global settings; this hook
|
|
1640
1871
|
// overrides the value with the per-model budgets configured for this model.
|
|
1641
|
-
// Registered after the ZINC hook so it sees the final payload.
|
|
1872
|
+
// Registered after the ZINC hook so it sees the final payload. Hook 0 has
|
|
1873
|
+
// already rewritten the payload model to the raw server id, so both raw
|
|
1874
|
+
// and compact ids are resolved against the registered model maps.
|
|
1642
1875
|
pi.on("before_provider_request", (event, ctx) => {
|
|
1643
1876
|
const payload = event.payload as Record<string, unknown>;
|
|
1644
1877
|
const modelId = typeof payload.model === "string" ? payload.model : undefined;
|
|
1645
1878
|
if (!modelId) return undefined;
|
|
1646
|
-
const
|
|
1879
|
+
const compactKey = compactModelIds.get(modelId) ?? modelId;
|
|
1880
|
+
const baseUrl = modelBaseUrls.get(compactKey);
|
|
1881
|
+
if (!supportsThinkingBudget(endpointKinds.get(baseUrl ?? "") as ServerKind | undefined)) return undefined;
|
|
1882
|
+
const budgets = config.modelOptions[compactKey]?.thinkingBudgets;
|
|
1647
1883
|
if (!budgets) return undefined;
|
|
1648
1884
|
const level = normalizeLevel(ctx.thinkingLevel ?? currentThinkingLevel);
|
|
1649
1885
|
if (!level) return undefined;
|
|
1650
1886
|
const value = budgets[level];
|
|
1651
1887
|
if (typeof value !== "number") return undefined;
|
|
1652
|
-
debugLog(`thinking budget for ${
|
|
1888
|
+
debugLog(`thinking budget for ${compactKey} [${level}] = ${value}`);
|
|
1653
1889
|
return { ...payload, [THINKING_BUDGET_FIELD]: value };
|
|
1654
1890
|
});
|
|
1655
1891
|
|
|
@@ -1669,18 +1905,22 @@ export default function (pi: ExtensionAPI) {
|
|
|
1669
1905
|
}
|
|
1670
1906
|
|
|
1671
1907
|
// ── Hook 3: header warmup capture (runs last, sees final payload) ──
|
|
1908
|
+
// Templates are keyed by the compact registered id (the same id passed to
|
|
1909
|
+
// warmupForModel); PromptWarmer re-resolves the raw server id per request.
|
|
1672
1910
|
pi.on("before_provider_request", (event, ctx) => {
|
|
1673
1911
|
if (!config.settings.warmup) return undefined;
|
|
1674
1912
|
const payload = event.payload as Record<string, unknown>;
|
|
1675
1913
|
const modelId = typeof payload?.model === "string" ? payload.model : undefined;
|
|
1676
|
-
const
|
|
1677
|
-
|
|
1914
|
+
const compactKey = modelId ? (compactModelIds.get(modelId) ?? modelId) : undefined;
|
|
1915
|
+
const baseUrl = compactKey ? modelBaseUrls.get(compactKey) : undefined;
|
|
1916
|
+
const capturePayload = compactKey ? { ...payload, model: compactKey } : payload;
|
|
1917
|
+
warmer.onProviderPayload(capturePayload, baseUrl, ctx.cwd);
|
|
1678
1918
|
return undefined;
|
|
1679
1919
|
});
|
|
1680
1920
|
|
|
1681
1921
|
// ── Command: /llamacpp-infra ───────────────────────────────────
|
|
1682
1922
|
pi.registerCommand("llamacpp-infra", {
|
|
1683
|
-
description: "llama.cpp-infra: discover llama.cpp/ZINC/DwarfStar models on any machine (config, scan, metrics…)",
|
|
1923
|
+
description: "llama.cpp-infra: discover llama.cpp/ZINC/DwarfStar/LM Studio models on any machine (config, scan, metrics…)",
|
|
1684
1924
|
getArgumentCompletions: (prefix) =>
|
|
1685
1925
|
["config", "scan", "status", "list", "metrics", "help"]
|
|
1686
1926
|
.filter((s) => s.startsWith(prefix))
|
|
@@ -1786,7 +2026,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1786
2026
|
if (!ep) return undefined;
|
|
1787
2027
|
// Find the raw entry whose registered id ends with the raw id fragment.
|
|
1788
2028
|
for (const [rawId, meta] of ep.meta) {
|
|
1789
|
-
if (m.
|
|
2029
|
+
if (m.serverModelId === rawId || m.id === rawId) return meta;
|
|
1790
2030
|
}
|
|
1791
2031
|
return undefined;
|
|
1792
2032
|
};
|
|
@@ -1814,7 +2054,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1814
2054
|
function showHelp(ctx: ExtensionContext) {
|
|
1815
2055
|
ctx.ui.notify(
|
|
1816
2056
|
[
|
|
1817
|
-
"🦙 llama.cpp-infra — models served by llama.cpp & variants (ZINC, DwarfStar/ds4, lucebox) on any machine",
|
|
2057
|
+
"🦙 llama.cpp-infra — models served by llama.cpp & variants (ZINC, DwarfStar/ds4, lucebox, LM Studio) on any machine",
|
|
1818
2058
|
"",
|
|
1819
2059
|
" /llamacpp-infra → quick status",
|
|
1820
2060
|
" /llamacpp-infra config → ⚙️ configure servers, budgets & settings",
|
|
@@ -1875,12 +2115,16 @@ export default function (pi: ExtensionAPI) {
|
|
|
1875
2115
|
"🦙 llama.cpp-infra",
|
|
1876
2116
|
"",
|
|
1877
2117
|
"Discovers models served by llama.cpp (single & router/multi-model),",
|
|
1878
|
-
"ZINC, DwarfStar (ds4-server) and
|
|
1879
|
-
"and registers them into pi's native /model list.",
|
|
2118
|
+
"ZINC, DwarfStar (ds4-server), lucebox and LM Studio on any number",
|
|
2119
|
+
"of machines, and registers them into pi's native /model list.",
|
|
1880
2120
|
"",
|
|
2121
|
+
"Models appear as compact ids: \"Name (host:port)\" — the raw GGUF",
|
|
2122
|
+
"path/alias is sent to the server automatically on every request.",
|
|
2123
|
+
"LM Studio uses its OpenAI-compatible /v1 endpoint and enriches",
|
|
2124
|
+
"metadata from /api/v1/models (or legacy /api/v0/models).",
|
|
1881
2125
|
"Per-model metadata: vision, drafter, model quant, KV cache quant.",
|
|
1882
2126
|
"Per-model thinking budgets via llama.cpp thinking_budget_tokens.",
|
|
1883
|
-
"Live throughput metrics from each server's /metrics endpoint.",
|
|
2127
|
+
"Live throughput metrics from each server's /metrics endpoint when exposed.",
|
|
1884
2128
|
"",
|
|
1885
2129
|
`Config: ${getConfigPath()}`,
|
|
1886
2130
|
].join("\n"),
|
|
@@ -2078,7 +2322,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2078
2322
|
ctx.ui.notify(`⚠️ A server with host "${trimmedHost}" already exists`, "warning");
|
|
2079
2323
|
return;
|
|
2080
2324
|
}
|
|
2081
|
-
const portsRaw = await ctx.ui.input("➕ Ports to probe", "e.g. 8000, 8080-8082");
|
|
2325
|
+
const portsRaw = await ctx.ui.input("➕ Ports to probe", "e.g. 1234, 8000, 8080-8082");
|
|
2082
2326
|
if (portsRaw === undefined) return;
|
|
2083
2327
|
const ports = parsePorts(portsRaw);
|
|
2084
2328
|
if (!ports) {
|
|
@@ -2087,7 +2331,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2087
2331
|
}
|
|
2088
2332
|
const label = await ctx.ui.input("🏷️ Label (optional)", trimmedHost);
|
|
2089
2333
|
if (label === undefined) return;
|
|
2090
|
-
const probeDs4 = await ctx.ui.confirm("🕵️ ds4 (DwarfStar) probe?", "Enable the chat-completions ping probe for this machine? (for DwarfStar/ds4-server hosts)");
|
|
2334
|
+
const probeDs4 = await ctx.ui.confirm("🕵️ ds4 (DwarfStar) probe?", "Enable the chat-completions ping probe for this machine? (for DwarfStar/ds4-server hosts; not needed for LM Studio)");
|
|
2091
2335
|
|
|
2092
2336
|
let id = idSafeHost(trimmedHost).replace(/[^a-z0-9.-]/g, "-");
|
|
2093
2337
|
let n = 2;
|
package/src/prompt-warmup.ts
CHANGED
|
@@ -248,6 +248,8 @@ export interface PromptWarmerOptions {
|
|
|
248
248
|
cacheFile?: string;
|
|
249
249
|
/** Devuelve el tipo de servidor ("llamacpp"|"lucebox"|"ds4"|"zinc"|"lmstudio") para una baseUrl. */
|
|
250
250
|
kindFor?: (baseUrl: string) => string | undefined;
|
|
251
|
+
/** Mapea el id registrado (compacto) al id crudo que espera el servidor. */
|
|
252
|
+
requestModelFor?: (modelId: string) => string;
|
|
251
253
|
/** Callback opcional para eventos de warmup (estado en footer, notify, ...). */
|
|
252
254
|
onEvent?: (ev: WarmupEvent) => void;
|
|
253
255
|
/** Log opcional. Por defecto silencioso (no-op); activa con PI_WARMUP_DEBUG=1. */
|
|
@@ -351,6 +353,7 @@ export class PromptWarmer {
|
|
|
351
353
|
readonly provider: string;
|
|
352
354
|
private cacheFile: string;
|
|
353
355
|
private kindFor: (baseUrl: string) => string | undefined;
|
|
356
|
+
private requestModelFor: (modelId: string) => string;
|
|
354
357
|
private onEvent?: (ev: WarmupEvent) => void;
|
|
355
358
|
private log: (msg: string) => void;
|
|
356
359
|
|
|
@@ -368,6 +371,7 @@ export class PromptWarmer {
|
|
|
368
371
|
this.cacheFile =
|
|
369
372
|
opts.cacheFile ?? path.join(homedir(), ".pi", "agent", `warmup-${opts.provider}.json`);
|
|
370
373
|
this.kindFor = opts.kindFor ?? (() => KIND_LLAMACPP);
|
|
374
|
+
this.requestModelFor = opts.requestModelFor ?? ((id: string) => id);
|
|
371
375
|
this.onEvent = opts.onEvent;
|
|
372
376
|
// Por defecto silencioso: escribir por stderr corrompe el render de la TUI
|
|
373
377
|
// de pi (trazas sucias en el editor / input box). Solo se loguea con
|
|
@@ -392,6 +396,15 @@ export class PromptWarmer {
|
|
|
392
396
|
}
|
|
393
397
|
}
|
|
394
398
|
|
|
399
|
+
/** Id crudo que espera el servidor para un id registrado (con fallback). */
|
|
400
|
+
private resolveRequestModel(modelId: string): string {
|
|
401
|
+
try {
|
|
402
|
+
return this.requestModelFor(modelId) || modelId;
|
|
403
|
+
} catch {
|
|
404
|
+
return modelId;
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
395
408
|
// ── Warmup ────────────────────────────────────────────────────────────────
|
|
396
409
|
|
|
397
410
|
/**
|
|
@@ -422,7 +435,7 @@ export class PromptWarmer {
|
|
|
422
435
|
let body: WarmupBody;
|
|
423
436
|
if (tpl) {
|
|
424
437
|
body = {
|
|
425
|
-
model: tpl.model,
|
|
438
|
+
model: this.resolveRequestModel(tpl.model),
|
|
426
439
|
messages: [...tpl.systemMessages, { role: "user", content: PLACEHOLDER_USER }],
|
|
427
440
|
max_tokens: 1,
|
|
428
441
|
temperature: 0,
|
|
@@ -432,7 +445,7 @@ export class PromptWarmer {
|
|
|
432
445
|
if (cachePromptSupported(tpl.kind)) body.cache_prompt = true;
|
|
433
446
|
} else if (systemPrompt && FALLBACK_ENABLED) {
|
|
434
447
|
body = {
|
|
435
|
-
model: model.id,
|
|
448
|
+
model: this.resolveRequestModel(model.id),
|
|
436
449
|
messages: [
|
|
437
450
|
{ role: "system", content: systemPrompt },
|
|
438
451
|
{ role: "user", content: PLACEHOLDER_USER },
|