pi-llamacpp-infra 1.0.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
- # llamacpp-infra — Discovery, Metrics & Control for llama.cpp Servers
1
+ # llamacpp-infra — Discovery, Metrics & Control for llama.cpp-family Servers
2
2
 
3
- **llamacpp-infra** turns pi into a first-class citizen of local llama.cpp infrastructure. It probes any number of machines — localhost, LAN or Tailscale — discovers every model served by llama.cpp and its variants, registers them into pi's native `/model` list, and gives you live Prometheus metrics, per-model thinking budgets, vision detection and a full configuration UI — all without leaving the pi prompt.
3
+ **llamacpp-infra** turns pi into a first-class citizen of local llama.cpp infrastructure. It probes any number of machines — localhost, LAN or Tailscale — discovers every model served by llama.cpp and its variants (including LM Studio), registers them into pi's native `/model` list, and gives you live Prometheus metrics, per-model thinking budgets, vision detection and a full configuration UI — all without leaving the pi prompt.
4
4
 
5
5
  ---
6
6
 
@@ -14,17 +14,20 @@ Every endpoint llamacpp-infra talks to runs llama.cpp or a direct variant:
14
14
  | **ZINC** | `owned_by: "zinc"` | Payload workaround: empty model field + tool normalization |
15
15
  | **DwarfStar / ds4** | Opt-in ping to `/v1/chat/completions` | antirez's ds4-server for DeepSeek V4 (per-server `probeDs4` flag) |
16
16
  | **lucebox** | `GET /props` with `server.name: "luce-*"` | DeepSeek dflash server with rich metadata |
17
+ | **LM Studio** | `GET /v1/models` + optional `/api/v1/models` metadata | Local OpenAI-compatible server backed by llama.cpp; default port `1234` |
17
18
 
18
- Anything else (LM Studio, vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-in providers for those.
19
+ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-in providers for those.
19
20
 
20
21
  ## Features
21
22
 
22
23
  - **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
24
+ - **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
23
25
  - **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
24
26
  - **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
25
27
  - **Live Prometheus metrics** — polls `/metrics` (or `/stats`) and renders a compact widget with instantaneous prompt/gen throughput; auto-activates for llamacpp-infra models only
26
28
  - **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
27
29
  - **Header warmup** — pre-caches the system prompt KV on llama.cpp-family servers so the first real request is faster
30
+ - **LM Studio support** — uses LM Studio's OpenAI-compatible `/v1` API, enriches names/context/quant/vision from `/api/v1/models` (or legacy `/api/v0/models`), and avoids llama.cpp-only request fields
28
31
  - **ZINC workaround** — ZINC rejects non-empty model IDs; the payload hook rewrites the request and normalizes tool definitions automatically
29
32
  - **Vision detection** — scans `/proc` for local llama-server processes launched with `--mmproj` and marks those models as image-capable; also reads server-reported `modalities` / `input_modalities`
30
33
  - **Native configuration UI** — everything configurable through `/llamacpp-infra config` with pi's native menus; no config file editing required
@@ -39,7 +42,7 @@ llamacpp-infra is a [pi package](https://pi.dev/packages): one extension (`src/i
39
42
  pi install git:github.com/noguerol/llamacpp-infra
40
43
 
41
44
  # Pin a tag/commit
42
- pi install git:github.com/noguerol/llamacpp-infra@v1.0.0
45
+ pi install git:github.com/noguerol/llamacpp-infra@v1.2.0
43
46
 
44
47
  # From npm
45
48
  pi install npm:pi-llamacpp-infra
@@ -58,18 +61,33 @@ pi remove npm:pi-llamacpp-infra
58
61
 
59
62
  > **Security:** pi packages run with full system access. Install only packages you trust and review the source.
60
63
 
61
- **Requirements:** a working pi installation and at least one llama.cpp-family server running somewhere accessible (localhost, LAN or Tailscale).
64
+ **Requirements:** a working pi installation and at least one llama.cpp-family server running somewhere accessible (localhost, LAN or Tailscale). LM Studio works when its local server is started (Developer tab or `lms server start`, usually on `http://localhost:1234/v1`).
62
65
 
63
66
  ## Quick Start
64
67
 
65
68
  ```
66
- /llamacpp-infra config # open the config menu → add your first server
69
+ /llamacpp-infra config # open the config menu → add your first server (LM Studio: port 1234)
67
70
  /llamacpp-infra scan # discover models now
68
71
  /llamacpp-infra list # see what was found
69
72
  ```
70
73
 
71
74
  That's it. On the next pi startup, llamacpp-infra probes your servers automatically and registers every model into `/model`. Switch models with `/model` as usual.
72
75
 
76
+ ### LM Studio quick setup
77
+
78
+ LM Studio exposes an OpenAI-compatible API on `/v1` (default `http://localhost:1234/v1`) and a richer local REST API for model metadata on `/api/v1/models` (older LM Studio versions used `/api/v0/models`). llamacpp-infra probes `/v1/models` as the source of usable model IDs and, when available, enriches them with LM Studio's context length, display name, quantization and vision capability.
79
+
80
+ ```bash
81
+ # Start LM Studio's local server (or use the Developer tab in the GUI)
82
+ lms server start
83
+
84
+ # In pi: add/select host 127.0.0.1 with port 1234, then scan
85
+ /llamacpp-infra config
86
+ /llamacpp-infra scan
87
+ ```
88
+
89
+ No special payload workaround is required: LM Studio accepts standard OpenAI chat-completions requests. llamacpp-infra deliberately does **not** send llama.cpp-only fields such as `cache_prompt` or `thinking_budget_tokens` to LM Studio.
90
+
73
91
  ## Commands
74
92
 
75
93
  | Command | Description |
@@ -102,12 +120,14 @@ Shows every discovered model with metadata badges:
102
120
  ```
103
121
  📋 Discovered models (8)
104
122
 
105
- 1. local:8080/Qwen3.6-27B-UD-Q3_K_XL 👁️ 🗜️ UD-Q3_K_XL
106
- 2. local:8081/DeepSeek-V4-Flash 🗜️ ROCMFP2
107
- 3. myserver:8080/Meta-Llama-3.1-8B 🚀 draft-model 🗜️ Q4_K_M
108
- 4. myserver:8081/gemma-3-4b-it 👁️ 🗜️ Q4_K_M
123
+ 1. Qwen3.6-27B-UD-Q3_K_XL (local:8080) 👁️ 🗜️ UD-Q3_K_XL
124
+ 2. DeepSeek-V4-Flash-ROCMFP2 (local:8081) 🗜️ ROCMFP2
125
+ 3. Meta-Llama-3.1-8B (myserver:8080) 🚀 draft-model 🗜️ Q4_K_M
126
+ 4. gemma-3-4b-it (myserver:8081) 👁️ 🗜️ Q4_K_M
109
127
  ```
110
128
 
129
+ The same compact id is what pi's `/model` picker shows, with the serving machine in parentheses.
130
+
111
131
  ### `/llamacpp-infra status`
112
132
 
113
133
  Detailed per-endpoint report:
@@ -135,7 +155,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
135
155
  "id": "local",
136
156
  "host": "127.0.0.1",
137
157
  "label": "Local",
138
- "ports": [8000, 8001, 8002, 8080, 8081, 8082],
158
+ "ports": [8000, 8001, 8002, 8080, 8081, 8082, 1234],
139
159
  "enabled": true,
140
160
  "probeDs4": false
141
161
  },
@@ -164,7 +184,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
164
184
  "metricsPollMs": 5000
165
185
  },
166
186
  "modelOptions": {
167
- "myserver:8080/Qwen3.6-27B": {
187
+ "Qwen3.6-27B (myserver:8080)": {
168
188
  "thinkingBudgets": {
169
189
  "minimal": 256,
170
190
  "low": 1024,
@@ -183,7 +203,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
183
203
  | `id` | required | Unique short id (used in model IDs and logs) |
184
204
  | `host` | required | Hostname, tailnet name or IP |
185
205
  | `label` | `host` | Friendly name shown in menus |
186
- | `ports` | required | Array of ports to probe |
206
+ | `ports` | required | Array of ports to probe (`1234` is LM Studio's usual local server port) |
187
207
  | `enabled` | `true` | Whether to probe this server |
188
208
  | `probeDs4` | `false` | Opt-in: ping `/v1/chat/completions` for DwarfStar/ds4 servers |
189
209
  | `apiKey` | — | Optional bearer token sent on discovery and per-model requests |
@@ -198,7 +218,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
198
218
  | `startupGraceMs` | `40000` | Keep trying at startup while nothing has answered |
199
219
  | `knownGoodFailLimit` | `3` | Consecutive failures before a live endpoint is dropped |
200
220
  | `detectVision` | `true` | Scan `/proc` for `--mmproj` + read server-reported modalities |
201
- | `prefixModelIds` | `true` | `host:port/model` format to avoid cross-server collisions |
221
+ | `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
202
222
  | `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
203
223
  | `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
204
224
  | `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
@@ -213,7 +233,15 @@ Levels: `minimal`, `low`, `medium`, `high`, `xhigh`, `max`.
213
233
 
214
234
  ## Model ID Format
215
235
 
216
- With `prefixModelIds: true` (default), every model ID is `host:port/model`, e.g. `myserver:8080/Qwen3.6-27B-UD-Q3_K_XL`. This avoids collisions when the same GGUF is served on multiple machines. Localhost servers (`127.0.0.1`, `localhost`) use `local:port/model` for readability.
236
+ Models are registered with compact display ids: `ModelName (host:port)`, e.g. `Qwen3.6-27B-UD-Q3_K_XL (myserver:8080)` — the compact model name plus the machine serving it in parentheses, matching how pi shows native provider models. Localhost servers (`127.0.0.1`, `localhost`) use `local:port` in the tag.
237
+
238
+ pi sends the compact id to the extension's request hook, which transparently rewrites it to the raw server-side id (the GGUF path, alias or router id the server advertised in `/v1/models`) before the request leaves pi. Config keys under `modelOptions` use the compact id; legacy `host:port/model` keys are migrated automatically on the first scan.
239
+
240
+ With `prefixModelIds: false` the machine tag is omitted (`ModelName`); it is re-added automatically only when two models would otherwise collide.
241
+
242
+ ### Thinking budgets in the UI
243
+
244
+ llama.cpp-family models are registered as reasoning models, exactly like a native pi provider: the footer shows `ModelName (host:port) • <level>`, the thinking selector offers levels with token estimates, and pi sends the configured `thinking_budget_tokens` budget on each request. Per-model budgets configured in the extension override pi's global per-level budgets.
217
245
 
218
246
  ## Live Metrics Widget
219
247
 
@@ -233,13 +261,13 @@ llamacpp-infra/
233
261
  ├── LICENSE # MIT
234
262
  ├── README.md
235
263
  └── src/
236
- ├── index.ts # Extension entry point (~2400 lines)
264
+ ├── index.ts # Extension entry point (~2600 lines)
237
265
  └── prompt-warmup.ts # Header warmup module (inlined, ~600 lines)
238
266
  ```
239
267
 
240
268
  Two-file extension with zero external dependencies (only pi's bundled `@earendil-works/pi-coding-agent` + Node built-ins):
241
269
 
242
- - **Discovery engine** — multi-server probing with timeouts, retry budgets, and per-server kind detection (llama.cpp, ZINC, DwarfStar, lucebox)
270
+ - **Discovery engine** — multi-server probing with timeouts, retry budgets, and per-server kind detection (llama.cpp, ZINC, DwarfStar, lucebox, LM Studio)
243
271
  - **Router support** — single-model and multi-model llama.cpp modes with per-model status, args parsing and metadata extraction
244
272
  - **Metrics subsystem** — Prometheus endpoint discovery, polling, and compact widget rendering
245
273
  - **Thinking budgets** — per-model per-level configuration with automatic `reasoning` registration
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "pi-llamacpp-infra",
3
- "version": "1.0.0",
4
- "description": "Discovery, metrics and control of llama.cpp-family servers for pi: probes any number of machines (localhost, LAN, Tailscale), registers every model into pi's native /model list, and provides live Prometheus metrics, per-model thinking budgets, vision detection and a native config UI. Supports llama.cpp, ZINC, DwarfStar/ds4 and lucebox.",
3
+ "version": "1.2.0",
4
+ "description": "Discovery, metrics and control of llama.cpp-family servers for pi: probes any number of machines (localhost, LAN, Tailscale), registers every model into pi's native /model list, and provides live Prometheus metrics, per-model thinking budgets, vision detection and a native config UI. Supports llama.cpp, ZINC, DwarfStar/ds4, lucebox and LM Studio.",
5
5
  "keywords": [
6
6
  "pi-package",
7
7
  "llamacpp",
@@ -11,7 +11,9 @@
11
11
  "metrics",
12
12
  "tailscale",
13
13
  "vision",
14
- "thinking"
14
+ "thinking",
15
+ "lm-studio",
16
+ "lmstudio"
15
17
  ],
16
18
  "author": "Javier Noguerol <https://github.com/noguerol>",
17
19
  "license": "MIT",
package/src/index.ts CHANGED
@@ -2,8 +2,8 @@
2
2
  * extension: llamacpp-infra
3
3
  * =========================
4
4
  * Discovery, metrics and control of models served by llama.cpp and its
5
- * variants (ZINC, DwarfStar/ds4, lucebox, OpenAI-compatible forks) on any
6
- * number of machines — localhost, LAN or Tailscale.
5
+ * variants (ZINC, DwarfStar/ds4, lucebox, LM Studio) on any number of
6
+ * machines — localhost, LAN or Tailscale.
7
7
  *
8
8
  * SCOPE
9
9
  * -----
@@ -15,8 +15,9 @@
15
15
  * ZINC — llama.cpp-compatible runtime (owned_by: "zinc")
16
16
  * DwarfStar — antirez's ds4-server for DeepSeek V4 (chat ping probe)
17
17
  * lucebox — DeepSeek dflash server (rich /props metadata)
18
+ * LM Studio — local OpenAI-compatible server backed by llama.cpp
18
19
  *
19
- * Anything else (LM Studio, vLLM, Ollama, cloud APIs…) is out of scope.
20
+ * Anything else (vLLM, Ollama, cloud APIs…) is out of scope.
20
21
  *
21
22
  * FEATURES
22
23
  * --------
@@ -47,6 +48,12 @@
47
48
  * 7. Header warmup (pre-cache system prompt KV on llama.cpp-family servers).
48
49
  * 8. ZINC workaround: ZINC rejects non-empty model ids; the payload hook
49
50
  * rewrites the request accordingly and normalizes tool definitions.
51
+ * 9. LM Studio support: discovers `/v1/models`, enriches metadata from
52
+ * `/api/v1/models` or legacy `/api/v0/models`, and avoids llama.cpp-only
53
+ * request fields such as `cache_prompt` / `thinking_budget_tokens`.
54
+ * 10. Compact model ids: models are registered as "Name (host:port)" — the
55
+ * raw server-side model id (GGUF path / alias) is restored automatically
56
+ * in before_provider_request before the request leaves pi.
50
57
  *
51
58
  * HISTORY
52
59
  * -------
@@ -149,6 +156,37 @@ interface LlamaCppMeta {
149
156
  n_ctx_train?: number;
150
157
  }
151
158
 
159
+ interface LmStudioModelInfo {
160
+ /** v1 REST API unique key; v0 uses `id` instead. */
161
+ key?: string;
162
+ id?: string;
163
+ display_name?: string;
164
+ /** v1: "llm" | "embedding"; v0: "llm" | "vlm" | "embeddings". */
165
+ type?: string;
166
+ publisher?: string;
167
+ architecture?: string | null;
168
+ arch?: string | null;
169
+ format?: string | null;
170
+ compatibility_type?: string | null;
171
+ quantization?: string | { name?: string | null; bits_per_weight?: number | null } | null;
172
+ state?: string;
173
+ max_context_length?: number;
174
+ loaded_instances?: Array<{ id?: string; config?: { context_length?: number; parallel?: number } }>;
175
+ capabilities?: {
176
+ vision?: boolean;
177
+ trained_for_tool_use?: boolean;
178
+ reasoning?: boolean | { allowed_options?: string[]; default?: string };
179
+ };
180
+ selected_variant?: string;
181
+ variants?: string[];
182
+ }
183
+
184
+ interface LmStudioModelsResponse {
185
+ models?: LmStudioModelInfo[];
186
+ data?: LmStudioModelInfo[];
187
+ object?: string;
188
+ }
189
+
152
190
  interface LlamaCppModel {
153
191
  id: string;
154
192
  name?: string;
@@ -170,6 +208,8 @@ interface LlamaCppModel {
170
208
  input_modalities?: string[];
171
209
  output_modalities?: string[];
172
210
  };
211
+ /** LM Studio REST metadata, when the local server exposes it. */
212
+ lmStudio?: LmStudioModelInfo;
173
213
  meta?: LlamaCppMeta;
174
214
  context_window?: number;
175
215
  context_length?: number;
@@ -210,7 +250,7 @@ interface ServerProps {
210
250
  cache_type_v?: string;
211
251
  }
212
252
 
213
- type ServerKind = "llamacpp" | "zinc" | "lucebox" | "dwarfstar";
253
+ type ServerKind = "llamacpp" | "zinc" | "lucebox" | "dwarfstar" | "lmstudio";
214
254
  type ServerMode = "single" | "router" | "unknown";
215
255
 
216
256
  /** Per-model metadata extracted during discovery. */
@@ -295,7 +335,7 @@ const DEFAULT_SERVERS: ServerConfig[] = [
295
335
  id: "local",
296
336
  host: "127.0.0.1",
297
337
  label: "Local",
298
- ports: [8000, 8001, 8002, 8080, 8081, 8082],
338
+ ports: [8000, 8001, 8002, 8080, 8081, 8082, 1234],
299
339
  enabled: true,
300
340
  probeDs4: false,
301
341
  },
@@ -681,22 +721,109 @@ async function fetchServerProps(
681
721
  return undefined;
682
722
  }
683
723
 
724
+ function lmStudioKey(info: LmStudioModelInfo): string | undefined {
725
+ return info.key ?? info.id;
726
+ }
727
+
728
+ function lmStudioQuantName(info: LmStudioModelInfo | undefined): string | undefined {
729
+ if (!info) return undefined;
730
+ const q = info.quantization;
731
+ if (typeof q === "string") return q || undefined;
732
+ return q?.name ?? undefined;
733
+ }
734
+
735
+ function lmStudioContextLength(info: LmStudioModelInfo | undefined): number | undefined {
736
+ if (!info) return undefined;
737
+ const loadedCtx = info.loaded_instances?.find((inst) => typeof inst.config?.context_length === "number")?.config?.context_length;
738
+ return loadedCtx ?? info.max_context_length;
739
+ }
740
+
741
+ function isLmStudioModelInfo(value: unknown): value is LmStudioModelInfo {
742
+ const m = value as LmStudioModelInfo | undefined;
743
+ if (!m || typeof m !== "object") return false;
744
+ return Boolean(
745
+ m.key ||
746
+ m.id ||
747
+ m.display_name ||
748
+ m.compatibility_type ||
749
+ m.format ||
750
+ m.loaded_instances ||
751
+ typeof m.max_context_length === "number" ||
752
+ m.capabilities,
753
+ );
754
+ }
755
+
756
+ async function fetchLmStudioCatalog(
757
+ baseUrl: string,
758
+ timeoutMs: number,
759
+ apiKey?: string,
760
+ ): Promise<LmStudioModelInfo[] | undefined> {
761
+ const rootUrl = baseUrl.replace(/\/v1\/?$/, "");
762
+ // LM Studio v1 is current; v0 is still common in older installations.
763
+ for (const path of ["/api/v1/models", "/api/v0/models"]) {
764
+ try {
765
+ const { status, body } = await httpGet(`${rootUrl}${path}`, timeoutMs, apiKey);
766
+ if (status < 200 || status >= 300) continue;
767
+ const payload = JSON.parse(body) as LmStudioModelsResponse;
768
+ const models = Array.isArray(payload.models) ? payload.models : Array.isArray(payload.data) ? payload.data : undefined;
769
+ if (models && models.length > 0 && models.some(isLmStudioModelInfo)) return models;
770
+ } catch {
771
+ // Not LM Studio, or an older server without the REST metadata endpoint.
772
+ }
773
+ }
774
+ return undefined;
775
+ }
776
+
777
+ function buildLmStudioCatalogMap(catalog: LmStudioModelInfo[]): Map<string, LmStudioModelInfo> {
778
+ const map = new Map<string, LmStudioModelInfo>();
779
+ const add = (key: string | undefined, info: LmStudioModelInfo) => {
780
+ const k = key?.trim().toLowerCase();
781
+ if (k && !map.has(k)) map.set(k, info);
782
+ };
783
+ for (const info of catalog) {
784
+ add(lmStudioKey(info), info);
785
+ add(info.key, info);
786
+ add(info.id, info);
787
+ add(info.selected_variant, info);
788
+ add(info.display_name, info);
789
+ for (const variant of info.variants ?? []) add(variant, info);
790
+ for (const inst of info.loaded_instances ?? []) add(inst.id, info);
791
+ }
792
+ return map;
793
+ }
794
+
795
+ function enrichWithLmStudioCatalog(models: LlamaCppModel[], catalog: LmStudioModelInfo[]): LlamaCppModel[] {
796
+ const byKey = buildLmStudioCatalogMap(catalog);
797
+ return models.map((model) => {
798
+ const rawId = String(model.id ?? "");
799
+ const info = byKey.get(rawId.toLowerCase()) ?? byKey.get(baseName(rawId).toLowerCase());
800
+ if (!info) return model;
801
+ return { ...model, display_name: info.display_name ?? model.display_name, lmStudio: info };
802
+ });
803
+ }
804
+
684
805
  /**
685
806
  * Detect the server kind for an endpoint:
686
807
  * 1. lucebox: /props with server.name "luce-*"
687
808
  * 2. DwarfStar: /props build_info "dwarf*"/"ds4*" (else via chat probe)
688
809
  * 3. ZINC: /v1/models with owned_by === "zinc"
689
- * 4. llama.cpp: anything else serving models
810
+ * 4. LM Studio: OpenAI-compatible /v1/models + REST /api/v1|v0/models
811
+ * 5. llama.cpp: anything else serving models
690
812
  */
691
813
  async function detectServerKind(
692
814
  baseUrl: string,
693
815
  models: LlamaCppModel[],
694
816
  props: ServerProps | undefined,
817
+ lmStudioCatalog?: LmStudioModelInfo[],
695
818
  ): Promise<ServerKind | "unknown"> {
819
+ void baseUrl;
696
820
  const serverName = String(props?.server?.name ?? props?.build_info ?? "").toLowerCase();
697
821
  if (serverName.startsWith("luce")) return "lucebox";
698
822
  if (serverName.startsWith("dwarf") || serverName.startsWith("ds4")) return "dwarfstar";
699
823
  if (models.length > 0 && models[0].owned_by === "zinc") return "zinc";
824
+ if (models.some((m) => String(m.owned_by ?? "").toLowerCase().includes("lmstudio"))) return "lmstudio";
825
+ if (lmStudioCatalog && lmStudioCatalog.length > 0) return "lmstudio";
826
+ if (models.some((m) => m.lmStudio)) return "lmstudio";
700
827
  if (models.length > 0) return "llamacpp";
701
828
  return "unknown";
702
829
  }
@@ -766,13 +893,14 @@ function buildModelMetadata(
766
893
  ): ModelMetadata {
767
894
  const meta: ModelMetadata = {};
768
895
 
769
- // ── Model quant (GGUF filename / router id) ──
896
+ // ── Model quant (GGUF filename / router id / LM Studio metadata) ──
770
897
  const sourcePath = entry.path ?? props?.model_path ?? rawId;
771
- meta.quant = extractQuantTag(sourcePath);
898
+ meta.quant = extractQuantTag(sourcePath) ?? lmStudioQuantName(entry.lmStudio)?.toUpperCase();
772
899
 
773
900
  // ── Vision ──
774
901
  if (entry.architecture?.input_modalities?.includes("image")) meta.vision = true;
775
902
  if (meta.vision === undefined && props?.modalities?.vision === true) meta.vision = true;
903
+ if (!meta.vision && (entry.lmStudio?.capabilities?.vision === true || entry.lmStudio?.type === "vlm")) meta.vision = true;
776
904
  if (!meta.vision && local?.hasMmproj) meta.vision = true;
777
905
 
778
906
  // ── Drafter (speculative decoding) ──
@@ -791,8 +919,9 @@ function buildModelMetadata(
791
919
  meta.cacheK = argsInfo?.cacheK ?? local?.cacheK ?? props?.cache_type_k?.toLowerCase();
792
920
  meta.cacheV = argsInfo?.cacheV ?? local?.cacheV ?? props?.cache_type_v?.toLowerCase();
793
921
 
794
- // ── Router load status ──
922
+ // ── Router / LM Studio load status ──
795
923
  if (entry.status?.value) meta.routerStatus = entry.status.value;
924
+ else if (entry.lmStudio?.state && entry.lmStudio.state !== "loaded") meta.routerStatus = entry.lmStudio.state;
796
925
 
797
926
  return meta;
798
927
  }
@@ -827,7 +956,7 @@ async function fetchModelsFromEndpoint(
827
956
  return ep;
828
957
  }
829
958
  const payload = JSON.parse(body) as LlamaCppModelsResponse;
830
- const models = payload.data ?? [];
959
+ let models = payload.data ?? [];
831
960
 
832
961
  // llama.cpp provides a parallel models[] array with friendly names
833
962
  const nameMap = new Map<string, string>();
@@ -837,18 +966,24 @@ async function fetchModelsFromEndpoint(
837
966
  }
838
967
  }
839
968
 
840
- // ── Mode detection ──
969
+ // ── Mode + server metadata detection ──
841
970
  // Router/multi-model entries carry path/status/architecture; the router's
842
- // own /props answers with role: "router".
843
- const isRouterShape = models.some((m) => m.path !== undefined || m.status !== undefined);
971
+ // own /props answers with role: "router". LM Studio answers OpenAI's
972
+ // /v1/models and exposes richer local metadata under /api/v1/models
973
+ // (legacy: /api/v0/models), without llama.cpp's /props endpoint.
844
974
  let props = await fetchServerProps(baseUrl, settings.discoveryTimeoutMs, srv.apiKey);
975
+ const lmStudioCatalog = props ? undefined : await fetchLmStudioCatalog(baseUrl, settings.discoveryTimeoutMs, srv.apiKey);
976
+ if (lmStudioCatalog) models = enrichWithLmStudioCatalog(models, lmStudioCatalog);
977
+ const isRouterShape = models.some((m) => m.path !== undefined || m.status !== undefined);
845
978
  const routerProps = props?.role === "router";
846
- const mode: ServerMode = isRouterShape || routerProps ? "router" : "single";
979
+ const mode: ServerMode = lmStudioCatalog
980
+ ? models.length > 1 ? "router" : "single"
981
+ : isRouterShape || routerProps ? "router" : "single";
847
982
  // In single-model mode props describes THE model; in router mode the root
848
983
  // props is the router itself (useless for per-model metadata).
849
984
  if (mode === "router") props = undefined;
850
985
 
851
- const kind = await detectServerKind(baseUrl, models, props);
986
+ const kind = await detectServerKind(baseUrl, models, props, lmStudioCatalog);
852
987
  // lucebox always enriches via /props (its own schema, even without router).
853
988
  if (kind === "lucebox") {
854
989
  props = (await fetchServerProps(baseUrl, settings.discoveryTimeoutMs, srv.apiKey)) ?? props;
@@ -891,6 +1026,10 @@ async function fetchModelsFromEndpoint(
891
1026
  // =============================================================================
892
1027
 
893
1028
  /** Compat profile per server kind. */
1029
+ function supportsThinkingBudget(kind: ServerKind | "unknown" | "auto" | undefined): boolean {
1030
+ return kind === "llamacpp" || kind === "lucebox";
1031
+ }
1032
+
894
1033
  function makeCompat(kind: ServerKind | "unknown" | "auto") {
895
1034
  const usageInStreaming = kind !== "zinc";
896
1035
  return {
@@ -900,8 +1039,9 @@ function makeCompat(kind: ServerKind | "unknown" | "auto") {
900
1039
  supportsUsageInStreaming: usageInStreaming,
901
1040
  supportsStrictMode: false,
902
1041
  // llama.cpp accepts a per-request `thinking_budget_tokens` cap.
1042
+ // LM Studio is OpenAI-compatible but does not accept llama.cpp-only fields.
903
1043
  // (Only honored by pi when the model is registered with reasoning: true.)
904
- ...(kind === "llamacpp" || kind === "lucebox" ? { thinkingTokenBudgetField: THINKING_BUDGET_FIELD } : {}),
1044
+ ...(supportsThinkingBudget(kind) ? { thinkingTokenBudgetField: THINKING_BUDGET_FIELD } : {}),
905
1045
  };
906
1046
  }
907
1047
 
@@ -917,6 +1057,8 @@ interface PiModel {
917
1057
  compat?: ReturnType<typeof makeCompat>;
918
1058
  headers?: Record<string, string>;
919
1059
  thinkingBudgets?: ThinkingBudgets;
1060
+ /** Raw model id expected by the server (compact ids are display-only). */
1061
+ serverModelId: string;
920
1062
  /** Metadata badges are applied to the name at build time. */
921
1063
  endpoint: { serverId: string; host: string; port: number; kind: ServerKind | "unknown" | "auto"; mode: ServerMode };
922
1064
  quant?: string;
@@ -949,8 +1091,9 @@ function toPiModel(
949
1091
  ): PiModel {
950
1092
  const rawId = String(model.id ?? "");
951
1093
  const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
952
- const usePrefix = settings.prefixModelIds;
953
- const prefixId = (id: string) => `${hostPort}/${id.replace(/^\/+/, "")}`;
1094
+ // Compact display id: "ModelName (host:port)". The raw server-side id is
1095
+ // kept separately (serverModelId) and restored in before_provider_request.
1096
+ const machineTag = settings.prefixModelIds ? ` (${hostPort})` : "";
954
1097
  const kind = ep.server;
955
1098
 
956
1099
  const common = {
@@ -966,22 +1109,41 @@ function toPiModel(
966
1109
  ...(srv.apiKey ? { headers: { Authorization: `Bearer ${srv.apiKey}` } } : {}),
967
1110
  };
968
1111
 
1112
+ // ── LM Studio: OpenAI-compatible runtime with rich REST model metadata ──
1113
+ if (kind === "lmstudio") {
1114
+ const info = model.lmStudio;
1115
+ const contextWindow = lmStudioContextLength(info) ?? model.context_window ?? model.context_length ?? 32768;
1116
+ const displayName = info?.display_name ? cleanModelName(info.display_name) : cleanModelName(rawId);
1117
+ return {
1118
+ ...common,
1119
+ id: `${displayName}${machineTag}`,
1120
+ serverModelId: rawId,
1121
+ name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
1122
+ // LM Studio exposes reasoning-capable models, but its OpenAI-compatible
1123
+ // endpoint does not use llama.cpp's thinking_budget_tokens field.
1124
+ reasoning: false,
1125
+ contextWindow,
1126
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
1127
+ compat: makeCompat(kind),
1128
+ };
1129
+ }
1130
+
969
1131
  // ── lucebox: alias as source of truth, rich metadata from /props ──
970
1132
  if (kind === "lucebox") {
971
1133
  const modelPath = ep.props?.model_path || rawId;
972
1134
  const alias = ep.props?.model_alias || rawId;
973
1135
  const nCtx = ep.props?.default_generation_settings?.n_ctx;
974
1136
  const contextWindow = model.context_length ?? nCtx ?? 8192;
975
- const budgets = config_ModelOptions()?.[usePrefix ? prefixId(alias) : alias]?.thinkingBudgets;
1137
+ const displayName = cleanModelName(baseName(alias !== rawId ? alias : modelPath));
976
1138
  return {
977
1139
  ...common,
978
- id: usePrefix ? prefixId(alias) : alias,
979
- name: `${cleanModelName(modelPath)}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
1140
+ id: `${displayName}${machineTag}`,
1141
+ serverModelId: alias,
1142
+ name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
980
1143
  reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
981
1144
  contextWindow,
982
1145
  maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
983
1146
  compat: makeCompat(kind),
984
- ...(budgets ? { thinkingBudgets: budgets } : {}),
985
1147
  };
986
1148
  }
987
1149
 
@@ -997,21 +1159,19 @@ function toPiModel(
997
1159
  const contextWindow =
998
1160
  model.meta?.n_ctx ?? model.meta?.n_ctx_train ?? model.context_window ?? model.context_length ?? 32768;
999
1161
 
1000
- // Models with configured thinking budgets must be registered as reasoning
1001
- // models so pi's thinking-level machinery (and budget field) engages.
1002
- const modelId = usePrefix ? prefixId(rawId) : rawId;
1003
- const budgets = config_ModelOptions()?.[modelId]?.thinkingBudgets;
1004
1162
  const isLlamaFamily = kind === "llamacpp";
1005
1163
 
1006
1164
  return {
1007
1165
  ...common,
1008
- id: modelId,
1166
+ id: `${displayName}${machineTag}`,
1167
+ serverModelId: rawId,
1009
1168
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
1010
- reasoning: isLlamaFamily || Boolean(budgets),
1169
+ // llama.cpp-family models behave like native pi reasoning models: the
1170
+ // footer shows the thinking level and pi sends the configured budget.
1171
+ reasoning: isLlamaFamily,
1011
1172
  contextWindow,
1012
1173
  maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
1013
1174
  compat: makeCompat(kind),
1014
- ...(budgets ? { thinkingBudgets: budgets } : {}),
1015
1175
  };
1016
1176
  }
1017
1177
 
@@ -1125,6 +1285,7 @@ function modelsSignature(endpoints: EndpointResult[]): string {
1125
1285
  .map((r) => {
1126
1286
  const models = r.models
1127
1287
  .filter((m) => {
1288
+ if (r.server === "lmstudio") return true;
1128
1289
  const st = r.meta.get(String(m.id ?? ""))?.routerStatus;
1129
1290
  return !st || st === "loaded" || rIncludeUnloaded;
1130
1291
  })
@@ -1160,6 +1321,22 @@ const zincModelIds = new Set<string>();
1160
1321
  const modelBaseUrls = new Map<string, string>();
1161
1322
  /** baseUrl → server kind (for the warmup request profile). */
1162
1323
  const endpointKinds = new Map<string, string>();
1324
+ /** Registered (compact) model id → raw server-side model id (payload rewrite). */
1325
+ const serverModelIds = new Map<string, string>();
1326
+ /** Raw server-side model id → registered (compact) model id. */
1327
+ const compactModelIds = new Map<string, string>();
1328
+
1329
+ /** Compact id registered in pi for a raw server model id (or the input). */
1330
+ function compactIdFor(modelId: string | undefined): string | undefined {
1331
+ if (!modelId) return undefined;
1332
+ return compactModelIds.get(modelId) ?? modelId;
1333
+ }
1334
+
1335
+ /** Raw server-side id to send in requests for a registered model id (or the input). */
1336
+ function rawIdFor(modelId: string | undefined): string | undefined {
1337
+ if (!modelId) return undefined;
1338
+ return serverModelIds.get(modelId) ?? modelId;
1339
+ }
1163
1340
 
1164
1341
  /**
1165
1342
  * Build pi models from a scan and (re)register the provider.
@@ -1167,7 +1344,10 @@ const endpointKinds = new Map<string, string>();
1167
1344
  */
1168
1345
  function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: InfraConfig): PiModel[] {
1169
1346
  zincModelIds.clear();
1347
+ serverModelIds.clear();
1348
+ compactModelIds.clear();
1170
1349
  const settings = config.settings;
1350
+ let configDirty = false;
1171
1351
 
1172
1352
  const piModels: PiModel[] = [];
1173
1353
  const seenIds = new Set<string>();
@@ -1177,7 +1357,11 @@ function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: In
1177
1357
  if (!srv) continue;
1178
1358
 
1179
1359
  // Router mode: skip unloaded models unless explicitly included.
1360
+ // LM Studio's /v1/models can expose embedding models too; this provider
1361
+ // registers chat/completions models only.
1180
1362
  const visibleModels = ep.models.filter((m) => {
1363
+ const lmType = m.lmStudio?.type;
1364
+ if (ep.server === "lmstudio") return lmType !== "embedding" && lmType !== "embeddings";
1181
1365
  const st = ep.meta.get(String(m.id ?? ""))?.routerStatus;
1182
1366
  return !st || st === "loaded" || settings.includeUnloadedRouterModels;
1183
1367
  });
@@ -1193,25 +1377,56 @@ function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: In
1193
1377
  const rawId = String(model.id ?? "");
1194
1378
  const modelMeta = ep.meta.get(rawId) ?? {};
1195
1379
  const pm = toPiModel(model, ep, srv, settings, modelMeta);
1380
+ const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
1196
1381
 
1197
- // ID collision guard: force the "host:port/" prefix on collisions,
1198
- // then a numeric suffix if even that collides.
1382
+ // ID collision guard: add the machine tag, then a numeric suffix.
1199
1383
  if (seenIds.has(pm.id)) {
1200
- pm.id = `${idSafeHost(srv.host)}:${ep.port}/${pm.id.replace(/^\/+/, "")}`;
1384
+ if (!pm.id.includes(`(${hostPort})`)) pm.id = `${pm.id} (${hostPort})`;
1201
1385
  let n = 2;
1202
1386
  while (seenIds.has(pm.id)) pm.id = `${pm.id}-${n++}`;
1203
1387
  }
1204
1388
  seenIds.add(pm.id);
1205
1389
 
1390
+ // Thinking budgets: resolve by the registered (compact) id or legacy
1391
+ // "host:port/model" keys, then migrate legacy keys forward.
1392
+ const opts = config_ModelOptions();
1393
+ let budgets = opts[pm.id]?.thinkingBudgets;
1394
+ if (!budgets) {
1395
+ for (const legacyKey of [`${hostPort}/${pm.serverModelId.replace(/^\/+/, "")}`, pm.serverModelId]) {
1396
+ const entry = opts[legacyKey];
1397
+ if (entry?.thinkingBudgets) {
1398
+ if (!opts[pm.id]) opts[pm.id] = entry;
1399
+ delete opts[legacyKey]; // migrate: compact id replaces host:port/model
1400
+ configDirty = true;
1401
+ budgets = entry.thinkingBudgets;
1402
+ break;
1403
+ }
1404
+ }
1405
+ }
1406
+ if (budgets && ep.server !== "lmstudio") {
1407
+ pm.thinkingBudgets = budgets;
1408
+ // Models with configured thinking budgets must be registered as
1409
+ // reasoning models so pi's thinking-level machinery engages.
1410
+ pm.reasoning = true;
1411
+ }
1412
+
1413
+ // Duplicate display names within one endpoint get the raw id appended.
1206
1414
  const candidateName = cleanModelName(String(model.name ?? model.id));
1207
1415
  if (nameCount.get(candidateName)! > 1) {
1208
- pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${rawId})`;
1416
+ pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${pm.serverModelId})`;
1209
1417
  }
1210
1418
  piModels.push(pm);
1211
- if (ep.server === "zinc") zincModelIds.add(pm.id);
1419
+ if (ep.server === "zinc") {
1420
+ zincModelIds.add(pm.id);
1421
+ zincModelIds.add(pm.serverModelId);
1422
+ }
1423
+ serverModelIds.set(pm.id, pm.serverModelId);
1424
+ if (!compactModelIds.has(pm.serverModelId)) compactModelIds.set(pm.serverModelId, pm.id);
1212
1425
  }
1213
1426
  }
1214
1427
 
1428
+ if (configDirty) saveConfig(config);
1429
+
1215
1430
  // Maps for hooks
1216
1431
  endpointKinds.clear();
1217
1432
  for (const ep of scan.endpoints) {
@@ -1281,6 +1496,7 @@ export default function (pi: ExtensionAPI) {
1281
1496
  provider: PROVIDER_NAME,
1282
1497
  cacheFile: join(homedir(), ".pi", "agent", "warmup-llamacpp-infra.json"),
1283
1498
  kindFor: (baseUrl) => endpointKinds.get(baseUrl),
1499
+ requestModelFor: (modelId) => rawIdFor(modelId) ?? modelId,
1284
1500
  onEvent: (ev) => warmupStatus.handle(ev),
1285
1501
  });
1286
1502
 
@@ -1587,6 +1803,21 @@ export default function (pi: ExtensionAPI) {
1587
1803
  registerEmptyProvider();
1588
1804
  void discoverAndRegister().then((r) => schedulePolling(r.shouldPoll));
1589
1805
 
1806
+ // ── Hook 0: compact registered id → raw server model id ─────────
1807
+ // Model ids registered in pi are compact display ids ("Name (host:port)");
1808
+ // llama.cpp-family servers expect the raw model path/alias they advertised
1809
+ // in /v1/models, so the payload model field is rewritten here. Registered
1810
+ // first so every later hook (and the server) sees the raw id.
1811
+ pi.on("before_provider_request", (event, _ctx) => {
1812
+ const payload = event.payload as Record<string, unknown>;
1813
+ const modelInPayload = typeof payload.model === "string" ? payload.model : undefined;
1814
+ if (!modelInPayload) return undefined;
1815
+ const raw = serverModelIds.get(modelInPayload);
1816
+ if (raw === undefined || raw === modelInPayload) return undefined;
1817
+ debugLog(`model id "${modelInPayload}" → "${raw}"`);
1818
+ return { ...payload, model: raw };
1819
+ });
1820
+
1590
1821
  // ── Hook 1: ZINC payload workaround ────────────────────────────
1591
1822
  // ZINC rejects non-empty model ids and is picky about tool formats.
1592
1823
  function isZincModel(modelInPayload: unknown): boolean {
@@ -1638,18 +1869,23 @@ export default function (pi: ExtensionAPI) {
1638
1869
  // ── Hook 2: per-model thinking budget (llama.cpp thinking_budget_tokens) ──
1639
1870
  // pi injects thinking_token budgets from its global settings; this hook
1640
1871
  // overrides the value with the per-model budgets configured for this model.
1641
- // Registered after the ZINC hook so it sees the final payload.
1872
+ // Registered after the ZINC hook so it sees the final payload. Hook 0 has
1873
+ // already rewritten the payload model to the raw server id, so both raw
1874
+ // and compact ids are resolved against the registered model maps.
1642
1875
  pi.on("before_provider_request", (event, ctx) => {
1643
1876
  const payload = event.payload as Record<string, unknown>;
1644
1877
  const modelId = typeof payload.model === "string" ? payload.model : undefined;
1645
1878
  if (!modelId) return undefined;
1646
- const budgets = config.modelOptions[modelId]?.thinkingBudgets;
1879
+ const compactKey = compactModelIds.get(modelId) ?? modelId;
1880
+ const baseUrl = modelBaseUrls.get(compactKey);
1881
+ if (!supportsThinkingBudget(endpointKinds.get(baseUrl ?? "") as ServerKind | undefined)) return undefined;
1882
+ const budgets = config.modelOptions[compactKey]?.thinkingBudgets;
1647
1883
  if (!budgets) return undefined;
1648
1884
  const level = normalizeLevel(ctx.thinkingLevel ?? currentThinkingLevel);
1649
1885
  if (!level) return undefined;
1650
1886
  const value = budgets[level];
1651
1887
  if (typeof value !== "number") return undefined;
1652
- debugLog(`thinking budget for ${modelId} [${level}] = ${value}`);
1888
+ debugLog(`thinking budget for ${compactKey} [${level}] = ${value}`);
1653
1889
  return { ...payload, [THINKING_BUDGET_FIELD]: value };
1654
1890
  });
1655
1891
 
@@ -1669,18 +1905,22 @@ export default function (pi: ExtensionAPI) {
1669
1905
  }
1670
1906
 
1671
1907
  // ── Hook 3: header warmup capture (runs last, sees final payload) ──
1908
+ // Templates are keyed by the compact registered id (the same id passed to
1909
+ // warmupForModel); PromptWarmer re-resolves the raw server id per request.
1672
1910
  pi.on("before_provider_request", (event, ctx) => {
1673
1911
  if (!config.settings.warmup) return undefined;
1674
1912
  const payload = event.payload as Record<string, unknown>;
1675
1913
  const modelId = typeof payload?.model === "string" ? payload.model : undefined;
1676
- const baseUrl = modelId ? modelBaseUrls.get(modelId) : undefined;
1677
- warmer.onProviderPayload(payload, baseUrl, ctx.cwd);
1914
+ const compactKey = modelId ? (compactModelIds.get(modelId) ?? modelId) : undefined;
1915
+ const baseUrl = compactKey ? modelBaseUrls.get(compactKey) : undefined;
1916
+ const capturePayload = compactKey ? { ...payload, model: compactKey } : payload;
1917
+ warmer.onProviderPayload(capturePayload, baseUrl, ctx.cwd);
1678
1918
  return undefined;
1679
1919
  });
1680
1920
 
1681
1921
  // ── Command: /llamacpp-infra ───────────────────────────────────
1682
1922
  pi.registerCommand("llamacpp-infra", {
1683
- description: "llama.cpp-infra: discover llama.cpp/ZINC/DwarfStar models on any machine (config, scan, metrics…)",
1923
+ description: "llama.cpp-infra: discover llama.cpp/ZINC/DwarfStar/LM Studio models on any machine (config, scan, metrics…)",
1684
1924
  getArgumentCompletions: (prefix) =>
1685
1925
  ["config", "scan", "status", "list", "metrics", "help"]
1686
1926
  .filter((s) => s.startsWith(prefix))
@@ -1786,7 +2026,7 @@ export default function (pi: ExtensionAPI) {
1786
2026
  if (!ep) return undefined;
1787
2027
  // Find the raw entry whose registered id ends with the raw id fragment.
1788
2028
  for (const [rawId, meta] of ep.meta) {
1789
- if (m.id.endsWith(`/${rawId.replace(/^\/+/, "")}`) || m.id === rawId) return meta;
2029
+ if (m.serverModelId === rawId || m.id === rawId) return meta;
1790
2030
  }
1791
2031
  return undefined;
1792
2032
  };
@@ -1814,7 +2054,7 @@ export default function (pi: ExtensionAPI) {
1814
2054
  function showHelp(ctx: ExtensionContext) {
1815
2055
  ctx.ui.notify(
1816
2056
  [
1817
- "🦙 llama.cpp-infra — models served by llama.cpp & variants (ZINC, DwarfStar/ds4, lucebox) on any machine",
2057
+ "🦙 llama.cpp-infra — models served by llama.cpp & variants (ZINC, DwarfStar/ds4, lucebox, LM Studio) on any machine",
1818
2058
  "",
1819
2059
  " /llamacpp-infra → quick status",
1820
2060
  " /llamacpp-infra config → ⚙️ configure servers, budgets & settings",
@@ -1875,12 +2115,16 @@ export default function (pi: ExtensionAPI) {
1875
2115
  "🦙 llama.cpp-infra",
1876
2116
  "",
1877
2117
  "Discovers models served by llama.cpp (single & router/multi-model),",
1878
- "ZINC, DwarfStar (ds4-server) and lucebox on any number of machines,",
1879
- "and registers them into pi's native /model list.",
2118
+ "ZINC, DwarfStar (ds4-server), lucebox and LM Studio on any number",
2119
+ "of machines, and registers them into pi's native /model list.",
1880
2120
  "",
2121
+ "Models appear as compact ids: \"Name (host:port)\" — the raw GGUF",
2122
+ "path/alias is sent to the server automatically on every request.",
2123
+ "LM Studio uses its OpenAI-compatible /v1 endpoint and enriches",
2124
+ "metadata from /api/v1/models (or legacy /api/v0/models).",
1881
2125
  "Per-model metadata: vision, drafter, model quant, KV cache quant.",
1882
2126
  "Per-model thinking budgets via llama.cpp thinking_budget_tokens.",
1883
- "Live throughput metrics from each server's /metrics endpoint.",
2127
+ "Live throughput metrics from each server's /metrics endpoint when exposed.",
1884
2128
  "",
1885
2129
  `Config: ${getConfigPath()}`,
1886
2130
  ].join("\n"),
@@ -2078,7 +2322,7 @@ export default function (pi: ExtensionAPI) {
2078
2322
  ctx.ui.notify(`⚠️ A server with host "${trimmedHost}" already exists`, "warning");
2079
2323
  return;
2080
2324
  }
2081
- const portsRaw = await ctx.ui.input("➕ Ports to probe", "e.g. 8000, 8080-8082");
2325
+ const portsRaw = await ctx.ui.input("➕ Ports to probe", "e.g. 1234, 8000, 8080-8082");
2082
2326
  if (portsRaw === undefined) return;
2083
2327
  const ports = parsePorts(portsRaw);
2084
2328
  if (!ports) {
@@ -2087,7 +2331,7 @@ export default function (pi: ExtensionAPI) {
2087
2331
  }
2088
2332
  const label = await ctx.ui.input("🏷️ Label (optional)", trimmedHost);
2089
2333
  if (label === undefined) return;
2090
- const probeDs4 = await ctx.ui.confirm("🕵️ ds4 (DwarfStar) probe?", "Enable the chat-completions ping probe for this machine? (for DwarfStar/ds4-server hosts)");
2334
+ const probeDs4 = await ctx.ui.confirm("🕵️ ds4 (DwarfStar) probe?", "Enable the chat-completions ping probe for this machine? (for DwarfStar/ds4-server hosts; not needed for LM Studio)");
2091
2335
 
2092
2336
  let id = idSafeHost(trimmedHost).replace(/[^a-z0-9.-]/g, "-");
2093
2337
  let n = 2;
@@ -248,6 +248,8 @@ export interface PromptWarmerOptions {
248
248
  cacheFile?: string;
249
249
  /** Devuelve el tipo de servidor ("llamacpp"|"lucebox"|"ds4"|"zinc"|"lmstudio") para una baseUrl. */
250
250
  kindFor?: (baseUrl: string) => string | undefined;
251
+ /** Mapea el id registrado (compacto) al id crudo que espera el servidor. */
252
+ requestModelFor?: (modelId: string) => string;
251
253
  /** Callback opcional para eventos de warmup (estado en footer, notify, ...). */
252
254
  onEvent?: (ev: WarmupEvent) => void;
253
255
  /** Log opcional. Por defecto silencioso (no-op); activa con PI_WARMUP_DEBUG=1. */
@@ -351,6 +353,7 @@ export class PromptWarmer {
351
353
  readonly provider: string;
352
354
  private cacheFile: string;
353
355
  private kindFor: (baseUrl: string) => string | undefined;
356
+ private requestModelFor: (modelId: string) => string;
354
357
  private onEvent?: (ev: WarmupEvent) => void;
355
358
  private log: (msg: string) => void;
356
359
 
@@ -368,6 +371,7 @@ export class PromptWarmer {
368
371
  this.cacheFile =
369
372
  opts.cacheFile ?? path.join(homedir(), ".pi", "agent", `warmup-${opts.provider}.json`);
370
373
  this.kindFor = opts.kindFor ?? (() => KIND_LLAMACPP);
374
+ this.requestModelFor = opts.requestModelFor ?? ((id: string) => id);
371
375
  this.onEvent = opts.onEvent;
372
376
  // Por defecto silencioso: escribir por stderr corrompe el render de la TUI
373
377
  // de pi (trazas sucias en el editor / input box). Solo se loguea con
@@ -392,6 +396,15 @@ export class PromptWarmer {
392
396
  }
393
397
  }
394
398
 
399
+ /** Id crudo que espera el servidor para un id registrado (con fallback). */
400
+ private resolveRequestModel(modelId: string): string {
401
+ try {
402
+ return this.requestModelFor(modelId) || modelId;
403
+ } catch {
404
+ return modelId;
405
+ }
406
+ }
407
+
395
408
  // ── Warmup ────────────────────────────────────────────────────────────────
396
409
 
397
410
  /**
@@ -422,7 +435,7 @@ export class PromptWarmer {
422
435
  let body: WarmupBody;
423
436
  if (tpl) {
424
437
  body = {
425
- model: tpl.model,
438
+ model: this.resolveRequestModel(tpl.model),
426
439
  messages: [...tpl.systemMessages, { role: "user", content: PLACEHOLDER_USER }],
427
440
  max_tokens: 1,
428
441
  temperature: 0,
@@ -432,7 +445,7 @@ export class PromptWarmer {
432
445
  if (cachePromptSupported(tpl.kind)) body.cache_prompt = true;
433
446
  } else if (systemPrompt && FALLBACK_ENABLED) {
434
447
  body = {
435
- model: model.id,
448
+ model: this.resolveRequestModel(model.id),
436
449
  messages: [
437
450
  { role: "system", content: systemPrompt },
438
451
  { role: "user", content: PLACEHOLDER_USER },