pi-mtplx 0.1.8 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -18,14 +18,14 @@ Restart Pi after installation.
18
18
  2. Download the model(s) you want, e.g. `mtplx install Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality`.
19
19
  3. Register a downloaded model with Pi via `/mtplx` → **Models** (or add it manually to `~/.pi/agent/mtplx-models.json`), then open `/model` to activate it.
20
20
 
21
- If you pick an MTPLX model that isn't installed, the server won't start — Pi will warn you. Use `mtplx list` to see what you've downloaded.
21
+ `/mtplx` → **Models** only scans models installed through the local MTPLX CLI. It never downloads model weights and does not discover models available only on a configured remote endpoint. If you pick an MTPLX model that isn't installed locally, Pi cannot auto-start it. Use `mtplx list` to see what you've downloaded.
22
22
 
23
23
  ## What happens on first run
24
24
 
25
25
  Once installed, Pi automatically manages your MTPLX workflow:
26
26
 
27
- - **Model discovery** — `/mtplx` → **Models** scans what you've downloaded (`mtplx list`) and registers any of them with Pi. No model is bundled or pre-hardcoded: each Pi model id and its capabilities (context window, vision, reasoning) are derived live from the installed artifact, so models beyond the MTPLX stock set work too.
28
- - **Auto-start** — The MTPLX server starts when you switch to an `mtplx` model and shuts down cleanly when Pi exits.
27
+ - **Model discovery** — `/mtplx` → **Models** scans what you've downloaded (`mtplx list`) and registers any of them with Pi. Each Pi model ID comes from MTPLX's own `quickstart --dry-run` plan, while capabilities (context window, vision, reasoning) are read from the installed artifact. Models beyond the MTPLX stock set work too.
28
+ - **Auto-start** — The MTPLX server starts when you switch to an `mtplx` model. Auto-start and auto-shutdown are on by default and can be changed in `/mtplx`.
29
29
  - **Token speed** — A `⚡N.N tk/s` indicator appears in the footer showing the generation speed of the last assistant turn.
30
30
 
31
31
  ## Commands
@@ -34,10 +34,15 @@ Run `/mtplx` to open an interactive menu:
34
34
 
35
35
  | Option | What it does |
36
36
  | -------- | ------------- |
37
- | **Toggle (on/off)** | Start or stop the MTPLX server |
37
+ | **Server** | Read-only status showing the model currently served at the configured endpoint, or that the endpoint is unavailable |
38
+ | **Toggle (on/off)** | Start or stop an MTPLX server launched by this Pi session; separately managed servers are left running |
39
+ | **API Key** | View a masked identifier for, or replace, the API key Pi uses for the local MTPLX server |
40
+ | **Endpoint** | Set the OpenAI-compatible MTPLX base URL; useful for a separately started server |
38
41
  | **Fan Curves** | Set the thermal profile (`default`, `smart`, `max`) |
42
+ | **Auto Shutdown Pi-Owned Server** | Choose whether Pi stops an MTPLX server it launched on `/quit` or a normal terminal-close shutdown (on by default); separately managed and remote servers are always left running |
43
+ | **Auto Start** | Choose whether Pi starts or switches MTPLX when loading an MTPLX model (on by default) |
39
44
  | **SSD Session Cache** | Enable or disable MTPLX's SSD-backed session cache for subsequent server starts (on by default) |
40
- | **Models** | Register or unregister models — ✓ means registered (click to unregister), ✗ means available (click to register) |
45
+ | **Models** | Register or unregister locally installed models — ✓ means registered (click to unregister), ✗ means available (click to register); it does not fetch or discover remote models |
41
46
  | **Uninstall** | Remove the `mtplx` provider from Pi's config |
42
47
 
43
48
  > **To activate a model** after registering or unregistering it, open **`/model`** (or `/scoped-models`). pi-mtplx writes Pi's config files immediately, but Pi loads them into memory on startup — so a **newly registered** MTPLX model only appears in `/model` after you restart Pi with **`/quit`** and relaunch it (`pi`). Unregistering/re-registering an existing model is picked up by opening `/model`, but a brand-new model id requires the restart.
@@ -46,21 +51,23 @@ Run `/mtplx` to open an interactive menu:
46
51
 
47
52
  ### Model registry
48
53
 
49
- Models are registered in `~/.pi/agent/mtplx-models.json`. Each entry maps a Pi model ID to an MTPLX artifact ref:
54
+ Models are registered in `~/.pi/agent/mtplx-models.json`. Each entry maps MTPLX's canonical served model ID to an MTPLX artifact ref:
50
55
 
51
56
  ```json
52
57
  {
53
- "mtplx-qwen3.8-27b-mtplx-optimized-quality": {
58
+ "mtplx-qwen38-27b-optimized-quality": {
54
59
  "ref": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality"
55
60
  }
56
61
  }
57
62
  ```
58
63
 
59
- Register new models via the `/mtplx` → **Models** menu, or add entries manually to this file. To activate a **newly added** model, restart Pi with `/quit` and relaunch it (`pi`), then open `/model` (or `/scoped-models`) — Pi reads this file at startup, so a brand-new model id won't show up in `/model` otherwise.
64
+ Register new models via the `/mtplx` → **Models** menu; it asks MTPLX for the exact served ID. Opening this menu also migrates older ref-derived IDs to their canonical MTPLX names. Restart Pi after a migration or registration, then open `/model` (or `/scoped-models`) to activate the model.
65
+
66
+ Local Forge outputs use their installed paths as artifact refs; downloaded repositories retain their `owner/model` refs. If an artifact cannot produce a serving plan, the menu shows MTPLX's error and skips that artifact while keeping the other models available.
60
67
 
61
68
  ### Fan mode
62
69
 
63
- Controls the thermal profile of your MTPLX server. Saved to `~/.pi/agent/mtplx-fanmode.json` and persists across restarts.
70
+ Controls the thermal profile used when Pi starts an MTPLX server. Saved to `~/.pi/agent/mtplx-fanmode.json` and persists across restarts. When Pi discovers a server that is already running—whether it is local, remote, or Pi-owned—it does not automatically change the fan curve; if the server reports a supported curve that differs from Pi's saved default, Pi adopts that curve as its new default for future Pi-started servers. Choosing `/mtplx` → **Fan Curves** is an explicit override and changes the fan curve of the active healthy server.
64
71
 
65
72
  | Mode | Behavior |
66
73
  | ------ | ---------- |
@@ -68,12 +75,41 @@ Controls the thermal profile of your MTPLX server. Saved to `~/.pi/agent/mtplx-f
68
75
  | `smart` | Default — adaptive thermal management |
69
76
  | `max` | Maximum fan, fastest inference |
70
77
 
78
+ ### Server authentication
79
+
80
+ pi-mtplx always uses an API key. It uses `providers.mtplx.apiKey` from `~/.pi/agent/models.json` when configured; otherwise it uses the local default `mtplx-local`. The resolved key is passed to Pi-managed MTPLX startup and sent with health, fan-control, and inference requests. If Pi already has a stored MTPLX API-key credential, pi-mtplx synchronizes it with this key at startup and whenever `/mtplx` → **API Key** saves a change. Restart Pi after changing the key so its in-memory inference provider reloads the configuration. The menu only shows a masked suffix for custom keys, never the full secret.
81
+
82
+ ### Standalone MTPLX
83
+
84
+ To use an MTPLX server that you start outside Pi, set `/mtplx` → **Auto Start** to off, then set `/mtplx` → **Endpoint** to that server's OpenAI base URL (for example, `http://127.0.0.1:8001/v1`). Restart Pi after changing it. The endpoint, health checks, and fan controls all use `providers.mtplx.baseUrl`; its default is `http://127.0.0.1:8000/v1`. pi-mtplx never stops a server it did not launch, including on normal Pi shutdown.
85
+
86
+ The served model ID must be the same ID selected in Pi. Before every MTPLX request, pi-mtplx verifies this through `/health` and reports both IDs if they differ. A separately managed server is never switched; start it with `--model-id <Pi model ID>` if its artifact's default identity differs. `/mtplx` → **Models** cannot discover a remote-only model, so add its exact ID to Pi's `mtplx` provider configuration manually before selecting it. Use the same API key in both the MTPLX launch command and `/mtplx` → **API Key**.
87
+
88
+ ### Auto start
89
+
90
+ Auto Start is on by default. Pi first checks the selected endpoint before starting anything. If a healthy MTPLX server is already there and serves the selected model, Pi uses it without replacing it. If it serves a different model, Pi switches it only when that server was started by the current Pi session; separately managed and remote servers are left alone, and Pi blocks the request with both model IDs. Pi starts MTPLX only when no MTPLX server is available at a loopback endpoint. With Auto Start off, Pi performs the same health and model-match check but never starts or switches a server; it instead tells you to start MTPLX at the configured endpoint.
91
+
92
+ ### Auto shutdown Pi-owned server
93
+
94
+ Auto shutdown is on by default. When enabled, pi-mtplx stops an MTPLX server it launched during the current Pi session when Pi exits via `/quit` or a normal terminal-close shutdown. It never stops a separately managed or remote server. Turn it off in `/mtplx` → **Auto Shutdown Pi-Owned Server** to leave the Pi-launched server running after Pi exits. It cannot handle abrupt termination such as `SIGKILL` or a power loss.
95
+
96
+ ## Remote endpoints
97
+
98
+ I do not have a remote server, so I was unable to personally test how pi-mtplx interacts with remote endpoints.
99
+
100
+ ## Reporting bugs
101
+
102
+ If you encounter a bug, please open a [GitHub Issue](../../issues) and let me know.
103
+
71
104
  ## Troubleshooting
72
105
 
73
106
  | Problem | Fix |
74
107
  | --------- | ----- |
75
108
  | **"MTPLX not started"** — Pi warns when you ask an MTPLX model to respond | Run `/mtplx` → **Toggle** to start the server |
76
109
  | **"No MTPLX models registered"** | Run `/mtplx` → **Models** to discover and register one |
110
+ | **"MTPLX is already running ... but Pi requested ..."** | A manually started MTPLX server is serving a different, unmanaged model. Stop it in the MTPLX app, or find it with `lsof -nP -iTCP:8000 -sTCP:LISTEN` and run `kill -TERM <PID>`, then retry so Pi can start and manage the selected model. |
111
+ | **"MTPLX rejected the API key"** | The running server requires a different key. Use `/mtplx` → **API Key** to enter its current key, then retry. |
112
+ | **"Connection error"** with standalone MTPLX | Pi's configured endpoint has no listener. Set `/mtplx` → **Endpoint** to the standalone server's exact URL (including `/v1`), then restart Pi. |
77
113
  | **MTPLX startup timed out after 180s** | Run `mtplx status --deep` for MTPLX-side diagnostics (model validation, memory, thermal) |
78
114
 
79
115
  ## License
@@ -6,33 +6,69 @@
6
6
  * and shared helpers (src/utils.ts) into Pi via before_agent_start,
7
7
  * agent_end and session_shutdown, plus the /mtplx command.
8
8
  */
9
- import { acquire, release, stopServer } from "../src/mtplx-process.ts";
10
- import { getFanMode, health, setFanMode, setFanModeValue } from "../src/mtplx-client.ts";
9
+ import { acquire, isOwnedByThisSession, release, stopServer, validateServer } from "../src/mtplx-process.ts";
10
+ import { authenticationFailureMessage, getFanMode, health, healthProbe, setFanMode, setFanModeValue } from "../src/mtplx-client.ts";
11
11
  import { MTPLX_PROVIDER, manageModels, removeModel, removePiMtplxProvider } from "../src/model-discovery.ts";
12
- import { FAN_MODES, isMtplxModel, loadSsdSessionCache, saveFanMode, saveSsdSessionCache, type FanMode } from "../src/utils.ts";
12
+ import { FAN_MODES, isMtplxModel, loadAutoShutdown, loadAutoStart, loadMtplxApiKey, loadMtplxEndpoint, loadSsdSessionCache, maskMtplxApiKey, mtplxEndpointFromBaseUrl, saveAutoShutdown, saveAutoStart, saveFanMode, saveMtplxApiKey, saveMtplxEndpoint, saveSsdSessionCache, syncMtplxStoredCredential, type FanMode } from "../src/utils.ts";
13
13
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
14
14
 
15
+ function syncFanModeFromRunningServer(fanMode: string | undefined): { previous: FanMode; current: FanMode } | undefined {
16
+ if (typeof fanMode !== "string" || !(FAN_MODES as readonly string[]).includes(fanMode)) return;
17
+ const current = fanMode as FanMode;
18
+ const previous = getFanMode();
19
+ if (current === previous) return;
20
+ setFanModeValue(current);
21
+ saveFanMode(current);
22
+ return { previous, current };
23
+ }
24
+
15
25
  export default function mtplxAutostart(pi: ExtensionAPI): void {
26
+ let lastServerNotice: string | undefined;
27
+
28
+ // Pi's runtime may already have loaded auth.json for this process, but
29
+ // syncing here ensures its next launch uses the same key as models.json.
30
+ syncMtplxStoredCredential();
31
+
16
32
  pi.registerCommand("mtplx", {
17
- description: "MTPLX — toggle the server, configure fan and SSD cache, manage models, or uninstall",
33
+ description: "MTPLX — view server status, configure its endpoint and lifecycle, manage models, or uninstall",
18
34
  handler: async (_args, ctx) => {
19
35
  const current = await health();
20
36
  const status = current ? "on" : "off";
37
+ const fanSync = current ? syncFanModeFromRunningServer(current.fan_mode) : undefined;
38
+ const server = current
39
+ ? `Server (serving: ${current.model})`
40
+ : "Server (unavailable — check endpoint or API key)";
21
41
  await ctx.ui.setStatus("mtplx", `MTPLX: ${status}`);
42
+ if (fanSync) {
43
+ ctx.ui.notify(
44
+ `MTPLX was already running with fan curve ${fanSync.current}, while Pi's saved default was ${fanSync.previous}. Pi left the server unchanged and updated its saved default to ${fanSync.current}.`,
45
+ "info",
46
+ );
47
+ }
22
48
  const ssdSessionCache = loadSsdSessionCache();
49
+ const autoShutdown = loadAutoShutdown();
50
+ const autoStart = loadAutoStart();
51
+ const apiKeyLabel = maskMtplxApiKey(loadMtplxApiKey());
52
+ const endpoint = loadMtplxEndpoint();
23
53
  const topChoices = [
54
+ server,
24
55
  `Toggle (${status})`,
56
+ `API Key (current: ${apiKeyLabel})`,
57
+ `Endpoint (current: ${endpoint.baseUrl})`,
25
58
  `Fan Curves (current: ${getFanMode()})`,
59
+ `Auto Shutdown Pi-Owned Server (current: ${autoShutdown ? "on" : "off"})`,
60
+ `Auto Start (current: ${autoStart ? "on" : "off"})`,
26
61
  `SSD Session Cache (current: ${ssdSessionCache ? "on" : "off"})`,
27
62
  "Models",
28
63
  "Uninstall (remove provider)",
29
64
  ];
30
65
  const top = await ctx.ui.select("MTPLX", topChoices, undefined);
31
66
  if (!top) return;
67
+ if (top === server) return;
32
68
  if (top.startsWith("Toggle")) {
33
69
  if (current) {
34
- await stopServer();
35
- ctx.ui.notify("MTPLX server stopped", "info");
70
+ const stopped = await stopServer();
71
+ ctx.ui.notify(stopped ? "MTPLX server stopped" : "MTPLX is managed separately; pi-mtplx left it running.", stopped ? "info" : "warning");
36
72
  } else if (ctx.model && isMtplxModel(ctx.model)) {
37
73
  await acquire(ctx.model.id);
38
74
  ctx.ui.notify(`MTPLX server started (${ctx.model.id})`, "info");
@@ -41,6 +77,57 @@ export default function mtplxAutostart(pi: ExtensionAPI): void {
41
77
  }
42
78
  return;
43
79
  }
80
+ if (top.startsWith("Endpoint")) {
81
+ const baseUrl = await ctx.ui.input("MTPLX endpoint", `Current: ${endpoint.baseUrl} — e.g. http://127.0.0.1:8001/v1`);
82
+ if (baseUrl === undefined) return;
83
+ try {
84
+ const url = new URL(baseUrl.trim());
85
+ if (url.protocol !== "http:" && url.protocol !== "https:") throw new Error("unsupported protocol");
86
+ } catch {
87
+ ctx.ui.notify("MTPLX endpoint was not updated. Enter an http(s) URL.", "error");
88
+ return;
89
+ }
90
+ if (mtplxEndpointFromBaseUrl(baseUrl).baseUrl === endpoint.baseUrl) {
91
+ ctx.ui.notify("MTPLX endpoint is unchanged.", "info");
92
+ return;
93
+ }
94
+ if (isOwnedByThisSession()) {
95
+ try {
96
+ await stopServer();
97
+ } catch (error) {
98
+ ctx.ui.notify(
99
+ `MTPLX endpoint was not changed because Pi could not stop its running server: ${error instanceof Error ? error.message : String(error)}`,
100
+ "error",
101
+ );
102
+ return;
103
+ }
104
+ ctx.ui.notify("Stopped Pi-owned MTPLX server before changing the endpoint.", "info");
105
+ }
106
+ if (!saveMtplxEndpoint(baseUrl)) {
107
+ ctx.ui.notify("MTPLX endpoint was not updated. Enter an http(s) URL.", "error");
108
+ return;
109
+ }
110
+ const probe = await healthProbe();
111
+ ctx.ui.notify(probe.health ? "MTPLX endpoint saved and verified. Restart Pi before inference uses it." : "MTPLX endpoint saved, but health verification failed. Check its host, port, and API key; then restart Pi.", probe.health ? "info" : "warning");
112
+ return;
113
+ }
114
+ if (top.startsWith("API Key")) {
115
+ const apiKey = await ctx.ui.input(`MTPLX API Key — current: ${apiKeyLabel}`, "Paste a custom key, or leave blank for the default (visible while typing)");
116
+ if (apiKey === undefined) return;
117
+ if (!saveMtplxApiKey(apiKey)) {
118
+ ctx.ui.notify("MTPLX API key was not updated. Check that models.json is valid.", "error");
119
+ return;
120
+ }
121
+ const probe = await healthProbe();
122
+ if (probe.health) {
123
+ ctx.ui.notify("MTPLX API key updated and verified against the running server. Restart Pi before inference requests use the new key.", "info");
124
+ } else if (probe.authenticationRejected) {
125
+ ctx.ui.notify(`MTPLX API key was saved, but verification failed: ${authenticationFailureMessage()} Restart Pi after correcting it.`, "error");
126
+ } else {
127
+ ctx.ui.notify("MTPLX API key updated. No running server was available to verify it. Restart Pi before inference requests use the new key.", "warning");
128
+ }
129
+ return;
130
+ }
44
131
  if (top.startsWith("Fan Curves")) {
45
132
  const choices = FAN_MODES.map((mode) => (mode === getFanMode() ? `${mode} (current)` : mode));
46
133
  const picked = await ctx.ui.select("MTPLX — fan mode for autostart", choices, undefined);
@@ -49,10 +136,35 @@ export default function mtplxAutostart(pi: ExtensionAPI): void {
49
136
  if (!(FAN_MODES as readonly string[]).includes(mode)) return;
50
137
  setFanModeValue(mode);
51
138
  saveFanMode(mode);
52
- if (current) {
53
- await setFanMode();
54
- }
55
- ctx.ui.notify(`MTPLX fan mode set to ${getFanMode()}`, "info");
139
+ if (current) await setFanMode();
140
+ ctx.ui.notify(
141
+ current
142
+ ? `MTPLX fan curve set to ${getFanMode()} and saved as Pi's default for future Pi-started servers.`
143
+ : `MTPLX fan default saved as ${getFanMode()}. It applies the next time Pi starts MTPLX.`,
144
+ "info",
145
+ );
146
+ }
147
+ if (top.startsWith("Auto Shutdown")) {
148
+ const choices = [
149
+ autoShutdown ? "Off" : "Off (current)",
150
+ autoShutdown ? "On (current)" : "On",
151
+ ];
152
+ const picked = await ctx.ui.select("MTPLX — stop the Pi-owned server when Pi exits", choices, undefined);
153
+ if (!picked) return;
154
+ const enabled = picked.startsWith("On");
155
+ saveAutoShutdown(enabled);
156
+ ctx.ui.notify(`Pi will ${enabled ? "stop its own MTPLX server" : "leave its own MTPLX server running"} when Pi exits.`, "info");
157
+ }
158
+ if (top.startsWith("Auto Start")) {
159
+ const choices = [
160
+ autoStart ? "Off" : "Off (current)",
161
+ autoStart ? "On (current)" : "On",
162
+ ];
163
+ const picked = await ctx.ui.select("MTPLX — autostart MTPLX server when loading an MTPLX model", choices, undefined);
164
+ if (!picked) return;
165
+ const enabled = picked.startsWith("On");
166
+ saveAutoStart(enabled);
167
+ ctx.ui.notify(`MTPLX will ${enabled ? "autostart" : "not autostart"} when loading an MTPLX model.`, "info");
56
168
  }
57
169
  if (top.startsWith("SSD Session Cache")) {
58
170
  const choices = [
@@ -84,13 +196,36 @@ export default function mtplxAutostart(pi: ExtensionAPI): void {
84
196
  },
85
197
  });
86
198
 
87
- pi.on("before_agent_start", async (_event, ctx) => {
199
+ // This must run in `input`, not `before_agent_start`: Pi reports errors
200
+ // from before_agent_start but still sends the prompt to the provider.
201
+ pi.on("input", async (_event, ctx) => {
88
202
  if (!ctx.model || !isMtplxModel(ctx.model)) return;
203
+ const autoStart = loadAutoStart();
89
204
  try {
90
- await acquire(ctx.model.id);
205
+ const server = autoStart
206
+ ? await acquire(ctx.model.id)
207
+ : await validateServer(ctx.model.id);
208
+ const fanSync = server.alreadyRunning ? syncFanModeFromRunningServer(server.fanMode) : undefined;
209
+ ctx.ui.setStatus("mtplx", `MTPLX: ${server.model}`);
210
+ if (server.alreadyRunning) {
211
+ const message = autoStart
212
+ ? `MTPLX is already running at ${server.endpoint.baseUrl}, serving ${JSON.stringify(server.model)}. Auto Start will not start MTPLX. The model selected in Pi matches; proceeding.`
213
+ : `MTPLX is available at ${server.endpoint.baseUrl}, serving ${JSON.stringify(server.model)}. It matches Pi's selected model; Auto Start is off, so Pi will proceed without managing the server.`;
214
+ const fanMessage = fanSync
215
+ ? ` Its fan curve is ${fanSync.current}, while Pi's saved default was ${fanSync.previous}; Pi left the server unchanged and updated its saved default to ${fanSync.current}.`
216
+ : "";
217
+ const notice = message + fanMessage;
218
+ if (notice !== lastServerNotice) ctx.ui.notify(notice, "info");
219
+ lastServerNotice = notice;
220
+ } else {
221
+ lastServerNotice = undefined;
222
+ }
91
223
  } catch (error) {
92
- throw new Error(`MTPLX request blocked: ${error instanceof Error ? error.message : String(error)}`);
224
+ lastServerNotice = undefined;
225
+ ctx.ui.notify(`MTPLX request blocked: ${error instanceof Error ? error.message : String(error)}`, "error");
226
+ return { action: "handled" };
93
227
  }
228
+ return { action: "continue" };
94
229
  });
95
230
 
96
231
  pi.on("agent_end", async (_event, ctx) => {
@@ -98,11 +233,11 @@ export default function mtplxAutostart(pi: ExtensionAPI): void {
98
233
  });
99
234
 
100
235
  pi.on("session_shutdown", async (event) => {
101
- if (event.reason !== "quit") return;
236
+ if (event.reason !== "quit" || !loadAutoShutdown()) return;
102
237
  try {
103
238
  await stopServer();
104
239
  } catch (error) {
105
- console.error(`MTPLX cleanup on quit failed: ${error instanceof Error ? error.message : String(error)}`);
240
+ console.error(`MTPLX auto-shutdown failed: ${error instanceof Error ? error.message : String(error)}`);
106
241
  }
107
242
  });
108
243
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-mtplx",
3
- "version": "0.1.8",
3
+ "version": "0.2.0",
4
4
  "description": "Zero-config MTPLX integration for Pi coding agent",
5
5
  "keywords": [
6
6
  "pi-package",
@@ -1,20 +1,18 @@
1
1
  /**
2
2
  * MTPLX model registry: Pi model id → installed MTPLX artifact ref.
3
3
  *
4
- * The refs are the artifact identifiers reported by `mtplx models --json`;
5
- * `--model-id` makes /health and /v1/models report Pi's id.
4
+ * The refs are the artifact identifiers reported by `mtplx list --json`.
5
+ * MTPLX's `quickstart --dry-run --json` supplies the canonical served id used
6
+ * by Pi, /health, and /v1/models.
6
7
  *
7
8
  * The extension ships no model weights and hard-codes no model. The registry
8
9
  * holds only models registered through `/mtplx`, persisted to
9
- * ~/.pi/agent/mtplx-models.json; each Pi model id is derived live from the
10
- * artifact's ref (see modelIdFromRef) so models.json, enabledModels and this
11
- * registry always agree on the id.
10
+ * ~/.pi/agent/mtplx-models.json; models.json, enabledModels and this registry
11
+ * use the same MTPLX-supplied id.
12
12
  *
13
- * Pi's model catalog is the USER's own ~/.pi/agent/models.json — a provider
14
- * config that pre-existed this package (created by `mtplx start pi` / the
15
- * /mtplx UI). The canonical provider name is MTPLX's own `mtplx`
16
- * (PI_PROVIDER_ID in MTPLX's mtplx/pi.py); this extension reads and writes
17
- * only that provider entry and leaves every other provider untouched.
13
+ * Pi's model catalog lives in ~/.pi/agent/models.json. The canonical provider
14
+ * name is `mtplx`; this extension reads and writes only that provider entry
15
+ * and leaves every other provider untouched.
18
16
  */
19
17
  import { readFileSync, writeFileSync, mkdirSync, existsSync } from "node:fs";
20
18
  import { homedir } from "node:os";
@@ -22,12 +20,11 @@ import { join } from "node:path";
22
20
  import { execFile } from "node:child_process";
23
21
  import { promisify } from "node:util";
24
22
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
25
- import { modelIdFromRef, displayNameFromId } from "./utils.ts";
23
+ import { commandError, displayNameFromId } from "./utils.ts";
26
24
 
27
25
  const execFileAsync = promisify(execFile);
28
26
 
29
- // Matches MTPLX's own PI_PROVIDER_ID ("mtplx") so the extension operates on
30
- // the exact provider block that `mtplx start pi` creates.
27
+ // Keep the provider name consistent across model registration and lifecycle handling.
31
28
  export const MTPLX_PROVIDER = "mtplx";
32
29
  const MODELS_FILE = join(homedir(), ".pi", "agent", "mtplx-models.json");
33
30
  const SETTINGS_FILE = join(homedir(), ".pi", "agent", "settings.json");
@@ -101,6 +98,9 @@ export function disableModelInSettings(modelId: string): boolean {
101
98
  }
102
99
  }
103
100
  type MtplxListedModel = { repo_id?: unknown; path?: unknown; name?: unknown };
101
+ type ResolvedMtplxModel = { model: MtplxListedModel; ref: string; id: string };
102
+
103
+ const servedIdCache = new Map<string, string>();
104
104
 
105
105
  export async function listMtplxModels(): Promise<MtplxListedModel[]> {
106
106
  try {
@@ -114,7 +114,64 @@ export async function listMtplxModels(): Promise<MtplxListedModel[]> {
114
114
  }
115
115
 
116
116
  export function listedIdentity(model: MtplxListedModel): string {
117
- return typeof model.repo_id === "string" && model.repo_id ? model.repo_id : typeof model.path === "string" ? model.path : "";
117
+ const repo = typeof model.repo_id === "string" ? model.repo_id : "";
118
+ // Forge artifacts have a bare directory name in repo_id. Quickstart treats
119
+ // that as a path relative to Pi's cwd, so use the installed path instead.
120
+ if (repo.includes("/")) return repo;
121
+ return typeof model.path === "string" && model.path ? model.path : repo;
122
+ }
123
+
124
+ /**
125
+ * MTPLX itself chooses the default served id. `mtplx list` reports artifact
126
+ * refs, not that id, so ask quickstart for its dry-run plan rather than trying
127
+ * to reproduce its naming rules in Pi.
128
+ */
129
+ export async function servedModelId(ref: string): Promise<string> {
130
+ const cached = servedIdCache.get(ref);
131
+ if (cached) return cached;
132
+ try {
133
+ const { stdout } = await execFileAsync("mtplx", ["quickstart", "--model", ref, "--dry-run", "--json"], { timeout: 30_000 });
134
+ const id = servedModelIdFromDryRun(JSON.parse(stdout) as unknown);
135
+ servedIdCache.set(ref, id);
136
+ return id;
137
+ } catch (error) {
138
+ // MTPLX reports CLI errors as JSON on stdout, which execFile's default
139
+ // error.message omits. Keep that diagnostic visible in the Models menu.
140
+ const stdout = (error as { stdout?: string } | null)?.stdout;
141
+ let detail: unknown;
142
+ try {
143
+ const payload = JSON.parse(stdout ?? "") as { detail?: unknown; error?: unknown };
144
+ detail = payload.detail || payload.error;
145
+ } catch { /* Fall back to stderr / process error below. */ }
146
+ throw new Error(typeof detail === "string" ? detail : commandError(error), { cause: error });
147
+ }
148
+ }
149
+
150
+ /** A broken or unfinished artifact must not block management of other models. */
151
+ export async function resolveInstalledModels(
152
+ installed: readonly MtplxListedModel[],
153
+ resolveId: (ref: string) => Promise<string> = servedModelId,
154
+ ): Promise<{ resolved: ResolvedMtplxModel[]; failures: string[] }> {
155
+ const resolved: ResolvedMtplxModel[] = [];
156
+ const failures: string[] = [];
157
+ for (const model of installed) {
158
+ const ref = listedIdentity(model);
159
+ if (!ref) continue;
160
+ try {
161
+ resolved.push({ model, ref, id: await resolveId(ref) });
162
+ } catch (error) {
163
+ failures.push(`${ref}: ${error instanceof Error ? error.message : String(error)}`);
164
+ }
165
+ }
166
+ return { resolved, failures };
167
+ }
168
+
169
+ /** Parse the model id from MTPLX's machine-readable quickstart plan. */
170
+ export function servedModelIdFromDryRun(plan: unknown): string {
171
+ if (typeof plan !== "object" || plan === null || typeof (plan as { model_id?: unknown }).model_id !== "string" || !(plan as { model_id: string }).model_id) {
172
+ throw new Error("MTPLX did not return a served model id");
173
+ }
174
+ return (plan as { model_id: string }).model_id;
118
175
  }
119
176
 
120
177
  /**
@@ -135,6 +192,68 @@ function catalogModelIds(): string[] {
135
192
  }
136
193
  }
137
194
 
195
+ /**
196
+ * Replace legacy ref-derived ids with MTPLX's own served ids. This runs only
197
+ * from the explicit Models menu, where the resulting Pi restart is expected.
198
+ */
199
+ function migrateRegisteredModelIds(models: readonly ResolvedMtplxModel[]): string[] {
200
+ const servedIds = new Map(models.map(({ ref, id }) => [ref, id]));
201
+ const changes = new Map<string, string>();
202
+ for (const [oldId, entry] of Object.entries(MTPLX_MODELS)) {
203
+ const newId = servedIds.get(entry.ref);
204
+ if (!newId || newId === oldId) continue;
205
+ const collision = MTPLX_MODELS[newId];
206
+ if (collision && collision.ref !== entry.ref) {
207
+ console.warn(`pi-mtplx: cannot migrate ${oldId} to ${newId}; that id is already assigned to a different artifact.`);
208
+ continue;
209
+ }
210
+ changes.set(oldId, newId);
211
+ }
212
+ if (changes.size === 0) return [];
213
+
214
+ try {
215
+ const modelsJsonPath = join(homedir(), ".pi", "agent", "models.json");
216
+ const catalog = JSON.parse(readFileSync(modelsJsonPath, "utf8")) as { providers?: Record<string, unknown> };
217
+ const provider = catalog.providers?.[MTPLX_PROVIDER] as { models?: Record<string, unknown>[] } | undefined;
218
+ if (provider && Array.isArray(provider.models)) {
219
+ const seen = new Set<string>();
220
+ provider.models = provider.models.flatMap((model) => {
221
+ const id = typeof model.id === "string" ? model.id : undefined;
222
+ const replacement = id ? changes.get(id) : undefined;
223
+ const nextId = replacement ?? id;
224
+ if (!nextId || seen.has(nextId)) return [];
225
+ seen.add(nextId);
226
+ return [{ ...model, id: nextId, ...(replacement ? { name: displayNameFromId(nextId) } : {}) }];
227
+ });
228
+ }
229
+ writeFileSync(modelsJsonPath, JSON.stringify(catalog, null, 2) + "\n");
230
+
231
+ if (existsSync(SETTINGS_FILE)) {
232
+ const settings = JSON.parse(readFileSync(SETTINGS_FILE, "utf8")) as { enabledModels?: unknown };
233
+ if (Array.isArray(settings.enabledModels)) {
234
+ settings.enabledModels = [...new Set(settings.enabledModels.map((entry) => {
235
+ if (typeof entry !== "string") return entry;
236
+ const id = entry.startsWith(`${MTPLX_PROVIDER}/`) ? entry.slice(MTPLX_PROVIDER.length + 1) : undefined;
237
+ return id && changes.has(id) ? `${MTPLX_PROVIDER}/${changes.get(id)}` : entry;
238
+ }))];
239
+ writeFileSync(SETTINGS_FILE, JSON.stringify(settings, null, 2) + "\n");
240
+ }
241
+ }
242
+
243
+ for (const [oldId, newId] of changes) {
244
+ const entry = MTPLX_MODELS[oldId];
245
+ if (!entry) continue;
246
+ MTPLX_MODELS[newId] = entry;
247
+ delete MTPLX_MODELS[oldId];
248
+ }
249
+ saveRegisteredModels();
250
+ return [...changes].map(([oldId, newId]) => `${oldId} → ${newId}`);
251
+ } catch (error) {
252
+ console.error(`pi-mtplx could not migrate MTPLX model ids: ${error instanceof Error ? error.message : String(error)}`);
253
+ return [];
254
+ }
255
+ }
256
+
138
257
  type ModelProfile = {
139
258
  contextWindow: number;
140
259
  maxTokens: number;
@@ -197,13 +316,17 @@ export async function manageModels(ctx: ExtensionContext): Promise<boolean> {
197
316
  ctx.ui.notify("No MTPLX models found. Install one with `mtplx install`.", "warning");
198
317
  return false;
199
318
  }
319
+ const { resolved, failures } = await resolveInstalledModels(installed);
320
+ if (failures.length > 0) {
321
+ ctx.ui.notify(`Skipped ${failures.length} MTPLX model(s) whose served ID could not be read:\n${failures.join("\n")}`, "warning");
322
+ }
323
+ if (resolved.length === 0) return false;
324
+ const migrated = migrateRegisteredModelIds(resolved);
325
+ if (migrated.length > 0) {
326
+ ctx.ui.notify(`Updated registered model IDs to MTPLX's canonical names. Restart Pi before selecting them.`, "info");
327
+ }
200
328
  const catalog = new Set(catalogModelIds());
201
- const choices = installed.map((model) => {
202
- const ref = listedIdentity(model);
203
- // Each artifact maps to exactly one Pi model id, derived live from its ref.
204
- // No hard-coded ids — the same id is written to models.json, enabledModels and
205
- // the registry so the three never drift apart.
206
- const id = modelIdFromRef(ref);
329
+ const choices = resolved.map(({ ref, id }) => {
207
330
  // ✓ means registered in Pi's catalog (models.json); ✗ means just installed in MTPLX.
208
331
  const mark = catalog.has(id) ? "✓" : "✗";
209
332
  return `${mark} ${id} — ${ref}`;
@@ -212,9 +335,10 @@ export async function manageModels(ctx: ExtensionContext): Promise<boolean> {
212
335
  const picked = await ctx.ui.select("MTPLX models — ✓ registered in Pi · ✗ available in MTPLX — registering a new model needs a Pi restart (/quit → pi) to show up in /model", choices, undefined);
213
336
  if (!picked || picked === "Cancel") return false;
214
337
  const ref = picked.split(" — ").slice(1).join(" — ");
215
- const modelId = modelIdFromRef(ref);
216
338
  // The installed-model record for the chosen artifact (used to read its live metadata).
217
- const chosen = installed.find((m) => listedIdentity(m) === ref);
339
+ const selected = resolved.find((entry) => entry.ref === ref);
340
+ if (!selected) return false;
341
+ const { model: chosen, id: modelId } = selected;
218
342
 
219
343
  if (catalog.has(modelId)) {
220
344
  // Unregister: remove from models.json, enabledModels, and the id→ref registry.
@@ -349,4 +473,4 @@ export function removePiMtplxProvider(): boolean {
349
473
  console.error(`pi-mtplx uninstall could not update models.json: ${error instanceof Error ? error.message : String(error)}`);
350
474
  return false;
351
475
  }
352
- }
476
+ }
@@ -1,12 +1,13 @@
1
1
  /**
2
- * Minimal HTTP client for the locally-running MTPLX server (OpenAI-compatible
3
- * endpoint on 127.0.0.1:8000). Only touches endpoints that MTPLX exposes:
2
+ * Minimal HTTP client for the configured MTPLX OpenAI-compatible endpoint.
3
+ * Only touches endpoints that MTPLX exposes:
4
4
  * GET /health
5
5
  * POST /v1/mtplx/thermal/fan_mode
6
6
  */
7
- import { HOST, PORT, type FanMode, loadFanMode } from "./utils.ts";
7
+ import { type FanMode, type MtplxEndpoint, loadFanMode, loadMtplxEndpoint, loadResolvedMtplxApiKey } from "./utils.ts";
8
8
 
9
9
  export type Health = { ok: true; model: string; model_path?: string; fan_mode?: string };
10
+ export type HealthProbe = { health: Health | undefined; authenticationRejected?: true };
10
11
 
11
12
  // Current fan curve for autostart and live updates. Loaded from disk so the
12
13
  // choice survives Pi restarts (see FANMODE_FILE in utils.ts).
@@ -20,37 +21,66 @@ export function setFanModeValue(mode: FanMode): void {
20
21
  fanMode = mode;
21
22
  }
22
23
 
23
- export async function health(): Promise<Health | undefined> {
24
+ /**
25
+ * Pi's own model requests use the configured key, or pi-mtplx's deterministic
26
+ * local fallback. Mirror that header for every lifecycle endpoint.
27
+ */
28
+ function authRequest(): { headers: Record<string, string> } {
29
+ return { headers: { authorization: `Bearer ${loadResolvedMtplxApiKey()}` } };
30
+ }
31
+
32
+ export function authenticationFailureMessage(): string {
33
+ return "MTPLX rejected Pi's API key. Update it via /mtplx → API Key.";
34
+ }
35
+
36
+ export async function healthProbe(endpoint = loadMtplxEndpoint()): Promise<HealthProbe> {
24
37
  const controller = new AbortController();
25
38
  const timeout = setTimeout(() => controller.abort(), 1_500);
26
39
  try {
27
- const response = await fetch(`http://${HOST}:${PORT}/health`, { signal: controller.signal });
28
- if (!response.ok) return undefined;
40
+ const auth = authRequest();
41
+ const response = await fetch(`${endpoint.origin}/health`, {
42
+ headers: auth.headers,
43
+ signal: controller.signal,
44
+ });
45
+ if (response.status === 401 || response.status === 403) {
46
+ return { health: undefined, authenticationRejected: true };
47
+ }
48
+ if (!response.ok) return { health: undefined };
29
49
  const body = (await response.json()) as { ok?: unknown; model?: unknown; model_path?: unknown; fan_mode?: unknown };
30
- if (body.ok !== true || typeof body.model !== "string") return undefined;
50
+ if (body.ok !== true || typeof body.model !== "string") return { health: undefined };
31
51
  return {
32
- ok: true,
33
- model: body.model,
34
- model_path: typeof body.model_path === "string" ? body.model_path : undefined,
35
- fan_mode: typeof body.fan_mode === "string" ? body.fan_mode : undefined,
52
+ health: {
53
+ ok: true,
54
+ model: body.model,
55
+ model_path: typeof body.model_path === "string" ? body.model_path : undefined,
56
+ fan_mode: typeof body.fan_mode === "string" ? body.fan_mode : undefined,
57
+ },
36
58
  };
37
59
  } catch {
38
- return undefined;
60
+ return { health: undefined };
39
61
  } finally {
40
62
  clearTimeout(timeout);
41
63
  }
42
64
  }
43
65
 
66
+ export async function health(endpoint?: MtplxEndpoint): Promise<Health | undefined> {
67
+ return (await healthProbe(endpoint)).health;
68
+ }
69
+
44
70
  export async function setFanMode(): Promise<void> {
45
71
  const controller = new AbortController();
46
72
  const timeout = setTimeout(() => controller.abort(), 10_000);
47
73
  try {
48
- const response = await fetch(`http://${HOST}:${PORT}/v1/mtplx/thermal/fan_mode`, {
74
+ const auth = authRequest();
75
+ const response = await fetch(`${loadMtplxEndpoint().origin}/v1/mtplx/thermal/fan_mode`, {
49
76
  method: "POST",
50
- headers: { "content-type": "application/json" },
77
+ headers: { "content-type": "application/json", ...auth.headers },
51
78
  body: JSON.stringify({ mode: fanMode }),
52
79
  signal: controller.signal,
53
80
  });
81
+ if (response.status === 401 || response.status === 403) {
82
+ throw new Error(authenticationFailureMessage());
83
+ }
54
84
  const body = (await response.json()) as { verified?: unknown; current_mode?: unknown; error?: unknown };
55
85
  if (!response.ok || body.verified !== true || body.current_mode !== fanMode) {
56
86
  throw new Error(`MTPLX fan mode could not be set to ${fanMode}: ${typeof body.error === "string" ? body.error : "unverified response"}`);
@@ -60,4 +90,4 @@ export async function setFanMode(): Promise<void> {
60
90
  } finally {
61
91
  clearTimeout(timeout);
62
92
  }
63
- }
93
+ }
@@ -7,21 +7,18 @@
7
7
  * Pi's model id — a positive fingerprint that a given healthy server on
8
8
  * the port is one this extension owns.
9
9
  * - Cleanup is therefore precise: a server only gets shut down if the
10
- * extension spawned it in this session, or if `/health` positively
11
- * identifies it as an MTPLX server serving one of the extension's model
12
- * ids. A foreign service on the port is never touched, and a manually
13
- * started MTPLX server (no matching --model-id fingerprint) is never
14
- * killed — it is simply served from as-is.
15
- * - Stop goes through `mtplx stop --host --port --json` (MTPLX's own
16
- * graceful-stop mechanism: SIGTERM → grace → SIGKILL), which targets the
17
- * server answering on that exact host:port, not arbitrary processes.
10
+ * extension spawned it in this Pi session. A manually or separately
11
+ * started server is never killed, even when it serves a registered model.
12
+ * - A server launched by this Pi session is stopped through its own detached
13
+ * process group. This avoids relying on a second `mtplx` CLI invocation
14
+ * during shutdown.
18
15
  */
19
16
  import { spawn, execFile } from "node:child_process";
20
17
  import { promisify } from "node:util";
21
18
  import type { ChildProcess } from "node:child_process";
22
- import { getFanMode, health, setFanMode, type Health } from "./mtplx-client.ts";
19
+ import { authenticationFailureMessage, getFanMode, health, healthProbe, setFanMode } from "./mtplx-client.ts";
23
20
  import { MTPLX_MODELS } from "./model-discovery.ts";
24
- import { HOST, PORT, READY_TIMEOUT_MS, POLL_MS, loadSsdSessionCache, sleep, commandError, portIsOccupied } from "./utils.ts";
21
+ import { READY_TIMEOUT_MS, POLL_MS, loadMtplxEndpoint, loadResolvedMtplxApiKey, loadSsdSessionCache, sleep, commandError, portIsOccupied, type MtplxEndpoint } from "./utils.ts";
25
22
 
26
23
  const execFileAsync = promisify(execFile);
27
24
 
@@ -42,84 +39,132 @@ export function startupError(cause: Error): Error {
42
39
  }
43
40
 
44
41
  /**
45
- * Shut down the server on HOST:PORT — but only if we can positively identify
46
- * it as MTPLX (via /health). A healthy, non-MTPLX listener is an error, never
47
- * a kill target.
42
+ * Shut down the server at the configured endpoint, but only when this Pi
43
+ * session launched it. A healthy external listener is never a kill target.
48
44
  */
49
- export async function stopServer(): Promise<void> {
50
- const current = await health();
45
+ export async function stopServer(): Promise<boolean> {
46
+ const endpoint = loadMtplxEndpoint();
47
+ const probe = await healthProbe();
48
+ const current = probe.health;
51
49
  if (!current) {
52
- if (await portIsOccupied()) {
53
- throw new Error(`MTPLX cannot use ${HOST}:${PORT}: another, non-MTPLX service is listening there.`);
50
+ if (await portIsOccupied(endpoint)) {
51
+ if (probe.authenticationRejected) throw new Error(authenticationFailureMessage());
52
+ throw new Error(`MTPLX cannot use ${endpoint.host}:${endpoint.port}: another, non-MTPLX service is listening there.`);
54
53
  }
55
54
  // Nothing listening (or not MTPLX): nothing to stop. Drop any stale handle.
56
55
  ownedChild = undefined;
57
- return;
56
+ return true;
58
57
  }
59
58
 
60
- // /health says an MTPLX server is answering here.
61
- if (!isOwnedByThisSession() && !currentModelIsOurs(current.model)) {
62
- // An MTPLX server the extension does not own (e.g. started manually by
63
- // the user). MTPLX exposes no reliable way to distinguish it from a
64
- // Pi-managed one at the port level beyond --model-id, so do not kill
65
- // it. If a Pi-owned one is still needed, the user can run /mtplx →
66
- // Toggle.
59
+ if (!isOwnedByThisSession()) {
60
+ // A separately started server can deliberately use the same model id as a
61
+ // Pi-managed one, so model identity is not ownership.
67
62
  console.warn(
68
- `pi-mtplx: leaving MTPLX server on ${HOST}:${PORT} (model ${JSON.stringify(current.model)}) untouched — not owned by this Pi session. Use /mtplx → Toggle to stop it.`,
63
+ `pi-mtplx: leaving MTPLX server on ${endpoint.host}:${endpoint.port} (model ${JSON.stringify(current.model)}) untouched — not owned by this Pi session.`,
69
64
  );
70
- return;
65
+ return false;
71
66
  }
72
67
 
73
- try {
74
- await execFileAsync("mtplx", ["stop", "--host", HOST, "--port", String(PORT), "--json"], { timeout: 20_000 });
75
- } catch (error) {
76
- throw new Error(`MTPLX shutdown failed: ${commandError(error)}`);
68
+ let stopError: unknown;
69
+ const childPid = ownedChild?.pid;
70
+ let signalledOwnedProcess = false;
71
+ if (isOwnedByThisSession() && childPid) {
72
+ try {
73
+ // `detached: true` gives this child its own POSIX process group. Signal
74
+ // that group so a quickstart wrapper and its server are stopped together.
75
+ process.kill(-childPid, "SIGTERM");
76
+ signalledOwnedProcess = true;
77
+ } catch (error) {
78
+ // The child may have already exited while its server survived. In that
79
+ // case, fall through to MTPLX's host-and-port stop command below.
80
+ if (!(error && typeof error === "object" && "code" in error && error.code === "ESRCH")) {
81
+ stopError = error;
82
+ }
83
+ }
84
+ }
85
+
86
+ if (!signalledOwnedProcess) {
87
+ try {
88
+ if (!endpoint.isLoopback) throw new Error("MTPLX endpoint is remote; pi-mtplx will not stop a server it did not launch.");
89
+ await execFileAsync("mtplx", ["stop", "--host", endpoint.host, "--port", String(endpoint.port), "--json"], { timeout: 20_000 });
90
+ stopError = undefined;
91
+ } catch (error) {
92
+ // Some MTPLX CLI failures occur after it has already signalled the server.
93
+ // Confirm the listener state before reporting shutdown as failed.
94
+ stopError = error;
95
+ }
77
96
  }
78
- ownedChild = undefined;
79
97
 
80
98
  const deadline = Date.now() + 20_000;
81
99
  while (Date.now() < deadline) {
82
- if (!(await health())) return;
100
+ if (!(await portIsOccupied(endpoint))) {
101
+ ownedChild = undefined;
102
+ return true;
103
+ }
83
104
  await sleep(POLL_MS);
84
105
  }
85
- throw new Error(`MTPLX shutdown timed out; ${HOST}:${PORT} is still healthy.`);
106
+ if (stopError) throw new Error(`MTPLX shutdown failed: ${commandError(stopError)}`);
107
+ if (signalledOwnedProcess) {
108
+ throw new Error(`MTPLX shutdown timed out after signalling Pi's server process; ${endpoint.host}:${endpoint.port} is still occupied.`);
109
+ }
110
+ throw new Error(`MTPLX shutdown timed out; ${endpoint.host}:${endpoint.port} is still occupied.`);
111
+ }
112
+
113
+ function modelMismatchError(runningModel: string, requestedModel: string, endpoint: MtplxEndpoint): Error {
114
+ return new Error(
115
+ `MTPLX is already running at ${endpoint.baseUrl}, serving ${JSON.stringify(runningModel)}, but Pi has ${JSON.stringify(requestedModel)} selected. ` +
116
+ `Pi will not replace the running server. Select ${JSON.stringify(runningModel)} in Pi, or restart MTPLX with \`--model-id ${requestedModel}\`, then try again.`,
117
+ );
86
118
  }
87
119
 
88
- /** Positive ownership fingerprint: /health model id belongs to this extension's registry. */
89
- function currentModelIsOurs(model: string): boolean {
90
- return Object.keys(MTPLX_MODELS).includes(model);
120
+ function autoStartDisabledUnavailableError(endpoint: MtplxEndpoint): Error {
121
+ return new Error(
122
+ `MTPLX is unavailable at ${endpoint.baseUrl}. Auto Start is off, so Pi will not start a server. ` +
123
+ "Start MTPLX at this endpoint or enable Auto Start, then try again.",
124
+ );
91
125
  }
92
126
 
127
+ export type ReadyServer = {
128
+ endpoint: MtplxEndpoint;
129
+ model: string;
130
+ fanMode?: string;
131
+ alreadyRunning: boolean;
132
+ };
133
+
93
134
  /**
94
135
  * Spawn the MTPLX server for a registered model and wait until /health
95
136
  * confirms it serves exactly that model. The child is spawned detached so it
96
137
  * outlives Pi's event-loop teardown, and unref'd so it never blocks Pi exit.
97
138
  */
98
139
  export async function startServer(modelId: string): Promise<void> {
140
+ const endpoint = loadMtplxEndpoint();
99
141
  const configured = MTPLX_MODELS[modelId];
100
142
  if (!configured) {
101
143
  throw new Error(`MTPLX model ${JSON.stringify(modelId)} is not mapped to an installed MTPLX artifact. Update the pi-mtplx model registry after adding it to Pi.`);
102
144
  }
103
145
 
146
+ const args = [
147
+ "quickstart",
148
+ "--model",
149
+ configured.ref,
150
+ "--model-id",
151
+ modelId,
152
+ "--fan-mode",
153
+ getFanMode(),
154
+ "--host",
155
+ endpoint.host,
156
+ "--port",
157
+ String(endpoint.port),
158
+ // Explicitly pass the user's persisted choice; this extension defaults it to on.
159
+ "--ssd-session-cache",
160
+ loadSsdSessionCache() ? "on" : "off",
161
+ ];
162
+ args.push("--api-key", loadResolvedMtplxApiKey());
163
+
104
164
  let exited: Error | undefined;
105
165
  const child = spawn(
106
166
  "mtplx",
107
- [
108
- "quickstart",
109
- "--model",
110
- configured.ref,
111
- "--model-id",
112
- modelId,
113
- "--fan-mode",
114
- getFanMode(),
115
- "--host",
116
- HOST,
117
- "--port",
118
- String(PORT),
119
- // Explicitly pass the user's persisted choice; this extension defaults it to on.
120
- "--ssd-session-cache",
121
- loadSsdSessionCache() ? "on" : "off",
122
- ],
167
+ args,
123
168
  { detached: true, stdio: "ignore" },
124
169
  );
125
170
  ownedChild = child;
@@ -146,30 +191,58 @@ export async function startServer(modelId: string): Promise<void> {
146
191
  }
147
192
 
148
193
  /**
149
- * Ensure the server on HOST:PORT serves `modelId`, transparently switching
150
- * models: stop the current (identified) server, then start the requested one.
194
+ * Confirm that the configured endpoint is healthy and serves `modelId`.
195
+ * Unlike `ensureServer`, this never starts, stops, or switches a server.
151
196
  */
152
- export async function ensureServer(modelId: string): Promise<void> {
153
- const current = await health();
197
+ export async function validateServer(modelId: string, endpoint = loadMtplxEndpoint()): Promise<ReadyServer> {
198
+ const probe = await healthProbe(endpoint);
199
+ const current = probe.health;
154
200
  if (current?.model === modelId) {
155
- if (current.fan_mode !== getFanMode()) await setFanMode();
156
- return;
201
+ return { endpoint, model: current.model, fanMode: current.fan_mode, alreadyRunning: true };
202
+ }
203
+ if (current) {
204
+ throw modelMismatchError(current.model, modelId, endpoint);
205
+ }
206
+ if (probe.authenticationRejected) throw new Error(authenticationFailureMessage());
207
+ throw autoStartDisabledUnavailableError(endpoint);
208
+ }
209
+
210
+ /**
211
+ * Ensure the configured endpoint serves `modelId`. A healthy separately
212
+ * managed server is authoritative: Pi reuses a matching one and never
213
+ * replaces a different model. A server started by this Pi session can be
214
+ * stopped and switched to the selected model. Pi starts a server only when
215
+ * the endpoint is unavailable or after switching its own server.
216
+ */
217
+ export async function ensureServer(modelId: string, endpoint = loadMtplxEndpoint()): Promise<ReadyServer> {
218
+ const probe = await healthProbe(endpoint);
219
+ const current = probe.health;
220
+ if (current) {
221
+ if (current.model === modelId) {
222
+ return { endpoint, model: current.model, fanMode: current.fan_mode, alreadyRunning: true };
223
+ }
224
+ if (!isOwnedByThisSession()) throw modelMismatchError(current.model, modelId, endpoint);
225
+ await stopServer();
226
+ }
227
+ if (await portIsOccupied(endpoint)) {
228
+ if (probe.authenticationRejected) throw new Error(authenticationFailureMessage());
229
+ throw new Error(`MTPLX cannot start because ${endpoint.host}:${endpoint.port} is occupied by a non-MTPLX service.`);
157
230
  }
158
- if (current) await stopServer();
159
- else if (await portIsOccupied()) {
160
- throw new Error(`MTPLX cannot start because ${HOST}:${PORT} is occupied by a non-MTPLX service.`);
231
+ if (!endpoint.isLoopback) {
232
+ throw new Error(`MTPLX endpoint ${endpoint.baseUrl} is unavailable. pi-mtplx only auto-starts loopback servers; start this endpoint separately.`);
161
233
  }
162
234
  await startServer(modelId);
163
235
  await setFanMode();
236
+ return { endpoint, model: modelId, fanMode: getFanMode(), alreadyRunning: false };
164
237
  }
165
238
 
166
- let transition: Promise<void> | undefined;
239
+ let transition: Promise<ReadyServer> | undefined;
167
240
  let transitionModel: string | undefined;
168
241
  let activeModel: string | undefined;
169
242
  let activeAgents = 0;
170
243
  let idleWaiters: Array<() => void> = [];
171
244
 
172
- export function ensureOnce(modelId: string): Promise<void> {
245
+ export function ensureOnce(modelId: string): Promise<ReadyServer> {
173
246
  if (transition) {
174
247
  if (transitionModel === modelId) return transition;
175
248
  return transition.then(() => ensureOnce(modelId));
@@ -182,7 +255,7 @@ export function ensureOnce(modelId: string): Promise<void> {
182
255
  return transition;
183
256
  }
184
257
 
185
- export async function acquire(modelId: string): Promise<void> {
258
+ export async function acquire(modelId: string): Promise<ReadyServer> {
186
259
  // Never replace a model while an already-admitted MTPLX agent is running.
187
260
  // Same-model requests share both this lease and any in-flight start Promise.
188
261
  while (activeAgents > 0 && activeModel !== modelId) {
@@ -194,7 +267,7 @@ export async function acquire(modelId: string): Promise<void> {
194
267
  activeModel = modelId;
195
268
  activeAgents += 1;
196
269
  try {
197
- await ensureOnce(modelId);
270
+ return await ensureOnce(modelId);
198
271
  } catch (error) {
199
272
  release();
200
273
  throw error;
package/src/utils.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Shared constants and small helpers for the pi-mtplx extension.
3
3
  */
4
- import { readFileSync, writeFileSync, mkdirSync } from "node:fs";
4
+ import { existsSync, readFileSync, writeFileSync, mkdirSync } from "node:fs";
5
5
  import { createConnection } from "node:net";
6
6
  import { homedir } from "node:os";
7
7
  import { join } from "node:path";
@@ -10,10 +10,55 @@ export const HOST = "127.0.0.1";
10
10
  export const PORT = 8000;
11
11
  export const READY_TIMEOUT_MS = 180_000;
12
12
  export const POLL_MS = 500;
13
+ export const PI_MODELS_FILE = join(homedir(), ".pi", "agent", "models.json");
14
+ export const PI_AUTH_FILE = join(homedir(), ".pi", "agent", "auth.json");
13
15
 
14
16
  export type FanMode = "default" | "smart" | "max";
15
17
  export const FAN_MODES: readonly FanMode[] = ["default", "smart", "max"];
16
18
  export const DEFAULT_SSD_SESSION_CACHE = true;
19
+ export const DEFAULT_AUTO_SHUTDOWN = true;
20
+ export const DEFAULT_AUTO_START = true;
21
+ export const DEFAULT_MTPLX_API_KEY = "mtplx-local";
22
+
23
+ export type MtplxEndpoint = {
24
+ baseUrl: string;
25
+ origin: string;
26
+ host: string;
27
+ port: number;
28
+ isLoopback: boolean;
29
+ };
30
+
31
+ /** Resolve the endpoint from Pi's provider configuration, with a local default. */
32
+ export function mtplxEndpointFromBaseUrl(baseUrl: unknown): MtplxEndpoint {
33
+ const fallback = `http://${HOST}:${PORT}/v1`;
34
+ let url: URL;
35
+ try {
36
+ url = new URL(typeof baseUrl === "string" && baseUrl.trim() ? baseUrl.trim() : fallback);
37
+ if (url.protocol !== "http:" && url.protocol !== "https:") throw new Error("unsupported protocol");
38
+ } catch {
39
+ url = new URL(fallback);
40
+ }
41
+ // URL.hostname retains brackets around IPv6 literals ("[::1]"). The
42
+ // MTPLX CLI and node:net both expect the bare address, and ::1 is local.
43
+ const host = url.hostname.replace(/^\[|\]$/g, "");
44
+ const port = Number(url.port || (url.protocol === "https:" ? 443 : 80));
45
+ return {
46
+ baseUrl: url.toString().replace(/\/$/, ""),
47
+ origin: url.origin,
48
+ host,
49
+ port,
50
+ isLoopback: host === "127.0.0.1" || host === "::1" || host.toLowerCase() === "localhost",
51
+ };
52
+ }
53
+
54
+ export function loadMtplxEndpoint(): MtplxEndpoint {
55
+ try {
56
+ const catalog = JSON.parse(readFileSync(PI_MODELS_FILE, "utf8")) as { providers?: { mtplx?: { baseUrl?: unknown } } };
57
+ return mtplxEndpointFromBaseUrl(catalog.providers?.mtplx?.baseUrl);
58
+ } catch {
59
+ return mtplxEndpointFromBaseUrl(undefined);
60
+ }
61
+ }
17
62
 
18
63
  // Fan mode ("fan curve") applied at autostart, live-updated from `/mtplx` while the
19
64
  // server runs. Persisted to disk so the choice survives Pi restarts: `fanMode` is a
@@ -66,17 +111,188 @@ export function saveSsdSessionCache(enabled: boolean): void {
66
111
  }
67
112
  }
68
113
 
69
- // Derive Pi's model id from an artifact ref: "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality"
70
- // → "mtplx-qwen38-27b-optimized-quality" (same scheme as the built-in entry).
71
- export function modelIdFromRef(ref: string): string {
72
- const slug = ref
73
- .split("/")
74
- .pop() ?? ref
75
- .replace(/^mtplx/i, "")
76
- .replace(/[^a-zA-Z0-9]+/g, "-")
77
- .replace(/-+/g, "-")
78
- .replace(/^-|-$/g, "");
79
- return `mtplx-${slug || "model"}`.toLowerCase();
114
+ // Whether Pi should stop the server during a normal process shutdown. Kept
115
+ // separate from the server's own settings so users can deliberately leave a
116
+ // loaded model available after Pi exits.
117
+ export const AUTO_SHUTDOWN_FILE = join(homedir(), ".pi", "agent", "mtplx-auto-shutdown.json");
118
+ export const AUTO_START_FILE = join(homedir(), ".pi", "agent", "mtplx-auto-start.json");
119
+
120
+ export function loadAutoStart(): boolean {
121
+ try {
122
+ const parsed = JSON.parse(readFileSync(AUTO_START_FILE, "utf8")) as { enabled?: unknown };
123
+ if (typeof parsed.enabled === "boolean") return parsed.enabled;
124
+ } catch {
125
+ // missing or corrupt file → fall back to the default
126
+ }
127
+ return DEFAULT_AUTO_START;
128
+ }
129
+
130
+ export function saveAutoStart(enabled: boolean): void {
131
+ try {
132
+ mkdirSync(join(homedir(), ".pi", "agent"), { recursive: true });
133
+ writeFileSync(AUTO_START_FILE, JSON.stringify({ enabled }, null, 2) + "\n");
134
+ } catch (error) {
135
+ console.error(`MTPLX could not persist auto-start preference: ${error instanceof Error ? error.message : String(error)}`);
136
+ }
137
+ }
138
+
139
+ export function loadAutoShutdown(): boolean {
140
+ try {
141
+ const parsed = JSON.parse(readFileSync(AUTO_SHUTDOWN_FILE, "utf8")) as { enabled?: unknown };
142
+ if (typeof parsed.enabled === "boolean") return parsed.enabled;
143
+ } catch {
144
+ // missing or corrupt file → default to cleaning up the Pi-managed server
145
+ }
146
+ return DEFAULT_AUTO_SHUTDOWN;
147
+ }
148
+
149
+ export function saveAutoShutdown(enabled: boolean): void {
150
+ try {
151
+ mkdirSync(join(homedir(), ".pi", "agent"), { recursive: true });
152
+ writeFileSync(AUTO_SHUTDOWN_FILE, JSON.stringify({ enabled }, null, 2) + "\n");
153
+ } catch (error) {
154
+ console.error(`MTPLX could not persist auto-shutdown preference: ${error instanceof Error ? error.message : String(error)}`);
155
+ }
156
+ }
157
+
158
+ /** Read the provider-level key Pi uses for requests to the local MTPLX server. */
159
+ export function mtplxApiKeyFromCatalog(catalog: unknown): string | undefined {
160
+ if (typeof catalog !== "object" || catalog === null) return undefined;
161
+ const providers = (catalog as { providers?: unknown }).providers;
162
+ if (typeof providers !== "object" || providers === null) return undefined;
163
+ const provider = (providers as { mtplx?: unknown }).mtplx;
164
+ if (typeof provider !== "object" || provider === null) return undefined;
165
+ const apiKey = (provider as { apiKey?: unknown }).apiKey;
166
+ return typeof apiKey === "string" && apiKey.trim() ? apiKey.trim() : undefined;
167
+ }
168
+
169
+ /**
170
+ * Lifecycle calls reread this on every request. Pi's inference provider has
171
+ * its own in-memory catalog, so a key changed here still requires a Pi restart
172
+ * before model requests use it.
173
+ */
174
+ export function loadMtplxApiKey(): string | undefined {
175
+ try {
176
+ return mtplxApiKeyFromCatalog(JSON.parse(readFileSync(PI_MODELS_FILE, "utf8")) as unknown);
177
+ } catch {
178
+ return undefined;
179
+ }
180
+ }
181
+
182
+ /** Every Pi-managed MTPLX connection has a deterministic key. */
183
+ export function resolveMtplxApiKey(apiKey: string | undefined): string {
184
+ return apiKey ?? DEFAULT_MTPLX_API_KEY;
185
+ }
186
+
187
+ export function loadResolvedMtplxApiKey(): string {
188
+ return resolveMtplxApiKey(loadMtplxApiKey());
189
+ }
190
+
191
+ /** A stable, non-secret identifier suitable for the interactive menu. */
192
+ export function maskMtplxApiKey(apiKey: string | undefined): string {
193
+ if (!apiKey) return `default (${DEFAULT_MTPLX_API_KEY})`;
194
+ return apiKey.length <= 4 ? "configured" : `••••${apiKey.slice(-4)}`;
195
+ }
196
+
197
+ /**
198
+ * Pi gives a stored `auth.json` API-key credential precedence over models.json.
199
+ * Keep an existing MTPLX credential in sync; do not create one, because the
200
+ * provider-level key is sufficient when no stored credential exists.
201
+ */
202
+ function syncStoredMtplxCredential(apiKey: string): void {
203
+ if (!existsSync(PI_AUTH_FILE)) return;
204
+ const parsed = JSON.parse(readFileSync(PI_AUTH_FILE, "utf8")) as Record<string, unknown>;
205
+ // Current Pi stores credentials directly as { "mtplx": { ... } }. Accept
206
+ // the older nested shape too, without changing either file's structure.
207
+ const credentials = typeof parsed.auth === "object" && parsed.auth !== null && !Array.isArray(parsed.auth)
208
+ ? parsed.auth as Record<string, unknown>
209
+ : parsed;
210
+ const credential = credentials.mtplx;
211
+ if (credential === undefined) return;
212
+ if (typeof credential !== "object" || credential === null || Array.isArray(credential) || (credential as { type?: unknown }).type !== "api_key") {
213
+ throw new Error("auth.json contains an MTPLX credential that is not an API key");
214
+ }
215
+ (credential as Record<string, unknown>).key = apiKey;
216
+ writeFileSync(PI_AUTH_FILE, JSON.stringify(parsed, null, 2) + "\n");
217
+ }
218
+
219
+ /** Reconcile Pi's stored MTPLX credential with the configured provider key. */
220
+ export function syncMtplxStoredCredential(): void {
221
+ try {
222
+ syncStoredMtplxCredential(loadResolvedMtplxApiKey());
223
+ } catch (error) {
224
+ console.error(`MTPLX could not synchronize Pi's stored API key: ${error instanceof Error ? error.message : String(error)}`);
225
+ }
226
+ }
227
+
228
+ /** Update Pi's MTPLX provider and any stored MTPLX credential to the same key. */
229
+ export function saveMtplxApiKey(apiKey: string): boolean {
230
+ const normalized = apiKey.trim();
231
+ const resolved = resolveMtplxApiKey(normalized || undefined);
232
+ try {
233
+ let catalog: Record<string, unknown> = {};
234
+ if (existsSync(PI_MODELS_FILE)) {
235
+ const parsed = JSON.parse(readFileSync(PI_MODELS_FILE, "utf8")) as unknown;
236
+ if (typeof parsed !== "object" || parsed === null || Array.isArray(parsed)) throw new Error("models.json is not an object");
237
+ catalog = parsed as Record<string, unknown>;
238
+ }
239
+ let providers = catalog.providers;
240
+ if (providers !== undefined && (typeof providers !== "object" || providers === null || Array.isArray(providers))) {
241
+ throw new Error("models.json providers is not an object");
242
+ }
243
+ const providerCatalog = (providers ??= {}) as Record<string, unknown>;
244
+ const existing = providerCatalog.mtplx;
245
+ if (existing !== undefined && (typeof existing !== "object" || existing === null || Array.isArray(existing))) {
246
+ throw new Error("models.json mtplx provider is not an object");
247
+ }
248
+ const provider = (existing ?? {
249
+ api: "openai-completions",
250
+ authHeader: true,
251
+ baseUrl: `http://${HOST}:${PORT}/v1`,
252
+ }) as Record<string, unknown>;
253
+ if (normalized) provider.apiKey = normalized;
254
+ else delete provider.apiKey;
255
+ providerCatalog.mtplx = provider;
256
+ catalog.providers = providerCatalog;
257
+ mkdirSync(join(homedir(), ".pi", "agent"), { recursive: true });
258
+ writeFileSync(PI_MODELS_FILE, JSON.stringify(catalog, null, 2) + "\n");
259
+ syncStoredMtplxCredential(resolved);
260
+ return true;
261
+ } catch (error) {
262
+ console.error(`MTPLX could not update API key: ${error instanceof Error ? error.message : String(error)}`);
263
+ return false;
264
+ }
265
+ }
266
+
267
+ /** Save the OpenAI-compatible base URL used by Pi and lifecycle health checks. */
268
+ export function saveMtplxEndpoint(baseUrl: string): boolean {
269
+ const normalized = baseUrl.trim();
270
+ if (!normalized) return false;
271
+ let endpoint: MtplxEndpoint;
272
+ try {
273
+ const url = new URL(normalized);
274
+ if (url.protocol !== "http:" && url.protocol !== "https:") return false;
275
+ endpoint = mtplxEndpointFromBaseUrl(normalized);
276
+ } catch {
277
+ return false;
278
+ }
279
+ try {
280
+ const catalog = existsSync(PI_MODELS_FILE)
281
+ ? JSON.parse(readFileSync(PI_MODELS_FILE, "utf8")) as Record<string, unknown>
282
+ : {};
283
+ if (typeof catalog !== "object" || catalog === null || Array.isArray(catalog)) throw new Error("models.json is not an object");
284
+ const providers = (catalog.providers ??= {}) as Record<string, unknown>;
285
+ const provider = (providers.mtplx ?? { api: "openai-completions", authHeader: true }) as Record<string, unknown>;
286
+ if (typeof provider !== "object" || provider === null || Array.isArray(provider)) throw new Error("models.json mtplx provider is not an object");
287
+ provider.baseUrl = endpoint.baseUrl;
288
+ providers.mtplx = provider;
289
+ mkdirSync(join(homedir(), ".pi", "agent"), { recursive: true });
290
+ writeFileSync(PI_MODELS_FILE, JSON.stringify(catalog, null, 2) + "\n");
291
+ return true;
292
+ } catch (error) {
293
+ console.error(`MTPLX could not update endpoint: ${error instanceof Error ? error.message : String(error)}`);
294
+ return false;
295
+ }
80
296
  }
81
297
 
82
298
  export function displayNameFromId(id: string): string {
@@ -86,14 +302,6 @@ export function displayNameFromId(id: string): string {
86
302
  .replace(/\b\w/g, (c) => c.toUpperCase());
87
303
  }
88
304
 
89
- export function slugFromId(id: string): string {
90
- return id
91
- .replace(/^mtplx-/, "")
92
- .replace(/[^a-zA-Z0-9]+/g, "-")
93
- .replace(/-+/g, "-")
94
- .replace(/^-|-$/g, "");
95
- }
96
-
97
305
  export function isMtplxModel(model: { provider: string; id: string } | undefined): boolean {
98
306
  // The provider name is MTPLX's own PI_PROVIDER_ID (`mtplx`), matching the
99
307
  // provider block that `mtplx start pi` writes to models.json.
@@ -111,9 +319,9 @@ export function commandError(error: unknown): string {
111
319
  return stderr || details.message || "unknown command failure";
112
320
  }
113
321
 
114
- export function portIsOccupied(): Promise<boolean> {
322
+ export function portIsOccupied(endpoint = loadMtplxEndpoint()): Promise<boolean> {
115
323
  return new Promise((resolve) => {
116
- const socket = createConnection({ host: HOST, port: PORT });
324
+ const socket = createConnection({ host: endpoint.host, port: endpoint.port });
117
325
  const done = (occupied: boolean) => {
118
326
  socket.destroy();
119
327
  resolve(occupied);