pi-llamacpp-infra 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -8
- package/package.json +1 -1
- package/src/index.ts +106 -27
- package/src/prompt-warmup.ts +15 -2
package/README.md
CHANGED
|
@@ -21,6 +21,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
|
|
|
21
21
|
## Features
|
|
22
22
|
|
|
23
23
|
- **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
|
|
24
|
+
- **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
|
|
24
25
|
- **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
|
|
25
26
|
- **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
|
|
26
27
|
- **Live Prometheus metrics** — polls `/metrics` (or `/stats`) and renders a compact widget with instantaneous prompt/gen throughput; auto-activates for llamacpp-infra models only
|
|
@@ -41,7 +42,7 @@ llamacpp-infra is a [pi package](https://pi.dev/packages): one extension (`src/i
|
|
|
41
42
|
pi install git:github.com/noguerol/llamacpp-infra
|
|
42
43
|
|
|
43
44
|
# Pin a tag/commit
|
|
44
|
-
pi install git:github.com/noguerol/llamacpp-infra@v1.
|
|
45
|
+
pi install git:github.com/noguerol/llamacpp-infra@v1.2.0
|
|
45
46
|
|
|
46
47
|
# From npm
|
|
47
48
|
pi install npm:pi-llamacpp-infra
|
|
@@ -119,12 +120,14 @@ Shows every discovered model with metadata badges:
|
|
|
119
120
|
```
|
|
120
121
|
📋 Discovered models (8)
|
|
121
122
|
|
|
122
|
-
1.
|
|
123
|
-
2.
|
|
124
|
-
3.
|
|
125
|
-
4.
|
|
123
|
+
1. Qwen3.6-27B-UD-Q3_K_XL (local:8080) 👁️ 🗜️ UD-Q3_K_XL
|
|
124
|
+
2. DeepSeek-V4-Flash-ROCMFP2 (local:8081) 🗜️ ROCMFP2
|
|
125
|
+
3. Meta-Llama-3.1-8B (myserver:8080) 🚀 draft-model 🗜️ Q4_K_M
|
|
126
|
+
4. gemma-3-4b-it (myserver:8081) 👁️ 🗜️ Q4_K_M
|
|
126
127
|
```
|
|
127
128
|
|
|
129
|
+
The same compact id is what pi's `/model` picker shows, with the serving machine in parentheses.
|
|
130
|
+
|
|
128
131
|
### `/llamacpp-infra status`
|
|
129
132
|
|
|
130
133
|
Detailed per-endpoint report:
|
|
@@ -181,7 +184,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
181
184
|
"metricsPollMs": 5000
|
|
182
185
|
},
|
|
183
186
|
"modelOptions": {
|
|
184
|
-
"
|
|
187
|
+
"Qwen3.6-27B (myserver:8080)": {
|
|
185
188
|
"thinkingBudgets": {
|
|
186
189
|
"minimal": 256,
|
|
187
190
|
"low": 1024,
|
|
@@ -215,7 +218,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
215
218
|
| `startupGraceMs` | `40000` | Keep trying at startup while nothing has answered |
|
|
216
219
|
| `knownGoodFailLimit` | `3` | Consecutive failures before a live endpoint is dropped |
|
|
217
220
|
| `detectVision` | `true` | Scan `/proc` for `--mmproj` + read server-reported modalities |
|
|
218
|
-
| `prefixModelIds` | `true` | `host:port
|
|
221
|
+
| `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
|
|
219
222
|
| `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
|
|
220
223
|
| `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
|
|
221
224
|
| `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
|
|
@@ -230,7 +233,15 @@ Levels: `minimal`, `low`, `medium`, `high`, `xhigh`, `max`.
|
|
|
230
233
|
|
|
231
234
|
## Model ID Format
|
|
232
235
|
|
|
233
|
-
|
|
236
|
+
Models are registered with compact display ids: `ModelName (host:port)`, e.g. `Qwen3.6-27B-UD-Q3_K_XL (myserver:8080)` — the compact model name plus the machine serving it in parentheses, matching how pi shows native provider models. Localhost servers (`127.0.0.1`, `localhost`) use `local:port` in the tag.
|
|
237
|
+
|
|
238
|
+
pi sends the compact id to the extension's request hook, which transparently rewrites it to the raw server-side id (the GGUF path, alias or router id the server advertised in `/v1/models`) before the request leaves pi. Config keys under `modelOptions` use the compact id; legacy `host:port/model` keys are migrated automatically on the first scan.
|
|
239
|
+
|
|
240
|
+
With `prefixModelIds: false` the machine tag is omitted (`ModelName`); it is re-added automatically only when two models would otherwise collide.
|
|
241
|
+
|
|
242
|
+
### Thinking budgets in the UI
|
|
243
|
+
|
|
244
|
+
llama.cpp-family models are registered as reasoning models, exactly like a native pi provider: the footer shows `ModelName (host:port) • <level>`, the thinking selector offers levels with token estimates, and pi sends the configured `thinking_budget_tokens` budget on each request. Per-model budgets configured in the extension override pi's global per-level budgets.
|
|
234
245
|
|
|
235
246
|
## Live Metrics Widget
|
|
236
247
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.2.0",
|
|
4
4
|
"description": "Discovery, metrics and control of llama.cpp-family servers for pi: probes any number of machines (localhost, LAN, Tailscale), registers every model into pi's native /model list, and provides live Prometheus metrics, per-model thinking budgets, vision detection and a native config UI. Supports llama.cpp, ZINC, DwarfStar/ds4, lucebox and LM Studio.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
package/src/index.ts
CHANGED
|
@@ -51,6 +51,9 @@
|
|
|
51
51
|
* 9. LM Studio support: discovers `/v1/models`, enriches metadata from
|
|
52
52
|
* `/api/v1/models` or legacy `/api/v0/models`, and avoids llama.cpp-only
|
|
53
53
|
* request fields such as `cache_prompt` / `thinking_budget_tokens`.
|
|
54
|
+
* 10. Compact model ids: models are registered as "Name (host:port)" — the
|
|
55
|
+
* raw server-side model id (GGUF path / alias) is restored automatically
|
|
56
|
+
* in before_provider_request before the request leaves pi.
|
|
54
57
|
*
|
|
55
58
|
* HISTORY
|
|
56
59
|
* -------
|
|
@@ -1054,6 +1057,8 @@ interface PiModel {
|
|
|
1054
1057
|
compat?: ReturnType<typeof makeCompat>;
|
|
1055
1058
|
headers?: Record<string, string>;
|
|
1056
1059
|
thinkingBudgets?: ThinkingBudgets;
|
|
1060
|
+
/** Raw model id expected by the server (compact ids are display-only). */
|
|
1061
|
+
serverModelId: string;
|
|
1057
1062
|
/** Metadata badges are applied to the name at build time. */
|
|
1058
1063
|
endpoint: { serverId: string; host: string; port: number; kind: ServerKind | "unknown" | "auto"; mode: ServerMode };
|
|
1059
1064
|
quant?: string;
|
|
@@ -1086,8 +1091,9 @@ function toPiModel(
|
|
|
1086
1091
|
): PiModel {
|
|
1087
1092
|
const rawId = String(model.id ?? "");
|
|
1088
1093
|
const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
|
|
1089
|
-
|
|
1090
|
-
|
|
1094
|
+
// Compact display id: "ModelName (host:port)". The raw server-side id is
|
|
1095
|
+
// kept separately (serverModelId) and restored in before_provider_request.
|
|
1096
|
+
const machineTag = settings.prefixModelIds ? ` (${hostPort})` : "";
|
|
1091
1097
|
const kind = ep.server;
|
|
1092
1098
|
|
|
1093
1099
|
const common = {
|
|
@@ -1108,10 +1114,10 @@ function toPiModel(
|
|
|
1108
1114
|
const info = model.lmStudio;
|
|
1109
1115
|
const contextWindow = lmStudioContextLength(info) ?? model.context_window ?? model.context_length ?? 32768;
|
|
1110
1116
|
const displayName = info?.display_name ? cleanModelName(info.display_name) : cleanModelName(rawId);
|
|
1111
|
-
const modelId = usePrefix ? prefixId(rawId) : rawId;
|
|
1112
1117
|
return {
|
|
1113
1118
|
...common,
|
|
1114
|
-
id:
|
|
1119
|
+
id: `${displayName}${machineTag}`,
|
|
1120
|
+
serverModelId: rawId,
|
|
1115
1121
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
1116
1122
|
// LM Studio exposes reasoning-capable models, but its OpenAI-compatible
|
|
1117
1123
|
// endpoint does not use llama.cpp's thinking_budget_tokens field.
|
|
@@ -1128,16 +1134,16 @@ function toPiModel(
|
|
|
1128
1134
|
const alias = ep.props?.model_alias || rawId;
|
|
1129
1135
|
const nCtx = ep.props?.default_generation_settings?.n_ctx;
|
|
1130
1136
|
const contextWindow = model.context_length ?? nCtx ?? 8192;
|
|
1131
|
-
const
|
|
1137
|
+
const displayName = cleanModelName(baseName(alias !== rawId ? alias : modelPath));
|
|
1132
1138
|
return {
|
|
1133
1139
|
...common,
|
|
1134
|
-
id:
|
|
1135
|
-
|
|
1140
|
+
id: `${displayName}${machineTag}`,
|
|
1141
|
+
serverModelId: alias,
|
|
1142
|
+
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
1136
1143
|
reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
|
|
1137
1144
|
contextWindow,
|
|
1138
1145
|
maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
|
|
1139
1146
|
compat: makeCompat(kind),
|
|
1140
|
-
...(budgets ? { thinkingBudgets: budgets } : {}),
|
|
1141
1147
|
};
|
|
1142
1148
|
}
|
|
1143
1149
|
|
|
@@ -1153,21 +1159,19 @@ function toPiModel(
|
|
|
1153
1159
|
const contextWindow =
|
|
1154
1160
|
model.meta?.n_ctx ?? model.meta?.n_ctx_train ?? model.context_window ?? model.context_length ?? 32768;
|
|
1155
1161
|
|
|
1156
|
-
// Models with configured thinking budgets must be registered as reasoning
|
|
1157
|
-
// models so pi's thinking-level machinery (and budget field) engages.
|
|
1158
|
-
const modelId = usePrefix ? prefixId(rawId) : rawId;
|
|
1159
|
-
const budgets = config_ModelOptions()?.[modelId]?.thinkingBudgets;
|
|
1160
1162
|
const isLlamaFamily = kind === "llamacpp";
|
|
1161
1163
|
|
|
1162
1164
|
return {
|
|
1163
1165
|
...common,
|
|
1164
|
-
id:
|
|
1166
|
+
id: `${displayName}${machineTag}`,
|
|
1167
|
+
serverModelId: rawId,
|
|
1165
1168
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
1166
|
-
reasoning:
|
|
1169
|
+
// llama.cpp-family models behave like native pi reasoning models: the
|
|
1170
|
+
// footer shows the thinking level and pi sends the configured budget.
|
|
1171
|
+
reasoning: isLlamaFamily,
|
|
1167
1172
|
contextWindow,
|
|
1168
1173
|
maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
|
|
1169
1174
|
compat: makeCompat(kind),
|
|
1170
|
-
...(budgets ? { thinkingBudgets: budgets } : {}),
|
|
1171
1175
|
};
|
|
1172
1176
|
}
|
|
1173
1177
|
|
|
@@ -1317,6 +1321,22 @@ const zincModelIds = new Set<string>();
|
|
|
1317
1321
|
const modelBaseUrls = new Map<string, string>();
|
|
1318
1322
|
/** baseUrl → server kind (for the warmup request profile). */
|
|
1319
1323
|
const endpointKinds = new Map<string, string>();
|
|
1324
|
+
/** Registered (compact) model id → raw server-side model id (payload rewrite). */
|
|
1325
|
+
const serverModelIds = new Map<string, string>();
|
|
1326
|
+
/** Raw server-side model id → registered (compact) model id. */
|
|
1327
|
+
const compactModelIds = new Map<string, string>();
|
|
1328
|
+
|
|
1329
|
+
/** Compact id registered in pi for a raw server model id (or the input). */
|
|
1330
|
+
function compactIdFor(modelId: string | undefined): string | undefined {
|
|
1331
|
+
if (!modelId) return undefined;
|
|
1332
|
+
return compactModelIds.get(modelId) ?? modelId;
|
|
1333
|
+
}
|
|
1334
|
+
|
|
1335
|
+
/** Raw server-side id to send in requests for a registered model id (or the input). */
|
|
1336
|
+
function rawIdFor(modelId: string | undefined): string | undefined {
|
|
1337
|
+
if (!modelId) return undefined;
|
|
1338
|
+
return serverModelIds.get(modelId) ?? modelId;
|
|
1339
|
+
}
|
|
1320
1340
|
|
|
1321
1341
|
/**
|
|
1322
1342
|
* Build pi models from a scan and (re)register the provider.
|
|
@@ -1324,7 +1344,10 @@ const endpointKinds = new Map<string, string>();
|
|
|
1324
1344
|
*/
|
|
1325
1345
|
function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: InfraConfig): PiModel[] {
|
|
1326
1346
|
zincModelIds.clear();
|
|
1347
|
+
serverModelIds.clear();
|
|
1348
|
+
compactModelIds.clear();
|
|
1327
1349
|
const settings = config.settings;
|
|
1350
|
+
let configDirty = false;
|
|
1328
1351
|
|
|
1329
1352
|
const piModels: PiModel[] = [];
|
|
1330
1353
|
const seenIds = new Set<string>();
|
|
@@ -1354,25 +1377,56 @@ function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: In
|
|
|
1354
1377
|
const rawId = String(model.id ?? "");
|
|
1355
1378
|
const modelMeta = ep.meta.get(rawId) ?? {};
|
|
1356
1379
|
const pm = toPiModel(model, ep, srv, settings, modelMeta);
|
|
1380
|
+
const hostPort = `${idSafeHost(srv.host)}:${ep.port}`;
|
|
1357
1381
|
|
|
1358
|
-
// ID collision guard:
|
|
1359
|
-
// then a numeric suffix if even that collides.
|
|
1382
|
+
// ID collision guard: add the machine tag, then a numeric suffix.
|
|
1360
1383
|
if (seenIds.has(pm.id)) {
|
|
1361
|
-
pm.id = `${
|
|
1384
|
+
if (!pm.id.includes(`(${hostPort})`)) pm.id = `${pm.id} (${hostPort})`;
|
|
1362
1385
|
let n = 2;
|
|
1363
1386
|
while (seenIds.has(pm.id)) pm.id = `${pm.id}-${n++}`;
|
|
1364
1387
|
}
|
|
1365
1388
|
seenIds.add(pm.id);
|
|
1366
1389
|
|
|
1390
|
+
// Thinking budgets: resolve by the registered (compact) id or legacy
|
|
1391
|
+
// "host:port/model" keys, then migrate legacy keys forward.
|
|
1392
|
+
const opts = config_ModelOptions();
|
|
1393
|
+
let budgets = opts[pm.id]?.thinkingBudgets;
|
|
1394
|
+
if (!budgets) {
|
|
1395
|
+
for (const legacyKey of [`${hostPort}/${pm.serverModelId.replace(/^\/+/, "")}`, pm.serverModelId]) {
|
|
1396
|
+
const entry = opts[legacyKey];
|
|
1397
|
+
if (entry?.thinkingBudgets) {
|
|
1398
|
+
if (!opts[pm.id]) opts[pm.id] = entry;
|
|
1399
|
+
delete opts[legacyKey]; // migrate: compact id replaces host:port/model
|
|
1400
|
+
configDirty = true;
|
|
1401
|
+
budgets = entry.thinkingBudgets;
|
|
1402
|
+
break;
|
|
1403
|
+
}
|
|
1404
|
+
}
|
|
1405
|
+
}
|
|
1406
|
+
if (budgets && ep.server !== "lmstudio") {
|
|
1407
|
+
pm.thinkingBudgets = budgets;
|
|
1408
|
+
// Models with configured thinking budgets must be registered as
|
|
1409
|
+
// reasoning models so pi's thinking-level machinery engages.
|
|
1410
|
+
pm.reasoning = true;
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
// Duplicate display names within one endpoint get the raw id appended.
|
|
1367
1414
|
const candidateName = cleanModelName(String(model.name ?? model.id));
|
|
1368
1415
|
if (nameCount.get(candidateName)! > 1) {
|
|
1369
|
-
pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${
|
|
1416
|
+
pm.name = `${candidateName}${badgeSuffix(modelMeta, settings.showBadgesInNames)} (${pm.serverModelId})`;
|
|
1370
1417
|
}
|
|
1371
1418
|
piModels.push(pm);
|
|
1372
|
-
if (ep.server === "zinc")
|
|
1419
|
+
if (ep.server === "zinc") {
|
|
1420
|
+
zincModelIds.add(pm.id);
|
|
1421
|
+
zincModelIds.add(pm.serverModelId);
|
|
1422
|
+
}
|
|
1423
|
+
serverModelIds.set(pm.id, pm.serverModelId);
|
|
1424
|
+
if (!compactModelIds.has(pm.serverModelId)) compactModelIds.set(pm.serverModelId, pm.id);
|
|
1373
1425
|
}
|
|
1374
1426
|
}
|
|
1375
1427
|
|
|
1428
|
+
if (configDirty) saveConfig(config);
|
|
1429
|
+
|
|
1376
1430
|
// Maps for hooks
|
|
1377
1431
|
endpointKinds.clear();
|
|
1378
1432
|
for (const ep of scan.endpoints) {
|
|
@@ -1442,6 +1496,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1442
1496
|
provider: PROVIDER_NAME,
|
|
1443
1497
|
cacheFile: join(homedir(), ".pi", "agent", "warmup-llamacpp-infra.json"),
|
|
1444
1498
|
kindFor: (baseUrl) => endpointKinds.get(baseUrl),
|
|
1499
|
+
requestModelFor: (modelId) => rawIdFor(modelId) ?? modelId,
|
|
1445
1500
|
onEvent: (ev) => warmupStatus.handle(ev),
|
|
1446
1501
|
});
|
|
1447
1502
|
|
|
@@ -1748,6 +1803,21 @@ export default function (pi: ExtensionAPI) {
|
|
|
1748
1803
|
registerEmptyProvider();
|
|
1749
1804
|
void discoverAndRegister().then((r) => schedulePolling(r.shouldPoll));
|
|
1750
1805
|
|
|
1806
|
+
// ── Hook 0: compact registered id → raw server model id ─────────
|
|
1807
|
+
// Model ids registered in pi are compact display ids ("Name (host:port)");
|
|
1808
|
+
// llama.cpp-family servers expect the raw model path/alias they advertised
|
|
1809
|
+
// in /v1/models, so the payload model field is rewritten here. Registered
|
|
1810
|
+
// first so every later hook (and the server) sees the raw id.
|
|
1811
|
+
pi.on("before_provider_request", (event, _ctx) => {
|
|
1812
|
+
const payload = event.payload as Record<string, unknown>;
|
|
1813
|
+
const modelInPayload = typeof payload.model === "string" ? payload.model : undefined;
|
|
1814
|
+
if (!modelInPayload) return undefined;
|
|
1815
|
+
const raw = serverModelIds.get(modelInPayload);
|
|
1816
|
+
if (raw === undefined || raw === modelInPayload) return undefined;
|
|
1817
|
+
debugLog(`model id "${modelInPayload}" → "${raw}"`);
|
|
1818
|
+
return { ...payload, model: raw };
|
|
1819
|
+
});
|
|
1820
|
+
|
|
1751
1821
|
// ── Hook 1: ZINC payload workaround ────────────────────────────
|
|
1752
1822
|
// ZINC rejects non-empty model ids and is picky about tool formats.
|
|
1753
1823
|
function isZincModel(modelInPayload: unknown): boolean {
|
|
@@ -1799,20 +1869,23 @@ export default function (pi: ExtensionAPI) {
|
|
|
1799
1869
|
// ── Hook 2: per-model thinking budget (llama.cpp thinking_budget_tokens) ──
|
|
1800
1870
|
// pi injects thinking_token budgets from its global settings; this hook
|
|
1801
1871
|
// overrides the value with the per-model budgets configured for this model.
|
|
1802
|
-
// Registered after the ZINC hook so it sees the final payload.
|
|
1872
|
+
// Registered after the ZINC hook so it sees the final payload. Hook 0 has
|
|
1873
|
+
// already rewritten the payload model to the raw server id, so both raw
|
|
1874
|
+
// and compact ids are resolved against the registered model maps.
|
|
1803
1875
|
pi.on("before_provider_request", (event, ctx) => {
|
|
1804
1876
|
const payload = event.payload as Record<string, unknown>;
|
|
1805
1877
|
const modelId = typeof payload.model === "string" ? payload.model : undefined;
|
|
1806
1878
|
if (!modelId) return undefined;
|
|
1807
|
-
const
|
|
1879
|
+
const compactKey = compactModelIds.get(modelId) ?? modelId;
|
|
1880
|
+
const baseUrl = modelBaseUrls.get(compactKey);
|
|
1808
1881
|
if (!supportsThinkingBudget(endpointKinds.get(baseUrl ?? "") as ServerKind | undefined)) return undefined;
|
|
1809
|
-
const budgets = config.modelOptions[
|
|
1882
|
+
const budgets = config.modelOptions[compactKey]?.thinkingBudgets;
|
|
1810
1883
|
if (!budgets) return undefined;
|
|
1811
1884
|
const level = normalizeLevel(ctx.thinkingLevel ?? currentThinkingLevel);
|
|
1812
1885
|
if (!level) return undefined;
|
|
1813
1886
|
const value = budgets[level];
|
|
1814
1887
|
if (typeof value !== "number") return undefined;
|
|
1815
|
-
debugLog(`thinking budget for ${
|
|
1888
|
+
debugLog(`thinking budget for ${compactKey} [${level}] = ${value}`);
|
|
1816
1889
|
return { ...payload, [THINKING_BUDGET_FIELD]: value };
|
|
1817
1890
|
});
|
|
1818
1891
|
|
|
@@ -1832,12 +1905,16 @@ export default function (pi: ExtensionAPI) {
|
|
|
1832
1905
|
}
|
|
1833
1906
|
|
|
1834
1907
|
// ── Hook 3: header warmup capture (runs last, sees final payload) ──
|
|
1908
|
+
// Templates are keyed by the compact registered id (the same id passed to
|
|
1909
|
+
// warmupForModel); PromptWarmer re-resolves the raw server id per request.
|
|
1835
1910
|
pi.on("before_provider_request", (event, ctx) => {
|
|
1836
1911
|
if (!config.settings.warmup) return undefined;
|
|
1837
1912
|
const payload = event.payload as Record<string, unknown>;
|
|
1838
1913
|
const modelId = typeof payload?.model === "string" ? payload.model : undefined;
|
|
1839
|
-
const
|
|
1840
|
-
|
|
1914
|
+
const compactKey = modelId ? (compactModelIds.get(modelId) ?? modelId) : undefined;
|
|
1915
|
+
const baseUrl = compactKey ? modelBaseUrls.get(compactKey) : undefined;
|
|
1916
|
+
const capturePayload = compactKey ? { ...payload, model: compactKey } : payload;
|
|
1917
|
+
warmer.onProviderPayload(capturePayload, baseUrl, ctx.cwd);
|
|
1841
1918
|
return undefined;
|
|
1842
1919
|
});
|
|
1843
1920
|
|
|
@@ -1949,7 +2026,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
1949
2026
|
if (!ep) return undefined;
|
|
1950
2027
|
// Find the raw entry whose registered id ends with the raw id fragment.
|
|
1951
2028
|
for (const [rawId, meta] of ep.meta) {
|
|
1952
|
-
if (m.
|
|
2029
|
+
if (m.serverModelId === rawId || m.id === rawId) return meta;
|
|
1953
2030
|
}
|
|
1954
2031
|
return undefined;
|
|
1955
2032
|
};
|
|
@@ -2041,6 +2118,8 @@ export default function (pi: ExtensionAPI) {
|
|
|
2041
2118
|
"ZINC, DwarfStar (ds4-server), lucebox and LM Studio on any number",
|
|
2042
2119
|
"of machines, and registers them into pi's native /model list.",
|
|
2043
2120
|
"",
|
|
2121
|
+
"Models appear as compact ids: \"Name (host:port)\" — the raw GGUF",
|
|
2122
|
+
"path/alias is sent to the server automatically on every request.",
|
|
2044
2123
|
"LM Studio uses its OpenAI-compatible /v1 endpoint and enriches",
|
|
2045
2124
|
"metadata from /api/v1/models (or legacy /api/v0/models).",
|
|
2046
2125
|
"Per-model metadata: vision, drafter, model quant, KV cache quant.",
|
package/src/prompt-warmup.ts
CHANGED
|
@@ -248,6 +248,8 @@ export interface PromptWarmerOptions {
|
|
|
248
248
|
cacheFile?: string;
|
|
249
249
|
/** Devuelve el tipo de servidor ("llamacpp"|"lucebox"|"ds4"|"zinc"|"lmstudio") para una baseUrl. */
|
|
250
250
|
kindFor?: (baseUrl: string) => string | undefined;
|
|
251
|
+
/** Mapea el id registrado (compacto) al id crudo que espera el servidor. */
|
|
252
|
+
requestModelFor?: (modelId: string) => string;
|
|
251
253
|
/** Callback opcional para eventos de warmup (estado en footer, notify, ...). */
|
|
252
254
|
onEvent?: (ev: WarmupEvent) => void;
|
|
253
255
|
/** Log opcional. Por defecto silencioso (no-op); activa con PI_WARMUP_DEBUG=1. */
|
|
@@ -351,6 +353,7 @@ export class PromptWarmer {
|
|
|
351
353
|
readonly provider: string;
|
|
352
354
|
private cacheFile: string;
|
|
353
355
|
private kindFor: (baseUrl: string) => string | undefined;
|
|
356
|
+
private requestModelFor: (modelId: string) => string;
|
|
354
357
|
private onEvent?: (ev: WarmupEvent) => void;
|
|
355
358
|
private log: (msg: string) => void;
|
|
356
359
|
|
|
@@ -368,6 +371,7 @@ export class PromptWarmer {
|
|
|
368
371
|
this.cacheFile =
|
|
369
372
|
opts.cacheFile ?? path.join(homedir(), ".pi", "agent", `warmup-${opts.provider}.json`);
|
|
370
373
|
this.kindFor = opts.kindFor ?? (() => KIND_LLAMACPP);
|
|
374
|
+
this.requestModelFor = opts.requestModelFor ?? ((id: string) => id);
|
|
371
375
|
this.onEvent = opts.onEvent;
|
|
372
376
|
// Por defecto silencioso: escribir por stderr corrompe el render de la TUI
|
|
373
377
|
// de pi (trazas sucias en el editor / input box). Solo se loguea con
|
|
@@ -392,6 +396,15 @@ export class PromptWarmer {
|
|
|
392
396
|
}
|
|
393
397
|
}
|
|
394
398
|
|
|
399
|
+
/** Id crudo que espera el servidor para un id registrado (con fallback). */
|
|
400
|
+
private resolveRequestModel(modelId: string): string {
|
|
401
|
+
try {
|
|
402
|
+
return this.requestModelFor(modelId) || modelId;
|
|
403
|
+
} catch {
|
|
404
|
+
return modelId;
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
395
408
|
// ── Warmup ────────────────────────────────────────────────────────────────
|
|
396
409
|
|
|
397
410
|
/**
|
|
@@ -422,7 +435,7 @@ export class PromptWarmer {
|
|
|
422
435
|
let body: WarmupBody;
|
|
423
436
|
if (tpl) {
|
|
424
437
|
body = {
|
|
425
|
-
model: tpl.model,
|
|
438
|
+
model: this.resolveRequestModel(tpl.model),
|
|
426
439
|
messages: [...tpl.systemMessages, { role: "user", content: PLACEHOLDER_USER }],
|
|
427
440
|
max_tokens: 1,
|
|
428
441
|
temperature: 0,
|
|
@@ -432,7 +445,7 @@ export class PromptWarmer {
|
|
|
432
445
|
if (cachePromptSupported(tpl.kind)) body.cache_prompt = true;
|
|
433
446
|
} else if (systemPrompt && FALLBACK_ENABLED) {
|
|
434
447
|
body = {
|
|
435
|
-
model: model.id,
|
|
448
|
+
model: this.resolveRequestModel(model.id),
|
|
436
449
|
messages: [
|
|
437
450
|
{ role: "system", content: systemPrompt },
|
|
438
451
|
{ role: "user", content: PLACEHOLDER_USER },
|