pi-llamacpp-infra 1.2.4 → 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/core.ts +125 -0
- package/src/index.ts +40 -1
- package/src/registration.ts +13 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.5",
|
|
4
4
|
"description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
package/src/core.ts
CHANGED
|
@@ -17,6 +17,7 @@ import type {
|
|
|
17
17
|
export const PROVIDER_NAME = "llamacpp-infra";
|
|
18
18
|
export const STATUS_KEY = "llamacpp-infra";
|
|
19
19
|
export const CONFIG_FILE = "llamacpp-infra.json";
|
|
20
|
+
export const MODELS_CACHE_FILE = "llamacpp-infra-models.json";
|
|
20
21
|
export const LEGACY_CONFIG_FILE = "local-models.json";
|
|
21
22
|
export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
|
|
22
23
|
export const DEFAULT_API_KEY = "no-auth";
|
|
@@ -113,6 +114,130 @@ export function saveConfig(config: InfraConfig): void {
|
|
|
113
114
|
}
|
|
114
115
|
}
|
|
115
116
|
|
|
117
|
+
// ── Last-known-models cache ────────────────────────────────────────────────
|
|
118
|
+
// Lets a freshly booted pi process register the provider with the models from
|
|
119
|
+
// the previous scan IMMEDIATELY (synchronously, during the extension factory),
|
|
120
|
+
// so CLI consumers that resolve --model at startup (e.g. trimegisto sub-agents
|
|
121
|
+
// spawned as `pi -p --no-session --model provider/model`) find the model before
|
|
122
|
+
// the async discovery re-scan completes. The async scan then refreshes this
|
|
123
|
+
// registration with live data as usual.
|
|
124
|
+
|
|
125
|
+
const MODELS_CACHE_VERSION = 1;
|
|
126
|
+
|
|
127
|
+
/** JSON-safe projection of a registered PiModel (no secrets; headers re-derived). */
|
|
128
|
+
interface CachedModelEntry {
|
|
129
|
+
id: string;
|
|
130
|
+
name: string;
|
|
131
|
+
baseUrl: string;
|
|
132
|
+
reasoning: boolean;
|
|
133
|
+
input: ("text" | "image")[];
|
|
134
|
+
contextWindow: number;
|
|
135
|
+
maxTokens: number;
|
|
136
|
+
serverModelId: string;
|
|
137
|
+
endpoint: import("./types.ts").PiModel["endpoint"];
|
|
138
|
+
thinkingBudgets?: import("./types.ts").ThinkingBudgets;
|
|
139
|
+
quant?: string;
|
|
140
|
+
cacheK?: string;
|
|
141
|
+
cacheV?: string;
|
|
142
|
+
drafter?: string;
|
|
143
|
+
routerStatus?: string;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
interface ModelsCacheFile {
|
|
147
|
+
version: number;
|
|
148
|
+
savedAt: string;
|
|
149
|
+
models: CachedModelEntry[];
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
export function getModelsCachePath(): string {
|
|
153
|
+
return join(getAgentDir(), MODELS_CACHE_FILE);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/** Persist the last successfully registered models for fast boot on next process. */
|
|
157
|
+
export function saveModelsCache(models: import("./types.ts").PiModel[]): void {
|
|
158
|
+
if (!models || models.length === 0) return;
|
|
159
|
+
try {
|
|
160
|
+
const file: ModelsCacheFile = {
|
|
161
|
+
version: MODELS_CACHE_VERSION,
|
|
162
|
+
savedAt: new Date().toISOString(),
|
|
163
|
+
models: models.map((m) => ({
|
|
164
|
+
id: m.id,
|
|
165
|
+
name: m.name,
|
|
166
|
+
baseUrl: m.baseUrl,
|
|
167
|
+
reasoning: m.reasoning,
|
|
168
|
+
input: m.input,
|
|
169
|
+
contextWindow: m.contextWindow,
|
|
170
|
+
maxTokens: m.maxTokens,
|
|
171
|
+
serverModelId: m.serverModelId,
|
|
172
|
+
endpoint: m.endpoint,
|
|
173
|
+
thinkingBudgets: m.thinkingBudgets,
|
|
174
|
+
quant: m.quant,
|
|
175
|
+
cacheK: m.cacheK,
|
|
176
|
+
cacheV: m.cacheV,
|
|
177
|
+
drafter: m.drafter,
|
|
178
|
+
routerStatus: m.routerStatus,
|
|
179
|
+
})),
|
|
180
|
+
};
|
|
181
|
+
const dir = getAgentDir();
|
|
182
|
+
if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
|
|
183
|
+
writeFileSync(getModelsCachePath(), JSON.stringify(file), "utf-8");
|
|
184
|
+
debugLog(`models cache saved (${models.length} model(s))`);
|
|
185
|
+
} catch (err) {
|
|
186
|
+
console.error(`[llamacpp-infra] Models cache save error: ${err instanceof Error ? err.message : String(err)}`);
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/** Load models cached by a previous scan, or null when absent/corrupt/empty. */
|
|
191
|
+
export function loadModelsCache(): import("./types.ts").PiModel[] | null {
|
|
192
|
+
try {
|
|
193
|
+
const path = getModelsCachePath();
|
|
194
|
+
if (!existsSync(path)) return null;
|
|
195
|
+
const raw = JSON.parse(readFileSync(path, "utf-8")) as Partial<ModelsCacheFile>;
|
|
196
|
+
if (raw?.version !== MODELS_CACHE_VERSION || !Array.isArray(raw.models) || raw.models.length === 0) return null;
|
|
197
|
+
return raw.models.map((c) => ({
|
|
198
|
+
id: c.id,
|
|
199
|
+
name: c.name,
|
|
200
|
+
baseUrl: c.baseUrl,
|
|
201
|
+
reasoning: c.reasoning,
|
|
202
|
+
input: Array.isArray(c.input) && c.input.length > 0 ? c.input : (["text"] as ("text" | "image")[]),
|
|
203
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
204
|
+
contextWindow: c.contextWindow,
|
|
205
|
+
maxTokens: c.maxTokens,
|
|
206
|
+
compat: makeCompat(c.endpoint?.kind),
|
|
207
|
+
serverModelId: c.serverModelId,
|
|
208
|
+
endpoint: c.endpoint,
|
|
209
|
+
thinkingBudgets: c.thinkingBudgets,
|
|
210
|
+
quant: c.quant,
|
|
211
|
+
cacheK: c.cacheK,
|
|
212
|
+
cacheV: c.cacheV,
|
|
213
|
+
drafter: c.drafter,
|
|
214
|
+
routerStatus: c.routerStatus,
|
|
215
|
+
}));
|
|
216
|
+
} catch (err) {
|
|
217
|
+
debugLog(`models cache load failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
218
|
+
return null;
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
/** Rebuild the shared maps (id↔raw, baseUrls, kinds, zinc) from cached models. */
|
|
223
|
+
export function applyCachedSharedState(models: import("./types.ts").PiModel[]): void {
|
|
224
|
+
shared.zincModelIds.clear();
|
|
225
|
+
shared.serverModelIds.clear();
|
|
226
|
+
shared.compactModelIds.clear();
|
|
227
|
+
shared.endpointKinds.clear();
|
|
228
|
+
shared.modelBaseUrls.clear();
|
|
229
|
+
for (const pm of models) {
|
|
230
|
+
if (pm.endpoint?.kind === "zinc") {
|
|
231
|
+
shared.zincModelIds.add(pm.id);
|
|
232
|
+
shared.zincModelIds.add(pm.serverModelId);
|
|
233
|
+
}
|
|
234
|
+
shared.serverModelIds.set(pm.id, pm.serverModelId);
|
|
235
|
+
if (!shared.compactModelIds.has(pm.serverModelId)) shared.compactModelIds.set(pm.serverModelId, pm.id);
|
|
236
|
+
shared.endpointKinds.set(pm.baseUrl, pm.endpoint?.kind ?? "unknown");
|
|
237
|
+
shared.modelBaseUrls.set(pm.id, pm.baseUrl);
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
116
241
|
// ── Compat profile ─────────────────────────────────────────────────────────
|
|
117
242
|
export function supportsThinkingBudget(kind: ServerKind | "unknown" | "auto" | undefined): boolean {
|
|
118
243
|
return kind === "llamacpp" || kind === "lucebox";
|
package/src/index.ts
CHANGED
|
@@ -9,9 +9,11 @@ import {
|
|
|
9
9
|
PROVIDER_NAME,
|
|
10
10
|
STATUS_KEY,
|
|
11
11
|
THINKING_BUDGET_FIELD,
|
|
12
|
+
applyCachedSharedState,
|
|
12
13
|
compactIdFor,
|
|
13
14
|
debugLog,
|
|
14
15
|
loadConfig,
|
|
16
|
+
loadModelsCache,
|
|
15
17
|
modelOptions,
|
|
16
18
|
normalizeLevel,
|
|
17
19
|
rawIdFor,
|
|
@@ -127,6 +129,43 @@ export default function (pi: ExtensionAPI) {
|
|
|
127
129
|
providerIsEmpty = true;
|
|
128
130
|
}
|
|
129
131
|
|
|
132
|
+
// ── Boot provider registration (synchronous, cache-first) ────────────
|
|
133
|
+
// Registers models from the last known scan IMMEDIATELY when a cache exists,
|
|
134
|
+
// so this process can resolve `--model llamacpp-infra/...` at startup without
|
|
135
|
+
// waiting for the async discovery re-scan (which refreshes right after).
|
|
136
|
+
// Falls back to the empty "scanning…" provider when there is no cache.
|
|
137
|
+
function registerBootProvider() {
|
|
138
|
+
const cached = loadModelsCache();
|
|
139
|
+
if (cached && cached.length > 0) {
|
|
140
|
+
try {
|
|
141
|
+
pi.unregisterProvider(PROVIDER_NAME);
|
|
142
|
+
} catch {
|
|
143
|
+
// not registered yet
|
|
144
|
+
}
|
|
145
|
+
// Re-derive per-server auth headers (never persisted with the cache).
|
|
146
|
+
for (const m of cached) {
|
|
147
|
+
const srv = config.servers.find((s) => s.id === m.endpoint?.serverId && s.host === m.endpoint?.host);
|
|
148
|
+
if (srv?.apiKey) m.headers = { Authorization: `Bearer ${srv.apiKey}` };
|
|
149
|
+
}
|
|
150
|
+
applyCachedSharedState(cached);
|
|
151
|
+
const bootSrv = config.servers.find((s) => s.id === cached[0].endpoint?.serverId);
|
|
152
|
+
pi.registerProvider(PROVIDER_NAME, {
|
|
153
|
+
name: `🦙 llama.cpp-infra (cached ${cached.length}, rescanning…)`,
|
|
154
|
+
baseUrl: cached[0].baseUrl,
|
|
155
|
+
apiKey: bootSrv?.apiKey || "no-auth",
|
|
156
|
+
api: "openai-completions",
|
|
157
|
+
streamSimple: createLongTimeoutOpenAICompletionsStream,
|
|
158
|
+
models: cached,
|
|
159
|
+
});
|
|
160
|
+
providerIsEmpty = false;
|
|
161
|
+
shared.registeredCount = cached.length;
|
|
162
|
+
shared.lastModels = cached;
|
|
163
|
+
debugLog(`boot: registered ${cached.length} cached model(s); async rescan will refresh`);
|
|
164
|
+
return;
|
|
165
|
+
}
|
|
166
|
+
registerEmptyProvider();
|
|
167
|
+
}
|
|
168
|
+
|
|
130
169
|
async function discoverAndRegister(): Promise<{ scan: import("./types.ts").ScanResult; shouldPoll: boolean }> {
|
|
131
170
|
try {
|
|
132
171
|
if (!extensionActive) {
|
|
@@ -474,7 +513,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
474
513
|
});
|
|
475
514
|
|
|
476
515
|
// ── Initial non-blocking registration ─────────────────────────────────
|
|
477
|
-
|
|
516
|
+
registerBootProvider();
|
|
478
517
|
void discoverAndRegister()
|
|
479
518
|
.then((r) => {
|
|
480
519
|
if (extensionActive) schedulePolling(r.shouldPoll);
|
package/src/registration.ts
CHANGED
|
@@ -9,6 +9,7 @@ import {
|
|
|
9
9
|
makeCompat,
|
|
10
10
|
modelOptions,
|
|
11
11
|
saveConfig,
|
|
12
|
+
saveModelsCache,
|
|
12
13
|
shared,
|
|
13
14
|
} from "./core.ts";
|
|
14
15
|
import type { ExtensionAPI } from "./types.ts";
|
|
@@ -128,7 +129,12 @@ function toPiModel(
|
|
|
128
129
|
* Build pi models from a scan and (re)register the provider.
|
|
129
130
|
* Returns the registered model list.
|
|
130
131
|
*/
|
|
131
|
-
export function buildAndRegisterProvider(
|
|
132
|
+
export function buildAndRegisterProvider(
|
|
133
|
+
pi: ExtensionAPI,
|
|
134
|
+
scan: ScanResult,
|
|
135
|
+
config: InfraConfig,
|
|
136
|
+
options?: { persistCache?: boolean },
|
|
137
|
+
): PiModel[] {
|
|
132
138
|
shared.zincModelIds.clear();
|
|
133
139
|
shared.serverModelIds.clear();
|
|
134
140
|
shared.compactModelIds.clear();
|
|
@@ -237,5 +243,11 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
|
|
|
237
243
|
models: piModels,
|
|
238
244
|
});
|
|
239
245
|
|
|
246
|
+
// Persist the last known good models so a fresh pi process can register them
|
|
247
|
+
// synchronously at boot (before async discovery completes) — required for CLI
|
|
248
|
+
// consumers that resolve --model right after startup (e.g. trimegisto spawns).
|
|
249
|
+
// Tests opt out so they never touch the real per-user cache.
|
|
250
|
+
if (piModels.length > 0 && options?.persistCache !== false) saveModelsCache(piModels);
|
|
251
|
+
|
|
240
252
|
return piModels;
|
|
241
253
|
}
|