pi-llamacpp-infra 1.2.3 → 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llamacpp-infra",
3
- "version": "1.2.3",
3
+ "version": "1.2.5",
4
4
  "description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
5
5
  "keywords": [
6
6
  "pi-package",
@@ -35,6 +35,7 @@
35
35
  "src"
36
36
  ],
37
37
  "peerDependencies": {
38
+ "@earendil-works/pi-ai": "*",
38
39
  "@earendil-works/pi-coding-agent": "*"
39
40
  }
40
41
  }
package/src/core.ts CHANGED
@@ -17,6 +17,7 @@ import type {
17
17
  export const PROVIDER_NAME = "llamacpp-infra";
18
18
  export const STATUS_KEY = "llamacpp-infra";
19
19
  export const CONFIG_FILE = "llamacpp-infra.json";
20
+ export const MODELS_CACHE_FILE = "llamacpp-infra-models.json";
20
21
  export const LEGACY_CONFIG_FILE = "local-models.json";
21
22
  export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
22
23
  export const DEFAULT_API_KEY = "no-auth";
@@ -113,6 +114,130 @@ export function saveConfig(config: InfraConfig): void {
113
114
  }
114
115
  }
115
116
 
117
+ // ── Last-known-models cache ────────────────────────────────────────────────
118
+ // Lets a freshly booted pi process register the provider with the models from
119
+ // the previous scan IMMEDIATELY (synchronously, during the extension factory),
120
+ // so CLI consumers that resolve --model at startup (e.g. trimegisto sub-agents
121
+ // spawned as `pi -p --no-session --model provider/model`) find the model before
122
+ // the async discovery re-scan completes. The async scan then refreshes this
123
+ // registration with live data as usual.
124
+
125
+ const MODELS_CACHE_VERSION = 1;
126
+
127
+ /** JSON-safe projection of a registered PiModel (no secrets; headers re-derived). */
128
+ interface CachedModelEntry {
129
+ id: string;
130
+ name: string;
131
+ baseUrl: string;
132
+ reasoning: boolean;
133
+ input: ("text" | "image")[];
134
+ contextWindow: number;
135
+ maxTokens: number;
136
+ serverModelId: string;
137
+ endpoint: import("./types.ts").PiModel["endpoint"];
138
+ thinkingBudgets?: import("./types.ts").ThinkingBudgets;
139
+ quant?: string;
140
+ cacheK?: string;
141
+ cacheV?: string;
142
+ drafter?: string;
143
+ routerStatus?: string;
144
+ }
145
+
146
+ interface ModelsCacheFile {
147
+ version: number;
148
+ savedAt: string;
149
+ models: CachedModelEntry[];
150
+ }
151
+
152
+ export function getModelsCachePath(): string {
153
+ return join(getAgentDir(), MODELS_CACHE_FILE);
154
+ }
155
+
156
+ /** Persist the last successfully registered models for fast boot on next process. */
157
+ export function saveModelsCache(models: import("./types.ts").PiModel[]): void {
158
+ if (!models || models.length === 0) return;
159
+ try {
160
+ const file: ModelsCacheFile = {
161
+ version: MODELS_CACHE_VERSION,
162
+ savedAt: new Date().toISOString(),
163
+ models: models.map((m) => ({
164
+ id: m.id,
165
+ name: m.name,
166
+ baseUrl: m.baseUrl,
167
+ reasoning: m.reasoning,
168
+ input: m.input,
169
+ contextWindow: m.contextWindow,
170
+ maxTokens: m.maxTokens,
171
+ serverModelId: m.serverModelId,
172
+ endpoint: m.endpoint,
173
+ thinkingBudgets: m.thinkingBudgets,
174
+ quant: m.quant,
175
+ cacheK: m.cacheK,
176
+ cacheV: m.cacheV,
177
+ drafter: m.drafter,
178
+ routerStatus: m.routerStatus,
179
+ })),
180
+ };
181
+ const dir = getAgentDir();
182
+ if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
183
+ writeFileSync(getModelsCachePath(), JSON.stringify(file), "utf-8");
184
+ debugLog(`models cache saved (${models.length} model(s))`);
185
+ } catch (err) {
186
+ console.error(`[llamacpp-infra] Models cache save error: ${err instanceof Error ? err.message : String(err)}`);
187
+ }
188
+ }
189
+
190
+ /** Load models cached by a previous scan, or null when absent/corrupt/empty. */
191
+ export function loadModelsCache(): import("./types.ts").PiModel[] | null {
192
+ try {
193
+ const path = getModelsCachePath();
194
+ if (!existsSync(path)) return null;
195
+ const raw = JSON.parse(readFileSync(path, "utf-8")) as Partial<ModelsCacheFile>;
196
+ if (raw?.version !== MODELS_CACHE_VERSION || !Array.isArray(raw.models) || raw.models.length === 0) return null;
197
+ return raw.models.map((c) => ({
198
+ id: c.id,
199
+ name: c.name,
200
+ baseUrl: c.baseUrl,
201
+ reasoning: c.reasoning,
202
+ input: Array.isArray(c.input) && c.input.length > 0 ? c.input : (["text"] as ("text" | "image")[]),
203
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
204
+ contextWindow: c.contextWindow,
205
+ maxTokens: c.maxTokens,
206
+ compat: makeCompat(c.endpoint?.kind),
207
+ serverModelId: c.serverModelId,
208
+ endpoint: c.endpoint,
209
+ thinkingBudgets: c.thinkingBudgets,
210
+ quant: c.quant,
211
+ cacheK: c.cacheK,
212
+ cacheV: c.cacheV,
213
+ drafter: c.drafter,
214
+ routerStatus: c.routerStatus,
215
+ }));
216
+ } catch (err) {
217
+ debugLog(`models cache load failed: ${err instanceof Error ? err.message : String(err)}`);
218
+ return null;
219
+ }
220
+ }
221
+
222
+ /** Rebuild the shared maps (id↔raw, baseUrls, kinds, zinc) from cached models. */
223
+ export function applyCachedSharedState(models: import("./types.ts").PiModel[]): void {
224
+ shared.zincModelIds.clear();
225
+ shared.serverModelIds.clear();
226
+ shared.compactModelIds.clear();
227
+ shared.endpointKinds.clear();
228
+ shared.modelBaseUrls.clear();
229
+ for (const pm of models) {
230
+ if (pm.endpoint?.kind === "zinc") {
231
+ shared.zincModelIds.add(pm.id);
232
+ shared.zincModelIds.add(pm.serverModelId);
233
+ }
234
+ shared.serverModelIds.set(pm.id, pm.serverModelId);
235
+ if (!shared.compactModelIds.has(pm.serverModelId)) shared.compactModelIds.set(pm.serverModelId, pm.id);
236
+ shared.endpointKinds.set(pm.baseUrl, pm.endpoint?.kind ?? "unknown");
237
+ shared.modelBaseUrls.set(pm.id, pm.baseUrl);
238
+ }
239
+ }
240
+
116
241
  // ── Compat profile ─────────────────────────────────────────────────────────
117
242
  export function supportsThinkingBudget(kind: ServerKind | "unknown" | "auto" | undefined): boolean {
118
243
  return kind === "llamacpp" || kind === "lucebox";
package/src/index.ts CHANGED
@@ -9,9 +9,11 @@ import {
9
9
  PROVIDER_NAME,
10
10
  STATUS_KEY,
11
11
  THINKING_BUDGET_FIELD,
12
+ applyCachedSharedState,
12
13
  compactIdFor,
13
14
  debugLog,
14
15
  loadConfig,
16
+ loadModelsCache,
15
17
  modelOptions,
16
18
  normalizeLevel,
17
19
  rawIdFor,
@@ -127,6 +129,43 @@ export default function (pi: ExtensionAPI) {
127
129
  providerIsEmpty = true;
128
130
  }
129
131
 
132
+ // ── Boot provider registration (synchronous, cache-first) ────────────
133
+ // Registers models from the last known scan IMMEDIATELY when a cache exists,
134
+ // so this process can resolve `--model llamacpp-infra/...` at startup without
135
+ // waiting for the async discovery re-scan (which refreshes right after).
136
+ // Falls back to the empty "scanning…" provider when there is no cache.
137
+ function registerBootProvider() {
138
+ const cached = loadModelsCache();
139
+ if (cached && cached.length > 0) {
140
+ try {
141
+ pi.unregisterProvider(PROVIDER_NAME);
142
+ } catch {
143
+ // not registered yet
144
+ }
145
+ // Re-derive per-server auth headers (never persisted with the cache).
146
+ for (const m of cached) {
147
+ const srv = config.servers.find((s) => s.id === m.endpoint?.serverId && s.host === m.endpoint?.host);
148
+ if (srv?.apiKey) m.headers = { Authorization: `Bearer ${srv.apiKey}` };
149
+ }
150
+ applyCachedSharedState(cached);
151
+ const bootSrv = config.servers.find((s) => s.id === cached[0].endpoint?.serverId);
152
+ pi.registerProvider(PROVIDER_NAME, {
153
+ name: `🦙 llama.cpp-infra (cached ${cached.length}, rescanning…)`,
154
+ baseUrl: cached[0].baseUrl,
155
+ apiKey: bootSrv?.apiKey || "no-auth",
156
+ api: "openai-completions",
157
+ streamSimple: createLongTimeoutOpenAICompletionsStream,
158
+ models: cached,
159
+ });
160
+ providerIsEmpty = false;
161
+ shared.registeredCount = cached.length;
162
+ shared.lastModels = cached;
163
+ debugLog(`boot: registered ${cached.length} cached model(s); async rescan will refresh`);
164
+ return;
165
+ }
166
+ registerEmptyProvider();
167
+ }
168
+
130
169
  async function discoverAndRegister(): Promise<{ scan: import("./types.ts").ScanResult; shouldPoll: boolean }> {
131
170
  try {
132
171
  if (!extensionActive) {
@@ -474,7 +513,7 @@ export default function (pi: ExtensionAPI) {
474
513
  });
475
514
 
476
515
  // ── Initial non-blocking registration ─────────────────────────────────
477
- registerEmptyProvider();
516
+ registerBootProvider();
478
517
  void discoverAndRegister()
479
518
  .then((r) => {
480
519
  if (extensionActive) schedulePolling(r.shouldPoll);
@@ -9,6 +9,7 @@ import {
9
9
  makeCompat,
10
10
  modelOptions,
11
11
  saveConfig,
12
+ saveModelsCache,
12
13
  shared,
13
14
  } from "./core.ts";
14
15
  import type { ExtensionAPI } from "./types.ts";
@@ -128,7 +129,12 @@ function toPiModel(
128
129
  * Build pi models from a scan and (re)register the provider.
129
130
  * Returns the registered model list.
130
131
  */
131
- export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, config: InfraConfig): PiModel[] {
132
+ export function buildAndRegisterProvider(
133
+ pi: ExtensionAPI,
134
+ scan: ScanResult,
135
+ config: InfraConfig,
136
+ options?: { persistCache?: boolean },
137
+ ): PiModel[] {
132
138
  shared.zincModelIds.clear();
133
139
  shared.serverModelIds.clear();
134
140
  shared.compactModelIds.clear();
@@ -237,5 +243,11 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
237
243
  models: piModels,
238
244
  });
239
245
 
246
+ // Persist the last known good models so a fresh pi process can register them
247
+ // synchronously at boot (before async discovery completes) — required for CLI
248
+ // consumers that resolve --model right after startup (e.g. trimegisto spawns).
249
+ // Tests opt out so they never touch the real per-user cache.
250
+ if (piModels.length > 0 && options?.persistCache !== false) saveModelsCache(piModels);
251
+
240
252
  return piModels;
241
253
  }
package/src/runtime.ts CHANGED
@@ -67,17 +67,24 @@ async function loadOpenAICompletionsStreamSimple(): Promise<StreamSimple> {
67
67
  if (!openAICompletionsStreamPromise) {
68
68
  openAICompletionsStreamPromise = (async () => {
69
69
  try {
70
- const mod = await import("@earendil-works/pi-ai/api/openai-completions");
71
- return (mod as { streamSimple: StreamSimple }).streamSimple;
70
+ // pi's extension loader aliases the pi-ai ROOT specifier to the compat
71
+ // entry (which re-exports openAICompletionsApi), so this import works in
72
+ // every pi runtime (jiti aliases / virtual modules / tsconfig paths).
73
+ // Subpath specifiers like "@earendil-works/pi-ai/api/openai-completions"
74
+ // are NOT aliased and only resolve inside a real node_modules install.
75
+ const mod = await import("@earendil-works/pi-ai");
76
+ const streams = (mod as { openAICompletionsApi?: () => { streamSimple: StreamSimple } }).openAICompletionsApi?.();
77
+ if (typeof streams?.streamSimple === "function") return streams.streamSimple;
72
78
  } catch {
73
- // In pi package installs, pi-ai may be nested under pi-coding-agent
74
- // instead of hoisted as a top-level dependency of this extension.
75
- const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
76
- const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
77
- const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
78
- const mod = await import(pathToFileURL(nestedModule).href);
79
- return (mod as { streamSimple: StreamSimple }).streamSimple;
79
+ // fall through to the nested-module lookup below
80
80
  }
81
+ // In real pi package installs, pi-ai may be nested under pi-coding-agent
82
+ // instead of hoisted as a top-level dependency of this extension.
83
+ const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
84
+ const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
85
+ const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
86
+ const nestedMod = await import(pathToFileURL(nestedModule).href);
87
+ return (nestedMod as { streamSimple: StreamSimple }).streamSimple;
81
88
  })();
82
89
  }
83
90
  return openAICompletionsStreamPromise;