pi-llamacpp-infra 1.2.2 → 1.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -25,6 +25,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
25
25
  - **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
26
26
  - **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
27
27
  - **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
28
+ - **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
28
29
  - **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
29
30
  - **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
30
31
  - **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
@@ -223,6 +224,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
223
224
  | `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
224
225
  | `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
225
226
  | `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
227
+ | output cap | `32768` | Registered per model as `maxTokens` unless the server reports an explicit `max_tokens`; still bounded by available context |
228
+ | request timeout | `1200000` | 20 minute timeout floor applied to llamacpp-infra OpenAI-compatible streams |
226
229
  | `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
227
230
  | `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
228
231
  | `metricsPollMs` | `5000` | How often `/metrics` is fetched |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llamacpp-infra",
3
- "version": "1.2.2",
3
+ "version": "1.2.4",
4
4
  "description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
5
5
  "keywords": [
6
6
  "pi-package",
@@ -35,6 +35,7 @@
35
35
  "src"
36
36
  ],
37
37
  "peerDependencies": {
38
+ "@earendil-works/pi-ai": "*",
38
39
  "@earendil-works/pi-coding-agent": "*"
39
40
  }
40
41
  }
package/src/core.ts CHANGED
@@ -21,6 +21,8 @@ export const LEGACY_CONFIG_FILE = "local-models.json";
21
21
  export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
22
22
  export const DEFAULT_API_KEY = "no-auth";
23
23
  export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
24
+ export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
25
+ export const DEFAULT_PROVIDER_TIMEOUT_MS = 20 * 60 * 1000;
24
26
 
25
27
  // ── Settings defaults ───────────────────────────────────────────────────────
26
28
  export const DEFAULT_SETTINGS: SettingsConfig = {
@@ -28,6 +30,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
28
30
  pollIntervalMs: 4000,
29
31
  pollMaxMs: 90_000,
30
32
  startupGraceMs: 40_000,
33
+ maxOutputTokens: DEFAULT_MAX_OUTPUT_TOKENS,
34
+ requestTimeoutMs: DEFAULT_PROVIDER_TIMEOUT_MS,
31
35
  knownGoodFailLimit: 3,
32
36
  detectVision: true,
33
37
  prefixModelIds: true,
package/src/index.ts CHANGED
@@ -20,6 +20,7 @@ import {
20
20
  shared,
21
21
  supportsThinkingBudget,
22
22
  } from "./core.ts";
23
+ import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
23
24
 
24
25
  export default function (pi: ExtensionAPI) {
25
26
  const config = loadConfig();
@@ -120,6 +121,7 @@ export default function (pi: ExtensionAPI) {
120
121
  baseUrl: `http://${first?.host ?? "127.0.0.1"}:${first?.ports[0] ?? 8080}/v1`,
121
122
  apiKey: first?.apiKey || "no-auth",
122
123
  api: "openai-completions",
124
+ streamSimple: createLongTimeoutOpenAICompletionsStream,
123
125
  models: [],
124
126
  });
125
127
  providerIsEmpty = true;
@@ -21,6 +21,7 @@ import type {
21
21
  ScanResult,
22
22
  } from "./types.ts";
23
23
  import { cleanModelName, lmStudioContextLength } from "./scan.ts";
24
+ import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
24
25
 
25
26
  /** Compact badge string appended to display names when enabled. */
26
27
  function badgeSuffix(
@@ -73,7 +74,7 @@ function toPiModel(
73
74
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
74
75
  reasoning: false,
75
76
  contextWindow,
76
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
77
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
77
78
  compat: makeCompat(kind),
78
79
  };
79
80
  }
@@ -92,7 +93,7 @@ function toPiModel(
92
93
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
93
94
  reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
94
95
  contextWindow,
95
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
96
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
96
97
  compat: makeCompat(kind),
97
98
  };
98
99
  }
@@ -118,7 +119,7 @@ function toPiModel(
118
119
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
119
120
  reasoning: isLlamaFamily,
120
121
  contextWindow,
121
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
122
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
122
123
  compat: makeCompat(kind),
123
124
  };
124
125
  }
@@ -232,6 +233,7 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
232
233
  baseUrl: defaultBaseUrl,
233
234
  apiKey: defaultApiKey,
234
235
  api: "openai-completions",
236
+ streamSimple: createLongTimeoutOpenAICompletionsStream,
235
237
  models: piModels,
236
238
  });
237
239
 
package/src/runtime.ts ADDED
@@ -0,0 +1,138 @@
1
+ // Runtime request defaults for the OpenAI-compatible llama.cpp provider.
2
+ // pi owns the normal SDK stream options, but this provider is for slow local
3
+ // models where long generations are expected. Keep a local floor so llamacpp-
4
+ // infra models don't inherit too-small global/default request timeouts.
5
+
6
+ import { dirname, join } from "node:path";
7
+ import { fileURLToPath, pathToFileURL } from "node:url";
8
+ import { DEFAULT_PROVIDER_TIMEOUT_MS, shared } from "./core.ts";
9
+
10
+ type AssistantMessageEvent = any;
11
+ type AssistantMessage = any;
12
+ type StreamOptions = Record<string, any> | undefined;
13
+ type StreamSimple = (model: any, context: any, options?: Record<string, any>) => AsyncIterable<AssistantMessageEvent> & { result?: () => Promise<AssistantMessage> };
14
+
15
+ class ForwardedAssistantMessageEventStream implements AsyncIterable<AssistantMessageEvent> {
16
+ private queue: AssistantMessageEvent[] = [];
17
+ private waiting: Array<(result: IteratorResult<AssistantMessageEvent>) => void> = [];
18
+ private done = false;
19
+ private resolveFinal!: (value: AssistantMessage) => void;
20
+ private readonly finalResult = new Promise<AssistantMessage>((resolve) => {
21
+ this.resolveFinal = resolve;
22
+ });
23
+
24
+ push(event: AssistantMessageEvent): void {
25
+ if (this.done) return;
26
+ if (event?.type === "done") {
27
+ this.done = true;
28
+ this.resolveFinal(event.message);
29
+ } else if (event?.type === "error") {
30
+ this.done = true;
31
+ this.resolveFinal(event.error);
32
+ }
33
+
34
+ const waiter = this.waiting.shift();
35
+ if (waiter) waiter({ value: event, done: false });
36
+ else this.queue.push(event);
37
+ }
38
+
39
+ end(result?: AssistantMessage): void {
40
+ this.done = true;
41
+ if (result !== undefined) this.resolveFinal(result);
42
+ while (this.waiting.length > 0) this.waiting.shift()?.({ value: undefined, done: true });
43
+ }
44
+
45
+ async *[Symbol.asyncIterator](): AsyncIterator<AssistantMessageEvent> {
46
+ for (;;) {
47
+ if (this.queue.length > 0) {
48
+ yield this.queue.shift();
49
+ } else if (this.done) {
50
+ return;
51
+ } else {
52
+ const next = await new Promise<IteratorResult<AssistantMessageEvent>>((resolve) => this.waiting.push(resolve));
53
+ if (next.done) return;
54
+ yield next.value;
55
+ }
56
+ }
57
+ }
58
+
59
+ result(): Promise<AssistantMessage> {
60
+ return this.finalResult;
61
+ }
62
+ }
63
+
64
+ let openAICompletionsStreamPromise: Promise<StreamSimple> | undefined;
65
+
66
+ async function loadOpenAICompletionsStreamSimple(): Promise<StreamSimple> {
67
+ if (!openAICompletionsStreamPromise) {
68
+ openAICompletionsStreamPromise = (async () => {
69
+ try {
70
+ // pi's extension loader aliases the pi-ai ROOT specifier to the compat
71
+ // entry (which re-exports openAICompletionsApi), so this import works in
72
+ // every pi runtime (jiti aliases / virtual modules / tsconfig paths).
73
+ // Subpath specifiers like "@earendil-works/pi-ai/api/openai-completions"
74
+ // are NOT aliased and only resolve inside a real node_modules install.
75
+ const mod = await import("@earendil-works/pi-ai");
76
+ const streams = (mod as { openAICompletionsApi?: () => { streamSimple: StreamSimple } }).openAICompletionsApi?.();
77
+ if (typeof streams?.streamSimple === "function") return streams.streamSimple;
78
+ } catch {
79
+ // fall through to the nested-module lookup below
80
+ }
81
+ // In real pi package installs, pi-ai may be nested under pi-coding-agent
82
+ // instead of hoisted as a top-level dependency of this extension.
83
+ const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
84
+ const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
85
+ const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
86
+ const nestedMod = await import(pathToFileURL(nestedModule).href);
87
+ return (nestedMod as { streamSimple: StreamSimple }).streamSimple;
88
+ })();
89
+ }
90
+ return openAICompletionsStreamPromise;
91
+ }
92
+
93
+ export function activeRequestTimeoutMs(): number {
94
+ const configured = shared.activeConfig?.settings?.requestTimeoutMs;
95
+ return typeof configured === "number" && Number.isFinite(configured) && configured > 0
96
+ ? configured
97
+ : DEFAULT_PROVIDER_TIMEOUT_MS;
98
+ }
99
+
100
+ export function withLocalRuntimeDefaults(options: StreamOptions, floorMs?: number): Record<string, any> {
101
+ const floor = typeof floorMs === "number" && Number.isFinite(floorMs) && floorMs > 0 ? floorMs : DEFAULT_PROVIDER_TIMEOUT_MS;
102
+ const currentTimeout = typeof options?.timeoutMs === "number" && Number.isFinite(options.timeoutMs) ? options.timeoutMs : undefined;
103
+ return {
104
+ ...(options ?? {}),
105
+ timeoutMs: currentTimeout === undefined ? floor : Math.max(currentTimeout, floor),
106
+ };
107
+ }
108
+
109
+ export function createLongTimeoutOpenAICompletionsStream(model: any, context: any, options?: Record<string, any>) {
110
+ const out = new ForwardedAssistantMessageEventStream();
111
+ void (async () => {
112
+ try {
113
+ const streamSimple = await loadOpenAICompletionsStreamSimple();
114
+ const inner = streamSimple(model, context, withLocalRuntimeDefaults(options, activeRequestTimeoutMs()));
115
+ for await (const event of inner) out.push(event);
116
+ if (typeof inner.result === "function") out.end(await inner.result());
117
+ else out.end();
118
+ } catch (err) {
119
+ const message = err instanceof Error ? err.message : String(err);
120
+ out.push({
121
+ type: "error",
122
+ reason: "error",
123
+ error: {
124
+ role: "assistant",
125
+ content: [],
126
+ api: model?.api ?? "openai-completions",
127
+ provider: model?.provider ?? "llamacpp-infra",
128
+ model: model?.id ?? "unknown",
129
+ usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
130
+ stopReason: "error",
131
+ errorMessage: message,
132
+ timestamp: Date.now(),
133
+ },
134
+ });
135
+ }
136
+ })();
137
+ return out;
138
+ }
package/src/types.ts CHANGED
@@ -31,6 +31,8 @@ export interface SettingsConfig {
31
31
  pollIntervalMs: number;
32
32
  pollMaxMs: number;
33
33
  startupGraceMs: number;
34
+ maxOutputTokens: number;
35
+ requestTimeoutMs: number;
34
36
  knownGoodFailLimit: number;
35
37
  detectVision: boolean;
36
38
  prefixModelIds: boolean;
package/src/ui.ts CHANGED
@@ -64,6 +64,15 @@ function formatCtx(tokens: number | undefined): string {
64
64
  return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : `${tokens}`;
65
65
  }
66
66
 
67
+ function formatTokens(n: number): string {
68
+ if (!Number.isFinite(n) || n <= 0) return "0";
69
+ if (n >= 1024) {
70
+ const k = n / 1024;
71
+ return Number.isInteger(k) ? `${k}k` : `${k.toFixed(1)}k`;
72
+ }
73
+ return `${n}`;
74
+ }
75
+
67
76
  function metadataBadges(m: ModelMetadata | undefined): string {
68
77
  if (!m) return "";
69
78
  const parts: string[] = [];
@@ -661,6 +670,16 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
661
670
  label: `💤 Include unloaded router models: ${s.includeUnloadedRouterModels ? "ON" : "OFF"}`,
662
671
  description: "router mode: list models not currently loaded",
663
672
  },
673
+ {
674
+ value: "maxtokens",
675
+ label: `📏 Max output tokens: ${formatTokens(s.maxOutputTokens)}`,
676
+ description: "per-model ceiling for outgoing max_tokens requests",
677
+ },
678
+ {
679
+ value: "timeoutfloor",
680
+ label: `⏱️ Request timeout: ${formatMs(s.requestTimeoutMs)}`,
681
+ description: "lower bound on OpenAI-completions stream idle timeout",
682
+ },
664
683
  {
665
684
  value: "warmup",
666
685
  label: `☕ Header warmup: ${s.warmup ? "ON" : "OFF"}`,
@@ -765,6 +784,25 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
765
784
  saveConfig(config);
766
785
  ctx.ui.notify(`☕ Header warmup ${config.settings.warmup ? "ON" : "OFF"}`, "info");
767
786
  break;
787
+ case "maxtokens": {
788
+ const v = await pickNumber("📏 Max output tokens", [4_096, 8_192, 16_384, 24_576, 32_768, 49_152, 65_536]);
789
+ if (v !== undefined) {
790
+ config.settings.maxOutputTokens = v;
791
+ saveConfig(config);
792
+ ctx.ui.notify(`📏 Max output tokens: ${v.toLocaleString()}`, "info");
793
+ await deps.rescan(ctx);
794
+ }
795
+ break;
796
+ }
797
+ case "timeoutfloor": {
798
+ const v = await pickNumber("⏱️ Request timeout", [60_000, 300_000, 600_000, 1_200_000, 1_800_000, 3_600_000]);
799
+ if (v !== undefined) {
800
+ config.settings.requestTimeoutMs = v;
801
+ saveConfig(config);
802
+ ctx.ui.notify(`⏱️ Request timeout: ${formatMs(v)}`, "info");
803
+ }
804
+ break;
805
+ }
768
806
  case "reset": {
769
807
  const ok = await ctx.ui.confirm("♻️ Reset settings", "Restore all discovery settings to their defaults?");
770
808
  if (ok) {