pi-llamacpp-infra 1.2.2 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/package.json +1 -1
- package/src/core.ts +4 -0
- package/src/index.ts +2 -0
- package/src/registration.ts +5 -3
- package/src/runtime.ts +131 -0
- package/src/types.ts +2 -0
- package/src/ui.ts +38 -0
package/README.md
CHANGED
|
@@ -25,6 +25,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
|
|
|
25
25
|
- **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
|
|
26
26
|
- **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
|
|
27
27
|
- **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
|
|
28
|
+
- **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
|
|
28
29
|
- **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
|
|
29
30
|
- **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
|
|
30
31
|
- **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
|
|
@@ -223,6 +224,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
223
224
|
| `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
|
|
224
225
|
| `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
|
|
225
226
|
| `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
|
|
227
|
+
| output cap | `32768` | Registered per model as `maxTokens` unless the server reports an explicit `max_tokens`; still bounded by available context |
|
|
228
|
+
| request timeout | `1200000` | 20 minute timeout floor applied to llamacpp-infra OpenAI-compatible streams |
|
|
226
229
|
| `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
|
|
227
230
|
| `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
|
|
228
231
|
| `metricsPollMs` | `5000` | How often `/metrics` is fetched |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.3",
|
|
4
4
|
"description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
package/src/core.ts
CHANGED
|
@@ -21,6 +21,8 @@ export const LEGACY_CONFIG_FILE = "local-models.json";
|
|
|
21
21
|
export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
|
|
22
22
|
export const DEFAULT_API_KEY = "no-auth";
|
|
23
23
|
export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
|
|
24
|
+
export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
|
|
25
|
+
export const DEFAULT_PROVIDER_TIMEOUT_MS = 20 * 60 * 1000;
|
|
24
26
|
|
|
25
27
|
// ── Settings defaults ───────────────────────────────────────────────────────
|
|
26
28
|
export const DEFAULT_SETTINGS: SettingsConfig = {
|
|
@@ -28,6 +30,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
|
|
|
28
30
|
pollIntervalMs: 4000,
|
|
29
31
|
pollMaxMs: 90_000,
|
|
30
32
|
startupGraceMs: 40_000,
|
|
33
|
+
maxOutputTokens: DEFAULT_MAX_OUTPUT_TOKENS,
|
|
34
|
+
requestTimeoutMs: DEFAULT_PROVIDER_TIMEOUT_MS,
|
|
31
35
|
knownGoodFailLimit: 3,
|
|
32
36
|
detectVision: true,
|
|
33
37
|
prefixModelIds: true,
|
package/src/index.ts
CHANGED
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
shared,
|
|
21
21
|
supportsThinkingBudget,
|
|
22
22
|
} from "./core.ts";
|
|
23
|
+
import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
|
|
23
24
|
|
|
24
25
|
export default function (pi: ExtensionAPI) {
|
|
25
26
|
const config = loadConfig();
|
|
@@ -120,6 +121,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
120
121
|
baseUrl: `http://${first?.host ?? "127.0.0.1"}:${first?.ports[0] ?? 8080}/v1`,
|
|
121
122
|
apiKey: first?.apiKey || "no-auth",
|
|
122
123
|
api: "openai-completions",
|
|
124
|
+
streamSimple: createLongTimeoutOpenAICompletionsStream,
|
|
123
125
|
models: [],
|
|
124
126
|
});
|
|
125
127
|
providerIsEmpty = true;
|
package/src/registration.ts
CHANGED
|
@@ -21,6 +21,7 @@ import type {
|
|
|
21
21
|
ScanResult,
|
|
22
22
|
} from "./types.ts";
|
|
23
23
|
import { cleanModelName, lmStudioContextLength } from "./scan.ts";
|
|
24
|
+
import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
|
|
24
25
|
|
|
25
26
|
/** Compact badge string appended to display names when enabled. */
|
|
26
27
|
function badgeSuffix(
|
|
@@ -73,7 +74,7 @@ function toPiModel(
|
|
|
73
74
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
74
75
|
reasoning: false,
|
|
75
76
|
contextWindow,
|
|
76
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
77
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
77
78
|
compat: makeCompat(kind),
|
|
78
79
|
};
|
|
79
80
|
}
|
|
@@ -92,7 +93,7 @@ function toPiModel(
|
|
|
92
93
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
93
94
|
reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
|
|
94
95
|
contextWindow,
|
|
95
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
96
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
96
97
|
compat: makeCompat(kind),
|
|
97
98
|
};
|
|
98
99
|
}
|
|
@@ -118,7 +119,7 @@ function toPiModel(
|
|
|
118
119
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
119
120
|
reasoning: isLlamaFamily,
|
|
120
121
|
contextWindow,
|
|
121
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
122
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
122
123
|
compat: makeCompat(kind),
|
|
123
124
|
};
|
|
124
125
|
}
|
|
@@ -232,6 +233,7 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
|
|
|
232
233
|
baseUrl: defaultBaseUrl,
|
|
233
234
|
apiKey: defaultApiKey,
|
|
234
235
|
api: "openai-completions",
|
|
236
|
+
streamSimple: createLongTimeoutOpenAICompletionsStream,
|
|
235
237
|
models: piModels,
|
|
236
238
|
});
|
|
237
239
|
|
package/src/runtime.ts
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
// Runtime request defaults for the OpenAI-compatible llama.cpp provider.
|
|
2
|
+
// pi owns the normal SDK stream options, but this provider is for slow local
|
|
3
|
+
// models where long generations are expected. Keep a local floor so llamacpp-
|
|
4
|
+
// infra models don't inherit too-small global/default request timeouts.
|
|
5
|
+
|
|
6
|
+
import { dirname, join } from "node:path";
|
|
7
|
+
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
8
|
+
import { DEFAULT_PROVIDER_TIMEOUT_MS, shared } from "./core.ts";
|
|
9
|
+
|
|
10
|
+
type AssistantMessageEvent = any;
|
|
11
|
+
type AssistantMessage = any;
|
|
12
|
+
type StreamOptions = Record<string, any> | undefined;
|
|
13
|
+
type StreamSimple = (model: any, context: any, options?: Record<string, any>) => AsyncIterable<AssistantMessageEvent> & { result?: () => Promise<AssistantMessage> };
|
|
14
|
+
|
|
15
|
+
class ForwardedAssistantMessageEventStream implements AsyncIterable<AssistantMessageEvent> {
|
|
16
|
+
private queue: AssistantMessageEvent[] = [];
|
|
17
|
+
private waiting: Array<(result: IteratorResult<AssistantMessageEvent>) => void> = [];
|
|
18
|
+
private done = false;
|
|
19
|
+
private resolveFinal!: (value: AssistantMessage) => void;
|
|
20
|
+
private readonly finalResult = new Promise<AssistantMessage>((resolve) => {
|
|
21
|
+
this.resolveFinal = resolve;
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
push(event: AssistantMessageEvent): void {
|
|
25
|
+
if (this.done) return;
|
|
26
|
+
if (event?.type === "done") {
|
|
27
|
+
this.done = true;
|
|
28
|
+
this.resolveFinal(event.message);
|
|
29
|
+
} else if (event?.type === "error") {
|
|
30
|
+
this.done = true;
|
|
31
|
+
this.resolveFinal(event.error);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const waiter = this.waiting.shift();
|
|
35
|
+
if (waiter) waiter({ value: event, done: false });
|
|
36
|
+
else this.queue.push(event);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
end(result?: AssistantMessage): void {
|
|
40
|
+
this.done = true;
|
|
41
|
+
if (result !== undefined) this.resolveFinal(result);
|
|
42
|
+
while (this.waiting.length > 0) this.waiting.shift()?.({ value: undefined, done: true });
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
async *[Symbol.asyncIterator](): AsyncIterator<AssistantMessageEvent> {
|
|
46
|
+
for (;;) {
|
|
47
|
+
if (this.queue.length > 0) {
|
|
48
|
+
yield this.queue.shift();
|
|
49
|
+
} else if (this.done) {
|
|
50
|
+
return;
|
|
51
|
+
} else {
|
|
52
|
+
const next = await new Promise<IteratorResult<AssistantMessageEvent>>((resolve) => this.waiting.push(resolve));
|
|
53
|
+
if (next.done) return;
|
|
54
|
+
yield next.value;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
result(): Promise<AssistantMessage> {
|
|
60
|
+
return this.finalResult;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
let openAICompletionsStreamPromise: Promise<StreamSimple> | undefined;
|
|
65
|
+
|
|
66
|
+
async function loadOpenAICompletionsStreamSimple(): Promise<StreamSimple> {
|
|
67
|
+
if (!openAICompletionsStreamPromise) {
|
|
68
|
+
openAICompletionsStreamPromise = (async () => {
|
|
69
|
+
try {
|
|
70
|
+
const mod = await import("@earendil-works/pi-ai/api/openai-completions");
|
|
71
|
+
return (mod as { streamSimple: StreamSimple }).streamSimple;
|
|
72
|
+
} catch {
|
|
73
|
+
// In pi package installs, pi-ai may be nested under pi-coding-agent
|
|
74
|
+
// instead of hoisted as a top-level dependency of this extension.
|
|
75
|
+
const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
|
|
76
|
+
const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
|
|
77
|
+
const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
|
|
78
|
+
const mod = await import(pathToFileURL(nestedModule).href);
|
|
79
|
+
return (mod as { streamSimple: StreamSimple }).streamSimple;
|
|
80
|
+
}
|
|
81
|
+
})();
|
|
82
|
+
}
|
|
83
|
+
return openAICompletionsStreamPromise;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export function activeRequestTimeoutMs(): number {
|
|
87
|
+
const configured = shared.activeConfig?.settings?.requestTimeoutMs;
|
|
88
|
+
return typeof configured === "number" && Number.isFinite(configured) && configured > 0
|
|
89
|
+
? configured
|
|
90
|
+
: DEFAULT_PROVIDER_TIMEOUT_MS;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function withLocalRuntimeDefaults(options: StreamOptions, floorMs?: number): Record<string, any> {
|
|
94
|
+
const floor = typeof floorMs === "number" && Number.isFinite(floorMs) && floorMs > 0 ? floorMs : DEFAULT_PROVIDER_TIMEOUT_MS;
|
|
95
|
+
const currentTimeout = typeof options?.timeoutMs === "number" && Number.isFinite(options.timeoutMs) ? options.timeoutMs : undefined;
|
|
96
|
+
return {
|
|
97
|
+
...(options ?? {}),
|
|
98
|
+
timeoutMs: currentTimeout === undefined ? floor : Math.max(currentTimeout, floor),
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function createLongTimeoutOpenAICompletionsStream(model: any, context: any, options?: Record<string, any>) {
|
|
103
|
+
const out = new ForwardedAssistantMessageEventStream();
|
|
104
|
+
void (async () => {
|
|
105
|
+
try {
|
|
106
|
+
const streamSimple = await loadOpenAICompletionsStreamSimple();
|
|
107
|
+
const inner = streamSimple(model, context, withLocalRuntimeDefaults(options, activeRequestTimeoutMs()));
|
|
108
|
+
for await (const event of inner) out.push(event);
|
|
109
|
+
if (typeof inner.result === "function") out.end(await inner.result());
|
|
110
|
+
else out.end();
|
|
111
|
+
} catch (err) {
|
|
112
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
113
|
+
out.push({
|
|
114
|
+
type: "error",
|
|
115
|
+
reason: "error",
|
|
116
|
+
error: {
|
|
117
|
+
role: "assistant",
|
|
118
|
+
content: [],
|
|
119
|
+
api: model?.api ?? "openai-completions",
|
|
120
|
+
provider: model?.provider ?? "llamacpp-infra",
|
|
121
|
+
model: model?.id ?? "unknown",
|
|
122
|
+
usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
|
|
123
|
+
stopReason: "error",
|
|
124
|
+
errorMessage: message,
|
|
125
|
+
timestamp: Date.now(),
|
|
126
|
+
},
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
})();
|
|
130
|
+
return out;
|
|
131
|
+
}
|
package/src/types.ts
CHANGED
package/src/ui.ts
CHANGED
|
@@ -64,6 +64,15 @@ function formatCtx(tokens: number | undefined): string {
|
|
|
64
64
|
return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : `${tokens}`;
|
|
65
65
|
}
|
|
66
66
|
|
|
67
|
+
function formatTokens(n: number): string {
|
|
68
|
+
if (!Number.isFinite(n) || n <= 0) return "0";
|
|
69
|
+
if (n >= 1024) {
|
|
70
|
+
const k = n / 1024;
|
|
71
|
+
return Number.isInteger(k) ? `${k}k` : `${k.toFixed(1)}k`;
|
|
72
|
+
}
|
|
73
|
+
return `${n}`;
|
|
74
|
+
}
|
|
75
|
+
|
|
67
76
|
function metadataBadges(m: ModelMetadata | undefined): string {
|
|
68
77
|
if (!m) return "";
|
|
69
78
|
const parts: string[] = [];
|
|
@@ -661,6 +670,16 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
|
|
|
661
670
|
label: `💤 Include unloaded router models: ${s.includeUnloadedRouterModels ? "ON" : "OFF"}`,
|
|
662
671
|
description: "router mode: list models not currently loaded",
|
|
663
672
|
},
|
|
673
|
+
{
|
|
674
|
+
value: "maxtokens",
|
|
675
|
+
label: `📏 Max output tokens: ${formatTokens(s.maxOutputTokens)}`,
|
|
676
|
+
description: "per-model ceiling for outgoing max_tokens requests",
|
|
677
|
+
},
|
|
678
|
+
{
|
|
679
|
+
value: "timeoutfloor",
|
|
680
|
+
label: `⏱️ Request timeout: ${formatMs(s.requestTimeoutMs)}`,
|
|
681
|
+
description: "lower bound on OpenAI-completions stream idle timeout",
|
|
682
|
+
},
|
|
664
683
|
{
|
|
665
684
|
value: "warmup",
|
|
666
685
|
label: `☕ Header warmup: ${s.warmup ? "ON" : "OFF"}`,
|
|
@@ -765,6 +784,25 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
|
|
|
765
784
|
saveConfig(config);
|
|
766
785
|
ctx.ui.notify(`☕ Header warmup ${config.settings.warmup ? "ON" : "OFF"}`, "info");
|
|
767
786
|
break;
|
|
787
|
+
case "maxtokens": {
|
|
788
|
+
const v = await pickNumber("📏 Max output tokens", [4_096, 8_192, 16_384, 24_576, 32_768, 49_152, 65_536]);
|
|
789
|
+
if (v !== undefined) {
|
|
790
|
+
config.settings.maxOutputTokens = v;
|
|
791
|
+
saveConfig(config);
|
|
792
|
+
ctx.ui.notify(`📏 Max output tokens: ${v.toLocaleString()}`, "info");
|
|
793
|
+
await deps.rescan(ctx);
|
|
794
|
+
}
|
|
795
|
+
break;
|
|
796
|
+
}
|
|
797
|
+
case "timeoutfloor": {
|
|
798
|
+
const v = await pickNumber("⏱️ Request timeout", [60_000, 300_000, 600_000, 1_200_000, 1_800_000, 3_600_000]);
|
|
799
|
+
if (v !== undefined) {
|
|
800
|
+
config.settings.requestTimeoutMs = v;
|
|
801
|
+
saveConfig(config);
|
|
802
|
+
ctx.ui.notify(`⏱️ Request timeout: ${formatMs(v)}`, "info");
|
|
803
|
+
}
|
|
804
|
+
break;
|
|
805
|
+
}
|
|
768
806
|
case "reset": {
|
|
769
807
|
const ok = await ctx.ui.confirm("♻️ Reset settings", "Restore all discovery settings to their defaults?");
|
|
770
808
|
if (ok) {
|