pi-llamacpp-infra 1.2.1 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -8
- package/package.json +1 -1
- package/src/core.ts +4 -0
- package/src/index.ts +5 -5
- package/src/registration.ts +5 -3
- package/src/runtime.ts +131 -0
- package/src/speed.ts +26 -21
- package/src/types.ts +2 -0
- package/src/ui.ts +38 -0
package/README.md
CHANGED
|
@@ -25,6 +25,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
|
|
|
25
25
|
- **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
|
|
26
26
|
- **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
|
|
27
27
|
- **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
|
|
28
|
+
- **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
|
|
28
29
|
- **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
|
|
29
30
|
- **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
|
|
30
31
|
- **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
|
|
@@ -223,6 +224,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
223
224
|
| `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
|
|
224
225
|
| `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
|
|
225
226
|
| `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
|
|
227
|
+
| output cap | `32768` | Registered per model as `maxTokens` unless the server reports an explicit `max_tokens`; still bounded by available context |
|
|
228
|
+
| request timeout | `1200000` | 20 minute timeout floor applied to llamacpp-infra OpenAI-compatible streams |
|
|
226
229
|
| `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
|
|
227
230
|
| `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
|
|
228
231
|
| `metricsPollMs` | `5000` | How often `/metrics` is fetched |
|
|
@@ -247,20 +250,20 @@ llama.cpp-family models are registered as reasoning models, exactly like a nativ
|
|
|
247
250
|
|
|
248
251
|
## Live Speed & Metrics (footer)
|
|
249
252
|
|
|
250
|
-
When enabled, the speed reading appears in the footer's status line (no extra terminal row) whenever the active model is from llamacpp-infra, updating constantly while tokens flow:
|
|
253
|
+
When enabled, the speed reading appears in the footer's status line (no extra terminal row) whenever the active model is from llamacpp-infra, updating constantly while tokens flow. Both entries are kept ultra-compact so they coexist with other extensions on pi's single status line (which truncates from the end):
|
|
251
254
|
|
|
252
255
|
```
|
|
253
|
-
🦙
|
|
254
|
-
🦙
|
|
255
|
-
🦙
|
|
256
|
-
🦙
|
|
257
|
-
🦙
|
|
256
|
+
🦙(12) ⚡… (before the first token)
|
|
257
|
+
🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (while streaming)
|
|
258
|
+
🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (just after the answer ends)
|
|
259
|
+
🦙(12) ⏸ (between turns)
|
|
260
|
+
🦙(12) ▶2 ⚡ 150 t/s 🔥 18.0 t/s (pi idle, server busy for other clients)
|
|
258
261
|
```
|
|
259
262
|
|
|
260
|
-
(
|
|
263
|
+
(`🦙(n)` is the extension's model-count status; both live on the same footer line, so no extra row is consumed.)
|
|
261
264
|
|
|
262
265
|
- **Client measurement (always, no `--metrics` needed)** — prefill speed = `prompt tokens ÷ (request → first token)` (pi's `usage.input`, OpenAI-style `prompt_tokens` as fallback); generation speed = a moving 1.5 s window over per-token arrival samples. Updated ~every 100 ms while a stream is live (throttled, and unchanged text is skipped, so the footer never churns).
|
|
263
|
-
- **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown; when the server is idle, the plain
|
|
266
|
+
- **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown (`▶n`); when the server is idle, the plain `⏸` reading returns.
|
|
264
267
|
|
|
265
268
|
## Architecture
|
|
266
269
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.3",
|
|
4
4
|
"description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
package/src/core.ts
CHANGED
|
@@ -21,6 +21,8 @@ export const LEGACY_CONFIG_FILE = "local-models.json";
|
|
|
21
21
|
export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
|
|
22
22
|
export const DEFAULT_API_KEY = "no-auth";
|
|
23
23
|
export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
|
|
24
|
+
export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
|
|
25
|
+
export const DEFAULT_PROVIDER_TIMEOUT_MS = 20 * 60 * 1000;
|
|
24
26
|
|
|
25
27
|
// ── Settings defaults ───────────────────────────────────────────────────────
|
|
26
28
|
export const DEFAULT_SETTINGS: SettingsConfig = {
|
|
@@ -28,6 +30,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
|
|
|
28
30
|
pollIntervalMs: 4000,
|
|
29
31
|
pollMaxMs: 90_000,
|
|
30
32
|
startupGraceMs: 40_000,
|
|
33
|
+
maxOutputTokens: DEFAULT_MAX_OUTPUT_TOKENS,
|
|
34
|
+
requestTimeoutMs: DEFAULT_PROVIDER_TIMEOUT_MS,
|
|
31
35
|
knownGoodFailLimit: 3,
|
|
32
36
|
detectVision: true,
|
|
33
37
|
prefixModelIds: true,
|
package/src/index.ts
CHANGED
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
shared,
|
|
21
21
|
supportsThinkingBudget,
|
|
22
22
|
} from "./core.ts";
|
|
23
|
+
import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
|
|
23
24
|
|
|
24
25
|
export default function (pi: ExtensionAPI) {
|
|
25
26
|
const config = loadConfig();
|
|
@@ -120,6 +121,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
120
121
|
baseUrl: `http://${first?.host ?? "127.0.0.1"}:${first?.ports[0] ?? 8080}/v1`,
|
|
121
122
|
apiKey: first?.apiKey || "no-auth",
|
|
122
123
|
api: "openai-completions",
|
|
124
|
+
streamSimple: createLongTimeoutOpenAICompletionsStream,
|
|
123
125
|
models: [],
|
|
124
126
|
});
|
|
125
127
|
providerIsEmpty = true;
|
|
@@ -225,7 +227,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
225
227
|
|
|
226
228
|
async function rescan(ctx?: ExtensionContext) {
|
|
227
229
|
if (!extensionActive) return;
|
|
228
|
-
if (ctxHasUI(ctx)) ctx.ui.setStatus(STATUS_KEY, "🔎
|
|
230
|
+
if (ctxHasUI(ctx)) ctx.ui.setStatus(STATUS_KEY, "🔎");
|
|
229
231
|
stopPolling();
|
|
230
232
|
registerEmptyProvider();
|
|
231
233
|
const r = await discoverAndRegister();
|
|
@@ -237,11 +239,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
237
239
|
function updateStatusFooter(ctx?: ExtensionContext) {
|
|
238
240
|
if (!ctxHasUI(ctx)) return;
|
|
239
241
|
if (shared.registeredCount > 0) {
|
|
240
|
-
|
|
241
|
-
const total = shared.lastScan?.serversTotal ?? 0;
|
|
242
|
-
ctx.ui.setStatus(STATUS_KEY, `🦙 ${shared.registeredCount} models · ${up}/${total} ✓`);
|
|
242
|
+
ctx.ui.setStatus(STATUS_KEY, `🦙(${shared.registeredCount})`);
|
|
243
243
|
} else if (shared.lastScan?.endpoints.some((e) => e.loading)) {
|
|
244
|
-
ctx.ui.setStatus(STATUS_KEY, "⏳
|
|
244
|
+
ctx.ui.setStatus(STATUS_KEY, "⏳");
|
|
245
245
|
} else if (shared.lastError) {
|
|
246
246
|
ctx.ui.setStatus(STATUS_KEY, "⚠️");
|
|
247
247
|
} else {
|
package/src/registration.ts
CHANGED
|
@@ -21,6 +21,7 @@ import type {
|
|
|
21
21
|
ScanResult,
|
|
22
22
|
} from "./types.ts";
|
|
23
23
|
import { cleanModelName, lmStudioContextLength } from "./scan.ts";
|
|
24
|
+
import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
|
|
24
25
|
|
|
25
26
|
/** Compact badge string appended to display names when enabled. */
|
|
26
27
|
function badgeSuffix(
|
|
@@ -73,7 +74,7 @@ function toPiModel(
|
|
|
73
74
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
74
75
|
reasoning: false,
|
|
75
76
|
contextWindow,
|
|
76
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
77
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
77
78
|
compat: makeCompat(kind),
|
|
78
79
|
};
|
|
79
80
|
}
|
|
@@ -92,7 +93,7 @@ function toPiModel(
|
|
|
92
93
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
93
94
|
reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
|
|
94
95
|
contextWindow,
|
|
95
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
96
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
96
97
|
compat: makeCompat(kind),
|
|
97
98
|
};
|
|
98
99
|
}
|
|
@@ -118,7 +119,7 @@ function toPiModel(
|
|
|
118
119
|
name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
|
|
119
120
|
reasoning: isLlamaFamily,
|
|
120
121
|
contextWindow,
|
|
121
|
-
maxTokens: model.max_tokens ?? Math.min(contextWindow,
|
|
122
|
+
maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
|
|
122
123
|
compat: makeCompat(kind),
|
|
123
124
|
};
|
|
124
125
|
}
|
|
@@ -232,6 +233,7 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
|
|
|
232
233
|
baseUrl: defaultBaseUrl,
|
|
233
234
|
apiKey: defaultApiKey,
|
|
234
235
|
api: "openai-completions",
|
|
236
|
+
streamSimple: createLongTimeoutOpenAICompletionsStream,
|
|
235
237
|
models: piModels,
|
|
236
238
|
});
|
|
237
239
|
|
package/src/runtime.ts
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
// Runtime request defaults for the OpenAI-compatible llama.cpp provider.
|
|
2
|
+
// pi owns the normal SDK stream options, but this provider is for slow local
|
|
3
|
+
// models where long generations are expected. Keep a local floor so llamacpp-
|
|
4
|
+
// infra models don't inherit too-small global/default request timeouts.
|
|
5
|
+
|
|
6
|
+
import { dirname, join } from "node:path";
|
|
7
|
+
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
8
|
+
import { DEFAULT_PROVIDER_TIMEOUT_MS, shared } from "./core.ts";
|
|
9
|
+
|
|
10
|
+
type AssistantMessageEvent = any;
|
|
11
|
+
type AssistantMessage = any;
|
|
12
|
+
type StreamOptions = Record<string, any> | undefined;
|
|
13
|
+
type StreamSimple = (model: any, context: any, options?: Record<string, any>) => AsyncIterable<AssistantMessageEvent> & { result?: () => Promise<AssistantMessage> };
|
|
14
|
+
|
|
15
|
+
class ForwardedAssistantMessageEventStream implements AsyncIterable<AssistantMessageEvent> {
|
|
16
|
+
private queue: AssistantMessageEvent[] = [];
|
|
17
|
+
private waiting: Array<(result: IteratorResult<AssistantMessageEvent>) => void> = [];
|
|
18
|
+
private done = false;
|
|
19
|
+
private resolveFinal!: (value: AssistantMessage) => void;
|
|
20
|
+
private readonly finalResult = new Promise<AssistantMessage>((resolve) => {
|
|
21
|
+
this.resolveFinal = resolve;
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
push(event: AssistantMessageEvent): void {
|
|
25
|
+
if (this.done) return;
|
|
26
|
+
if (event?.type === "done") {
|
|
27
|
+
this.done = true;
|
|
28
|
+
this.resolveFinal(event.message);
|
|
29
|
+
} else if (event?.type === "error") {
|
|
30
|
+
this.done = true;
|
|
31
|
+
this.resolveFinal(event.error);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const waiter = this.waiting.shift();
|
|
35
|
+
if (waiter) waiter({ value: event, done: false });
|
|
36
|
+
else this.queue.push(event);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
end(result?: AssistantMessage): void {
|
|
40
|
+
this.done = true;
|
|
41
|
+
if (result !== undefined) this.resolveFinal(result);
|
|
42
|
+
while (this.waiting.length > 0) this.waiting.shift()?.({ value: undefined, done: true });
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
async *[Symbol.asyncIterator](): AsyncIterator<AssistantMessageEvent> {
|
|
46
|
+
for (;;) {
|
|
47
|
+
if (this.queue.length > 0) {
|
|
48
|
+
yield this.queue.shift();
|
|
49
|
+
} else if (this.done) {
|
|
50
|
+
return;
|
|
51
|
+
} else {
|
|
52
|
+
const next = await new Promise<IteratorResult<AssistantMessageEvent>>((resolve) => this.waiting.push(resolve));
|
|
53
|
+
if (next.done) return;
|
|
54
|
+
yield next.value;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
result(): Promise<AssistantMessage> {
|
|
60
|
+
return this.finalResult;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
let openAICompletionsStreamPromise: Promise<StreamSimple> | undefined;
|
|
65
|
+
|
|
66
|
+
async function loadOpenAICompletionsStreamSimple(): Promise<StreamSimple> {
|
|
67
|
+
if (!openAICompletionsStreamPromise) {
|
|
68
|
+
openAICompletionsStreamPromise = (async () => {
|
|
69
|
+
try {
|
|
70
|
+
const mod = await import("@earendil-works/pi-ai/api/openai-completions");
|
|
71
|
+
return (mod as { streamSimple: StreamSimple }).streamSimple;
|
|
72
|
+
} catch {
|
|
73
|
+
// In pi package installs, pi-ai may be nested under pi-coding-agent
|
|
74
|
+
// instead of hoisted as a top-level dependency of this extension.
|
|
75
|
+
const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
|
|
76
|
+
const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
|
|
77
|
+
const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
|
|
78
|
+
const mod = await import(pathToFileURL(nestedModule).href);
|
|
79
|
+
return (mod as { streamSimple: StreamSimple }).streamSimple;
|
|
80
|
+
}
|
|
81
|
+
})();
|
|
82
|
+
}
|
|
83
|
+
return openAICompletionsStreamPromise;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export function activeRequestTimeoutMs(): number {
|
|
87
|
+
const configured = shared.activeConfig?.settings?.requestTimeoutMs;
|
|
88
|
+
return typeof configured === "number" && Number.isFinite(configured) && configured > 0
|
|
89
|
+
? configured
|
|
90
|
+
: DEFAULT_PROVIDER_TIMEOUT_MS;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function withLocalRuntimeDefaults(options: StreamOptions, floorMs?: number): Record<string, any> {
|
|
94
|
+
const floor = typeof floorMs === "number" && Number.isFinite(floorMs) && floorMs > 0 ? floorMs : DEFAULT_PROVIDER_TIMEOUT_MS;
|
|
95
|
+
const currentTimeout = typeof options?.timeoutMs === "number" && Number.isFinite(options.timeoutMs) ? options.timeoutMs : undefined;
|
|
96
|
+
return {
|
|
97
|
+
...(options ?? {}),
|
|
98
|
+
timeoutMs: currentTimeout === undefined ? floor : Math.max(currentTimeout, floor),
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function createLongTimeoutOpenAICompletionsStream(model: any, context: any, options?: Record<string, any>) {
|
|
103
|
+
const out = new ForwardedAssistantMessageEventStream();
|
|
104
|
+
void (async () => {
|
|
105
|
+
try {
|
|
106
|
+
const streamSimple = await loadOpenAICompletionsStreamSimple();
|
|
107
|
+
const inner = streamSimple(model, context, withLocalRuntimeDefaults(options, activeRequestTimeoutMs()));
|
|
108
|
+
for await (const event of inner) out.push(event);
|
|
109
|
+
if (typeof inner.result === "function") out.end(await inner.result());
|
|
110
|
+
else out.end();
|
|
111
|
+
} catch (err) {
|
|
112
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
113
|
+
out.push({
|
|
114
|
+
type: "error",
|
|
115
|
+
reason: "error",
|
|
116
|
+
error: {
|
|
117
|
+
role: "assistant",
|
|
118
|
+
content: [],
|
|
119
|
+
api: model?.api ?? "openai-completions",
|
|
120
|
+
provider: model?.provider ?? "llamacpp-infra",
|
|
121
|
+
model: model?.id ?? "unknown",
|
|
122
|
+
usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
|
|
123
|
+
stopReason: "error",
|
|
124
|
+
errorMessage: message,
|
|
125
|
+
timestamp: Date.now(),
|
|
126
|
+
},
|
|
127
|
+
});
|
|
128
|
+
}
|
|
129
|
+
})();
|
|
130
|
+
return out;
|
|
131
|
+
}
|
package/src/speed.ts
CHANGED
|
@@ -6,10 +6,15 @@
|
|
|
6
6
|
// prefill t/s = prompt tokens / (before_provider_request → first token)
|
|
7
7
|
// gen t/s = moving window over per-token arrival samples
|
|
8
8
|
//
|
|
9
|
+
// Footer format (kept ultra-compact so it coexists with other extensions on
|
|
10
|
+
// pi's single status line, which is truncated at the end):
|
|
11
|
+
//
|
|
12
|
+
// ⚡ {prefill} t/s 🔥 {gen} t/s
|
|
13
|
+
//
|
|
9
14
|
// The server-side /metrics polling (metrics.ts) only supplements this:
|
|
10
15
|
// when the client is idle but the server reports other clients processing,
|
|
11
|
-
// their rate is shown
|
|
12
|
-
// no --metrics endpoint.
|
|
16
|
+
// their rate is shown as `▶{n} ⚡ … t/s 🔥 … t/s`. The client measurement
|
|
17
|
+
// works even when the server has no --metrics endpoint.
|
|
13
18
|
|
|
14
19
|
import { debugLog, METRICS_STATUS_KEY } from "./core.ts";
|
|
15
20
|
import type { AssistantMessageEvent, ExtensionContext, ServerMetricsState, ThemeFg } from "./types.ts";
|
|
@@ -126,58 +131,58 @@ export function createSpeedTracker(deps: SpeedDeps) {
|
|
|
126
131
|
return tokens / (span / 1000);
|
|
127
132
|
}
|
|
128
133
|
|
|
129
|
-
function buildLine(ctx: ExtensionContext | undefined, now: number): string
|
|
134
|
+
function buildLine(ctx: ExtensionContext | undefined, now: number): string {
|
|
130
135
|
const f = fg(ctx);
|
|
131
|
-
const parts: string[] = [
|
|
136
|
+
const parts: string[] = [];
|
|
132
137
|
|
|
133
138
|
switch (state) {
|
|
134
139
|
case "prefill":
|
|
135
|
-
parts.push(f("warning", "
|
|
140
|
+
parts.push(f("warning", "⚡…"));
|
|
136
141
|
break;
|
|
137
142
|
|
|
138
143
|
case "streaming": {
|
|
144
|
+
// Prefill rate of the previous call (only known once it ended).
|
|
145
|
+
if (lastPrefillTps !== undefined) {
|
|
146
|
+
parts.push(f("muted", `⚡ ${Math.round(lastPrefillTps)} t/s`));
|
|
147
|
+
}
|
|
139
148
|
const gen = genRateAt(now);
|
|
140
149
|
if (gen !== undefined) {
|
|
141
150
|
lastGenTps = gen;
|
|
142
151
|
parts.push(f(rateColor(gen, "gen"), `🔥 ${formatRate(gen)} t/s`));
|
|
143
152
|
}
|
|
144
|
-
if (lastPrefillTps !== undefined) {
|
|
145
|
-
parts.push(f("muted", `⚡ ${Math.round(lastPrefillTps)}`));
|
|
146
|
-
}
|
|
147
|
-
parts.push(f("muted", `${tokenCount} tok`));
|
|
148
153
|
break;
|
|
149
154
|
}
|
|
150
155
|
|
|
151
156
|
case "done": {
|
|
152
|
-
if (lastGenTps !== undefined) {
|
|
153
|
-
parts.push(f("muted", `🔥 ${formatRate(lastGenTps)} t/s`));
|
|
154
|
-
}
|
|
155
157
|
if (lastPrefillTps !== undefined) {
|
|
156
|
-
parts.push(f("
|
|
158
|
+
parts.push(f(rateColor(lastPrefillTps, "prefill"), `⚡ ${Math.round(lastPrefillTps)} t/s`));
|
|
159
|
+
}
|
|
160
|
+
if (lastGenTps !== undefined) {
|
|
161
|
+
parts.push(f(rateColor(lastGenTps, "gen"), `🔥 ${formatRate(lastGenTps)} t/s`));
|
|
157
162
|
}
|
|
158
|
-
parts.push(f("muted", `${tokenCount} tok`));
|
|
159
163
|
break;
|
|
160
164
|
}
|
|
161
165
|
|
|
162
166
|
case "idle": {
|
|
163
167
|
// Server supplement: this endpoint is busy for *other* clients.
|
|
164
168
|
if (server && server.processing > 0) {
|
|
165
|
-
parts.push(f("success",
|
|
169
|
+
parts.push(f("success", `▶${server.processing}`));
|
|
166
170
|
if (server.promptTps !== undefined && server.promptTps > 0) {
|
|
167
|
-
parts.push(f(rateColor(server.promptTps, "prefill"), `⚡ ${Math.round(server.promptTps)}`));
|
|
171
|
+
parts.push(f(rateColor(server.promptTps, "prefill"), `⚡ ${Math.round(server.promptTps)} t/s`));
|
|
168
172
|
}
|
|
169
173
|
if (server.genTps !== undefined && server.genTps > 0) {
|
|
170
|
-
parts.push(f(rateColor(server.genTps, "gen"), `🔥 ${formatRate(server.genTps)}`));
|
|
174
|
+
parts.push(f(rateColor(server.genTps, "gen"), `🔥 ${formatRate(server.genTps)} t/s`));
|
|
171
175
|
}
|
|
172
|
-
parts.push(f("muted", "server"));
|
|
173
176
|
} else {
|
|
174
|
-
parts.push(f("muted", "⏸
|
|
177
|
+
parts.push(f("muted", "⏸"));
|
|
175
178
|
}
|
|
176
179
|
break;
|
|
177
180
|
}
|
|
178
181
|
}
|
|
179
182
|
|
|
180
|
-
|
|
183
|
+
// A stream is live but no rate is computable yet (first ~300 ms of a call).
|
|
184
|
+
if (parts.length === 0) parts.push(f("muted", "🔥…"));
|
|
185
|
+
return parts.join(" ");
|
|
181
186
|
}
|
|
182
187
|
|
|
183
188
|
function render(ctx: ExtensionContext | undefined, now: number, force = false): void {
|
|
@@ -186,7 +191,7 @@ export function createSpeedTracker(deps: SpeedDeps) {
|
|
|
186
191
|
return;
|
|
187
192
|
}
|
|
188
193
|
if (!deps.hasUI(ctx) || !deps.isOurs(ctx)) return;
|
|
189
|
-
const line = buildLine(ctx, now)
|
|
194
|
+
const line = buildLine(ctx, now);
|
|
190
195
|
if (!force && now - lastRenderAt < RENDER_THROTTLE_MS) return;
|
|
191
196
|
lastRenderAt = now;
|
|
192
197
|
// No visual change → skip (avoids a full UI re-render on the footer).
|
package/src/types.ts
CHANGED
package/src/ui.ts
CHANGED
|
@@ -64,6 +64,15 @@ function formatCtx(tokens: number | undefined): string {
|
|
|
64
64
|
return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : `${tokens}`;
|
|
65
65
|
}
|
|
66
66
|
|
|
67
|
+
function formatTokens(n: number): string {
|
|
68
|
+
if (!Number.isFinite(n) || n <= 0) return "0";
|
|
69
|
+
if (n >= 1024) {
|
|
70
|
+
const k = n / 1024;
|
|
71
|
+
return Number.isInteger(k) ? `${k}k` : `${k.toFixed(1)}k`;
|
|
72
|
+
}
|
|
73
|
+
return `${n}`;
|
|
74
|
+
}
|
|
75
|
+
|
|
67
76
|
function metadataBadges(m: ModelMetadata | undefined): string {
|
|
68
77
|
if (!m) return "";
|
|
69
78
|
const parts: string[] = [];
|
|
@@ -661,6 +670,16 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
|
|
|
661
670
|
label: `💤 Include unloaded router models: ${s.includeUnloadedRouterModels ? "ON" : "OFF"}`,
|
|
662
671
|
description: "router mode: list models not currently loaded",
|
|
663
672
|
},
|
|
673
|
+
{
|
|
674
|
+
value: "maxtokens",
|
|
675
|
+
label: `📏 Max output tokens: ${formatTokens(s.maxOutputTokens)}`,
|
|
676
|
+
description: "per-model ceiling for outgoing max_tokens requests",
|
|
677
|
+
},
|
|
678
|
+
{
|
|
679
|
+
value: "timeoutfloor",
|
|
680
|
+
label: `⏱️ Request timeout: ${formatMs(s.requestTimeoutMs)}`,
|
|
681
|
+
description: "lower bound on OpenAI-completions stream idle timeout",
|
|
682
|
+
},
|
|
664
683
|
{
|
|
665
684
|
value: "warmup",
|
|
666
685
|
label: `☕ Header warmup: ${s.warmup ? "ON" : "OFF"}`,
|
|
@@ -765,6 +784,25 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
|
|
|
765
784
|
saveConfig(config);
|
|
766
785
|
ctx.ui.notify(`☕ Header warmup ${config.settings.warmup ? "ON" : "OFF"}`, "info");
|
|
767
786
|
break;
|
|
787
|
+
case "maxtokens": {
|
|
788
|
+
const v = await pickNumber("📏 Max output tokens", [4_096, 8_192, 16_384, 24_576, 32_768, 49_152, 65_536]);
|
|
789
|
+
if (v !== undefined) {
|
|
790
|
+
config.settings.maxOutputTokens = v;
|
|
791
|
+
saveConfig(config);
|
|
792
|
+
ctx.ui.notify(`📏 Max output tokens: ${v.toLocaleString()}`, "info");
|
|
793
|
+
await deps.rescan(ctx);
|
|
794
|
+
}
|
|
795
|
+
break;
|
|
796
|
+
}
|
|
797
|
+
case "timeoutfloor": {
|
|
798
|
+
const v = await pickNumber("⏱️ Request timeout", [60_000, 300_000, 600_000, 1_200_000, 1_800_000, 3_600_000]);
|
|
799
|
+
if (v !== undefined) {
|
|
800
|
+
config.settings.requestTimeoutMs = v;
|
|
801
|
+
saveConfig(config);
|
|
802
|
+
ctx.ui.notify(`⏱️ Request timeout: ${formatMs(v)}`, "info");
|
|
803
|
+
}
|
|
804
|
+
break;
|
|
805
|
+
}
|
|
768
806
|
case "reset": {
|
|
769
807
|
const ok = await ctx.ui.confirm("♻️ Reset settings", "Restore all discovery settings to their defaults?");
|
|
770
808
|
if (ok) {
|