pi-llamacpp-infra 1.2.1 → 1.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -25,6 +25,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
25
25
  - **Multi-machine discovery** — configurable list of servers (host, ports, API key, options); probes all of them at startup and on demand
26
26
  - **Compact model ids** — models appear as `Name (host:port)` in pi's `/model` picker, like a native provider; the raw GGUF path/alias is sent to the server automatically on every request
27
27
  - **Single-model & router modes** — llama.cpp single-model mode (one GGUF per instance) and router mode (multiple models per server, with per-model status and args)
28
+ - **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
28
29
  - **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
29
30
  - **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
30
31
  - **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
@@ -223,6 +224,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
223
224
  | `prefixModelIds` | `true` | Append the machine tag `(host:port)` to model ids; OFF keeps bare names and only disambiguates collisions |
224
225
  | `showBadgesInNames` | `true` | Append 👁️🚀💤 badges to model display names |
225
226
  | `includeUnloadedRouterModels` | `false` | Router mode: list models that are not currently loaded |
227
+ | output cap | `32768` | Registered per model as `maxTokens` unless the server reports an explicit `max_tokens`; still bounded by available context |
228
+ | request timeout | `1200000` | 20 minute timeout floor applied to llamacpp-infra OpenAI-compatible streams |
226
229
  | `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
227
230
  | `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
228
231
  | `metricsPollMs` | `5000` | How often `/metrics` is fetched |
@@ -247,20 +250,20 @@ llama.cpp-family models are registered as reasoning models, exactly like a nativ
247
250
 
248
251
  ## Live Speed & Metrics (footer)
249
252
 
250
- When enabled, the speed reading appears in the footer's status line (no extra terminal row) whenever the active model is from llamacpp-infra, updating constantly while tokens flow:
253
+ When enabled, the speed reading appears in the footer's status line (no extra terminal row) whenever the active model is from llamacpp-infra, updating constantly while tokens flow. Both entries are kept ultra-compact so they coexist with other extensions on pi's single status line (which truncates from the end):
251
254
 
252
255
  ```
253
- 🦙 12 models · 3/3 ✓ 📊 · ⚡ prefill… (before the first token)
254
- 🦙 12 models · 3/3 ✓ 📊 · 🔥 38.1 t/s · ⚡ 420 · 1.2k tok (while streaming)
255
- 🦙 12 models · 3/3 ✓ 📊 · 🔥 38.1 t/s · ⚡ 420 · 1.2k tok (just after the answer ends)
256
- 🦙 12 models · 3/3 ✓ 📊 · ⏸ idle (between turns)
257
- 🦙 12 models · 3/3 ✓ 📊 · ▶ 2 · ⚡ 150 · 🔥 18.0 · server (pi idle, server busy for other clients)
256
+ 🦙(12) ⚡… (before the first token)
257
+ 🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (while streaming)
258
+ 🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (just after the answer ends)
259
+ 🦙(12) ⏸ (between turns)
260
+ 🦙(12) ▶2 ⚡ 150 t/s 🔥 18.0 t/s (pi idle, server busy for other clients)
258
261
  ```
259
262
 
260
- (The `🦙 …` prefix is the extension's model-count status; both live on the same footer line, so no extra row is consumed.)
263
+ (`🦙(n)` is the extension's model-count status; both live on the same footer line, so no extra row is consumed.)
261
264
 
262
265
  - **Client measurement (always, no `--metrics` needed)** — prefill speed = `prompt tokens ÷ (request → first token)` (pi's `usage.input`, OpenAI-style `prompt_tokens` as fallback); generation speed = a moving 1.5 s window over per-token arrival samples. Updated ~every 100 ms while a stream is live (throttled, and unchanged text is skipped, so the footer never churns).
263
- - **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown; when the server is idle, the plain `⏸ idle` reading returns.
266
+ - **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown (`▶n`); when the server is idle, the plain `⏸` reading returns.
264
267
 
265
268
  ## Architecture
266
269
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llamacpp-infra",
3
- "version": "1.2.1",
3
+ "version": "1.2.3",
4
4
  "description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
5
5
  "keywords": [
6
6
  "pi-package",
package/src/core.ts CHANGED
@@ -21,6 +21,8 @@ export const LEGACY_CONFIG_FILE = "local-models.json";
21
21
  export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
22
22
  export const DEFAULT_API_KEY = "no-auth";
23
23
  export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
24
+ export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
25
+ export const DEFAULT_PROVIDER_TIMEOUT_MS = 20 * 60 * 1000;
24
26
 
25
27
  // ── Settings defaults ───────────────────────────────────────────────────────
26
28
  export const DEFAULT_SETTINGS: SettingsConfig = {
@@ -28,6 +30,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
28
30
  pollIntervalMs: 4000,
29
31
  pollMaxMs: 90_000,
30
32
  startupGraceMs: 40_000,
33
+ maxOutputTokens: DEFAULT_MAX_OUTPUT_TOKENS,
34
+ requestTimeoutMs: DEFAULT_PROVIDER_TIMEOUT_MS,
31
35
  knownGoodFailLimit: 3,
32
36
  detectVision: true,
33
37
  prefixModelIds: true,
package/src/index.ts CHANGED
@@ -20,6 +20,7 @@ import {
20
20
  shared,
21
21
  supportsThinkingBudget,
22
22
  } from "./core.ts";
23
+ import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
23
24
 
24
25
  export default function (pi: ExtensionAPI) {
25
26
  const config = loadConfig();
@@ -120,6 +121,7 @@ export default function (pi: ExtensionAPI) {
120
121
  baseUrl: `http://${first?.host ?? "127.0.0.1"}:${first?.ports[0] ?? 8080}/v1`,
121
122
  apiKey: first?.apiKey || "no-auth",
122
123
  api: "openai-completions",
124
+ streamSimple: createLongTimeoutOpenAICompletionsStream,
123
125
  models: [],
124
126
  });
125
127
  providerIsEmpty = true;
@@ -225,7 +227,7 @@ export default function (pi: ExtensionAPI) {
225
227
 
226
228
  async function rescan(ctx?: ExtensionContext) {
227
229
  if (!extensionActive) return;
228
- if (ctxHasUI(ctx)) ctx.ui.setStatus(STATUS_KEY, "🔎 scanning…");
230
+ if (ctxHasUI(ctx)) ctx.ui.setStatus(STATUS_KEY, "🔎");
229
231
  stopPolling();
230
232
  registerEmptyProvider();
231
233
  const r = await discoverAndRegister();
@@ -237,11 +239,9 @@ export default function (pi: ExtensionAPI) {
237
239
  function updateStatusFooter(ctx?: ExtensionContext) {
238
240
  if (!ctxHasUI(ctx)) return;
239
241
  if (shared.registeredCount > 0) {
240
- const up = shared.lastScan?.serversUp ?? 0;
241
- const total = shared.lastScan?.serversTotal ?? 0;
242
- ctx.ui.setStatus(STATUS_KEY, `🦙 ${shared.registeredCount} models · ${up}/${total} ✓`);
242
+ ctx.ui.setStatus(STATUS_KEY, `🦙(${shared.registeredCount})`);
243
243
  } else if (shared.lastScan?.endpoints.some((e) => e.loading)) {
244
- ctx.ui.setStatus(STATUS_KEY, "⏳ loading…");
244
+ ctx.ui.setStatus(STATUS_KEY, "⏳");
245
245
  } else if (shared.lastError) {
246
246
  ctx.ui.setStatus(STATUS_KEY, "⚠️");
247
247
  } else {
@@ -21,6 +21,7 @@ import type {
21
21
  ScanResult,
22
22
  } from "./types.ts";
23
23
  import { cleanModelName, lmStudioContextLength } from "./scan.ts";
24
+ import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
24
25
 
25
26
  /** Compact badge string appended to display names when enabled. */
26
27
  function badgeSuffix(
@@ -73,7 +74,7 @@ function toPiModel(
73
74
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
74
75
  reasoning: false,
75
76
  contextWindow,
76
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
77
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
77
78
  compat: makeCompat(kind),
78
79
  };
79
80
  }
@@ -92,7 +93,7 @@ function toPiModel(
92
93
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
93
94
  reasoning: ep.props?.capabilities?.reasoning_supported ?? true,
94
95
  contextWindow,
95
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
96
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
96
97
  compat: makeCompat(kind),
97
98
  };
98
99
  }
@@ -118,7 +119,7 @@ function toPiModel(
118
119
  name: `${displayName}${badgeSuffix(modelMeta, settings.showBadgesInNames)}`,
119
120
  reasoning: isLlamaFamily,
120
121
  contextWindow,
121
- maxTokens: model.max_tokens ?? Math.min(contextWindow, 8192),
122
+ maxTokens: model.max_tokens ?? Math.min(contextWindow, settings.maxOutputTokens),
122
123
  compat: makeCompat(kind),
123
124
  };
124
125
  }
@@ -232,6 +233,7 @@ export function buildAndRegisterProvider(pi: ExtensionAPI, scan: ScanResult, con
232
233
  baseUrl: defaultBaseUrl,
233
234
  apiKey: defaultApiKey,
234
235
  api: "openai-completions",
236
+ streamSimple: createLongTimeoutOpenAICompletionsStream,
235
237
  models: piModels,
236
238
  });
237
239
 
package/src/runtime.ts ADDED
@@ -0,0 +1,131 @@
1
+ // Runtime request defaults for the OpenAI-compatible llama.cpp provider.
2
+ // pi owns the normal SDK stream options, but this provider is for slow local
3
+ // models where long generations are expected. Keep a local floor so llamacpp-
4
+ // infra models don't inherit too-small global/default request timeouts.
5
+
6
+ import { dirname, join } from "node:path";
7
+ import { fileURLToPath, pathToFileURL } from "node:url";
8
+ import { DEFAULT_PROVIDER_TIMEOUT_MS, shared } from "./core.ts";
9
+
10
+ type AssistantMessageEvent = any;
11
+ type AssistantMessage = any;
12
+ type StreamOptions = Record<string, any> | undefined;
13
+ type StreamSimple = (model: any, context: any, options?: Record<string, any>) => AsyncIterable<AssistantMessageEvent> & { result?: () => Promise<AssistantMessage> };
14
+
15
+ class ForwardedAssistantMessageEventStream implements AsyncIterable<AssistantMessageEvent> {
16
+ private queue: AssistantMessageEvent[] = [];
17
+ private waiting: Array<(result: IteratorResult<AssistantMessageEvent>) => void> = [];
18
+ private done = false;
19
+ private resolveFinal!: (value: AssistantMessage) => void;
20
+ private readonly finalResult = new Promise<AssistantMessage>((resolve) => {
21
+ this.resolveFinal = resolve;
22
+ });
23
+
24
+ push(event: AssistantMessageEvent): void {
25
+ if (this.done) return;
26
+ if (event?.type === "done") {
27
+ this.done = true;
28
+ this.resolveFinal(event.message);
29
+ } else if (event?.type === "error") {
30
+ this.done = true;
31
+ this.resolveFinal(event.error);
32
+ }
33
+
34
+ const waiter = this.waiting.shift();
35
+ if (waiter) waiter({ value: event, done: false });
36
+ else this.queue.push(event);
37
+ }
38
+
39
+ end(result?: AssistantMessage): void {
40
+ this.done = true;
41
+ if (result !== undefined) this.resolveFinal(result);
42
+ while (this.waiting.length > 0) this.waiting.shift()?.({ value: undefined, done: true });
43
+ }
44
+
45
+ async *[Symbol.asyncIterator](): AsyncIterator<AssistantMessageEvent> {
46
+ for (;;) {
47
+ if (this.queue.length > 0) {
48
+ yield this.queue.shift();
49
+ } else if (this.done) {
50
+ return;
51
+ } else {
52
+ const next = await new Promise<IteratorResult<AssistantMessageEvent>>((resolve) => this.waiting.push(resolve));
53
+ if (next.done) return;
54
+ yield next.value;
55
+ }
56
+ }
57
+ }
58
+
59
+ result(): Promise<AssistantMessage> {
60
+ return this.finalResult;
61
+ }
62
+ }
63
+
64
+ let openAICompletionsStreamPromise: Promise<StreamSimple> | undefined;
65
+
66
+ async function loadOpenAICompletionsStreamSimple(): Promise<StreamSimple> {
67
+ if (!openAICompletionsStreamPromise) {
68
+ openAICompletionsStreamPromise = (async () => {
69
+ try {
70
+ const mod = await import("@earendil-works/pi-ai/api/openai-completions");
71
+ return (mod as { streamSimple: StreamSimple }).streamSimple;
72
+ } catch {
73
+ // In pi package installs, pi-ai may be nested under pi-coding-agent
74
+ // instead of hoisted as a top-level dependency of this extension.
75
+ const piIndexUrl = import.meta.resolve("@earendil-works/pi-coding-agent");
76
+ const piPackageDir = dirname(dirname(fileURLToPath(piIndexUrl)));
77
+ const nestedModule = join(piPackageDir, "node_modules", "@earendil-works", "pi-ai", "dist", "api", "openai-completions.js");
78
+ const mod = await import(pathToFileURL(nestedModule).href);
79
+ return (mod as { streamSimple: StreamSimple }).streamSimple;
80
+ }
81
+ })();
82
+ }
83
+ return openAICompletionsStreamPromise;
84
+ }
85
+
86
+ export function activeRequestTimeoutMs(): number {
87
+ const configured = shared.activeConfig?.settings?.requestTimeoutMs;
88
+ return typeof configured === "number" && Number.isFinite(configured) && configured > 0
89
+ ? configured
90
+ : DEFAULT_PROVIDER_TIMEOUT_MS;
91
+ }
92
+
93
+ export function withLocalRuntimeDefaults(options: StreamOptions, floorMs?: number): Record<string, any> {
94
+ const floor = typeof floorMs === "number" && Number.isFinite(floorMs) && floorMs > 0 ? floorMs : DEFAULT_PROVIDER_TIMEOUT_MS;
95
+ const currentTimeout = typeof options?.timeoutMs === "number" && Number.isFinite(options.timeoutMs) ? options.timeoutMs : undefined;
96
+ return {
97
+ ...(options ?? {}),
98
+ timeoutMs: currentTimeout === undefined ? floor : Math.max(currentTimeout, floor),
99
+ };
100
+ }
101
+
102
+ export function createLongTimeoutOpenAICompletionsStream(model: any, context: any, options?: Record<string, any>) {
103
+ const out = new ForwardedAssistantMessageEventStream();
104
+ void (async () => {
105
+ try {
106
+ const streamSimple = await loadOpenAICompletionsStreamSimple();
107
+ const inner = streamSimple(model, context, withLocalRuntimeDefaults(options, activeRequestTimeoutMs()));
108
+ for await (const event of inner) out.push(event);
109
+ if (typeof inner.result === "function") out.end(await inner.result());
110
+ else out.end();
111
+ } catch (err) {
112
+ const message = err instanceof Error ? err.message : String(err);
113
+ out.push({
114
+ type: "error",
115
+ reason: "error",
116
+ error: {
117
+ role: "assistant",
118
+ content: [],
119
+ api: model?.api ?? "openai-completions",
120
+ provider: model?.provider ?? "llamacpp-infra",
121
+ model: model?.id ?? "unknown",
122
+ usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, totalTokens: 0, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
123
+ stopReason: "error",
124
+ errorMessage: message,
125
+ timestamp: Date.now(),
126
+ },
127
+ });
128
+ }
129
+ })();
130
+ return out;
131
+ }
package/src/speed.ts CHANGED
@@ -6,10 +6,15 @@
6
6
  // prefill t/s = prompt tokens / (before_provider_request → first token)
7
7
  // gen t/s = moving window over per-token arrival samples
8
8
  //
9
+ // Footer format (kept ultra-compact so it coexists with other extensions on
10
+ // pi's single status line, which is truncated at the end):
11
+ //
12
+ // ⚡ {prefill} t/s 🔥 {gen} t/s
13
+ //
9
14
  // The server-side /metrics polling (metrics.ts) only supplements this:
10
15
  // when the client is idle but the server reports other clients processing,
11
- // their rate is shown. The client measurement works even when the server has
12
- // no --metrics endpoint.
16
+ // their rate is shown as `▶{n} ⚡ … t/s 🔥 … t/s`. The client measurement
17
+ // works even when the server has no --metrics endpoint.
13
18
 
14
19
  import { debugLog, METRICS_STATUS_KEY } from "./core.ts";
15
20
  import type { AssistantMessageEvent, ExtensionContext, ServerMetricsState, ThemeFg } from "./types.ts";
@@ -126,58 +131,58 @@ export function createSpeedTracker(deps: SpeedDeps) {
126
131
  return tokens / (span / 1000);
127
132
  }
128
133
 
129
- function buildLine(ctx: ExtensionContext | undefined, now: number): string[] {
134
+ function buildLine(ctx: ExtensionContext | undefined, now: number): string {
130
135
  const f = fg(ctx);
131
- const parts: string[] = [f("accent", "📊")];
136
+ const parts: string[] = [];
132
137
 
133
138
  switch (state) {
134
139
  case "prefill":
135
- parts.push(f("warning", "⚡ prefill…"));
140
+ parts.push(f("warning", "⚡…"));
136
141
  break;
137
142
 
138
143
  case "streaming": {
144
+ // Prefill rate of the previous call (only known once it ended).
145
+ if (lastPrefillTps !== undefined) {
146
+ parts.push(f("muted", `⚡ ${Math.round(lastPrefillTps)} t/s`));
147
+ }
139
148
  const gen = genRateAt(now);
140
149
  if (gen !== undefined) {
141
150
  lastGenTps = gen;
142
151
  parts.push(f(rateColor(gen, "gen"), `🔥 ${formatRate(gen)} t/s`));
143
152
  }
144
- if (lastPrefillTps !== undefined) {
145
- parts.push(f("muted", `⚡ ${Math.round(lastPrefillTps)}`));
146
- }
147
- parts.push(f("muted", `${tokenCount} tok`));
148
153
  break;
149
154
  }
150
155
 
151
156
  case "done": {
152
- if (lastGenTps !== undefined) {
153
- parts.push(f("muted", `🔥 ${formatRate(lastGenTps)} t/s`));
154
- }
155
157
  if (lastPrefillTps !== undefined) {
156
- parts.push(f("muted", `⚡ ${Math.round(lastPrefillTps)}`));
158
+ parts.push(f(rateColor(lastPrefillTps, "prefill"), `⚡ ${Math.round(lastPrefillTps)} t/s`));
159
+ }
160
+ if (lastGenTps !== undefined) {
161
+ parts.push(f(rateColor(lastGenTps, "gen"), `🔥 ${formatRate(lastGenTps)} t/s`));
157
162
  }
158
- parts.push(f("muted", `${tokenCount} tok`));
159
163
  break;
160
164
  }
161
165
 
162
166
  case "idle": {
163
167
  // Server supplement: this endpoint is busy for *other* clients.
164
168
  if (server && server.processing > 0) {
165
- parts.push(f("success", `▶ ${server.processing}`));
169
+ parts.push(f("success", `▶${server.processing}`));
166
170
  if (server.promptTps !== undefined && server.promptTps > 0) {
167
- parts.push(f(rateColor(server.promptTps, "prefill"), `⚡ ${Math.round(server.promptTps)}`));
171
+ parts.push(f(rateColor(server.promptTps, "prefill"), `⚡ ${Math.round(server.promptTps)} t/s`));
168
172
  }
169
173
  if (server.genTps !== undefined && server.genTps > 0) {
170
- parts.push(f(rateColor(server.genTps, "gen"), `🔥 ${formatRate(server.genTps)}`));
174
+ parts.push(f(rateColor(server.genTps, "gen"), `🔥 ${formatRate(server.genTps)} t/s`));
171
175
  }
172
- parts.push(f("muted", "server"));
173
176
  } else {
174
- parts.push(f("muted", "⏸ idle"));
177
+ parts.push(f("muted", "⏸"));
175
178
  }
176
179
  break;
177
180
  }
178
181
  }
179
182
 
180
- return [parts.join(" · ")];
183
+ // A stream is live but no rate is computable yet (first ~300 ms of a call).
184
+ if (parts.length === 0) parts.push(f("muted", "🔥…"));
185
+ return parts.join(" ");
181
186
  }
182
187
 
183
188
  function render(ctx: ExtensionContext | undefined, now: number, force = false): void {
@@ -186,7 +191,7 @@ export function createSpeedTracker(deps: SpeedDeps) {
186
191
  return;
187
192
  }
188
193
  if (!deps.hasUI(ctx) || !deps.isOurs(ctx)) return;
189
- const line = buildLine(ctx, now).join(" · ");
194
+ const line = buildLine(ctx, now);
190
195
  if (!force && now - lastRenderAt < RENDER_THROTTLE_MS) return;
191
196
  lastRenderAt = now;
192
197
  // No visual change → skip (avoids a full UI re-render on the footer).
package/src/types.ts CHANGED
@@ -31,6 +31,8 @@ export interface SettingsConfig {
31
31
  pollIntervalMs: number;
32
32
  pollMaxMs: number;
33
33
  startupGraceMs: number;
34
+ maxOutputTokens: number;
35
+ requestTimeoutMs: number;
34
36
  knownGoodFailLimit: number;
35
37
  detectVision: boolean;
36
38
  prefixModelIds: boolean;
package/src/ui.ts CHANGED
@@ -64,6 +64,15 @@ function formatCtx(tokens: number | undefined): string {
64
64
  return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : `${tokens}`;
65
65
  }
66
66
 
67
+ function formatTokens(n: number): string {
68
+ if (!Number.isFinite(n) || n <= 0) return "0";
69
+ if (n >= 1024) {
70
+ const k = n / 1024;
71
+ return Number.isInteger(k) ? `${k}k` : `${k.toFixed(1)}k`;
72
+ }
73
+ return `${n}`;
74
+ }
75
+
67
76
  function metadataBadges(m: ModelMetadata | undefined): string {
68
77
  if (!m) return "";
69
78
  const parts: string[] = [];
@@ -661,6 +670,16 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
661
670
  label: `💤 Include unloaded router models: ${s.includeUnloadedRouterModels ? "ON" : "OFF"}`,
662
671
  description: "router mode: list models not currently loaded",
663
672
  },
673
+ {
674
+ value: "maxtokens",
675
+ label: `📏 Max output tokens: ${formatTokens(s.maxOutputTokens)}`,
676
+ description: "per-model ceiling for outgoing max_tokens requests",
677
+ },
678
+ {
679
+ value: "timeoutfloor",
680
+ label: `⏱️ Request timeout: ${formatMs(s.requestTimeoutMs)}`,
681
+ description: "lower bound on OpenAI-completions stream idle timeout",
682
+ },
664
683
  {
665
684
  value: "warmup",
666
685
  label: `☕ Header warmup: ${s.warmup ? "ON" : "OFF"}`,
@@ -765,6 +784,25 @@ async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<vo
765
784
  saveConfig(config);
766
785
  ctx.ui.notify(`☕ Header warmup ${config.settings.warmup ? "ON" : "OFF"}`, "info");
767
786
  break;
787
+ case "maxtokens": {
788
+ const v = await pickNumber("📏 Max output tokens", [4_096, 8_192, 16_384, 24_576, 32_768, 49_152, 65_536]);
789
+ if (v !== undefined) {
790
+ config.settings.maxOutputTokens = v;
791
+ saveConfig(config);
792
+ ctx.ui.notify(`📏 Max output tokens: ${v.toLocaleString()}`, "info");
793
+ await deps.rescan(ctx);
794
+ }
795
+ break;
796
+ }
797
+ case "timeoutfloor": {
798
+ const v = await pickNumber("⏱️ Request timeout", [60_000, 300_000, 600_000, 1_200_000, 1_800_000, 3_600_000]);
799
+ if (v !== undefined) {
800
+ config.settings.requestTimeoutMs = v;
801
+ saveConfig(config);
802
+ ctx.ui.notify(`⏱️ Request timeout: ${formatMs(v)}`, "info");
803
+ }
804
+ break;
805
+ }
768
806
  case "reset": {
769
807
  const ok = await ctx.ui.confirm("♻️ Reset settings", "Restore all discovery settings to their defaults?");
770
808
  if (ok) {