pi-llamacpp-infra 1.2.5 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -28,6 +28,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
28
28
  - **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
29
29
  - **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
30
30
  - **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
31
+ - **Energy cost (💰)** — local models report no per-token cost, so llamacpp-infra estimates the **electricity** the inference consumed: configure each machine's power draw (kW) and tariff (per kWh) once, and every assistant message carries a realistic `usage.cost.total` (kW × €/kWh × measured request time) — pi's native cost footer, session stats and any consumer reading usage (e.g. trimegisto agents) all show it. Currency selector: USD / EUR / GBP / CNY
31
32
  - **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
32
33
  - **Header warmup** — pre-caches the system prompt KV on llama.cpp-family servers so the first real request is faster
33
34
  - **LM Studio support** — uses LM Studio's OpenAI-compatible `/v1` API, enriches names/context/quant/vision from `/api/v1/models` (or legacy `/api/v0/models`), and avoids llama.cpp-only request fields
@@ -160,7 +161,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
160
161
  "label": "Local",
161
162
  "ports": [8000, 8001, 8002, 8080, 8081, 8082, 1234],
162
163
  "enabled": true,
163
- "probeDs4": false
164
+ "probeDs4": false,
165
+ "costProfile": { "kW": 0.15, "ratePerKwh": 0.21, "label": "bruma 27B" }
164
166
  },
165
167
  {
166
168
  "id": "myserver",
@@ -184,7 +186,9 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
184
186
  "includeUnloadedRouterModels": false,
185
187
  "warmup": true,
186
188
  "metricsEnabled": true,
187
- "metricsPollMs": 5000
189
+ "metricsPollMs": 5000,
190
+ "currency": "eur",
191
+ "costTracking": true
188
192
  },
189
193
  "modelOptions": {
190
194
  "Qwen3.6-27B (myserver:8080)": {
@@ -210,6 +214,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
210
214
  | `enabled` | `true` | Whether to probe this server |
211
215
  | `probeDs4` | `false` | Opt-in: ping `/v1/chat/completions` for DwarfStar/ds4 servers |
212
216
  | `apiKey` | — | Optional bearer token sent on discovery and per-model requests |
217
+ | `costProfile` | — | `{ kW, ratePerKwh }` energy-cost profile for this machine; enables 💰 estimation (see [Energy Cost](#energy-cost-)) |
213
218
 
214
219
  ### Settings
215
220
 
@@ -229,6 +234,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
229
234
  | `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
230
235
  | `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
231
236
  | `metricsPollMs` | `5000` | How often `/metrics` is fetched |
237
+ | `currency` | `eur` | Display currency for energy costs: `usd` / `eur` / `gbp` / `cny` |
238
+ | `costTracking` | `true` | Accumulate 💰 energy cost and inject it into `usage.cost.total` |
232
239
 
233
240
  ### Thinking budgets
234
241
 
@@ -258,6 +265,7 @@ When enabled, the speed reading appears in the footer's status line (no extra te
258
265
  🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (just after the answer ends)
259
266
  🦙(12) ⏸ (between turns)
260
267
  🦙(12) ▶2 ⚡ 150 t/s 🔥 18.0 t/s (pi idle, server busy for other clients)
268
+ 🦙(12) ⏸ 💰3.2c (energy cost of this session, after a turn)
261
269
  ```
262
270
 
263
271
  (`🦙(n)` is the extension's model-count status; both live on the same footer line, so no extra row is consumed.)
@@ -265,6 +273,41 @@ When enabled, the speed reading appears in the footer's status line (no extra te
265
273
  - **Client measurement (always, no `--metrics` needed)** — prefill speed = `prompt tokens ÷ (request → first token)` (pi's `usage.input`, OpenAI-style `prompt_tokens` as fallback); generation speed = a moving 1.5 s window over per-token arrival samples. Updated ~every 100 ms while a stream is live (throttled, and unchanged text is skipped, so the footer never churns).
266
274
  - **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown (`▶n`); when the server is idle, the plain `⏸` reading returns.
267
275
 
276
+ ## Energy Cost (💰)
277
+
278
+ Cloud providers report their own per-message cost, so pi's native cost footer is always right for them. Local llama.cpp-family servers report nothing — so llamacpp-infra estimates the **electricity** the inference consumed and feeds it into the standard usage pipeline:
279
+
280
+ ```
281
+ cost = (requestMs / 3_600_000) × kW × tariff
282
+ ```
283
+
284
+ Every assistant message therefore carries a realistic `usage.cost.total`: pi's own cost display/session stats show it, and any consumer that reads usage (e.g. **trimegisto** sub-agents, which accumulate `usage.cost.total` per message into their dashboard) gets it for free — no per-tool cost logic needed anywhere else.
285
+
286
+ ### Setup
287
+
288
+ ```
289
+ /llamacpp-infra config → 💰 Energy cost
290
+ ```
291
+
292
+ 1. **Currency** — USD, EUR, GBP or CNY. It only changes the display unit; tariffs are stored per kWh in that currency.
293
+ 2. **Per-server power draw & tariff** — the power draw belongs to the *machine*, so every server (machine) gets one profile: `kW` during inference (e.g. `0.15` for 150 W — GPU TDP + idle draw is a good approximation) and the electricity price per kWh. All models served by that machine inherit it.
294
+
295
+ Once set, each provider request is timed (`before_provider_request` → assistant `message_end`, partial/aborted requests included) and charged. The footer shows the session total as `💰` (e.g. `💰3.2c` = 3.2 euro-cents; `¢`/`c`/`p`/`分` per currency); enable/disable anytime from the same menu.
296
+
297
+ > **Parallel agents & accuracy** — when several clients hammer the same server in parallel, each client's wall time is the server time it actually consumed (decode is interleaved), so the session total approximates the machine's inference energy. Good for cost visibility; not a metering-grade measurement.
298
+
299
+ Config lives in `~/.pi/agent/llamacpp-infra.json`:
300
+
301
+ ```json
302
+ {
303
+ "servers": [
304
+ { "id": "local", "host": "127.0.0.1", "ports": [8000, 8080, 8081], "enabled": true,
305
+ "costProfile": { "kW": 0.15, "ratePerKwh": 0.21, "label": "bruma 27B" } }
306
+ ],
307
+ "settings": { "currency": "eur", "costTracking": true }
308
+ }
309
+ ```
310
+
268
311
  ## Architecture
269
312
 
270
313
  ```
@@ -280,6 +323,8 @@ llamacpp-infra/
280
323
  ├── registration.ts # Scan → pi-model mapping + provider registration (lazy).
281
324
  ├── metrics.ts # Server /metrics poller → ServerMetricsState (lazy; only if `metricsEnabled`).
282
325
  ├── speed.ts # Client-side speed tracker + footer status line (lazy; only if `metricsEnabled`).
326
+ ├── cost.ts # Cost profiles, currency formatting, energy math (static import; tiny).
327
+ ├── cost-tracker.ts # Energy-cost tracker: times requests, accumulates, feeds usage.cost (static).
283
328
  ├── ui.ts # /llamacpp-infra subcommands, menus, status, help (lazy).
284
329
  └── prompt-warmup.ts # Header warmup: capture + cache system prompt KV (lazy; only if `warmup`).
285
330
  ```
@@ -292,6 +337,7 @@ Module load profile:
292
337
  | `scan.ts` + `registration.ts` | First discovery (dynamic) | ~27 KB |
293
338
  | `prompt-warmup.ts` | Primed at load if `warmup` enabled; not loaded when disabled | ~15 KB; skipped entirely when `warmup` is OFF |
294
339
  | `metrics.ts` + `speed.ts` | Primed at load if `metricsEnabled`; not loaded when disabled | ~18 KB; skipped entirely when `metricsEnabled` is OFF |
340
+ | `cost.ts` + `cost-tracker.ts` | Static at load (tiny; needed in sub-agent processes too, which never fire `session_start`) | ~4 KB |
295
341
  | `ui.ts` | First `/llamacpp-infra …` command (dynamic) | ~32 KB |
296
342
 
297
343
  Zero external npm dependencies (only pi's bundled `@earendil-works/pi-coding-agent` + Node built-ins).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llamacpp-infra",
3
- "version": "1.2.5",
3
+ "version": "1.3.0",
4
4
  "description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
5
5
  "keywords": [
6
6
  "pi-package",
package/src/core.ts CHANGED
@@ -20,6 +20,7 @@ export const CONFIG_FILE = "llamacpp-infra.json";
20
20
  export const MODELS_CACHE_FILE = "llamacpp-infra-models.json";
21
21
  export const LEGACY_CONFIG_FILE = "local-models.json";
22
22
  export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
23
+ export const COST_STATUS_KEY = "llamacpp-infra-cost";
23
24
  export const DEFAULT_API_KEY = "no-auth";
24
25
  export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
25
26
  export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
@@ -41,6 +42,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
41
42
  metricsPollMs: 5000,
42
43
  includeUnloadedRouterModels: false,
43
44
  showBadgesInNames: true,
45
+ currency: "eur",
46
+ costTracking: true,
44
47
  };
45
48
 
46
49
  export const DEFAULT_SERVERS: ServerConfig[] = [
@@ -65,6 +68,50 @@ export function getConfigPath(): string {
65
68
  return join(getAgentDir(), CONFIG_FILE);
66
69
  }
67
70
 
71
+ /** Validate/normalize a raw cost profile from disk. */
72
+ export function parseCostProfile(raw: unknown): import("./types.ts").CostProfile | undefined {
73
+ if (!raw || typeof raw !== "object") return undefined;
74
+ const r = raw as Record<string, unknown>;
75
+ const kW = typeof r.kW === "number" ? r.kW : parseFloat(String(r.kW ?? ""));
76
+ const rate = typeof r.ratePerKwh === "number" ? r.ratePerKwh : parseFloat(String(r.ratePerKwh ?? ""));
77
+ if (!Number.isFinite(kW) || kW <= 0 || !Number.isFinite(rate) || rate <= 0) return undefined;
78
+ const p: import("./types.ts").CostProfile = { kW, ratePerKwh: rate };
79
+ if (typeof r.label === "string" && r.label) p.label = r.label;
80
+ if (typeof r.host === "string" && r.host) p.host = r.host;
81
+ if (typeof r.pattern === "string" && r.pattern) p.pattern = r.pattern;
82
+ return p;
83
+ }
84
+
85
+ const CURRENCY_CODES = new Set(["usd", "eur", "gbp", "cny"]);
86
+
87
+ /** Normalize a raw settings object onto the defaults (unknown keys dropped, bad values repaired). */
88
+ export function normalizeSettings(raw: unknown): SettingsConfig {
89
+ const r = (raw && typeof raw === "object" ? raw : {}) as Record<string, unknown>;
90
+ const s: SettingsConfig = { ...DEFAULT_SETTINGS };
91
+ for (const key of Object.keys(DEFAULT_SETTINGS) as Array<keyof SettingsConfig>) {
92
+ const v = r[key];
93
+ if (typeof v === typeof DEFAULT_SETTINGS[key]) (s as Record<string, unknown>)[key] = v;
94
+ }
95
+ if (typeof r.currency === "string" && CURRENCY_CODES.has(r.currency as string)) s.currency = r.currency as SettingsConfig["currency"];
96
+ return s;
97
+ }
98
+
99
+ function normalizeServer(s: Record<string, unknown>): import("./types.ts").ServerConfig {
100
+ return {
101
+ id: String(s.id ?? "local"),
102
+ host: String(s.host ?? "127.0.0.1"),
103
+ ...(typeof s.label === "string" && s.label ? { label: s.label } : {}),
104
+ ports: Array.isArray(s.ports) ? s.ports.filter((p): p is number => typeof p === "number") : [],
105
+ enabled: typeof s.enabled === "boolean" ? s.enabled : true,
106
+ ...(s.probeDs4 === true ? { probeDs4: true } : {}),
107
+ ...(typeof s.apiKey === "string" && s.apiKey ? { apiKey: s.apiKey } : {}),
108
+ ...(() => {
109
+ const p = parseCostProfile(s.costProfile);
110
+ return p ? { costProfile: p } : {};
111
+ })(),
112
+ };
113
+ }
114
+
68
115
  export function loadConfig(): InfraConfig {
69
116
  const defaults: InfraConfig = {
70
117
  servers: DEFAULT_SERVERS,
@@ -76,8 +123,10 @@ export function loadConfig(): InfraConfig {
76
123
  try {
77
124
  const raw = JSON.parse(readFileSync(path, "utf-8")) as Partial<InfraConfig>;
78
125
  return {
79
- servers: Array.isArray(raw.servers) ? raw.servers : defaults.servers,
80
- settings: { ...DEFAULT_SETTINGS, ...(raw.settings ?? {}) },
126
+ servers: Array.isArray(raw.servers)
127
+ ? raw.servers.map((s) => normalizeServer(s as Record<string, unknown>))
128
+ : defaults.servers,
129
+ settings: normalizeSettings(raw.settings),
81
130
  modelOptions: raw.modelOptions ?? {},
82
131
  };
83
132
  } catch (err) {
@@ -90,8 +139,10 @@ export function loadConfig(): InfraConfig {
90
139
  try {
91
140
  const raw = JSON.parse(readFileSync(legacyPath, "utf-8")) as Partial<InfraConfig>;
92
141
  const migrated: InfraConfig = {
93
- servers: Array.isArray(raw.servers) ? raw.servers : defaults.servers,
94
- settings: { ...DEFAULT_SETTINGS, ...(raw.settings ?? {}) },
142
+ servers: Array.isArray(raw.servers)
143
+ ? raw.servers.map((s) => normalizeServer(s as Record<string, unknown>))
144
+ : defaults.servers,
145
+ settings: normalizeSettings(raw.settings),
95
146
  modelOptions: raw.modelOptions ?? {},
96
147
  };
97
148
  debugLog(`migrated legacy config from ${legacyPath}`);
@@ -0,0 +1,147 @@
1
+ // Energy-cost tracker: measures inference time of the active llamacpp-infra
2
+ // model from pi's stream events and accumulates the electricity cost.
3
+ //
4
+ // cost per request = (requestMs / 3_600_000) × kW × tariff
5
+ //
6
+ // Timing: before_provider_request → message_end (assistant). This covers
7
+ // prefill + decode for one LLM call. Partial requests (errors/aborts) are
8
+ // still charged for the time actually consumed.
9
+ //
10
+ // The tracker exposes the accumulated cost so the footer can show it
11
+ // (⚡3.2¢) and the message_end hook can inject it into usage.cost, which pi
12
+ // then displays in its native cost footer — no trimegisto required.
13
+
14
+ import { COST_STATUS_KEY, debugLog } from "./core.ts";
15
+ import { formatCost } from "./cost.ts";
16
+ import type { CostProfile, Currency, ExtensionContext } from "./types.ts";
17
+
18
+
19
+ export interface CostDeps {
20
+ /** Whether the extension is still active. */
21
+ isActive: () => boolean;
22
+ /** Whether the ctx has a UI. */
23
+ hasUI: (ctx: ExtensionContext | undefined) => boolean;
24
+ /** Whether the current session model belongs to this provider. */
25
+ isOurs: (ctx: ExtensionContext | undefined) => boolean;
26
+ /** Whether cost tracking is enabled in settings. */
27
+ enabled: () => boolean;
28
+ /** Display currency. */
29
+ currency: () => Currency;
30
+ /** Resolve the cost profile for the current model (null = no estimation). */
31
+ profileFor: (ctx: ExtensionContext | undefined) => CostProfile | null;
32
+ }
33
+
34
+ export interface CostSnapshot {
35
+ /** Accumulated cost for the current session (in the configured currency). */
36
+ total: number;
37
+ /** Accumulated inference time in ms. */
38
+ ms: number;
39
+ /** Whether a profile is active (estimation on). */
40
+ active: boolean;
41
+ /** In-flight request start (ms epoch) or 0. */
42
+ inFlight: number;
43
+ }
44
+
45
+ /** Render the accumulated cost as a compact footer suffix, e.g. "💰3.2¢". */
46
+ export function formatCostSuffix(snap: CostSnapshot, currency: Currency): string {
47
+ if (!snap.active || snap.total <= 0) return "";
48
+ return `💰${formatCost(snap.total, currency)}`;
49
+ }
50
+
51
+ /** Push the cost line to pi's status footer (or clear it). */
52
+ function renderCostStatus(ctx: ExtensionContext | undefined, snap: CostSnapshot, currency: Currency): void {
53
+ if (!ctx || !ctx.ui) return;
54
+ const text = formatCostSuffix(snap, currency);
55
+ try {
56
+ ctx.ui.setStatus(COST_STATUS_KEY, text || undefined);
57
+ } catch {
58
+ // ignore
59
+ }
60
+ }
61
+
62
+ export function createCostTracker(deps: CostDeps) {
63
+ let reqStartAt: number | undefined;
64
+ let totalMs = 0;
65
+ let totalCost = 0;
66
+ /** Live ctx for callbacks without one. */
67
+ let lastCtx: ExtensionContext | undefined;
68
+
69
+ function rememberCtx(ctx: ExtensionContext | undefined): void {
70
+ if (ctx && deps.hasUI(ctx)) lastCtx = ctx;
71
+ }
72
+
73
+ function reset(): void {
74
+ reqStartAt = undefined;
75
+ totalMs = 0;
76
+ totalCost = 0;
77
+ }
78
+
79
+ /** Charge a finished (or aborted) request; returns the cost of this request. */
80
+ function chargeRequest(ctx: ExtensionContext | undefined, endAt: number): number {
81
+ if (reqStartAt === undefined) return 0;
82
+ const ms = Math.max(0, endAt - reqStartAt);
83
+ reqStartAt = undefined;
84
+ const profile = deps.profileFor(ctx ?? lastCtx);
85
+ if (!profile || ms <= 0) return 0;
86
+ const cost = (ms / 3_600_000) * profile.kW * profile.ratePerKwh;
87
+ totalMs += ms;
88
+ totalCost += cost;
89
+ debugLog(`cost: +${ms}ms → +${cost.toFixed(6)} (session ${totalCost.toFixed(6)})`);
90
+ return cost;
91
+ }
92
+
93
+ return {
94
+ /** before_provider_request: a new LLM call starts. */
95
+ onRequest(ctx: ExtensionContext | undefined, at: number): void {
96
+ if (!deps.isActive()) return;
97
+ rememberCtx(ctx);
98
+ if (!deps.isOurs(ctx) || !deps.enabled()) {
99
+ reset();
100
+ return;
101
+ }
102
+ // A new request supersedes any uncharged one (shouldn't happen, but be safe).
103
+ if (reqStartAt !== undefined) chargeRequest(ctx, at);
104
+ reqStartAt = at;
105
+ },
106
+
107
+ /**
108
+ * message_end (assistant): finalize the call and charge it.
109
+ * Returns the cost of THIS request (0 when no profile/off), so callers
110
+ * can add just the delta to usage.cost.total of that message.
111
+ */
112
+ onMessageEnd(ctx: ExtensionContext | undefined, message: unknown, at: number): number {
113
+ if (!deps.isActive()) return 0;
114
+ rememberCtx(ctx);
115
+ if (!deps.isOurs(ctx)) return 0;
116
+ const role = (message as { role?: string } | undefined)?.role;
117
+ if (role !== "assistant") return 0;
118
+ const charged = chargeRequest(ctx, at);
119
+ renderCostStatus(ctx ?? lastCtx, this.snapshot(), deps.currency());
120
+ return charged;
121
+ },
122
+
123
+ /** turn_end: nothing more to do (requests are charged at message_end). */
124
+ onTurnEnd(_ctx: ExtensionContext | undefined): void {
125
+ // no-op: kept for API symmetry with the speed tracker
126
+ },
127
+
128
+ /** Current accumulated view. */
129
+ snapshot(): CostSnapshot {
130
+ const profile = deps.profileFor(lastCtx);
131
+ return {
132
+ total: totalCost,
133
+ ms: totalMs,
134
+ active: !!profile && totalCost > 0,
135
+ inFlight: reqStartAt ?? 0,
136
+ };
137
+ },
138
+
139
+ /** Reset the session accumulator (model switch / session start). */
140
+ reset(ctx?: ExtensionContext): void {
141
+ reset();
142
+ renderCostStatus(ctx ?? lastCtx, this.snapshot(), deps.currency());
143
+ },
144
+ };
145
+ }
146
+
147
+ export type CostTracker = ReturnType<typeof createCostTracker>;
package/src/cost.ts ADDED
@@ -0,0 +1,95 @@
1
+ // Energy-cost estimation for local inference.
2
+ //
3
+ // Local servers report no per-token cost, so the honest cost is the
4
+ // electricity the inference consumed:
5
+ //
6
+ // cost = (inferenceMs / 3_600_000) × kW × tariff
7
+ //
8
+ // Profiles are configured per server (the power draw belongs to the machine,
9
+ // not the model). The currency selector (settings.currency) only changes the
10
+ // display unit — the stored tariff is always per kWh in the chosen currency.
11
+
12
+ import type { Currency, CostProfile, ServerConfig } from "./types.ts";
13
+
14
+ // ── Currency display ────────────────────────────────────────────────────────
15
+ export const CURRENCIES: Array<{ code: Currency; symbol: string; cent: string; label: string }> = [
16
+ { code: "usd", symbol: "$", cent: "¢", label: "USD ($)" },
17
+ { code: "eur", symbol: "€", cent: "c", label: "EUR (€)" },
18
+ { code: "gbp", symbol: "£", cent: "p", label: "GBP (£)" },
19
+ { code: "cny", symbol: "¥", cent: "分", label: "CNY (¥)" },
20
+ ];
21
+
22
+ export function currencySymbol(code: Currency | undefined): string {
23
+ return CURRENCIES.find((c) => c.code === code)?.symbol ?? "€";
24
+ }
25
+
26
+ /** Cent/fractional marker of a currency: $→¢, €→c, £→p, ¥→分. */
27
+ export function currencyCent(code: Currency | undefined): string {
28
+ return CURRENCIES.find((c) => c.code === code)?.cent ?? "c";
29
+ }
30
+
31
+ /**
32
+ * Format an amount in the chosen currency, compact:
33
+ * 1.234 → "$1.23" (≥ 1 unit: symbol + 2 decimals)
34
+ * 0.0315 → "3.2¢" (usd) / "3.2c" (eur) / "3.2p" (gbp) / "3.2分" (cny)
35
+ * 0.0004 → "0.4m¢" (sub-cent: milli-units)
36
+ */
37
+ export function formatCost(amount: number, code: Currency | undefined): string {
38
+ const sym = currencySymbol(code);
39
+ if (amount >= 1) return `${sym}${amount.toFixed(2)}`;
40
+ const cent = currencyCent(code);
41
+ if (amount >= 0.001) return `${(amount * 100).toFixed(1)}${cent}`;
42
+ return `${(amount * 1000).toFixed(1)}m${cent}`;
43
+ }
44
+
45
+ // ── Profile resolution ──────────────────────────────────────────────────────
46
+ /**
47
+ * Find the cost profile that applies to a model, or null when none match.
48
+ *
49
+ * A profile is attached to a server (ServerConfig.costProfile): the power
50
+ * draw is a property of the machine, so all models served by that server
51
+ * share it. Resolution order:
52
+ * 1. server match (endpoint.serverId → ServerConfig.id)
53
+ * 2. host fallback: endpoint host equals the server's host (covers scans
54
+ * whose serverId drifted from the config, e.g. cache boot)
55
+ * 3. per-profile `pattern`: substring match on the model id, for machines
56
+ * that serve very different workloads (rare; kept for flexibility)
57
+ */
58
+ export function resolveCostProfile(
59
+ model: { id?: string; endpoint?: { serverId?: string; host?: string } },
60
+ servers: ServerConfig[],
61
+ ): CostProfile | null {
62
+ if (!model?.id) return null;
63
+ const endpoint = model.endpoint;
64
+
65
+ // 1. Server id match (primary).
66
+ if (endpoint?.serverId) {
67
+ const srv = servers.find((s) => s.id === endpoint.serverId);
68
+ if (srv?.costProfile && srv.costProfile.kW > 0) return srv.costProfile;
69
+ }
70
+ // 2. Host fallback.
71
+ if (endpoint?.host) {
72
+ for (const srv of servers) {
73
+ if (srv.costProfile && srv.costProfile.kW > 0 && srv.host === endpoint.host) return srv.costProfile;
74
+ }
75
+ }
76
+ // 3. Model-id pattern (rare; only when explicitly set).
77
+ for (const srv of servers) {
78
+ const p = srv.costProfile;
79
+ if (p && p.kW > 0 && p.pattern && model.id.toLowerCase().includes(p.pattern.toLowerCase())) return p;
80
+ }
81
+ return null;
82
+ }
83
+
84
+ // ── Cost math ───────────────────────────────────────────────────────────────
85
+ /** Energy cost (in the configured currency) of `ms` of inference. */
86
+ export function energyCost(profile: CostProfile, ms: number): number {
87
+ if (!profile || ms <= 0) return 0;
88
+ const kwh = (ms / 3_600_000) * profile.kW;
89
+ return kwh * profile.ratePerKwh;
90
+ }
91
+
92
+ /** Format a profile for display, e.g. "0.15 kW @ 0.21 €/kWh". */
93
+ export function formatProfile(p: CostProfile, code: Currency | undefined): string {
94
+ return `${p.kW} kW @ ${p.ratePerKwh} ${currencySymbol(code)}/kWh`;
95
+ }
package/src/index.ts CHANGED
@@ -23,6 +23,8 @@ import {
23
23
  supportsThinkingBudget,
24
24
  } from "./core.ts";
25
25
  import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
26
+ import { resolveCostProfile } from "./cost.ts";
27
+ import { createCostTracker, type CostTracker } from "./cost-tracker.ts";
26
28
 
27
29
  export default function (pi: ExtensionAPI) {
28
30
  const config = loadConfig();
@@ -77,6 +79,28 @@ export default function (pi: ExtensionAPI) {
77
79
  let metricsApi: ReturnType<MetricsModule["createMetrics"]> | undefined;
78
80
  let speedApi: ReturnType<SpeedModule["createSpeedTracker"]> | undefined;
79
81
 
82
+ /**
83
+ * Energy-cost tracker. Created SYNCHRONOUSLY at factory time (not lazily
84
+ * like the speed tracker) because cost must be measured inside sub-agent
85
+ * processes too (`pi -p --no-session`), which never fire session_start:
86
+ * the message_end hook needs it on the very first LLM call. Gating is done
87
+ * via deps.enabled(), so toggling costTracking ON at runtime works.
88
+ */
89
+ const costApi: CostTracker = createCostTracker({
90
+ isActive: () => extensionActive,
91
+ hasUI: ctxHasUI,
92
+ isOurs: (ctx) => {
93
+ try {
94
+ return ctx?.model?.provider === PROVIDER_NAME;
95
+ } catch {
96
+ return false;
97
+ }
98
+ },
99
+ enabled: () => config.settings.costTracking,
100
+ currency: () => config.settings.currency,
101
+ profileFor: (ctx) => costProfileFor(ctx),
102
+ });
103
+
80
104
  // ── Internal state ────────────────────────────────────────────────────
81
105
  let lastSignature: string | undefined;
82
106
  let pollTimer: ReturnType<typeof setTimeout> | undefined;
@@ -102,6 +126,25 @@ export default function (pi: ExtensionAPI) {
102
126
 
103
127
  const epKey = (host: string, port: number) => `${host}:${port}`;
104
128
 
129
+ /** Cost profile for the model a ctx is currently talking to (null = no estimation). */
130
+ function costProfileFor(ctx: ExtensionContext | undefined): import("./types.ts").CostProfile | null {
131
+ try {
132
+ const m = ctx?.model as { id?: string; provider?: string } | undefined;
133
+ if (!m?.id) return null;
134
+ const baseUrl = shared.modelBaseUrls.get(m.id);
135
+ const ep = shared.lastScan?.endpoints.find((e) => e.baseUrl === baseUrl);
136
+ return resolveCostProfile(
137
+ {
138
+ id: m.id,
139
+ endpoint: ep ? { serverId: ep.serverId, host: ep.host, port: ep.port } : undefined,
140
+ },
141
+ config.servers,
142
+ );
143
+ } catch {
144
+ return null;
145
+ }
146
+ }
147
+
105
148
  function safeSystemPrompt(ctx: { getSystemPrompt?: () => string }): string | undefined {
106
149
  try {
107
150
  return ctx.getSystemPrompt?.();
@@ -309,9 +352,10 @@ export default function (pi: ExtensionAPI) {
309
352
  if (config.settings.metricsEnabled) metricsApi.start(ctx);
310
353
  }
311
354
 
312
- // ── Live speed: per-call prefill timing (before id rewrite) ──────────
355
+ // ── Live speed + energy cost: per-call timing (before id rewrite) ─────
313
356
  pi.on("before_provider_request", (_event, ctx) => {
314
357
  speedApi?.onRequest(ctx, Date.now());
358
+ costApi?.onRequest(ctx, Date.now());
315
359
  return undefined;
316
360
  });
317
361
 
@@ -322,6 +366,21 @@ export default function (pi: ExtensionAPI) {
322
366
 
323
367
  pi.on("message_end", (event, ctx) => {
324
368
  speedApi?.onMessageEnd(ctx, event.message, Date.now());
369
+
370
+ // Inject the energy cost of THIS request into usage.cost.total so pi's
371
+ // native cost footer, session stats and any consumer reading usage
372
+ // (e.g. trimegisto agents, which accumulate usage.cost.total per
373
+ // message) see a realistic cost for local models. message_end fires
374
+ // before persistence, so mutating event.message in place is enough.
375
+ const msg = event.message as { role?: string; usage?: { cost?: { total?: number } } } | undefined;
376
+ if (msg?.role === "assistant" && config.settings.costTracking) {
377
+ const requestCost = costApi?.onMessageEnd(ctx, event.message, Date.now()) ?? 0;
378
+ if (requestCost > 0 && msg.usage?.cost) {
379
+ msg.usage.cost.total = (msg.usage.cost.total ?? 0) + requestCost;
380
+ return { message: event.message };
381
+ }
382
+ }
383
+ return undefined;
325
384
  });
326
385
 
327
386
  pi.on("turn_end", (_event, ctx) => {
@@ -530,10 +589,15 @@ export default function (pi: ExtensionAPI) {
530
589
  w.warmer.warmupForModel(ctx.model, safeSystemPrompt(ctx), ctx.cwd);
531
590
  }
532
591
  currentThinkingLevel = ctx.thinkingLevel;
533
- if (config.settings.metricsEnabled) {
534
- await ensureSpeed().then((s) => s?.start(ctx));
535
- const m = await ensureMetrics();
536
- m.start(ctx);
592
+ try {
593
+ if (config.settings.metricsEnabled) {
594
+ await ensureSpeed().then((s) => s?.start(ctx));
595
+ const m = await ensureMetrics();
596
+ m.start(ctx);
597
+ }
598
+ } catch (err) {
599
+ // Tracking must never break session start (or spam the TUI with traces).
600
+ debugLog(`session_start tracking error ignored: ${err instanceof Error ? err.message : String(err)}`);
537
601
  }
538
602
  if (!ctxHasUI(ctx)) return;
539
603
  void discoverAndRegister()
@@ -561,10 +625,16 @@ export default function (pi: ExtensionAPI) {
561
625
  w.warmer.warmupForModel(event.model, safeSystemPrompt(ctx), ctx.cwd);
562
626
  }
563
627
  metricsApi?.resetForModelSwitch();
564
- if (config.settings.metricsEnabled) {
565
- await ensureSpeed().then((s) => s?.start(ctx));
566
- const m = await ensureMetrics();
567
- m.start(ctx);
628
+ costApi?.reset(ctx);
629
+ try {
630
+ if (config.settings.metricsEnabled) {
631
+ await ensureSpeed().then((s) => s?.start(ctx));
632
+ const m = await ensureMetrics();
633
+ m.start(ctx);
634
+ }
635
+ } catch (err) {
636
+ // Tracking must never break model switches (or spam the TUI with traces).
637
+ debugLog(`model_select tracking error ignored: ${err instanceof Error ? err.message : String(err)}`);
568
638
  }
569
639
  });
570
640
 
package/src/types.ts CHANGED
@@ -2,6 +2,24 @@
2
2
 
3
3
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
4
4
 
5
+ export type Currency = "usd" | "eur" | "gbp" | "cny";
6
+
7
+ /**
8
+ * Energy-cost profile for one machine: power draw during inference (kW) and
9
+ * electricity tariff (per kWh, in the configured currency). Cost of a request
10
+ * = (ms / 3_600_000) × kW × ratePerKwh. `host`/`pattern` refine matching when
11
+ * needed (they are optional; the serverId association is the primary key).
12
+ */
13
+ export interface CostProfile {
14
+ label?: string;
15
+ /** Power draw during inference, in kilowatts. 0/absent = estimation off. */
16
+ kW: number;
17
+ /** Electricity tariff per kWh (in the configured currency). */
18
+ ratePerKwh: number;
19
+ host?: string;
20
+ pattern?: string;
21
+ }
22
+
5
23
  /** One machine that may serve llama.cpp-family models on one or more ports. */
6
24
  export interface ServerConfig {
7
25
  id: string;
@@ -11,6 +29,8 @@ export interface ServerConfig {
11
29
  enabled: boolean;
12
30
  probeDs4?: boolean;
13
31
  apiKey?: string;
32
+ /** Energy-cost profile for this machine (kW + tariff). */
33
+ costProfile?: CostProfile;
14
34
  }
15
35
 
16
36
  /** Thinking budget (tokens) per pi thinking level, llama.cpp-style. */
@@ -41,6 +61,10 @@ export interface SettingsConfig {
41
61
  metricsPollMs: number;
42
62
  includeUnloadedRouterModels: boolean;
43
63
  showBadgesInNames: boolean;
64
+ /** Display currency for energy costs (default: eur). */
65
+ currency: Currency;
66
+ /** Whether the footer shows accumulated energy cost (default: true). */
67
+ costTracking: boolean;
44
68
  }
45
69
 
46
70
  export interface InfraConfig {
package/src/ui.ts CHANGED
@@ -19,6 +19,7 @@ import type {
19
19
  ThinkingBudgets,
20
20
  } from "./types.ts";
21
21
  import { fetchModelsFromEndpoint, scanLocalServers } from "./scan.ts";
22
+ import { CURRENCIES, formatProfile, currencySymbol } from "./cost.ts";
22
23
 
23
24
  // ── Small UI helpers ───────────────────────────────────────────────────────
24
25
  async function selectFrom<T>(
@@ -221,6 +222,11 @@ export async function showConfigMenu(ctx: ExtensionContext, deps: UiDeps): Promi
221
222
  { value: "models", label: "📋 Discovered models", description: `${shared.registeredCount} currently registered` },
222
223
  { value: "test", label: "🧪 Test connectivity", description: "probe every endpoint and show latency" },
223
224
  { value: "budgets", label: "🧠 Thinking budgets", description: `${budgetCount} model(s) with budgets` },
225
+ {
226
+ value: "cost",
227
+ label: `💰 Energy cost: ${config.settings.costTracking ? "ON" : "OFF"}`,
228
+ description: `currency ${currencySymbol(config.settings.currency)} · ${config.servers.filter((s) => s.costProfile && s.costProfile.kW > 0).length} server(s) with kW`,
229
+ },
224
230
  {
225
231
  value: "metrics",
226
232
  label: `📈 Live speed & metrics: ${config.settings.metricsEnabled ? "ON" : "OFF"}`,
@@ -248,6 +254,9 @@ export async function showConfigMenu(ctx: ExtensionContext, deps: UiDeps): Promi
248
254
  case "budgets":
249
255
  await showThinkingBudgetsMenu(ctx, deps);
250
256
  break;
257
+ case "cost":
258
+ await showCostMenu(ctx, deps);
259
+ break;
251
260
  case "metrics":
252
261
  await showMetricsMenu(ctx, deps);
253
262
  break;
@@ -619,6 +628,105 @@ async function showMetricsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<voi
619
628
  }
620
629
  }
621
630
 
631
+ // ── Energy-cost menu ────────────────────────────────────────────────────
632
+ async function showCostMenu(ctx: ExtensionContext, deps: UiDeps): Promise<void> {
633
+ const config = shared.activeConfig!;
634
+ for (;;) {
635
+ const withKw = config.servers.filter((s) => s.costProfile && s.costProfile.kW > 0);
636
+ const action = await selectFrom(ctx, `💰 Energy cost (${currencySymbol(config.settings.currency)})`, [
637
+ {
638
+ value: "toggle",
639
+ label: config.settings.costTracking ? "🔴 Disable cost tracking" : "🟢 Enable cost tracking",
640
+ description: "accumulate electricity cost of local inference (footer + usage.cost)",
641
+ },
642
+ {
643
+ value: "currency",
644
+ label: `💱 Currency: ${CURRENCIES.find((c) => c.code === config.settings.currency)?.label ?? config.settings.currency}`,
645
+ description: "display unit for costs and tariffs",
646
+ },
647
+ {
648
+ value: "servers",
649
+ label: "🖥️ Per-server kW / tariff",
650
+ description: `${withKw.length} server(s) configured`,
651
+ },
652
+ { value: "__back", label: "← Back", description: "" },
653
+ ]);
654
+ if (action === undefined || action === "__back") return;
655
+
656
+ if (action === "toggle") {
657
+ config.settings.costTracking = !config.settings.costTracking;
658
+ saveConfig(config);
659
+ deps.updateStatusFooter(ctx);
660
+ ctx.ui.notify(`💰 Cost tracking: ${config.settings.costTracking ? "ON" : "OFF"}`, "info");
661
+ }
662
+ else if (action === "currency") {
663
+ const cur = await selectFrom(ctx, "💱 Select currency", CURRENCIES.map((c) => ({ value: c.code, label: c.label })));
664
+ if (cur) {
665
+ config.settings.currency = cur;
666
+ saveConfig(config);
667
+ ctx.ui.notify(`💱 Currency: ${CURRENCIES.find((c) => c.code === cur)?.label}`, "info");
668
+ }
669
+ }
670
+ else if (action === "servers") {
671
+ await showServerCostMenu(ctx, config);
672
+ }
673
+ }
674
+ }
675
+
676
+ async function showServerCostMenu(ctx: ExtensionContext, config: NonNullable<typeof shared.activeConfig>): Promise<void> {
677
+ for (;;) {
678
+ const items = config.servers.map((srv) => ({
679
+ value: srv.id,
680
+ label: `${srv.costProfile?.kW ? "💰" : "·"} ${serverLabel(srv)}`,
681
+ description: srv.costProfile?.kW ? formatProfile(srv.costProfile, config.settings.currency) : `${srv.host} — no kW set`,
682
+ }));
683
+ items.push({ value: "__back", label: "← Back", description: "" });
684
+ const picked = await selectFrom(ctx, "💰 Per-server energy cost — select a machine", items);
685
+ if (!picked || picked === "__back") return;
686
+ const srv = config.servers.find((s) => s.id === picked);
687
+ if (!srv) return;
688
+
689
+ const p = srv.costProfile;
690
+ const sub = await selectFrom(ctx, `💰 ${serverLabel(srv)} (${srv.host})`, [
691
+ { value: "kW", label: p?.kW ? `⚡ Power draw: ${p.kW} kW` : "⚡ Set power draw (kW)", description: "W consumed during inference (e.g. 0.15 for 150 W)" },
692
+ { value: "rate", label: p?.ratePerKwh ? `🧾 Tariff: ${p.ratePerKwh} ${currencySymbol(config.settings.currency)}/kWh` : `🧾 Set tariff (${currencySymbol(config.settings.currency)}/kWh)`, description: "electricity price per kWh" },
693
+ { value: "label", label: p?.label ? `🏷️ Label: ${p.label}` : "🏷️ Set label", description: "optional, shown in menus" },
694
+ ...(p?.kW ? [{ value: "clear", label: "🗑️ Remove cost profile", description: "stop estimating cost for this machine" }] : []),
695
+ { value: "__back", label: "← Back", description: "" },
696
+ ]);
697
+ if (!sub || sub === "__back") continue;
698
+
699
+ if (sub === "kW") {
700
+ const raw = await ctx.ui.input("⚡ Power draw in kW (e.g. 0.15 = 150 W)", p?.kW ? String(p.kW) : "0.15");
701
+ const kW = parseFloat(raw ?? "");
702
+ if (isNaN(kW) || kW <= 0) { ctx.ui.notify("❌ Invalid kW", "error"); continue; }
703
+ srv.costProfile = { ...(srv.costProfile ?? { ratePerKwh: 0.2 }), kW };
704
+ saveConfig(config);
705
+ ctx.ui.notify(`⚡ ${serverLabel(srv)}: ${kW} kW`, "info");
706
+ }
707
+ else if (sub === "rate") {
708
+ const raw = await ctx.ui.input(`🧾 Tariff in ${currencySymbol(config.settings.currency)}/kWh (e.g. 0.21)`, p?.ratePerKwh ? String(p.ratePerKwh) : "0.21");
709
+ const rate = parseFloat(raw ?? "");
710
+ if (isNaN(rate) || rate <= 0) { ctx.ui.notify("❌ Invalid tariff", "error"); continue; }
711
+ srv.costProfile = { ...(srv.costProfile ?? { kW: 0.15 }), ratePerKwh: rate };
712
+ saveConfig(config);
713
+ ctx.ui.notify(`🧾 ${serverLabel(srv)}: ${rate} ${currencySymbol(config.settings.currency)}/kWh`, "info");
714
+ }
715
+ else if (sub === "label") {
716
+ const raw = await ctx.ui.input("🏷️ Label", p?.label ?? "");
717
+ if (raw !== undefined) {
718
+ srv.costProfile = { ...(srv.costProfile ?? { kW: 0.15, ratePerKwh: 0.21 }), label: raw.trim() || undefined };
719
+ saveConfig(config);
720
+ }
721
+ }
722
+ else if (sub === "clear") {
723
+ delete srv.costProfile;
724
+ saveConfig(config);
725
+ ctx.ui.notify(`🗑️ ${serverLabel(srv)}: cost profile removed`, "info");
726
+ }
727
+ }
728
+ }
729
+
622
730
  // ── Settings menu ──────────────────────────────────────────────────────────
623
731
  async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<void> {
624
732
  const config = shared.activeConfig!;