pi-llamacpp-infra 1.2.5 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -2
- package/package.json +1 -1
- package/src/core.ts +55 -4
- package/src/cost-tracker.ts +147 -0
- package/src/cost.ts +95 -0
- package/src/index.ts +79 -9
- package/src/types.ts +24 -0
- package/src/ui.ts +108 -0
package/README.md
CHANGED
|
@@ -28,6 +28,7 @@ Anything else (vLLM, Ollama, cloud APIs…) is out of scope — use pi's built-i
|
|
|
28
28
|
- **Long local generations** — discovered models are registered with up to **32,768 output tokens** (bounded by the model/server context) and llamacpp-infra OpenAI-compatible requests enforce a **20 minute** timeout floor so slow local runs don't get cut early by pi defaults
|
|
29
29
|
- **Per-model metadata badges** — 👁️ vision (mmproj / modalities), 🚀 drafter (speculative decoding), 🗜️ quant tag from GGUF filename, 🧠 KV cache quantization (from server args or `/proc`)
|
|
30
30
|
- **Live speed & metrics** — a constantly updating footer reading of the active model's prefill (⚡) and generation (🔥) token speed, measured straight from the stream (per token, ~10 updates/s); when pi is idle it also mirrors other clients the server's `/metrics` endpoint reports. Lives in the footer's status line, so no extra terminal row is taken. Works even without `--metrics`
|
|
31
|
+
- **Energy cost (💰)** — local models report no per-token cost, so llamacpp-infra estimates the **electricity** the inference consumed: configure each machine's power draw (kW) and tariff (per kWh) once, and every assistant message carries a realistic `usage.cost.total` (kW × €/kWh × measured request time) — pi's native cost footer, session stats and any consumer reading usage (e.g. trimegisto agents) all show it. Currency selector: USD / EUR / GBP / CNY
|
|
31
32
|
- **Thinking budgets** — llama.cpp accepts `thinking_budget_tokens` per request; configure budgets per thinking level (minimal/low/medium/high/xhigh/max) per model; models with budgets are registered with reasoning enabled
|
|
32
33
|
- **Header warmup** — pre-caches the system prompt KV on llama.cpp-family servers so the first real request is faster
|
|
33
34
|
- **LM Studio support** — uses LM Studio's OpenAI-compatible `/v1` API, enriches names/context/quant/vision from `/api/v1/models` (or legacy `/api/v0/models`), and avoids llama.cpp-only request fields
|
|
@@ -160,7 +161,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
160
161
|
"label": "Local",
|
|
161
162
|
"ports": [8000, 8001, 8002, 8080, 8081, 8082, 1234],
|
|
162
163
|
"enabled": true,
|
|
163
|
-
"probeDs4": false
|
|
164
|
+
"probeDs4": false,
|
|
165
|
+
"costProfile": { "kW": 0.15, "ratePerKwh": 0.21, "label": "bruma 27B" }
|
|
164
166
|
},
|
|
165
167
|
{
|
|
166
168
|
"id": "myserver",
|
|
@@ -184,7 +186,9 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
184
186
|
"includeUnloadedRouterModels": false,
|
|
185
187
|
"warmup": true,
|
|
186
188
|
"metricsEnabled": true,
|
|
187
|
-
"metricsPollMs": 5000
|
|
189
|
+
"metricsPollMs": 5000,
|
|
190
|
+
"currency": "eur",
|
|
191
|
+
"costTracking": true
|
|
188
192
|
},
|
|
189
193
|
"modelOptions": {
|
|
190
194
|
"Qwen3.6-27B (myserver:8080)": {
|
|
@@ -210,6 +214,7 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
210
214
|
| `enabled` | `true` | Whether to probe this server |
|
|
211
215
|
| `probeDs4` | `false` | Opt-in: ping `/v1/chat/completions` for DwarfStar/ds4 servers |
|
|
212
216
|
| `apiKey` | — | Optional bearer token sent on discovery and per-model requests |
|
|
217
|
+
| `costProfile` | — | `{ kW, ratePerKwh }` energy-cost profile for this machine; enables 💰 estimation (see [Energy Cost](#energy-cost-)) |
|
|
213
218
|
|
|
214
219
|
### Settings
|
|
215
220
|
|
|
@@ -229,6 +234,8 @@ Everything is configurable through the UI, but the persisted file is `~/.pi/agen
|
|
|
229
234
|
| `warmup` | `true` | Pre-cache system prompt KV on llama.cpp servers |
|
|
230
235
|
| `metricsEnabled` | `true` | Show live speed & metrics in the footer for llamacpp-infra models |
|
|
231
236
|
| `metricsPollMs` | `5000` | How often `/metrics` is fetched |
|
|
237
|
+
| `currency` | `eur` | Display currency for energy costs: `usd` / `eur` / `gbp` / `cny` |
|
|
238
|
+
| `costTracking` | `true` | Accumulate 💰 energy cost and inject it into `usage.cost.total` |
|
|
232
239
|
|
|
233
240
|
### Thinking budgets
|
|
234
241
|
|
|
@@ -258,6 +265,7 @@ When enabled, the speed reading appears in the footer's status line (no extra te
|
|
|
258
265
|
🦙(12) ⚡ 420 t/s 🔥 38.1 t/s (just after the answer ends)
|
|
259
266
|
🦙(12) ⏸ (between turns)
|
|
260
267
|
🦙(12) ▶2 ⚡ 150 t/s 🔥 18.0 t/s (pi idle, server busy for other clients)
|
|
268
|
+
🦙(12) ⏸ 💰3.2c (energy cost of this session, after a turn)
|
|
261
269
|
```
|
|
262
270
|
|
|
263
271
|
(`🦙(n)` is the extension's model-count status; both live on the same footer line, so no extra row is consumed.)
|
|
@@ -265,6 +273,41 @@ When enabled, the speed reading appears in the footer's status line (no extra te
|
|
|
265
273
|
- **Client measurement (always, no `--metrics` needed)** — prefill speed = `prompt tokens ÷ (request → first token)` (pi's `usage.input`, OpenAI-style `prompt_tokens` as fallback); generation speed = a moving 1.5 s window over per-token arrival samples. Updated ~every 100 ms while a stream is live (throttled, and unchanged text is skipped, so the footer never churns).
|
|
266
274
|
- **Server supplement (only when pi is idle)** — the poller fetches the server's Prometheus `/metrics` endpoint (or JSON `/stats`) every `metricsPollMs` (default 5 s). If the server reports other clients processing, their ⚡/🔥 rates are shown (`▶n`); when the server is idle, the plain `⏸` reading returns.
|
|
267
275
|
|
|
276
|
+
## Energy Cost (💰)
|
|
277
|
+
|
|
278
|
+
Cloud providers report their own per-message cost, so pi's native cost footer is always right for them. Local llama.cpp-family servers report nothing — so llamacpp-infra estimates the **electricity** the inference consumed and feeds it into the standard usage pipeline:
|
|
279
|
+
|
|
280
|
+
```
|
|
281
|
+
cost = (requestMs / 3_600_000) × kW × tariff
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
Every assistant message therefore carries a realistic `usage.cost.total`: pi's own cost display/session stats show it, and any consumer that reads usage (e.g. **trimegisto** sub-agents, which accumulate `usage.cost.total` per message into their dashboard) gets it for free — no per-tool cost logic needed anywhere else.
|
|
285
|
+
|
|
286
|
+
### Setup
|
|
287
|
+
|
|
288
|
+
```
|
|
289
|
+
/llamacpp-infra config → 💰 Energy cost
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
1. **Currency** — USD, EUR, GBP or CNY. It only changes the display unit; tariffs are stored per kWh in that currency.
|
|
293
|
+
2. **Per-server power draw & tariff** — the power draw belongs to the *machine*, so every server (machine) gets one profile: `kW` during inference (e.g. `0.15` for 150 W — GPU TDP + idle draw is a good approximation) and the electricity price per kWh. All models served by that machine inherit it.
|
|
294
|
+
|
|
295
|
+
Once set, each provider request is timed (`before_provider_request` → assistant `message_end`, partial/aborted requests included) and charged. The footer shows the session total as `💰` (e.g. `💰3.2c` = 3.2 euro-cents; `¢`/`c`/`p`/`分` per currency); enable/disable anytime from the same menu.
|
|
296
|
+
|
|
297
|
+
> **Parallel agents & accuracy** — when several clients hammer the same server in parallel, each client's wall time is the server time it actually consumed (decode is interleaved), so the session total approximates the machine's inference energy. Good for cost visibility; not a metering-grade measurement.
|
|
298
|
+
|
|
299
|
+
Config lives in `~/.pi/agent/llamacpp-infra.json`:
|
|
300
|
+
|
|
301
|
+
```json
|
|
302
|
+
{
|
|
303
|
+
"servers": [
|
|
304
|
+
{ "id": "local", "host": "127.0.0.1", "ports": [8000, 8080, 8081], "enabled": true,
|
|
305
|
+
"costProfile": { "kW": 0.15, "ratePerKwh": 0.21, "label": "bruma 27B" } }
|
|
306
|
+
],
|
|
307
|
+
"settings": { "currency": "eur", "costTracking": true }
|
|
308
|
+
}
|
|
309
|
+
```
|
|
310
|
+
|
|
268
311
|
## Architecture
|
|
269
312
|
|
|
270
313
|
```
|
|
@@ -280,6 +323,8 @@ llamacpp-infra/
|
|
|
280
323
|
├── registration.ts # Scan → pi-model mapping + provider registration (lazy).
|
|
281
324
|
├── metrics.ts # Server /metrics poller → ServerMetricsState (lazy; only if `metricsEnabled`).
|
|
282
325
|
├── speed.ts # Client-side speed tracker + footer status line (lazy; only if `metricsEnabled`).
|
|
326
|
+
├── cost.ts # Cost profiles, currency formatting, energy math (static import; tiny).
|
|
327
|
+
├── cost-tracker.ts # Energy-cost tracker: times requests, accumulates, feeds usage.cost (static).
|
|
283
328
|
├── ui.ts # /llamacpp-infra subcommands, menus, status, help (lazy).
|
|
284
329
|
└── prompt-warmup.ts # Header warmup: capture + cache system prompt KV (lazy; only if `warmup`).
|
|
285
330
|
```
|
|
@@ -292,6 +337,7 @@ Module load profile:
|
|
|
292
337
|
| `scan.ts` + `registration.ts` | First discovery (dynamic) | ~27 KB |
|
|
293
338
|
| `prompt-warmup.ts` | Primed at load if `warmup` enabled; not loaded when disabled | ~15 KB; skipped entirely when `warmup` is OFF |
|
|
294
339
|
| `metrics.ts` + `speed.ts` | Primed at load if `metricsEnabled`; not loaded when disabled | ~18 KB; skipped entirely when `metricsEnabled` is OFF |
|
|
340
|
+
| `cost.ts` + `cost-tracker.ts` | Static at load (tiny; needed in sub-agent processes too, which never fire `session_start`) | ~4 KB |
|
|
295
341
|
| `ui.ts` | First `/llamacpp-infra …` command (dynamic) | ~32 KB |
|
|
296
342
|
|
|
297
343
|
Zero external npm dependencies (only pi's bundled `@earendil-works/pi-coding-agent` + Node built-ins).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llamacpp-infra",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"description": "Discovery, metrics and control of llama.cpp-family servers for pi (llama.cpp, ZINC, DwarfStar/ds4, lucebox, LM Studio): scan machines (localhost, LAN, Tailscale), register models, live metrics, per-model thinking budgets, vision detection, native config UI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi-package",
|
package/src/core.ts
CHANGED
|
@@ -20,6 +20,7 @@ export const CONFIG_FILE = "llamacpp-infra.json";
|
|
|
20
20
|
export const MODELS_CACHE_FILE = "llamacpp-infra-models.json";
|
|
21
21
|
export const LEGACY_CONFIG_FILE = "local-models.json";
|
|
22
22
|
export const METRICS_STATUS_KEY = "llamacpp-infra-speed";
|
|
23
|
+
export const COST_STATUS_KEY = "llamacpp-infra-cost";
|
|
23
24
|
export const DEFAULT_API_KEY = "no-auth";
|
|
24
25
|
export const THINKING_BUDGET_FIELD = "thinking_budget_tokens";
|
|
25
26
|
export const DEFAULT_MAX_OUTPUT_TOKENS = 32_768;
|
|
@@ -41,6 +42,8 @@ export const DEFAULT_SETTINGS: SettingsConfig = {
|
|
|
41
42
|
metricsPollMs: 5000,
|
|
42
43
|
includeUnloadedRouterModels: false,
|
|
43
44
|
showBadgesInNames: true,
|
|
45
|
+
currency: "eur",
|
|
46
|
+
costTracking: true,
|
|
44
47
|
};
|
|
45
48
|
|
|
46
49
|
export const DEFAULT_SERVERS: ServerConfig[] = [
|
|
@@ -65,6 +68,50 @@ export function getConfigPath(): string {
|
|
|
65
68
|
return join(getAgentDir(), CONFIG_FILE);
|
|
66
69
|
}
|
|
67
70
|
|
|
71
|
+
/** Validate/normalize a raw cost profile from disk. */
|
|
72
|
+
export function parseCostProfile(raw: unknown): import("./types.ts").CostProfile | undefined {
|
|
73
|
+
if (!raw || typeof raw !== "object") return undefined;
|
|
74
|
+
const r = raw as Record<string, unknown>;
|
|
75
|
+
const kW = typeof r.kW === "number" ? r.kW : parseFloat(String(r.kW ?? ""));
|
|
76
|
+
const rate = typeof r.ratePerKwh === "number" ? r.ratePerKwh : parseFloat(String(r.ratePerKwh ?? ""));
|
|
77
|
+
if (!Number.isFinite(kW) || kW <= 0 || !Number.isFinite(rate) || rate <= 0) return undefined;
|
|
78
|
+
const p: import("./types.ts").CostProfile = { kW, ratePerKwh: rate };
|
|
79
|
+
if (typeof r.label === "string" && r.label) p.label = r.label;
|
|
80
|
+
if (typeof r.host === "string" && r.host) p.host = r.host;
|
|
81
|
+
if (typeof r.pattern === "string" && r.pattern) p.pattern = r.pattern;
|
|
82
|
+
return p;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const CURRENCY_CODES = new Set(["usd", "eur", "gbp", "cny"]);
|
|
86
|
+
|
|
87
|
+
/** Normalize a raw settings object onto the defaults (unknown keys dropped, bad values repaired). */
|
|
88
|
+
export function normalizeSettings(raw: unknown): SettingsConfig {
|
|
89
|
+
const r = (raw && typeof raw === "object" ? raw : {}) as Record<string, unknown>;
|
|
90
|
+
const s: SettingsConfig = { ...DEFAULT_SETTINGS };
|
|
91
|
+
for (const key of Object.keys(DEFAULT_SETTINGS) as Array<keyof SettingsConfig>) {
|
|
92
|
+
const v = r[key];
|
|
93
|
+
if (typeof v === typeof DEFAULT_SETTINGS[key]) (s as Record<string, unknown>)[key] = v;
|
|
94
|
+
}
|
|
95
|
+
if (typeof r.currency === "string" && CURRENCY_CODES.has(r.currency as string)) s.currency = r.currency as SettingsConfig["currency"];
|
|
96
|
+
return s;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
function normalizeServer(s: Record<string, unknown>): import("./types.ts").ServerConfig {
|
|
100
|
+
return {
|
|
101
|
+
id: String(s.id ?? "local"),
|
|
102
|
+
host: String(s.host ?? "127.0.0.1"),
|
|
103
|
+
...(typeof s.label === "string" && s.label ? { label: s.label } : {}),
|
|
104
|
+
ports: Array.isArray(s.ports) ? s.ports.filter((p): p is number => typeof p === "number") : [],
|
|
105
|
+
enabled: typeof s.enabled === "boolean" ? s.enabled : true,
|
|
106
|
+
...(s.probeDs4 === true ? { probeDs4: true } : {}),
|
|
107
|
+
...(typeof s.apiKey === "string" && s.apiKey ? { apiKey: s.apiKey } : {}),
|
|
108
|
+
...(() => {
|
|
109
|
+
const p = parseCostProfile(s.costProfile);
|
|
110
|
+
return p ? { costProfile: p } : {};
|
|
111
|
+
})(),
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
68
115
|
export function loadConfig(): InfraConfig {
|
|
69
116
|
const defaults: InfraConfig = {
|
|
70
117
|
servers: DEFAULT_SERVERS,
|
|
@@ -76,8 +123,10 @@ export function loadConfig(): InfraConfig {
|
|
|
76
123
|
try {
|
|
77
124
|
const raw = JSON.parse(readFileSync(path, "utf-8")) as Partial<InfraConfig>;
|
|
78
125
|
return {
|
|
79
|
-
servers: Array.isArray(raw.servers)
|
|
80
|
-
|
|
126
|
+
servers: Array.isArray(raw.servers)
|
|
127
|
+
? raw.servers.map((s) => normalizeServer(s as Record<string, unknown>))
|
|
128
|
+
: defaults.servers,
|
|
129
|
+
settings: normalizeSettings(raw.settings),
|
|
81
130
|
modelOptions: raw.modelOptions ?? {},
|
|
82
131
|
};
|
|
83
132
|
} catch (err) {
|
|
@@ -90,8 +139,10 @@ export function loadConfig(): InfraConfig {
|
|
|
90
139
|
try {
|
|
91
140
|
const raw = JSON.parse(readFileSync(legacyPath, "utf-8")) as Partial<InfraConfig>;
|
|
92
141
|
const migrated: InfraConfig = {
|
|
93
|
-
servers: Array.isArray(raw.servers)
|
|
94
|
-
|
|
142
|
+
servers: Array.isArray(raw.servers)
|
|
143
|
+
? raw.servers.map((s) => normalizeServer(s as Record<string, unknown>))
|
|
144
|
+
: defaults.servers,
|
|
145
|
+
settings: normalizeSettings(raw.settings),
|
|
95
146
|
modelOptions: raw.modelOptions ?? {},
|
|
96
147
|
};
|
|
97
148
|
debugLog(`migrated legacy config from ${legacyPath}`);
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
// Energy-cost tracker: measures inference time of the active llamacpp-infra
|
|
2
|
+
// model from pi's stream events and accumulates the electricity cost.
|
|
3
|
+
//
|
|
4
|
+
// cost per request = (requestMs / 3_600_000) × kW × tariff
|
|
5
|
+
//
|
|
6
|
+
// Timing: before_provider_request → message_end (assistant). This covers
|
|
7
|
+
// prefill + decode for one LLM call. Partial requests (errors/aborts) are
|
|
8
|
+
// still charged for the time actually consumed.
|
|
9
|
+
//
|
|
10
|
+
// The tracker exposes the accumulated cost so the footer can show it
|
|
11
|
+
// (⚡3.2¢) and the message_end hook can inject it into usage.cost, which pi
|
|
12
|
+
// then displays in its native cost footer — no trimegisto required.
|
|
13
|
+
|
|
14
|
+
import { COST_STATUS_KEY, debugLog } from "./core.ts";
|
|
15
|
+
import { formatCost } from "./cost.ts";
|
|
16
|
+
import type { CostProfile, Currency, ExtensionContext } from "./types.ts";
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
export interface CostDeps {
|
|
20
|
+
/** Whether the extension is still active. */
|
|
21
|
+
isActive: () => boolean;
|
|
22
|
+
/** Whether the ctx has a UI. */
|
|
23
|
+
hasUI: (ctx: ExtensionContext | undefined) => boolean;
|
|
24
|
+
/** Whether the current session model belongs to this provider. */
|
|
25
|
+
isOurs: (ctx: ExtensionContext | undefined) => boolean;
|
|
26
|
+
/** Whether cost tracking is enabled in settings. */
|
|
27
|
+
enabled: () => boolean;
|
|
28
|
+
/** Display currency. */
|
|
29
|
+
currency: () => Currency;
|
|
30
|
+
/** Resolve the cost profile for the current model (null = no estimation). */
|
|
31
|
+
profileFor: (ctx: ExtensionContext | undefined) => CostProfile | null;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export interface CostSnapshot {
|
|
35
|
+
/** Accumulated cost for the current session (in the configured currency). */
|
|
36
|
+
total: number;
|
|
37
|
+
/** Accumulated inference time in ms. */
|
|
38
|
+
ms: number;
|
|
39
|
+
/** Whether a profile is active (estimation on). */
|
|
40
|
+
active: boolean;
|
|
41
|
+
/** In-flight request start (ms epoch) or 0. */
|
|
42
|
+
inFlight: number;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Render the accumulated cost as a compact footer suffix, e.g. "💰3.2¢". */
|
|
46
|
+
export function formatCostSuffix(snap: CostSnapshot, currency: Currency): string {
|
|
47
|
+
if (!snap.active || snap.total <= 0) return "";
|
|
48
|
+
return `💰${formatCost(snap.total, currency)}`;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Push the cost line to pi's status footer (or clear it). */
|
|
52
|
+
function renderCostStatus(ctx: ExtensionContext | undefined, snap: CostSnapshot, currency: Currency): void {
|
|
53
|
+
if (!ctx || !ctx.ui) return;
|
|
54
|
+
const text = formatCostSuffix(snap, currency);
|
|
55
|
+
try {
|
|
56
|
+
ctx.ui.setStatus(COST_STATUS_KEY, text || undefined);
|
|
57
|
+
} catch {
|
|
58
|
+
// ignore
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function createCostTracker(deps: CostDeps) {
|
|
63
|
+
let reqStartAt: number | undefined;
|
|
64
|
+
let totalMs = 0;
|
|
65
|
+
let totalCost = 0;
|
|
66
|
+
/** Live ctx for callbacks without one. */
|
|
67
|
+
let lastCtx: ExtensionContext | undefined;
|
|
68
|
+
|
|
69
|
+
function rememberCtx(ctx: ExtensionContext | undefined): void {
|
|
70
|
+
if (ctx && deps.hasUI(ctx)) lastCtx = ctx;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function reset(): void {
|
|
74
|
+
reqStartAt = undefined;
|
|
75
|
+
totalMs = 0;
|
|
76
|
+
totalCost = 0;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Charge a finished (or aborted) request; returns the cost of this request. */
|
|
80
|
+
function chargeRequest(ctx: ExtensionContext | undefined, endAt: number): number {
|
|
81
|
+
if (reqStartAt === undefined) return 0;
|
|
82
|
+
const ms = Math.max(0, endAt - reqStartAt);
|
|
83
|
+
reqStartAt = undefined;
|
|
84
|
+
const profile = deps.profileFor(ctx ?? lastCtx);
|
|
85
|
+
if (!profile || ms <= 0) return 0;
|
|
86
|
+
const cost = (ms / 3_600_000) * profile.kW * profile.ratePerKwh;
|
|
87
|
+
totalMs += ms;
|
|
88
|
+
totalCost += cost;
|
|
89
|
+
debugLog(`cost: +${ms}ms → +${cost.toFixed(6)} (session ${totalCost.toFixed(6)})`);
|
|
90
|
+
return cost;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
return {
|
|
94
|
+
/** before_provider_request: a new LLM call starts. */
|
|
95
|
+
onRequest(ctx: ExtensionContext | undefined, at: number): void {
|
|
96
|
+
if (!deps.isActive()) return;
|
|
97
|
+
rememberCtx(ctx);
|
|
98
|
+
if (!deps.isOurs(ctx) || !deps.enabled()) {
|
|
99
|
+
reset();
|
|
100
|
+
return;
|
|
101
|
+
}
|
|
102
|
+
// A new request supersedes any uncharged one (shouldn't happen, but be safe).
|
|
103
|
+
if (reqStartAt !== undefined) chargeRequest(ctx, at);
|
|
104
|
+
reqStartAt = at;
|
|
105
|
+
},
|
|
106
|
+
|
|
107
|
+
/**
|
|
108
|
+
* message_end (assistant): finalize the call and charge it.
|
|
109
|
+
* Returns the cost of THIS request (0 when no profile/off), so callers
|
|
110
|
+
* can add just the delta to usage.cost.total of that message.
|
|
111
|
+
*/
|
|
112
|
+
onMessageEnd(ctx: ExtensionContext | undefined, message: unknown, at: number): number {
|
|
113
|
+
if (!deps.isActive()) return 0;
|
|
114
|
+
rememberCtx(ctx);
|
|
115
|
+
if (!deps.isOurs(ctx)) return 0;
|
|
116
|
+
const role = (message as { role?: string } | undefined)?.role;
|
|
117
|
+
if (role !== "assistant") return 0;
|
|
118
|
+
const charged = chargeRequest(ctx, at);
|
|
119
|
+
renderCostStatus(ctx ?? lastCtx, this.snapshot(), deps.currency());
|
|
120
|
+
return charged;
|
|
121
|
+
},
|
|
122
|
+
|
|
123
|
+
/** turn_end: nothing more to do (requests are charged at message_end). */
|
|
124
|
+
onTurnEnd(_ctx: ExtensionContext | undefined): void {
|
|
125
|
+
// no-op: kept for API symmetry with the speed tracker
|
|
126
|
+
},
|
|
127
|
+
|
|
128
|
+
/** Current accumulated view. */
|
|
129
|
+
snapshot(): CostSnapshot {
|
|
130
|
+
const profile = deps.profileFor(lastCtx);
|
|
131
|
+
return {
|
|
132
|
+
total: totalCost,
|
|
133
|
+
ms: totalMs,
|
|
134
|
+
active: !!profile && totalCost > 0,
|
|
135
|
+
inFlight: reqStartAt ?? 0,
|
|
136
|
+
};
|
|
137
|
+
},
|
|
138
|
+
|
|
139
|
+
/** Reset the session accumulator (model switch / session start). */
|
|
140
|
+
reset(ctx?: ExtensionContext): void {
|
|
141
|
+
reset();
|
|
142
|
+
renderCostStatus(ctx ?? lastCtx, this.snapshot(), deps.currency());
|
|
143
|
+
},
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
export type CostTracker = ReturnType<typeof createCostTracker>;
|
package/src/cost.ts
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
// Energy-cost estimation for local inference.
|
|
2
|
+
//
|
|
3
|
+
// Local servers report no per-token cost, so the honest cost is the
|
|
4
|
+
// electricity the inference consumed:
|
|
5
|
+
//
|
|
6
|
+
// cost = (inferenceMs / 3_600_000) × kW × tariff
|
|
7
|
+
//
|
|
8
|
+
// Profiles are configured per server (the power draw belongs to the machine,
|
|
9
|
+
// not the model). The currency selector (settings.currency) only changes the
|
|
10
|
+
// display unit — the stored tariff is always per kWh in the chosen currency.
|
|
11
|
+
|
|
12
|
+
import type { Currency, CostProfile, ServerConfig } from "./types.ts";
|
|
13
|
+
|
|
14
|
+
// ── Currency display ────────────────────────────────────────────────────────
|
|
15
|
+
export const CURRENCIES: Array<{ code: Currency; symbol: string; cent: string; label: string }> = [
|
|
16
|
+
{ code: "usd", symbol: "$", cent: "¢", label: "USD ($)" },
|
|
17
|
+
{ code: "eur", symbol: "€", cent: "c", label: "EUR (€)" },
|
|
18
|
+
{ code: "gbp", symbol: "£", cent: "p", label: "GBP (£)" },
|
|
19
|
+
{ code: "cny", symbol: "¥", cent: "分", label: "CNY (¥)" },
|
|
20
|
+
];
|
|
21
|
+
|
|
22
|
+
export function currencySymbol(code: Currency | undefined): string {
|
|
23
|
+
return CURRENCIES.find((c) => c.code === code)?.symbol ?? "€";
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** Cent/fractional marker of a currency: $→¢, €→c, £→p, ¥→分. */
|
|
27
|
+
export function currencyCent(code: Currency | undefined): string {
|
|
28
|
+
return CURRENCIES.find((c) => c.code === code)?.cent ?? "c";
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Format an amount in the chosen currency, compact:
|
|
33
|
+
* 1.234 → "$1.23" (≥ 1 unit: symbol + 2 decimals)
|
|
34
|
+
* 0.0315 → "3.2¢" (usd) / "3.2c" (eur) / "3.2p" (gbp) / "3.2分" (cny)
|
|
35
|
+
* 0.0004 → "0.4m¢" (sub-cent: milli-units)
|
|
36
|
+
*/
|
|
37
|
+
export function formatCost(amount: number, code: Currency | undefined): string {
|
|
38
|
+
const sym = currencySymbol(code);
|
|
39
|
+
if (amount >= 1) return `${sym}${amount.toFixed(2)}`;
|
|
40
|
+
const cent = currencyCent(code);
|
|
41
|
+
if (amount >= 0.001) return `${(amount * 100).toFixed(1)}${cent}`;
|
|
42
|
+
return `${(amount * 1000).toFixed(1)}m${cent}`;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// ── Profile resolution ──────────────────────────────────────────────────────
|
|
46
|
+
/**
|
|
47
|
+
* Find the cost profile that applies to a model, or null when none match.
|
|
48
|
+
*
|
|
49
|
+
* A profile is attached to a server (ServerConfig.costProfile): the power
|
|
50
|
+
* draw is a property of the machine, so all models served by that server
|
|
51
|
+
* share it. Resolution order:
|
|
52
|
+
* 1. server match (endpoint.serverId → ServerConfig.id)
|
|
53
|
+
* 2. host fallback: endpoint host equals the server's host (covers scans
|
|
54
|
+
* whose serverId drifted from the config, e.g. cache boot)
|
|
55
|
+
* 3. per-profile `pattern`: substring match on the model id, for machines
|
|
56
|
+
* that serve very different workloads (rare; kept for flexibility)
|
|
57
|
+
*/
|
|
58
|
+
export function resolveCostProfile(
|
|
59
|
+
model: { id?: string; endpoint?: { serverId?: string; host?: string } },
|
|
60
|
+
servers: ServerConfig[],
|
|
61
|
+
): CostProfile | null {
|
|
62
|
+
if (!model?.id) return null;
|
|
63
|
+
const endpoint = model.endpoint;
|
|
64
|
+
|
|
65
|
+
// 1. Server id match (primary).
|
|
66
|
+
if (endpoint?.serverId) {
|
|
67
|
+
const srv = servers.find((s) => s.id === endpoint.serverId);
|
|
68
|
+
if (srv?.costProfile && srv.costProfile.kW > 0) return srv.costProfile;
|
|
69
|
+
}
|
|
70
|
+
// 2. Host fallback.
|
|
71
|
+
if (endpoint?.host) {
|
|
72
|
+
for (const srv of servers) {
|
|
73
|
+
if (srv.costProfile && srv.costProfile.kW > 0 && srv.host === endpoint.host) return srv.costProfile;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
// 3. Model-id pattern (rare; only when explicitly set).
|
|
77
|
+
for (const srv of servers) {
|
|
78
|
+
const p = srv.costProfile;
|
|
79
|
+
if (p && p.kW > 0 && p.pattern && model.id.toLowerCase().includes(p.pattern.toLowerCase())) return p;
|
|
80
|
+
}
|
|
81
|
+
return null;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// ── Cost math ───────────────────────────────────────────────────────────────
|
|
85
|
+
/** Energy cost (in the configured currency) of `ms` of inference. */
|
|
86
|
+
export function energyCost(profile: CostProfile, ms: number): number {
|
|
87
|
+
if (!profile || ms <= 0) return 0;
|
|
88
|
+
const kwh = (ms / 3_600_000) * profile.kW;
|
|
89
|
+
return kwh * profile.ratePerKwh;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Format a profile for display, e.g. "0.15 kW @ 0.21 €/kWh". */
|
|
93
|
+
export function formatProfile(p: CostProfile, code: Currency | undefined): string {
|
|
94
|
+
return `${p.kW} kW @ ${p.ratePerKwh} ${currencySymbol(code)}/kWh`;
|
|
95
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -23,6 +23,8 @@ import {
|
|
|
23
23
|
supportsThinkingBudget,
|
|
24
24
|
} from "./core.ts";
|
|
25
25
|
import { createLongTimeoutOpenAICompletionsStream } from "./runtime.ts";
|
|
26
|
+
import { resolveCostProfile } from "./cost.ts";
|
|
27
|
+
import { createCostTracker, type CostTracker } from "./cost-tracker.ts";
|
|
26
28
|
|
|
27
29
|
export default function (pi: ExtensionAPI) {
|
|
28
30
|
const config = loadConfig();
|
|
@@ -77,6 +79,28 @@ export default function (pi: ExtensionAPI) {
|
|
|
77
79
|
let metricsApi: ReturnType<MetricsModule["createMetrics"]> | undefined;
|
|
78
80
|
let speedApi: ReturnType<SpeedModule["createSpeedTracker"]> | undefined;
|
|
79
81
|
|
|
82
|
+
/**
|
|
83
|
+
* Energy-cost tracker. Created SYNCHRONOUSLY at factory time (not lazily
|
|
84
|
+
* like the speed tracker) because cost must be measured inside sub-agent
|
|
85
|
+
* processes too (`pi -p --no-session`), which never fire session_start:
|
|
86
|
+
* the message_end hook needs it on the very first LLM call. Gating is done
|
|
87
|
+
* via deps.enabled(), so toggling costTracking ON at runtime works.
|
|
88
|
+
*/
|
|
89
|
+
const costApi: CostTracker = createCostTracker({
|
|
90
|
+
isActive: () => extensionActive,
|
|
91
|
+
hasUI: ctxHasUI,
|
|
92
|
+
isOurs: (ctx) => {
|
|
93
|
+
try {
|
|
94
|
+
return ctx?.model?.provider === PROVIDER_NAME;
|
|
95
|
+
} catch {
|
|
96
|
+
return false;
|
|
97
|
+
}
|
|
98
|
+
},
|
|
99
|
+
enabled: () => config.settings.costTracking,
|
|
100
|
+
currency: () => config.settings.currency,
|
|
101
|
+
profileFor: (ctx) => costProfileFor(ctx),
|
|
102
|
+
});
|
|
103
|
+
|
|
80
104
|
// ── Internal state ────────────────────────────────────────────────────
|
|
81
105
|
let lastSignature: string | undefined;
|
|
82
106
|
let pollTimer: ReturnType<typeof setTimeout> | undefined;
|
|
@@ -102,6 +126,25 @@ export default function (pi: ExtensionAPI) {
|
|
|
102
126
|
|
|
103
127
|
const epKey = (host: string, port: number) => `${host}:${port}`;
|
|
104
128
|
|
|
129
|
+
/** Cost profile for the model a ctx is currently talking to (null = no estimation). */
|
|
130
|
+
function costProfileFor(ctx: ExtensionContext | undefined): import("./types.ts").CostProfile | null {
|
|
131
|
+
try {
|
|
132
|
+
const m = ctx?.model as { id?: string; provider?: string } | undefined;
|
|
133
|
+
if (!m?.id) return null;
|
|
134
|
+
const baseUrl = shared.modelBaseUrls.get(m.id);
|
|
135
|
+
const ep = shared.lastScan?.endpoints.find((e) => e.baseUrl === baseUrl);
|
|
136
|
+
return resolveCostProfile(
|
|
137
|
+
{
|
|
138
|
+
id: m.id,
|
|
139
|
+
endpoint: ep ? { serverId: ep.serverId, host: ep.host, port: ep.port } : undefined,
|
|
140
|
+
},
|
|
141
|
+
config.servers,
|
|
142
|
+
);
|
|
143
|
+
} catch {
|
|
144
|
+
return null;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
105
148
|
function safeSystemPrompt(ctx: { getSystemPrompt?: () => string }): string | undefined {
|
|
106
149
|
try {
|
|
107
150
|
return ctx.getSystemPrompt?.();
|
|
@@ -309,9 +352,10 @@ export default function (pi: ExtensionAPI) {
|
|
|
309
352
|
if (config.settings.metricsEnabled) metricsApi.start(ctx);
|
|
310
353
|
}
|
|
311
354
|
|
|
312
|
-
// ── Live speed: per-call
|
|
355
|
+
// ── Live speed + energy cost: per-call timing (before id rewrite) ─────
|
|
313
356
|
pi.on("before_provider_request", (_event, ctx) => {
|
|
314
357
|
speedApi?.onRequest(ctx, Date.now());
|
|
358
|
+
costApi?.onRequest(ctx, Date.now());
|
|
315
359
|
return undefined;
|
|
316
360
|
});
|
|
317
361
|
|
|
@@ -322,6 +366,21 @@ export default function (pi: ExtensionAPI) {
|
|
|
322
366
|
|
|
323
367
|
pi.on("message_end", (event, ctx) => {
|
|
324
368
|
speedApi?.onMessageEnd(ctx, event.message, Date.now());
|
|
369
|
+
|
|
370
|
+
// Inject the energy cost of THIS request into usage.cost.total so pi's
|
|
371
|
+
// native cost footer, session stats and any consumer reading usage
|
|
372
|
+
// (e.g. trimegisto agents, which accumulate usage.cost.total per
|
|
373
|
+
// message) see a realistic cost for local models. message_end fires
|
|
374
|
+
// before persistence, so mutating event.message in place is enough.
|
|
375
|
+
const msg = event.message as { role?: string; usage?: { cost?: { total?: number } } } | undefined;
|
|
376
|
+
if (msg?.role === "assistant" && config.settings.costTracking) {
|
|
377
|
+
const requestCost = costApi?.onMessageEnd(ctx, event.message, Date.now()) ?? 0;
|
|
378
|
+
if (requestCost > 0 && msg.usage?.cost) {
|
|
379
|
+
msg.usage.cost.total = (msg.usage.cost.total ?? 0) + requestCost;
|
|
380
|
+
return { message: event.message };
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
return undefined;
|
|
325
384
|
});
|
|
326
385
|
|
|
327
386
|
pi.on("turn_end", (_event, ctx) => {
|
|
@@ -530,10 +589,15 @@ export default function (pi: ExtensionAPI) {
|
|
|
530
589
|
w.warmer.warmupForModel(ctx.model, safeSystemPrompt(ctx), ctx.cwd);
|
|
531
590
|
}
|
|
532
591
|
currentThinkingLevel = ctx.thinkingLevel;
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
592
|
+
try {
|
|
593
|
+
if (config.settings.metricsEnabled) {
|
|
594
|
+
await ensureSpeed().then((s) => s?.start(ctx));
|
|
595
|
+
const m = await ensureMetrics();
|
|
596
|
+
m.start(ctx);
|
|
597
|
+
}
|
|
598
|
+
} catch (err) {
|
|
599
|
+
// Tracking must never break session start (or spam the TUI with traces).
|
|
600
|
+
debugLog(`session_start tracking error ignored: ${err instanceof Error ? err.message : String(err)}`);
|
|
537
601
|
}
|
|
538
602
|
if (!ctxHasUI(ctx)) return;
|
|
539
603
|
void discoverAndRegister()
|
|
@@ -561,10 +625,16 @@ export default function (pi: ExtensionAPI) {
|
|
|
561
625
|
w.warmer.warmupForModel(event.model, safeSystemPrompt(ctx), ctx.cwd);
|
|
562
626
|
}
|
|
563
627
|
metricsApi?.resetForModelSwitch();
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
628
|
+
costApi?.reset(ctx);
|
|
629
|
+
try {
|
|
630
|
+
if (config.settings.metricsEnabled) {
|
|
631
|
+
await ensureSpeed().then((s) => s?.start(ctx));
|
|
632
|
+
const m = await ensureMetrics();
|
|
633
|
+
m.start(ctx);
|
|
634
|
+
}
|
|
635
|
+
} catch (err) {
|
|
636
|
+
// Tracking must never break model switches (or spam the TUI with traces).
|
|
637
|
+
debugLog(`model_select tracking error ignored: ${err instanceof Error ? err.message : String(err)}`);
|
|
568
638
|
}
|
|
569
639
|
});
|
|
570
640
|
|
package/src/types.ts
CHANGED
|
@@ -2,6 +2,24 @@
|
|
|
2
2
|
|
|
3
3
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
4
4
|
|
|
5
|
+
export type Currency = "usd" | "eur" | "gbp" | "cny";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Energy-cost profile for one machine: power draw during inference (kW) and
|
|
9
|
+
* electricity tariff (per kWh, in the configured currency). Cost of a request
|
|
10
|
+
* = (ms / 3_600_000) × kW × ratePerKwh. `host`/`pattern` refine matching when
|
|
11
|
+
* needed (they are optional; the serverId association is the primary key).
|
|
12
|
+
*/
|
|
13
|
+
export interface CostProfile {
|
|
14
|
+
label?: string;
|
|
15
|
+
/** Power draw during inference, in kilowatts. 0/absent = estimation off. */
|
|
16
|
+
kW: number;
|
|
17
|
+
/** Electricity tariff per kWh (in the configured currency). */
|
|
18
|
+
ratePerKwh: number;
|
|
19
|
+
host?: string;
|
|
20
|
+
pattern?: string;
|
|
21
|
+
}
|
|
22
|
+
|
|
5
23
|
/** One machine that may serve llama.cpp-family models on one or more ports. */
|
|
6
24
|
export interface ServerConfig {
|
|
7
25
|
id: string;
|
|
@@ -11,6 +29,8 @@ export interface ServerConfig {
|
|
|
11
29
|
enabled: boolean;
|
|
12
30
|
probeDs4?: boolean;
|
|
13
31
|
apiKey?: string;
|
|
32
|
+
/** Energy-cost profile for this machine (kW + tariff). */
|
|
33
|
+
costProfile?: CostProfile;
|
|
14
34
|
}
|
|
15
35
|
|
|
16
36
|
/** Thinking budget (tokens) per pi thinking level, llama.cpp-style. */
|
|
@@ -41,6 +61,10 @@ export interface SettingsConfig {
|
|
|
41
61
|
metricsPollMs: number;
|
|
42
62
|
includeUnloadedRouterModels: boolean;
|
|
43
63
|
showBadgesInNames: boolean;
|
|
64
|
+
/** Display currency for energy costs (default: eur). */
|
|
65
|
+
currency: Currency;
|
|
66
|
+
/** Whether the footer shows accumulated energy cost (default: true). */
|
|
67
|
+
costTracking: boolean;
|
|
44
68
|
}
|
|
45
69
|
|
|
46
70
|
export interface InfraConfig {
|
package/src/ui.ts
CHANGED
|
@@ -19,6 +19,7 @@ import type {
|
|
|
19
19
|
ThinkingBudgets,
|
|
20
20
|
} from "./types.ts";
|
|
21
21
|
import { fetchModelsFromEndpoint, scanLocalServers } from "./scan.ts";
|
|
22
|
+
import { CURRENCIES, formatProfile, currencySymbol } from "./cost.ts";
|
|
22
23
|
|
|
23
24
|
// ── Small UI helpers ───────────────────────────────────────────────────────
|
|
24
25
|
async function selectFrom<T>(
|
|
@@ -221,6 +222,11 @@ export async function showConfigMenu(ctx: ExtensionContext, deps: UiDeps): Promi
|
|
|
221
222
|
{ value: "models", label: "📋 Discovered models", description: `${shared.registeredCount} currently registered` },
|
|
222
223
|
{ value: "test", label: "🧪 Test connectivity", description: "probe every endpoint and show latency" },
|
|
223
224
|
{ value: "budgets", label: "🧠 Thinking budgets", description: `${budgetCount} model(s) with budgets` },
|
|
225
|
+
{
|
|
226
|
+
value: "cost",
|
|
227
|
+
label: `💰 Energy cost: ${config.settings.costTracking ? "ON" : "OFF"}`,
|
|
228
|
+
description: `currency ${currencySymbol(config.settings.currency)} · ${config.servers.filter((s) => s.costProfile && s.costProfile.kW > 0).length} server(s) with kW`,
|
|
229
|
+
},
|
|
224
230
|
{
|
|
225
231
|
value: "metrics",
|
|
226
232
|
label: `📈 Live speed & metrics: ${config.settings.metricsEnabled ? "ON" : "OFF"}`,
|
|
@@ -248,6 +254,9 @@ export async function showConfigMenu(ctx: ExtensionContext, deps: UiDeps): Promi
|
|
|
248
254
|
case "budgets":
|
|
249
255
|
await showThinkingBudgetsMenu(ctx, deps);
|
|
250
256
|
break;
|
|
257
|
+
case "cost":
|
|
258
|
+
await showCostMenu(ctx, deps);
|
|
259
|
+
break;
|
|
251
260
|
case "metrics":
|
|
252
261
|
await showMetricsMenu(ctx, deps);
|
|
253
262
|
break;
|
|
@@ -619,6 +628,105 @@ async function showMetricsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<voi
|
|
|
619
628
|
}
|
|
620
629
|
}
|
|
621
630
|
|
|
631
|
+
// ── Energy-cost menu ────────────────────────────────────────────────────
|
|
632
|
+
async function showCostMenu(ctx: ExtensionContext, deps: UiDeps): Promise<void> {
|
|
633
|
+
const config = shared.activeConfig!;
|
|
634
|
+
for (;;) {
|
|
635
|
+
const withKw = config.servers.filter((s) => s.costProfile && s.costProfile.kW > 0);
|
|
636
|
+
const action = await selectFrom(ctx, `💰 Energy cost (${currencySymbol(config.settings.currency)})`, [
|
|
637
|
+
{
|
|
638
|
+
value: "toggle",
|
|
639
|
+
label: config.settings.costTracking ? "🔴 Disable cost tracking" : "🟢 Enable cost tracking",
|
|
640
|
+
description: "accumulate electricity cost of local inference (footer + usage.cost)",
|
|
641
|
+
},
|
|
642
|
+
{
|
|
643
|
+
value: "currency",
|
|
644
|
+
label: `💱 Currency: ${CURRENCIES.find((c) => c.code === config.settings.currency)?.label ?? config.settings.currency}`,
|
|
645
|
+
description: "display unit for costs and tariffs",
|
|
646
|
+
},
|
|
647
|
+
{
|
|
648
|
+
value: "servers",
|
|
649
|
+
label: "🖥️ Per-server kW / tariff",
|
|
650
|
+
description: `${withKw.length} server(s) configured`,
|
|
651
|
+
},
|
|
652
|
+
{ value: "__back", label: "← Back", description: "" },
|
|
653
|
+
]);
|
|
654
|
+
if (action === undefined || action === "__back") return;
|
|
655
|
+
|
|
656
|
+
if (action === "toggle") {
|
|
657
|
+
config.settings.costTracking = !config.settings.costTracking;
|
|
658
|
+
saveConfig(config);
|
|
659
|
+
deps.updateStatusFooter(ctx);
|
|
660
|
+
ctx.ui.notify(`💰 Cost tracking: ${config.settings.costTracking ? "ON" : "OFF"}`, "info");
|
|
661
|
+
}
|
|
662
|
+
else if (action === "currency") {
|
|
663
|
+
const cur = await selectFrom(ctx, "💱 Select currency", CURRENCIES.map((c) => ({ value: c.code, label: c.label })));
|
|
664
|
+
if (cur) {
|
|
665
|
+
config.settings.currency = cur;
|
|
666
|
+
saveConfig(config);
|
|
667
|
+
ctx.ui.notify(`💱 Currency: ${CURRENCIES.find((c) => c.code === cur)?.label}`, "info");
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
else if (action === "servers") {
|
|
671
|
+
await showServerCostMenu(ctx, config);
|
|
672
|
+
}
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
async function showServerCostMenu(ctx: ExtensionContext, config: NonNullable<typeof shared.activeConfig>): Promise<void> {
|
|
677
|
+
for (;;) {
|
|
678
|
+
const items = config.servers.map((srv) => ({
|
|
679
|
+
value: srv.id,
|
|
680
|
+
label: `${srv.costProfile?.kW ? "💰" : "·"} ${serverLabel(srv)}`,
|
|
681
|
+
description: srv.costProfile?.kW ? formatProfile(srv.costProfile, config.settings.currency) : `${srv.host} — no kW set`,
|
|
682
|
+
}));
|
|
683
|
+
items.push({ value: "__back", label: "← Back", description: "" });
|
|
684
|
+
const picked = await selectFrom(ctx, "💰 Per-server energy cost — select a machine", items);
|
|
685
|
+
if (!picked || picked === "__back") return;
|
|
686
|
+
const srv = config.servers.find((s) => s.id === picked);
|
|
687
|
+
if (!srv) return;
|
|
688
|
+
|
|
689
|
+
const p = srv.costProfile;
|
|
690
|
+
const sub = await selectFrom(ctx, `💰 ${serverLabel(srv)} (${srv.host})`, [
|
|
691
|
+
{ value: "kW", label: p?.kW ? `⚡ Power draw: ${p.kW} kW` : "⚡ Set power draw (kW)", description: "W consumed during inference (e.g. 0.15 for 150 W)" },
|
|
692
|
+
{ value: "rate", label: p?.ratePerKwh ? `🧾 Tariff: ${p.ratePerKwh} ${currencySymbol(config.settings.currency)}/kWh` : `🧾 Set tariff (${currencySymbol(config.settings.currency)}/kWh)`, description: "electricity price per kWh" },
|
|
693
|
+
{ value: "label", label: p?.label ? `🏷️ Label: ${p.label}` : "🏷️ Set label", description: "optional, shown in menus" },
|
|
694
|
+
...(p?.kW ? [{ value: "clear", label: "🗑️ Remove cost profile", description: "stop estimating cost for this machine" }] : []),
|
|
695
|
+
{ value: "__back", label: "← Back", description: "" },
|
|
696
|
+
]);
|
|
697
|
+
if (!sub || sub === "__back") continue;
|
|
698
|
+
|
|
699
|
+
if (sub === "kW") {
|
|
700
|
+
const raw = await ctx.ui.input("⚡ Power draw in kW (e.g. 0.15 = 150 W)", p?.kW ? String(p.kW) : "0.15");
|
|
701
|
+
const kW = parseFloat(raw ?? "");
|
|
702
|
+
if (isNaN(kW) || kW <= 0) { ctx.ui.notify("❌ Invalid kW", "error"); continue; }
|
|
703
|
+
srv.costProfile = { ...(srv.costProfile ?? { ratePerKwh: 0.2 }), kW };
|
|
704
|
+
saveConfig(config);
|
|
705
|
+
ctx.ui.notify(`⚡ ${serverLabel(srv)}: ${kW} kW`, "info");
|
|
706
|
+
}
|
|
707
|
+
else if (sub === "rate") {
|
|
708
|
+
const raw = await ctx.ui.input(`🧾 Tariff in ${currencySymbol(config.settings.currency)}/kWh (e.g. 0.21)`, p?.ratePerKwh ? String(p.ratePerKwh) : "0.21");
|
|
709
|
+
const rate = parseFloat(raw ?? "");
|
|
710
|
+
if (isNaN(rate) || rate <= 0) { ctx.ui.notify("❌ Invalid tariff", "error"); continue; }
|
|
711
|
+
srv.costProfile = { ...(srv.costProfile ?? { kW: 0.15 }), ratePerKwh: rate };
|
|
712
|
+
saveConfig(config);
|
|
713
|
+
ctx.ui.notify(`🧾 ${serverLabel(srv)}: ${rate} ${currencySymbol(config.settings.currency)}/kWh`, "info");
|
|
714
|
+
}
|
|
715
|
+
else if (sub === "label") {
|
|
716
|
+
const raw = await ctx.ui.input("🏷️ Label", p?.label ?? "");
|
|
717
|
+
if (raw !== undefined) {
|
|
718
|
+
srv.costProfile = { ...(srv.costProfile ?? { kW: 0.15, ratePerKwh: 0.21 }), label: raw.trim() || undefined };
|
|
719
|
+
saveConfig(config);
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
else if (sub === "clear") {
|
|
723
|
+
delete srv.costProfile;
|
|
724
|
+
saveConfig(config);
|
|
725
|
+
ctx.ui.notify(`🗑️ ${serverLabel(srv)}: cost profile removed`, "info");
|
|
726
|
+
}
|
|
727
|
+
}
|
|
728
|
+
}
|
|
729
|
+
|
|
622
730
|
// ── Settings menu ──────────────────────────────────────────────────────────
|
|
623
731
|
async function showSettingsMenu(ctx: ExtensionContext, deps: UiDeps): Promise<void> {
|
|
624
732
|
const config = shared.activeConfig!;
|