pi-zro-provider 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.ts ADDED
@@ -0,0 +1,1110 @@
1
+ /**
2
+ * Zro Provider Extension
3
+ *
4
+ * Registers Zro (zro.moonmath.ai) as a custom provider using the
5
+ * openai-completions API. Base URL: https://zro.moonmath.ai/v1
6
+ *
7
+ * Model metadata comes from Zro's CLI model catalog, GET /api/cli/models —
8
+ * the same endpoint `zro models` uses. It provides canonical ids, display
9
+ * names, context/output limits, and per-model reasoning effort levels (each
10
+ * level pairs an id with a piLevel). patch.json remains available for
11
+ * verified endpoint regressions, but currently contains no overrides.
12
+ *
13
+ * Model resolution strategy: Stale-While-Revalidate
14
+ * 1. Serve stale immediately: disk cache → embedded models.json (zero-latency)
15
+ * 2. Revalidate in background: live API /api/cli/models → merge with embedded → cache → hot-swap
16
+ * 3. patch.json + custom-models.json applied on top of whichever source won
17
+ *
18
+ * Merge order: [live|cache|embedded] → apply patch.json → merge custom-models.json
19
+ *
20
+ * Footer Status Widget:
21
+ * A below-editor line shows Zro session + account state:
22
+ *
23
+ * ⚡ $0.42 · 7 req Pro ◆ $12.34 avail · $2.50 pack · 1.2k req/30d
24
+ * └─ session activity ──┘ └──────────────── account / quota ─────────────┘
25
+ *
26
+ * The left side reports what this session has spent/sent: request count
27
+ * always, plus token totals and, when the response exposes one, a cost
28
+ * extension on each chat completion — pi requests
29
+ * stream_options.include_usage, so the final SSE chunk carries usage and we
30
+ * read it off a teed response stream, no polling). The right side reports
31
+ * the plan name, total available spend, usage-pack balance, and 30-day
32
+ * request/token activity from Zro's /api/cli/status endpoint (the same
33
+ * view `zro status` prints), plus a request-rate atom captured from
34
+ * response headers when present. The right side compresses across
35
+ * progressive tiers as the terminal narrows. The balance flips to a ⚠
36
+ * warning at/below lowBalanceUsd.
37
+ *
38
+ * Lifecycle (mirrors pi-neuralwatt-provider): nothing renders before this
39
+ * session's first Zro turn completes, so fresh sessions and other
40
+ * providers' sessions see no half-empty line. Account status is prefetched
41
+ * on session start or model select when a Zro model is active, so the
42
+ * first turn ends with data already cached. It is polled again on pi's
43
+ * agent_settled event (fires only once no automatic retry, compaction, or
44
+ * queued continuation can follow) — and nowhere else, so sessions without
45
+ * Zro turns make zero status-related API calls. Between polls the balance
46
+ * moves optimistically: each turn's captured spend is deducted from the
47
+ * last available-spend value at turn_end so the account line tracks spend
48
+ * live; the agent_settled poll reconciles any drift.
49
+ *
50
+ * Display Configuration:
51
+ * Create ~/.pi/agent/extensions/zro.json:
52
+ * {
53
+ * "session": "widget", // "widget" | "statusbar" | "off"
54
+ * "account": "widget", // "widget" | "statusbar" | "off"
55
+ * "hideOnOtherProvider": true, // hide when a non-Zro model is active
56
+ * "lowBalanceUsd": 10 // warn threshold, null/false disables
57
+ * }
58
+ *
59
+ * - "widget" (default): rendered in the below-editor status line
60
+ * - "statusbar": rendered in the built-in pi status bar
61
+ * - "off": hidden entirely (account=off also skips status fetches)
62
+ *
63
+ * Manage interactively with /zro-status, or non-interactively:
64
+ * /zro-status session widget|statusbar|off
65
+ * /zro-status account widget|statusbar|off
66
+ * /zro-status hide true|false
67
+ * /zro-status lowBalance <usd>|off
68
+ * /zro-status refresh (re-fetch account status now)
69
+ * /zro-status reset
70
+ *
71
+ * Usage:
72
+ * # Option 1: Use your existing zro CLI login (no extra key needed)
73
+ * # Run `zro login` once; this extension reads ~/.config/zro/credentials.json
74
+ * # (XDG_CONFIG_HOME aware) as a fallback.
75
+ *
76
+ * # Option 2: Store in auth.json (recommended)
77
+ * # Add to ~/.pi/agent/auth.json:
78
+ * # "zro": { "type": "api_key", "key": "your-api-key" }
79
+ *
80
+ * # Option 3: Set as environment variable
81
+ * export ZRO_API_KEY=your-api-key
82
+ *
83
+ * # Run pi with the extension
84
+ * pi -e /path/to/pi-zro-provider
85
+ *
86
+ * Then use /model to select from available models.
87
+ *
88
+ * @see https://zro.moonmath.ai
89
+ */
90
+
91
+ import { clampThinkingLevel, streamOpenAICompletions } from "@earendil-works/pi-ai/compat";
92
+ import type { AssistantMessageEventStream, SimpleStreamOptions } from "@earendil-works/pi-ai/compat";
93
+ import { getAgentDir, type ExtensionAPI, type ExtensionContext, type ModelRegistry } from "@earendil-works/pi-coding-agent";
94
+ import modelsData from "./models.json" with { type: "json" };
95
+ import customModelsData from "./custom-models.json" with { type: "json" };
96
+ import patchData from "./patch.json" with { type: "json" };
97
+ import deprecatedData from "./deprecated-models.json" with { type: "json" };
98
+ import {
99
+ applyOptimisticSpend,
100
+ buildAccountTiers,
101
+ buildSessionLine,
102
+ coerceStatusConfig,
103
+ DEFAULT_STATUS_CONFIG,
104
+ EMPTY_ACCOUNT,
105
+ EMPTY_SESSION_STATS,
106
+ StatusLineWidget,
107
+ accountHasData,
108
+ type AccountState,
109
+ type SessionStats,
110
+ type StatusConfig,
111
+ } from "./status";
112
+ import fs from "fs";
113
+ import { homedir } from "os";
114
+ import path from "path";
115
+
116
+ // ─── Types ────────────────────────────────────────────────────────────────────
117
+
118
+ interface JsonModel {
119
+ id: string;
120
+ name: string;
121
+ reasoning: boolean;
122
+ input: ("text" | "image")[];
123
+ cost: {
124
+ input: number;
125
+ output: number;
126
+ cacheRead: number;
127
+ cacheWrite: number;
128
+ };
129
+ contextWindow: number;
130
+ maxTokens: number;
131
+ thinkingLevelMap?: Record<string, string | null>;
132
+ compat?: {
133
+ supportsDeveloperRole?: boolean;
134
+ supportsStore?: boolean;
135
+ maxTokensField?: "max_completion_tokens" | "max_tokens";
136
+ thinkingFormat?: "openai" | "zai" | "qwen" | "qwen-chat-template" | "deepseek" | "openrouter" | "baseten" | "string-thinking" | "together" | "ant-ling" | "chat-template";
137
+ supportsReasoningEffort?: boolean;
138
+ requiresReasoningContentOnAssistantMessages?: boolean;
139
+ };
140
+ }
141
+
142
+ interface PatchEntry {
143
+ name?: string;
144
+ reasoning?: boolean;
145
+ input?: ("text" | "image")[];
146
+ cost?: {
147
+ input?: number;
148
+ output?: number;
149
+ cacheRead?: number;
150
+ cacheWrite?: number;
151
+ };
152
+ contextWindow?: number;
153
+ maxTokens?: number;
154
+ thinkingLevelMap?: Record<string, string | null>;
155
+ compat?: Record<string, unknown>;
156
+ }
157
+
158
+ type PatchData = Record<string, PatchEntry>;
159
+
160
+ // ─── Patch Application ────────────────────────────────────────────────────────
161
+
162
+ function applyPatch(model: JsonModel, patch: PatchEntry): JsonModel {
163
+ const result = { ...model };
164
+
165
+ if (patch.name !== undefined) result.name = patch.name;
166
+ if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
167
+ if (patch.input !== undefined) result.input = patch.input;
168
+ if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
169
+ if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
170
+ if (patch.thinkingLevelMap !== undefined) result.thinkingLevelMap = { ...patch.thinkingLevelMap };
171
+
172
+ if (patch.cost) {
173
+ result.cost = {
174
+ input: patch.cost.input ?? result.cost.input,
175
+ output: patch.cost.output ?? result.cost.output,
176
+ cacheRead: patch.cost.cacheRead ?? result.cost.cacheRead,
177
+ cacheWrite: patch.cost.cacheWrite ?? result.cost.cacheWrite,
178
+ };
179
+ }
180
+ if (patch.compat) {
181
+ result.compat = { ...(result.compat || {}), ...patch.compat };
182
+ }
183
+
184
+ if (!result.reasoning && result.compat?.thinkingFormat) {
185
+ delete result.compat.thinkingFormat;
186
+ }
187
+ if (!result.reasoning && result.thinkingLevelMap) {
188
+ delete result.thinkingLevelMap;
189
+ }
190
+ if (result.compat && Object.keys(result.compat).length === 0) {
191
+ delete result.compat;
192
+ }
193
+
194
+ return result;
195
+ }
196
+
197
+ /** Full pipeline: base models → patch → custom → result */
198
+ function buildModels(base: JsonModel[], custom: JsonModel[], patch: PatchData): JsonModel[] {
199
+ const modelMap = new Map<string, JsonModel>();
200
+
201
+ // Seed with the base list plus grace-period deprecated models so patch.json
202
+ // entries apply to deprecated models exactly as while the model was live
203
+ // (withDeprecated keeps live data on id conflicts).
204
+ for (const model of withDeprecated(base)) {
205
+ modelMap.set(model.id, model);
206
+ }
207
+
208
+ for (const [id, patchEntry] of Object.entries(patch)) {
209
+ const existing = modelMap.get(id);
210
+ if (existing) {
211
+ modelMap.set(id, applyPatch(existing, patchEntry));
212
+ }
213
+ }
214
+
215
+ for (const model of custom) {
216
+ const existing = modelMap.get(model.id);
217
+ const patchEntry = patch[model.id];
218
+ if (existing && patchEntry) {
219
+ modelMap.set(model.id, applyPatch(model, patchEntry));
220
+ } else if (existing) {
221
+ modelMap.set(model.id, model);
222
+ } else if (patchEntry) {
223
+ modelMap.set(model.id, applyPatch(model, patchEntry));
224
+ } else {
225
+ modelMap.set(model.id, model);
226
+ }
227
+ }
228
+
229
+ return Array.from(modelMap.values());
230
+ }
231
+
232
+ // ─── Stale-While-Revalidate Model Sync ────────────────────────────────────────
233
+
234
+ const PROVIDER_ID = "zro";
235
+ // Endpoint root is overridable like zro's own CLI (ZRO_ENDPOINT_ROOT).
236
+ const ENDPOINT_ROOT = (process.env.ZRO_ENDPOINT_ROOT || "https://zro.moonmath.ai").replace(/\/+$/, "");
237
+ const BASE_URL = `${ENDPOINT_ROOT}/v1`;
238
+ const MODELS_URL = `${ENDPOINT_ROOT}/api/cli/models`;
239
+ const STATUS_URL = `${ENDPOINT_ROOT}/api/cli/status`;
240
+ const CACHE_DIR = path.join(getAgentDir(), "cache");
241
+ const CACHE_PATH = path.join(CACHE_DIR, `${PROVIDER_ID}-models.json`);
242
+ const LIVE_FETCH_TIMEOUT_MS = 8000;
243
+
244
+ // pi thinking levels → provider level ids. Matches the transform zro's own
245
+ // pi adapter applies: each catalog reasoning level publishes a piLevel
246
+ // (off|minimal|low|medium|high|xhigh) and the id the inference proxy
247
+ // expects, e.g. glm-5.2 → { off: "none", high: "high", xhigh: "max" }.
248
+ const PI_THINKING_LEVELS = ["off", "minimal", "low", "medium", "high", "xhigh"] as const;
249
+
250
+ function buildZroThinkingLevelMap(levels: any[]): Record<string, string | null> {
251
+ const result: Record<string, string | null> = {
252
+ off: null,
253
+ minimal: null,
254
+ low: null,
255
+ medium: null,
256
+ high: null,
257
+ xhigh: null,
258
+ };
259
+ for (const level of levels) {
260
+ if (
261
+ level &&
262
+ typeof level === "object" &&
263
+ typeof level.id === "string" &&
264
+ level.id.trim() &&
265
+ typeof level.piLevel === "string" &&
266
+ Object.prototype.hasOwnProperty.call(result, level.piLevel)
267
+ ) {
268
+ result[level.piLevel] = level.id;
269
+ }
270
+ }
271
+ return result;
272
+ }
273
+
274
+ /** Transform a model from Zro's CLI model catalog (/api/cli/models). */
275
+ function transformApiModel(apiModel: any): JsonModel | null {
276
+ if (typeof apiModel.id !== "string" || apiModel.id.length === 0) return null;
277
+
278
+ const levels = Array.isArray(apiModel?.reasoning?.levels) ? apiModel.reasoning.levels : [];
279
+
280
+ return {
281
+ id: apiModel.id,
282
+ name: apiModel.displayName || apiModel.id,
283
+ reasoning: true,
284
+ thinkingLevelMap: buildZroThinkingLevelMap(levels),
285
+ input: ["text"],
286
+ cost: {
287
+ input: 0,
288
+ output: 0,
289
+ cacheRead: 0,
290
+ cacheWrite: 0,
291
+ },
292
+ contextWindow: apiModel.contextWindow || 0,
293
+ maxTokens: apiModel.maxOutputTokens || apiModel.contextWindow || 0,
294
+ compat: {
295
+ supportsDeveloperRole: false,
296
+ supportsReasoningEffort: true,
297
+ maxTokensField: "max_tokens",
298
+ },
299
+ };
300
+ }
301
+
302
+ async function fetchLiveModels(apiKey: string, signal?: AbortSignal): Promise<JsonModel[] | null> {
303
+ try {
304
+ const response = await fetch(MODELS_URL, {
305
+ headers: { Authorization: `Bearer ${apiKey}` },
306
+ signal: signal ? AbortSignal.any([AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS), signal]) : AbortSignal.timeout(LIVE_FETCH_TIMEOUT_MS),
307
+ });
308
+ if (!response.ok) return null;
309
+ const data = await response.json();
310
+ const apiModels = Array.isArray(data) ? data : (data.models || data.data || []);
311
+ if (!Array.isArray(apiModels) || apiModels.length === 0) return null;
312
+ return apiModels.map(transformApiModel).filter((m): m is JsonModel => m !== null);
313
+ } catch {
314
+ return null;
315
+ }
316
+ }
317
+
318
+ function loadCachedModels(): JsonModel[] | null {
319
+ try {
320
+ const data = JSON.parse(fs.readFileSync(CACHE_PATH, "utf8"));
321
+ return Array.isArray(data) ? data : null;
322
+ } catch {
323
+ return null;
324
+ }
325
+ }
326
+
327
+ function cacheModels(models: JsonModel[]): void {
328
+ try {
329
+ fs.mkdirSync(CACHE_DIR, { recursive: true });
330
+ fs.writeFileSync(CACHE_PATH, JSON.stringify(models, null, 2) + "\n");
331
+ } catch {
332
+ // Cache write failure is non-fatal
333
+ }
334
+ }
335
+
336
+ function mergeWithEmbedded(liveModels: JsonModel[], embeddedModels: JsonModel[]): JsonModel[] {
337
+ const embeddedMap = new Map(embeddedModels.map(m => [m.id, m]));
338
+ const seen = new Set<string>();
339
+ const result: JsonModel[] = [];
340
+ for (const liveModel of liveModels) {
341
+ const embedded = embeddedMap.get(liveModel.id);
342
+ seen.add(liveModel.id);
343
+ if (embedded) {
344
+ // The live catalog is authoritative for context/output limits;
345
+ // curation (name/thinkingLevelMap/compat) still wins via ...embedded.
346
+ result.push({
347
+ ...liveModel,
348
+ ...embedded,
349
+ cost: liveModel.cost,
350
+ contextWindow: liveModel.contextWindow || embedded.contextWindow,
351
+ maxTokens: liveModel.maxTokens || embedded.maxTokens,
352
+ });
353
+ } else {
354
+ result.push(liveModel);
355
+ }
356
+ }
357
+ // Append any embedded models that the live API didn't return
358
+ for (const em of embeddedModels) {
359
+ if (!seen.has(em.id)) {
360
+ result.push(em);
361
+ }
362
+ }
363
+ return result;
364
+ }
365
+
366
+ // Grace period for delisted models. When the provider API stops listing a
367
+ // model, update-models.js moves its last-known definition into
368
+ // deprecated-models.json (stamped with deprecatedAt) instead of dropping it.
369
+ // For 14 days the model keeps working here so in-flight sessions and saved
370
+ // model settings do not break; afterwards it is evicted permanently.
371
+ const DEPRECATED_MODEL_TTL_MS = 14 * 24 * 60 * 60 * 1000;
372
+
373
+ // Grace-period deprecated models with deprecation metadata stripped.
374
+ function activeDeprecatedModels(): JsonModel[] {
375
+ const now = Date.now();
376
+ const result: JsonModel[] = [];
377
+ for (const entry of Object.values(deprecatedData as Record<string, JsonModel & { deprecatedAt?: string }>)) {
378
+ if (!entry?.id) continue;
379
+ const removedAt = Date.parse(entry.deprecatedAt ?? "");
380
+ if (Number.isNaN(removedAt) || now - removedAt > DEPRECATED_MODEL_TTL_MS) continue;
381
+ const model = { ...entry } as JsonModel & { deprecatedAt?: string };
382
+ delete model.deprecatedAt;
383
+ result.push(model);
384
+ }
385
+ return result;
386
+ }
387
+
388
+ // Append grace-period deprecated models the list does not already have (live data wins).
389
+ function withDeprecated(models: JsonModel[]): JsonModel[] {
390
+ const seen = new Set(models.map((m) => m.id));
391
+ const extras = activeDeprecatedModels().filter((m) => !seen.has(m.id));
392
+ return extras.length > 0 ? [...models, ...extras] : models;
393
+ }
394
+
395
+ function loadStaleModels(embeddedModels: JsonModel[]): JsonModel[] {
396
+ const cached = loadCachedModels();
397
+ if (!cached || cached.length === 0) return embeddedModels;
398
+
399
+ // Merge embedded models that are missing from cache (newly added models)
400
+ const cachedMap = new Map(cached.map(m => [m.id, m]));
401
+ for (const em of embeddedModels) {
402
+ if (!cachedMap.has(em.id)) {
403
+ cached.push(em);
404
+ }
405
+ }
406
+ return cached;
407
+ }
408
+
409
+ async function revalidateModels(apiKey: string | undefined, embeddedModels: JsonModel[], signal?: AbortSignal): Promise<JsonModel[] | null> {
410
+ if (!apiKey) return null;
411
+ const liveModels = await fetchLiveModels(apiKey, signal);
412
+ if (!liveModels || liveModels.length === 0) return null;
413
+ const merged = mergeWithEmbedded(liveModels, embeddedModels);
414
+ cacheModels(merged);
415
+ return merged;
416
+ }
417
+
418
+ // ─── API Key Resolution (via ModelRegistry) ────────────────────────────────────
419
+
420
+ let cachedApiKey: string | undefined;
421
+ let revalidateAbort: AbortController | null = null;
422
+
423
+ /** The zro CLI stores its login at ~/.config/zro/credentials.json (XDG aware). */
424
+ function storedZroCliKey(): string | undefined {
425
+ try {
426
+ const configRoot = process.env.XDG_CONFIG_HOME || path.join(homedir(), ".config");
427
+ const parsed = JSON.parse(fs.readFileSync(path.join(configRoot, "zro", "credentials.json"), "utf8"));
428
+ return typeof parsed?.apiKey === "string" && parsed.apiKey.trim() ? parsed.apiKey : undefined;
429
+ } catch {
430
+ return undefined;
431
+ }
432
+ }
433
+
434
+ async function resolveApiKey(modelRegistry: ModelRegistry): Promise<void> {
435
+ cachedApiKey =
436
+ (await modelRegistry.getApiKeyForProvider(PROVIDER_ID) ?? undefined) ||
437
+ (process.env.ZRO_API_KEY ?? undefined) ||
438
+ storedZroCliKey();
439
+ }
440
+
441
+ // ─── Status Display Configuration ──────────────────────────────────────────────
442
+
443
+ const CONFIG_PATH = path.join(getAgentDir(), "extensions", "zro.json");
444
+
445
+ let statusConfig: StatusConfig = { ...DEFAULT_STATUS_CONFIG };
446
+
447
+ function loadStatusConfig(): StatusConfig {
448
+ try {
449
+ const raw = JSON.parse(fs.readFileSync(CONFIG_PATH, "utf8"));
450
+ statusConfig = coerceStatusConfig(raw);
451
+ } catch {
452
+ // Missing or unreadable file → defaults
453
+ }
454
+ return statusConfig;
455
+ }
456
+
457
+ function writeStatusConfig(): void {
458
+ try {
459
+ let raw: Record<string, unknown> = {};
460
+ try {
461
+ const existing = JSON.parse(fs.readFileSync(CONFIG_PATH, "utf8"));
462
+ if (existing && typeof existing === "object" && !Array.isArray(existing)) raw = existing;
463
+ } catch {
464
+ // No existing file — start fresh
465
+ }
466
+ raw.session = statusConfig.session;
467
+ raw.account = statusConfig.account;
468
+ raw.hideOnOtherProvider = statusConfig.hideOnOtherProvider;
469
+ raw.lowBalanceUsd = statusConfig.lowBalanceUsd;
470
+ fs.mkdirSync(path.dirname(CONFIG_PATH), { recursive: true });
471
+ fs.writeFileSync(CONFIG_PATH, JSON.stringify(raw, null, 2) + "\n");
472
+ } catch {
473
+ // Config write failure is non-fatal — the in-memory config still applies
474
+ }
475
+ }
476
+
477
+ loadStatusConfig();
478
+
479
+ // ─── Response Metadata Capture ────────────────────────────────────────────────
480
+ // The custom streamSimple below wraps fetch per request (never globalThis —
481
+ // concurrent main/helper requests would clobber a global patch). For every
482
+ // /chat/completions response we capture x-ratelimit-* headers and tee the
483
+ // body: one copy goes to pi's OpenAI streaming layer, the other is scanned
484
+ // for the final usage chunk (tokens, spend extension) that closes the stream.
485
+
486
+ const sessionStats: SessionStats = { ...EMPTY_SESSION_STATS };
487
+ const account: AccountState = { ...EMPTY_ACCOUNT };
488
+
489
+ // Per-turn pending state — teed streams settle asynchronously, so capture
490
+ // lands in pending* and is committed at turn_end.
491
+ let pendingRequests = 0;
492
+ let pendingTokens = 0;
493
+ let pendingSpend = 0;
494
+ let pendingSawUsage = false;
495
+ let pendingSawOutOfCredits = false;
496
+ let outOfCreditsNotified = false;
497
+
498
+ const teeReaders = new Set<Promise<void>>();
499
+
500
+ function trackTeeReader(promise: Promise<void>): void {
501
+ teeReaders.add(promise);
502
+ const release = () => { teeReaders.delete(promise); };
503
+ promise.then(release, release);
504
+ }
505
+
506
+ function settleTeeReaders(): Promise<void> {
507
+ if (teeReaders.size === 0) return Promise.resolve();
508
+ const pending = Array.from(teeReaders);
509
+ return Promise.allSettled(pending).then(() => undefined);
510
+ }
511
+
512
+ function captureRateLimitHeaders(headers: Headers): void {
513
+ const limit = Number(headers.get("x-ratelimit-limit"));
514
+ const remaining = Number(headers.get("x-ratelimit-remaining"));
515
+ if (Number.isFinite(limit) && Number.isFinite(remaining)) {
516
+ account.rate = { limit, remaining, capturedAt: Date.now() };
517
+ return;
518
+ }
519
+ // Fallback: hourly budget headers some proxies forward
520
+ const limitHour = Number(headers.get("x-ratelimit-limit-hour"));
521
+ const remainingHour = Number(headers.get("x-ratelimit-remaining-hour"));
522
+ if (Number.isFinite(limitHour) && Number.isFinite(remainingHour)) {
523
+ account.rate = { limit: limitHour, remaining: remainingHour, capturedAt: Date.now() };
524
+ }
525
+ }
526
+
527
+ /** Extract spend/token data from a parsed completion chunk/body's usage object. */
528
+ function captureUsage(obj: any): void {
529
+ const usage = obj?.usage;
530
+ if (typeof usage !== "object" || usage === null) return;
531
+ const input = usage.prompt_tokens;
532
+ const output = usage.completion_tokens;
533
+ if (typeof input === "number" && Number.isFinite(input)) pendingTokens += input;
534
+ if (typeof output === "number" && Number.isFinite(output)) pendingTokens += output;
535
+ // Metered proxies expose a cost extension (usage.total_cost / usage.cost.usd);
536
+ // Zro may or may not — token totals are the always-available fallback.
537
+ const total = usage.total_cost ?? usage.cost?.usd ?? usage.cost?.total;
538
+ if (typeof total === "number" && Number.isFinite(total)) {
539
+ pendingSpend += Math.max(0, total);
540
+ }
541
+ pendingSawUsage = true;
542
+ }
543
+
544
+ /** Scan a teed response for the final usage chunk (SSE) or JSON body usage. */
545
+ async function readUsageFromTee(body: ReadableStream<Uint8Array>): Promise<void> {
546
+ const reader = body.getReader();
547
+ const decoder = new TextDecoder();
548
+ let buffer = "";
549
+
550
+ const processLine = (line: string): void => {
551
+ const trimmed = line.trim();
552
+ if (!trimmed.startsWith("data: ")) return;
553
+ const payload = trimmed.slice(6);
554
+ if (payload === "[DONE]") return;
555
+ try {
556
+ captureUsage(JSON.parse(payload));
557
+ } catch {
558
+ // Not JSON or no usage — benign
559
+ }
560
+ };
561
+
562
+ try {
563
+ while (true) {
564
+ const { done, value } = await reader.read();
565
+ if (done) break;
566
+ buffer += decoder.decode(value, { stream: true });
567
+ const lines = buffer.split("\n");
568
+ buffer = lines.pop() || "";
569
+ for (const line of lines) processLine(line);
570
+ }
571
+ } catch {
572
+ // Tee stream may error if the main stream is aborted — that's fine
573
+ }
574
+
575
+ const trailing = (buffer + decoder.decode(new Uint8Array(0), { stream: false })).trim();
576
+ if (trailing) {
577
+ if (trailing.startsWith("data: ")) {
578
+ processLine(trailing);
579
+ } else if (trailing.startsWith("{")) {
580
+ try {
581
+ captureUsage(JSON.parse(trailing));
582
+ } catch {
583
+ // Partial non-SSE body — ignore
584
+ }
585
+ }
586
+ }
587
+
588
+ try {
589
+ reader.releaseLock();
590
+ } catch {
591
+ // Ignore
592
+ }
593
+ }
594
+
595
+ // ─── Custom Streaming Provider ────────────────────────────────────────────────
596
+
597
+ function streamZro(
598
+ model: any,
599
+ context: any,
600
+ options?: SimpleStreamOptions,
601
+ ): AssistantMessageEventStream {
602
+ const apiKey = (options as any)?.apiKey || cachedApiKey || storedZroCliKey() || "";
603
+ if (!apiKey) {
604
+ throw new Error(
605
+ `No API key for Zro. Run \`zro login\`, add it to ~/.pi/agent/auth.json, ` +
606
+ `set ZRO_API_KEY env var, or use --api-key.`,
607
+ );
608
+ }
609
+
610
+ const zroModel = { ...model, api: "openai-completions", baseUrl: model.baseUrl || BASE_URL };
611
+
612
+ // pi hands the user's thinking selection as options.reasoning (a raw
613
+ // ThinkingLevel); streamOpenAICompletions only reads reasoningEffort.
614
+ // Replicate pi-ai's clamp+convert so levels reach the request body.
615
+ const clampedReasoning = options?.reasoning ? clampThinkingLevel(zroModel, options.reasoning) : undefined;
616
+ const reasoningEffort = clampedReasoning === "off" ? undefined : clampedReasoning;
617
+ const { reasoning: _reasoning, ...streamOptions } = (options ?? {}) as any;
618
+
619
+ // Per-request fetch wrapper: owns its interceptor, safe under concurrency.
620
+ const upstreamFetch = (streamOptions as any).fetch ?? globalThis.fetch;
621
+ const metaFetch = async (input: RequestInfo | URL, init?: RequestInit) => {
622
+ const response = await upstreamFetch(input as any, init);
623
+ const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
624
+ if (!url.includes("/chat/completions")) return response;
625
+
626
+ pendingRequests += 1;
627
+ captureRateLimitHeaders(response.headers);
628
+ if (response.status === 402) pendingSawOutOfCredits = true;
629
+ if (!response.ok || !response.body) return response;
630
+
631
+ const [bodyForSdk, bodyForMeta] = response.body.tee();
632
+ trackTeeReader(readUsageFromTee(bodyForMeta));
633
+ return new Response(bodyForSdk, {
634
+ headers: response.headers,
635
+ status: response.status,
636
+ statusText: response.statusText,
637
+ });
638
+ };
639
+
640
+ return streamOpenAICompletions(zroModel, context, {
641
+ ...streamOptions,
642
+ fetch: metaFetch,
643
+ reasoningEffort,
644
+ apiKey,
645
+ } as any);
646
+ }
647
+
648
+ // ─── Account Metadata Fetching ────────────────────────────────────────────────
649
+
650
+ const STATUS_MIN_INTERVAL_MS = 15_000;
651
+ const ACCOUNT_FETCH_TIMEOUT_MS = 8_000;
652
+
653
+ let statusAbort: AbortController | null = null;
654
+ // Bumped on every session_start; async continuations compare against this to
655
+ // drop work belonging to a replaced session (its ctx is stale and throws).
656
+ let statusEpoch = 0;
657
+ let lastStatusFetchAt = 0;
658
+ let statusInFlight: Promise<void> | null = null;
659
+ let metaFetched = false;
660
+
661
+ async function fetchJsonGet(url: string, apiKey: string, signal?: AbortSignal): Promise<any | null> {
662
+ try {
663
+ const response = await fetch(url, {
664
+ headers: { Authorization: `Bearer ${apiKey}` },
665
+ signal: signal
666
+ ? AbortSignal.any([AbortSignal.timeout(ACCOUNT_FETCH_TIMEOUT_MS), signal])
667
+ : AbortSignal.timeout(ACCOUNT_FETCH_TIMEOUT_MS),
668
+ });
669
+ if (!response.ok) return null;
670
+ return await response.json();
671
+ } catch {
672
+ return null;
673
+ }
674
+ }
675
+
676
+ /** Plan/spend/pack/activity from /api/cli/status — the `zro status` view. Throttled unless forced. */
677
+ function refreshAccountStatus(apiKey: string | undefined, signal: AbortSignal | undefined, force: boolean): Promise<void> {
678
+ if (!apiKey) return Promise.resolve();
679
+ if (!force && Date.now() - lastStatusFetchAt < STATUS_MIN_INTERVAL_MS) return Promise.resolve();
680
+ lastStatusFetchAt = Date.now();
681
+ if (statusInFlight) return statusInFlight;
682
+ statusInFlight = (async () => {
683
+ try {
684
+ const data = await fetchJsonGet(STATUS_URL, apiKey, signal);
685
+ if (data === null) return;
686
+ const billing = data?.billing;
687
+ if (billing && typeof billing === "object") {
688
+ if (typeof billing.totalRemaining === "number") account.availableUsd = billing.totalRemaining;
689
+ if (typeof billing.usagePacks?.remaining === "number") account.usagePackUsd = billing.usagePacks.remaining;
690
+ const planName = billing.plan?.name;
691
+ if (typeof planName === "string" && planName.trim()) account.planName = planName.trim();
692
+ }
693
+ const activity = data?.activity30d;
694
+ if (activity && typeof activity === "object") {
695
+ if (typeof activity.requests === "number") account.activity30dRequests = activity.requests;
696
+ if (typeof activity.totalTokens === "number") account.activity30dTokens = activity.totalTokens;
697
+ }
698
+ if (typeof data?.key?.alias === "string" && data.key.alias.trim()) account.keyAlias = data.key.alias.trim();
699
+ metaFetched = true;
700
+ } finally {
701
+ statusInFlight = null;
702
+ }
703
+ })();
704
+ return statusInFlight;
705
+ }
706
+
707
+ // ─── Status Rendering ─────────────────────────────────────────────────────────
708
+
709
+ const WIDGET_KEY = "zro";
710
+ const STATUS_KEY_SESSION = "zro-session";
711
+ const STATUS_KEY_ACCOUNT = "zro-account";
712
+
713
+ function currentProviderId(ctx: ExtensionContext): string | undefined {
714
+ // ctx.model is a getter that can throw on stale contexts
715
+ try {
716
+ return (ctx.model as any)?.provider as string | undefined;
717
+ } catch {
718
+ return undefined;
719
+ }
720
+ }
721
+
722
+ function isStaleCtxError(err: unknown): boolean {
723
+ return err instanceof Error && err.message.includes("This extension ctx is stale");
724
+ }
725
+
726
+ // Render entry point: swallows the stale-ctx throw so a refresh racing a
727
+ // session replacement (newSession/fork/switchSession/reload) can't crash pi.
728
+ function updateStatus(ctx: ExtensionContext): void {
729
+ try {
730
+ renderStatus(ctx);
731
+ } catch (err) {
732
+ if (!isStaleCtxError(err)) throw err;
733
+ }
734
+ }
735
+
736
+ // Re-render once an async refresh lands, unless the session was replaced
737
+ // meanwhile (epoch bump) — its ctx is stale and the render is obsolete anyway.
738
+ function updateStatusAfter(promise: Promise<void>, ctx: ExtensionContext): void {
739
+ const epoch = statusEpoch;
740
+ void promise.then(() => {
741
+ if (epoch === statusEpoch) updateStatus(ctx);
742
+ });
743
+ }
744
+
745
+ function renderStatus(ctx: ExtensionContext): void {
746
+ const provider = currentProviderId(ctx);
747
+ const hiddenByOtherProvider =
748
+ statusConfig.hideOnOtherProvider && provider !== undefined && provider !== PROVIDER_ID;
749
+
750
+ const clearAll = () => {
751
+ ctx.ui.setStatus(STATUS_KEY_SESSION, undefined);
752
+ ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
753
+ ctx.ui.setWidget(WIDGET_KEY, undefined);
754
+ };
755
+
756
+ if (hiddenByOtherProvider) {
757
+ clearAll();
758
+ return;
759
+ }
760
+
761
+ const hasActivity = sessionStats.requests > 0 || sessionStats.tokens > 0 || sessionStats.spend > 0;
762
+ const sessionLine = statusConfig.session !== "off" ? buildSessionLine(sessionStats) : undefined;
763
+ // Show only after Zro activity this session (like pi-neuralwatt):
764
+ // no empty-gap line on fresh sessions, no stale account glare on other
765
+ // providers' sessions.
766
+ const accountVisible = statusConfig.account !== "off" && accountHasData(account) && hasActivity;
767
+ const lowBalance =
768
+ statusConfig.lowBalanceUsd !== null && account.availableUsd !== null && account.availableUsd <= statusConfig.lowBalanceUsd;
769
+ const accTiers = accountVisible ? buildAccountTiers(account, lowBalance) : [];
770
+
771
+ // Status bar (built-in footer slots)
772
+ const sBar = statusConfig.session === "statusbar" ? sessionLine : undefined;
773
+ const aBar = statusConfig.account === "statusbar" && accountVisible ? accTiers[0] : undefined;
774
+ if (sBar && aBar) {
775
+ // Combined to avoid eating two footer slots
776
+ ctx.ui.setStatus(STATUS_KEY_SESSION, ctx.ui.theme.fg(lowBalance ? "warning" : "dim", `${sBar} · ${aBar}`));
777
+ ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
778
+ } else {
779
+ ctx.ui.setStatus(STATUS_KEY_SESSION, sBar ? ctx.ui.theme.fg("dim", sBar) : undefined);
780
+ ctx.ui.setStatus(STATUS_KEY_ACCOUNT, aBar ? ctx.ui.theme.fg(lowBalance ? "warning" : "dim", aBar) : undefined);
781
+ }
782
+
783
+ // Below-editor widget (two-zone, width-aware)
784
+ const leftW = statusConfig.session === "widget" ? sessionLine : undefined;
785
+ const rightW = statusConfig.account === "widget" && accountVisible ? accTiers : undefined;
786
+ if (leftW !== undefined || (rightW !== undefined && rightW.length > 0)) {
787
+ ctx.ui.setWidget(
788
+ WIDGET_KEY,
789
+ (_tui: any, theme: any) => new StatusLineWidget(theme, leftW ?? "", rightW ?? [], lowBalance),
790
+ { placement: "belowEditor" },
791
+ );
792
+ } else {
793
+ ctx.ui.setWidget(WIDGET_KEY, undefined);
794
+ }
795
+ }
796
+
797
+ function resetStatusState(): void {
798
+ sessionStats.requests = 0;
799
+ sessionStats.tokens = 0;
800
+ sessionStats.spend = 0;
801
+ Object.assign(account, EMPTY_ACCOUNT);
802
+ pendingRequests = 0;
803
+ pendingTokens = 0;
804
+ pendingSpend = 0;
805
+ pendingSawUsage = false;
806
+ pendingSawOutOfCredits = false;
807
+ outOfCreditsNotified = false;
808
+ lastStatusFetchAt = 0;
809
+ metaFetched = false;
810
+ }
811
+
812
+ /** Commit per-turn pending capture into session state (after tees settle). */
813
+ function commitPending(ctx: ExtensionContext): void {
814
+ if (!pendingSawUsage && pendingRequests === 0) return;
815
+ sessionStats.requests += pendingRequests;
816
+ sessionStats.tokens += pendingTokens;
817
+ sessionStats.spend += pendingSpend;
818
+
819
+ // Optimistic balance: deduct this turn's observed spend so the account
820
+ // line ticks down per turn with zero extra API calls. Every status poll
821
+ // overwrites account.availableUsd (never adjusts), so this cannot
822
+ // double-count; the agent_settled poll reconciles any drift.
823
+ applyOptimisticSpend(account, pendingSpend);
824
+
825
+ pendingRequests = 0;
826
+ pendingTokens = 0;
827
+ pendingSpend = 0;
828
+ pendingSawUsage = false;
829
+
830
+ if (pendingSawOutOfCredits) {
831
+ pendingSawOutOfCredits = false;
832
+ // Re-fetch now so the balance reflects exhaustion immediately
833
+ updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
834
+ if (!outOfCreditsNotified && ctx.hasUI) {
835
+ outOfCreditsNotified = true;
836
+ ctx.ui.notify("Zro has run out of available spend — top up at zro.moonmath.ai", "error");
837
+ }
838
+ }
839
+ }
840
+
841
+ // ─── Status Command ────────────────────────────────────────────────────────────
842
+
843
+ function statusSummary(): string {
844
+ const lb = statusConfig.lowBalanceUsd === null ? "off" : `${statusConfig.lowBalanceUsd}`;
845
+ return `session=${statusConfig.session}, account=${statusConfig.account}, hideOnOtherProvider=${statusConfig.hideOnOtherProvider}, lowBalanceUsd=${lb}`;
846
+ }
847
+
848
+ const STATUS_USAGE =
849
+ "Usage: /zro-status [session|account widget|statusbar|off · hide true|false · lowBalance <usd>|off · refresh · reset]";
850
+
851
+ async function handleStatusCommand(args: string, ctx: ExtensionContext): Promise<void> {
852
+ const tokens = args.trim().split(/\s+/).filter(Boolean);
853
+
854
+ if (tokens.length === 0) {
855
+ if (!ctx.hasUI) {
856
+ ctx.ui.notify(statusSummary(), "info");
857
+ return;
858
+ }
859
+ await configureStatusInteractive(ctx);
860
+ return;
861
+ }
862
+
863
+ const [rawKey, rawValue] = tokens;
864
+ const key = rawKey.toLowerCase();
865
+ const value = rawValue?.toLowerCase();
866
+
867
+ if (key === "refresh") {
868
+ metaFetched = false;
869
+ await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true);
870
+ updateStatus(ctx);
871
+ const bal = account.availableUsd !== null ? `$${account.availableUsd} avail` : "unknown";
872
+ ctx.ui.notify(`Zro account: ${bal}. ${statusSummary()}`, "info");
873
+ return;
874
+ }
875
+
876
+ if (key === "reset" && tokens.length === 1) {
877
+ statusConfig = { ...DEFAULT_STATUS_CONFIG };
878
+ writeStatusConfig();
879
+ updateStatus(ctx);
880
+ ctx.ui.notify(`Zro status reset. ${statusSummary()}`, "info");
881
+ return;
882
+ }
883
+
884
+ if ((key === "session" || key === "account") && tokens.length === 2) {
885
+ if (value !== "widget" && value !== "statusbar" && value !== "off") {
886
+ ctx.ui.notify(STATUS_USAGE, "error");
887
+ return;
888
+ }
889
+ statusConfig[key] = value;
890
+ writeStatusConfig();
891
+ if (value !== "off" && key === "account") {
892
+ // Turning account on: make sure we have data to show
893
+ updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
894
+ }
895
+ updateStatus(ctx);
896
+ ctx.ui.notify(`Zro ${key} line: ${value}. ${statusSummary()}`, "info");
897
+ return;
898
+ }
899
+
900
+ if ((key === "hide" || key === "hideonotherprovider") && tokens.length === 2) {
901
+ if (value !== "true" && value !== "false") {
902
+ ctx.ui.notify(STATUS_USAGE, "error");
903
+ return;
904
+ }
905
+ statusConfig.hideOnOtherProvider = value === "true";
906
+ writeStatusConfig();
907
+ updateStatus(ctx);
908
+ ctx.ui.notify(`Zro status. ${statusSummary()}`, "info");
909
+ return;
910
+ }
911
+
912
+ if (key === "lowbalance" && tokens.length === 2) {
913
+ if (value === "off") {
914
+ statusConfig.lowBalanceUsd = null;
915
+ } else {
916
+ const n = Number(value);
917
+ if (!Number.isFinite(n) || n <= 0) {
918
+ ctx.ui.notify(STATUS_USAGE, "error");
919
+ return;
920
+ }
921
+ statusConfig.lowBalanceUsd = n;
922
+ }
923
+ writeStatusConfig();
924
+ updateStatus(ctx);
925
+ ctx.ui.notify(`Zro status. ${statusSummary()}`, "info");
926
+ return;
927
+ }
928
+
929
+ ctx.ui.notify(STATUS_USAGE, "error");
930
+ }
931
+
932
+ async function configureStatusInteractive(ctx: ExtensionContext): Promise<void> {
933
+ const modes = ["widget", "statusbar", "off"] as const;
934
+ const nextMode = (m: string) => modes[(modes.indexOf(m as any) + 1) % modes.length];
935
+
936
+ for (;;) {
937
+ const lb = statusConfig.lowBalanceUsd === null ? "off" : `$${statusConfig.lowBalanceUsd}`;
938
+ const sessionOpt = `Session line (spend/tokens/requests): ${statusConfig.session}`;
939
+ const accountOpt = `Account line (plan/balance/packs/activity): ${statusConfig.account}`;
940
+ const hideOpt = `Hide on other providers: ${statusConfig.hideOnOtherProvider ? "on" : "off"}`;
941
+ const lbOpt = `Low-balance warning: ${lb}`;
942
+ const refreshOpt = "Refresh account status now";
943
+ const doneOpt = "Done";
944
+
945
+ const choice = await ctx.ui.select("Zro footer status", [
946
+ sessionOpt,
947
+ accountOpt,
948
+ hideOpt,
949
+ lbOpt,
950
+ refreshOpt,
951
+ doneOpt,
952
+ ]);
953
+
954
+ if (choice === undefined || choice === doneOpt) {
955
+ updateStatus(ctx);
956
+ return;
957
+ }
958
+ if (choice === sessionOpt) {
959
+ statusConfig.session = nextMode(statusConfig.session);
960
+ writeStatusConfig();
961
+ continue;
962
+ }
963
+ if (choice === accountOpt) {
964
+ statusConfig.account = nextMode(statusConfig.account);
965
+ writeStatusConfig();
966
+ if (statusConfig.account !== "off") {
967
+ updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true), ctx);
968
+ }
969
+ continue;
970
+ }
971
+ if (choice === hideOpt) {
972
+ statusConfig.hideOnOtherProvider = !statusConfig.hideOnOtherProvider;
973
+ writeStatusConfig();
974
+ updateStatus(ctx);
975
+ continue;
976
+ }
977
+ if (choice === lbOpt) {
978
+ const presets = ["off", "1", "5", "10", "25", "50", "100"];
979
+ const current = statusConfig.lowBalanceUsd === null ? "off" : String(statusConfig.lowBalanceUsd);
980
+ const ordered = presets.includes(current) ? presets : [current, ...presets];
981
+ const pick = await ctx.ui.select("Warn at/below balance (USD)", ordered);
982
+ if (pick !== undefined) {
983
+ statusConfig.lowBalanceUsd = pick === "off" ? null : Number(pick);
984
+ writeStatusConfig();
985
+ updateStatus(ctx);
986
+ }
987
+ continue;
988
+ }
989
+ if (choice === refreshOpt) {
990
+ metaFetched = false;
991
+ await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, true);
992
+ updateStatus(ctx);
993
+ continue;
994
+ }
995
+ }
996
+ }
997
+
998
+ // ─── Extension Entry Point ────────────────────────────────────────────────────
999
+
1000
+ // The currently-registered model list — starts stale, hot-swapped when the
1001
+ // live catalog lands. Provider identity funnels through makeProviderConfig so
1002
+ // the stream handler and models never desync.
1003
+ let currentModels: JsonModel[] = [];
1004
+
1005
+ function makeProviderConfig(models: JsonModel[] = currentModels) {
1006
+ return {
1007
+ baseUrl: BASE_URL,
1008
+ apiKey: "$ZRO_API_KEY",
1009
+ // Custom API name so our streamSimple registers as its own handler and
1010
+ // never shadows pi's built-in openai-completions pipeline for other
1011
+ // providers. streamZro delegates to pi-ai's OpenAI-compat streamer.
1012
+ api: "zro",
1013
+ headers: { "User-Agent": "pi-coding-agent" },
1014
+ models,
1015
+ streamSimple: streamZro,
1016
+ };
1017
+ }
1018
+
1019
+ export default function (pi: ExtensionAPI) {
1020
+ const embeddedModels = modelsData as JsonModel[];
1021
+ const customModels = customModelsData as JsonModel[];
1022
+ const patches = patchData as PatchData;
1023
+
1024
+ const staleBase = loadStaleModels(embeddedModels);
1025
+ const staleModels = buildModels(staleBase, customModels, patches);
1026
+ currentModels = staleModels;
1027
+
1028
+ pi.registerProvider(PROVIDER_ID, makeProviderConfig(staleModels));
1029
+
1030
+ pi.registerCommand("zro-status", {
1031
+ description: "Configure the Zro footer status (session spend, account balance, packs, activity)",
1032
+ handler: async (args, ctx) => {
1033
+ await handleStatusCommand(args, ctx);
1034
+ },
1035
+ });
1036
+
1037
+ pi.on("session_start", async (_event, ctx) => {
1038
+ const epoch = ++statusEpoch;
1039
+ revalidateAbort?.abort();
1040
+ revalidateAbort = new AbortController();
1041
+ const signal = revalidateAbort.signal;
1042
+ statusAbort?.abort();
1043
+ statusAbort = new AbortController();
1044
+ const statusSignal = statusAbort.signal;
1045
+
1046
+ loadStatusConfig();
1047
+ resetStatusState();
1048
+ updateStatus(ctx); // clears any carryover; activity-gated, renders nothing yet
1049
+ // Re-register so our identity (custom api + streamSimple) always wins
1050
+ // over anything that touched provider registration during load.
1051
+ pi.registerProvider(PROVIDER_ID, makeProviderConfig());
1052
+
1053
+ resolveApiKey(ctx.modelRegistry).then(() => {
1054
+ // A session replacement while the key resolved invalidated the
1055
+ // captured ctx (fast-resume, /new, /fork); nothing below may touch it.
1056
+ if (epoch !== statusEpoch) return;
1057
+ // Prefetch account status only when a Zro model is active
1058
+ // (pi-neuralwatt also prefetches so the first turn ends with data, but
1059
+ // gating here avoids API calls in sessions that never use the provider).
1060
+ if (currentProviderId(ctx) === PROVIDER_ID) {
1061
+ updateStatusAfter(refreshAccountStatus(cachedApiKey, statusSignal, true), ctx);
1062
+ }
1063
+ revalidateModels(cachedApiKey, embeddedModels, signal).then((freshBase) => {
1064
+ if (freshBase && epoch === statusEpoch && !signal.aborted) {
1065
+ currentModels = buildModels(freshBase, customModels, patches);
1066
+ pi.registerProvider(PROVIDER_ID, makeProviderConfig());
1067
+ }
1068
+ });
1069
+ });
1070
+ });
1071
+
1072
+ pi.on("model_select", (event, ctx) => {
1073
+ updateStatus(ctx);
1074
+ const model: any = (event as any).model;
1075
+ if (model?.provider === PROVIDER_ID && cachedApiKey) {
1076
+ updateStatusAfter(refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false), ctx);
1077
+ }
1078
+ });
1079
+
1080
+ pi.on("turn_end", async (_event, ctx) => {
1081
+ // Ensure every concurrent response tee has landed before committing.
1082
+ await settleTeeReaders();
1083
+ commitPending(ctx);
1084
+ // If the session_start/model_select status fetch raced or failed, retry
1085
+ // once we have real activity so the very first turn shows the balance.
1086
+ if (sessionStats.requests > 0 && account.availableUsd === null && !metaFetched) {
1087
+ await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false);
1088
+ }
1089
+ updateStatus(ctx);
1090
+ });
1091
+
1092
+ // agent_settled (not agent_end): fires only when no automatic retry,
1093
+ // compaction, or queued continuation can follow — the one moment polling
1094
+ // /api/cli/status is both fresh and not redundant. Gated on session activity
1095
+ // so sessions without Zro turns make zero API calls here.
1096
+ pi.on("agent_settled", async (_event, ctx) => {
1097
+ if (sessionStats.requests > 0 || sessionStats.tokens > 0 || sessionStats.spend > 0) {
1098
+ await refreshAccountStatus(cachedApiKey, statusAbort?.signal ?? undefined, false);
1099
+ updateStatus(ctx);
1100
+ }
1101
+ });
1102
+
1103
+ pi.on("session_shutdown", (_event, ctx) => {
1104
+ revalidateAbort?.abort();
1105
+ statusAbort?.abort();
1106
+ ctx.ui.setStatus(STATUS_KEY_SESSION, undefined);
1107
+ ctx.ui.setStatus(STATUS_KEY_ACCOUNT, undefined);
1108
+ ctx.ui.setWidget(WIDGET_KEY, undefined);
1109
+ });
1110
+ }