@genee/omp-opsx-addon 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1200 @@
1
+ /**
2
+ * System-level shared usage cache — persisted to `~/.omp/cache/` so multiple
3
+ * OMP processes share one fetch cycle instead of hammering provider APIs.
4
+ *
5
+ * Fetching is **per-provider independent**. Each provider tracks its own
6
+ * last-fetch/attempt timestamps and its own pending/loading flag, so:
7
+ * - a `turn_end` poke marks ONLY the current model's provider dirty (direct
8
+ * fetcher → one HTTP call; never touches the bulk API);
9
+ * - the tick fetches only dirty providers whose minimum-interval floor has
10
+ * elapsed, concurrently — a slow provider never blocks another;
11
+ * - a provider being fetched shows a `⟳` loading hint (header) while its
12
+ * previous report stays visible, and an unfetched logged-in placeholder
13
+ * shows `⟳` instead of `—` until the first fetch lands.
14
+ *
15
+ * Triggering is a **dirty-flag model**: every "fetch is due" signal — a local
16
+ * `turn_end`, another process's consumption landing on disk, a forced trigger
17
+ * (reset-imminent / exhausted-unconfirmed), or the idle fallback period — only
18
+ * marks the provider dirty. The actual fetch happens in one place: `fetchDue`,
19
+ * run on every render tick, honoring a per-provider minimum interval floor
20
+ * (`minFetchIntervalMs`, default 10s) and skipping in-flight providers. A
21
+ * failed fetch re-marks the provider dirty so the next floor-eligible tick
22
+ * retries. The idle fallback (`idleDirtyMs`, default 60s) marks every
23
+ * provider whose last SUCCESSFUL fetch is older than the period, so quiet
24
+ * sessions still refresh at most once per 60s per provider.
25
+ *
26
+ * One `renderTimer` ticks every `renderTickMs` (~1s). Each firing runs
27
+ * `fetchDue` and renders the widget (`onTick`), and — when the absolute
28
+ * elapsed time since the previous data tick reaches `tickMs` (~5s) — also
29
+ * reads disk (`syncFromDisk`), evaluates forced/idle triggers, and fetches.
30
+ * Absolute time (not tick counting) decides refreshes, so a suspended machine
31
+ * (closed laptop lid) refreshes immediately on its first tick after wake
32
+ * instead of losing cycles.
33
+ * The standalone {@link pokeProvider} entry marks providers dirty and runs the
34
+ * same `fetchDue` kernel immediately (optionally bypassing the floor with
35
+ * `force`, for decision paths that await fresh data).
36
+
37
+ * The cache is split into TWO files under `~/.omp/cache/`: a REALTIME file
38
+ * (`omp-opsx-addon-usage-realtime.json`, `sessionUsage` only) and a FETCH file
39
+ * (`omp-opsx-addon-usage-fetch.json`, `reports` + `fetchedByProvider` +
40
+ * `fetchedAt` only). The legacy single file is migrated on first start and
41
+ * removed. Both files share the same cross-process LOCKED read-modify-write
42
+ * protocol: an exclusive lock file (`<file>.lock`, stolen when older than 5s)
43
+ * serializes concurrent writers; each writer re-reads the file inside the lock
44
+ * and lands the serialized payload via an atomic tmp+rename. Readers take no
45
+ * lock — the rename swap guarantees they see a complete old-or-new file.
46
+ * Concurrent sessions consuming in the same second therefore queue their
47
+ * writes and keep their per-second increments under the same absolute
48
+ * unix-second keys instead of clobbering each other's `sessionUsage` entries.
49
+ * Realtime entries whose `updatedAt` is older than `REALTIME_TTL_MS` are
50
+ * reclaimed on write (crash-residue self-healing).
51
+ */
52
+
53
+ import { existsSync, readFileSync, writeFileSync, mkdirSync, renameSync, statSync, openSync, writeSync, closeSync, rmSync } from 'fs';
54
+ import { join, dirname, extname } from 'path';
55
+ import { homedir } from 'os';
56
+ import type { UsageReport } from '@oh-my-pi/pi-ai';
57
+ import type { AuthStorageLike, ProviderHealth, DirectFetcher, ApiKeyResolver } from './usage-resolver.js';
58
+ import { resolveProviderReports, buildHealthMap, canonicalizeProvider, listDetectableProviders } from './usage-resolver.js';
59
+ import { RedisRelay } from './usage-redis-probe.js';
60
+ import type { RedisTransport } from './usage-redis-client.js';
61
+ import type { UsageBucket } from './usage-estimator.js';
62
+ import { normalizeBucket } from './usage-estimator.js';
63
+ import type { ProviderUsageTotals } from './session-usage.js';
64
+ interface SharedUsageState {
65
+ reports: UsageReport[];
66
+ health: Map<string, ProviderHealth>;
67
+ fetchedAt: number;
68
+ /** Canonical ids of providers with an in-flight fetch (loading state). */
69
+ loadingProviders: ReadonlySet<string>;
70
+ }
71
+
72
+ /** Realtime cache file: GLOBAL cross-session aggregates (totals + per-second). */
73
+ interface RealtimeFile {
74
+ /** provider → cumulative bucketed totals (summed across every session). */
75
+ totals: Record<string, UsageBucket>;
76
+ /** unix second → provider → tokens consumed during that second (global sum). */
77
+ seconds: Record<string, Record<string, number>>;
78
+ }
79
+
80
+ /** Fetch cache file: provider reports + per-provider freshness timestamps. */
81
+ interface FetchFile {
82
+ fetchedAt: number;
83
+ reports: UsageReport[];
84
+ fetchedByProvider?: Record<string, number>;
85
+ }
86
+
87
+ /** Legacy single-file cache shape — migration source only (per-session entries). */
88
+ interface LegacyCacheFile {
89
+ fetchedAt: number;
90
+ reports: UsageReport[];
91
+ fetchedByProvider?: Record<string, number>;
92
+ sessionUsage?: Record<string, { perProvider?: Record<string, ProviderUsageTotals | number>; updatedAt?: number }>;
93
+ }
94
+
95
+ const CACHE_DIR = join(homedir(), '.omp', 'cache');
96
+ /** Legacy single cache file — kept ONLY as the one-time migration source. */
97
+ const CACHE_FILE = join(CACHE_DIR, 'omp-opsx-addon-usage.json');
98
+ /** Realtime cache file: global totals + per-second aggregates. */
99
+ export const CACHE_REALTIME_FILE = join(CACHE_DIR, 'omp-opsx-addon-usage-realtime.json');
100
+ /** Fetch cache file: `reports` + `fetchedByProvider` + `fetchedAt` only. */
101
+ export const CACHE_FETCH_FILE = join(CACHE_DIR, 'omp-opsx-addon-usage-fetch.json');
102
+
103
+ /** Default data-refresh cadence (~5s) — absolute elapsed time checked on each render tick. */
104
+ export const DEFAULT_TICK_MS = 5_000;
105
+ /** Default render cadence — the countdown advances every second. */
106
+ export const DEFAULT_RENDER_TICK_MS = 1_000;
107
+ /** Default minimum interval between fetch attempts per provider (10s). */
108
+ export const DEFAULT_MIN_FETCH_INTERVAL_MS = 10_000;
109
+ /** Default idle fallback period (60s): providers with no consumption still re-fetch. */
110
+ export const DEFAULT_IDLE_DIRTY_MS = 60_000;
111
+ /** Threshold for a "即将重置" forced fetch: any window resetting within this. */
112
+ export const RESET_IMMINENT_MS = 60_000;
113
+
114
+ /** Options for {@link startSharedPoller}. */
115
+ export interface StartPollerOptions {
116
+ /** Data-refresh cadence (ms of absolute elapsed time between refreshes). Default 5s. */
117
+ tickMs?: number;
118
+ /** Render cadence (countdown seconds). Default 1s; data refreshes are checked on this tick. */
119
+ renderTickMs?: number;
120
+ /** Minimum interval between fetch attempts per provider. Default 10s. */
121
+ minFetchIntervalMs?: number;
122
+ /** Idle fallback: providers whose last successful fetch is older than this get marked dirty. Default 60s. */
123
+ idleDirtyMs?: number;
124
+ /** Fired once at the end of every tick (after syncFromDisk + any fetch). */
125
+ onTick?: () => void;
126
+ /** Whether the calling session owns a real UI surface. A non-UI session (in-process task subagent whose ctx.ui is a no-op) must NOT steal the tick-render callback from the UI-owning session. */
127
+ ownsUi?: boolean;
128
+ /** Optional local Redis relay (write-through mirror + read prefetch + reconcile). `enabled: false` or omitted keeps the pure file path. */
129
+ redis?: RedisPollerOptions;
130
+ }
131
+
132
+ /** Redis relay sub-config passed to {@link startSharedPoller} (mirrors `RedisRelayOptions`). */
133
+ export interface RedisPollerOptions {
134
+ /** Total switch. `false` disables probe/relay entirely (pure file path). */
135
+ enabled: boolean;
136
+ host: string;
137
+ port: number;
138
+ connectTimeoutMs: number;
139
+ commandTimeoutMs: number;
140
+ /** Reconcile cadence (absolute elapsed time between per-key overwrites). */
141
+ reconcileIntervalMs: number;
142
+ probeIntervalMs: number;
143
+ unhealthyRetryMs: number;
144
+ ttlMs: number;
145
+ /** Test seam: injected into the underlying RedisClient. */
146
+ transport?: RedisTransport;
147
+
148
+ }
149
+
150
+ /** Per-provider last-SUCCESSFUL-fetch timestamp (canonical id → epoch ms). */
151
+ const lastFetchedByProvider = new Map<string, number>();
152
+ /** Per-provider last-ATTEMPT timestamp (canonical id → epoch ms; gates the retry floor). */
153
+ const lastAttemptByProvider = new Map<string, number>();
154
+ /** Canonical ids marked dirty: a fetch is due (triggered, not yet consumed). */
155
+ const dirtyProviders = new Set<string>();
156
+ /** Canonical ids currently mid-fetch (the loading set). */
157
+ const pendingProviders = new Set<string>();
158
+ /** Runtime fetch floor — set by `startSharedPoller` from `StartPollerOptions`. */
159
+ let minFetchIntervalMs = DEFAULT_MIN_FETCH_INTERVAL_MS;
160
+ /** Runtime idle fallback period — set by `startSharedPoller` from `StartPollerOptions`. */
161
+ let idleDirtyMs = DEFAULT_IDLE_DIRTY_MS;
162
+
163
+ /**
164
+ * Providers whose most recent fetch attempt produced NO fresh report (direct
165
+ * fetcher returned null or threw). Any existing health for them is
166
+ * UNCONFIRMED — it may be a stale snapshot (e.g. an old "exhausted" report
167
+ * that a healthy provider like deepseek never produced itself). Selection
168
+ * still reads their current health, but the auto-refresh path re-attempts
169
+ * these providers every tick (bypassing TTL) so a stale report can't keep
170
+ * blocking selection for long.
171
+ */
172
+ const unconfirmedProviders = new Set<string>();
173
+ /**
174
+ * Providers whose exhausted health THIS process confirmed via a fresh fetch.
175
+ * Guards `syncFromDisk` from re-flagging a provider right after `fetchProvider`
176
+ * confirmed it and wrote the (exhausted) report to disk — without it, disk
177
+ * reads would re-add the provider to {@link unconfirmedProviders} and hammer it.
178
+ */
179
+ const confirmedThisProcess = new Set<string>();
180
+ /** Canonical ids of providers covered by a direct fetcher (the fresh source). */
181
+ const directFetcherIds = new Set<string>();
182
+ let cachedReports: UsageReport[] = [];
183
+ let cachedHealth: Map<string, ProviderHealth> = new Map();
184
+ /** Single interval: renders every tick AND checks absolute elapsed time for the data refresh. */
185
+ let renderTimer: NodeJS.Timeout | null = null;
186
+ /**
187
+ * Tick render callback, shared with the standalone {@link pokeProvider} path.
188
+ * `startSharedPoller` stores the caller's `onTick` here; both tick and
189
+ * `pokeProvider` read this ref so there is a single source of truth for the
190
+ * registered render callback regardless of which path fires.
191
+ */
192
+ let onTickRef: (() => void) | undefined;
193
+ let startInFlight: Promise<void> | null = null;
194
+ /**
195
+ * Process-global tick-owner claim. The harness loads this plugin as a fresh
196
+ * module instance per session via `import(path?mtime=…)`, so a module-level
197
+ * claim is invisible to sibling instances sharing this JS realm — a subagent
198
+ * instance would otherwise misread `hasTickOwner()` as false and steal the
199
+ * owner's render callback + recorder key (double-counting on disk merge).
200
+ * The claim therefore lives on `globalThis`, keyed with the plugin name.
201
+ * The local {@link onTickRef} still holds the actual render callback (only
202
+ * meaningful in the module graph that registered it); the ownership check
203
+ * reads this shared claim instead.
204
+ */
205
+ const TICK_OWNER_KEY = Symbol.for('omp-opsx-addon.usage-poller.tickOwner');
206
+ function setTickOwner(cb: (() => void) | undefined): void {
207
+ (globalThis as Record<symbol, unknown>)[TICK_OWNER_KEY] = cb;
208
+ }
209
+ function getTickOwner(): (() => void) | undefined {
210
+ return (globalThis as Record<symbol, unknown>)[TICK_OWNER_KEY] as (() => void) | undefined;
211
+ }
212
+
213
+ /**
214
+ * Local Redis relay (write-through mirror / prefetch / reconcile). `null` =
215
+ * disabled or not yet configured — every path falls back to the realtime file.
216
+ */
217
+ let relay: RedisRelay | null = null;
218
+ /**
219
+ * Redis-prefetched read cache (healthy mode only). Stat-gated on the realtime
220
+ * file with the same ino/mtimeMs/size triple as the file-side caches: the
221
+ * prefetch step refreshes it when the file changes; the synchronous getters
222
+ * only ever return this cache while healthy.
223
+ */
224
+ let redisUsageCache: {
225
+ ino: number;
226
+ mtimeMs: number;
227
+ size: number;
228
+ totals: Record<string, UsageBucket>;
229
+ perSecond: Record<string, Record<string, number>>;
230
+ } | null = null;
231
+ /** Reconcile cadence set at relay creation (default 30s, design D5). */
232
+ let reconcileIntervalMs = 30_000;
233
+
234
+
235
+ // Test seams: the cache files are resolved lazily so tests can redirect I/O to
236
+ // temp paths without touching the real `~/.omp/cache`. `null` = production path.
237
+ let realtimeFileOverride: string | null = null;
238
+ let fetchFileOverride: string | null = null;
239
+ let legacyFileOverride: string | null = null;
240
+ const resolveRealtimeFile = (): string => realtimeFileOverride ?? CACHE_REALTIME_FILE;
241
+ const resolveFetchFile = (): string => fetchFileOverride ?? CACHE_FETCH_FILE;
242
+ const resolveLegacyCacheFile = (): string => legacyFileOverride ?? CACHE_FILE;
243
+
244
+ /** Stat the realtime file for the relay prefetch gate; null when absent/unreadable. */
245
+ function statRealtimeForRelay(): { ino: number; mtimeMs: number; size: number } | null {
246
+ try {
247
+ const st = statSync(resolveRealtimeFile());
248
+ return { ino: st.ino, mtimeMs: st.mtimeMs, size: st.size };
249
+ } catch {
250
+ return null;
251
+ }
252
+ }
253
+
254
+ /** Derive a sibling path (`<base>-<suffix><ext>`) from a realtime path. */
255
+ function deriveSiblingPath(realtimePath: string, suffix: string): string {
256
+ const ext = extname(realtimePath);
257
+ const base = ext ? realtimePath.slice(0, -ext.length) : realtimePath;
258
+ return `${base}-${suffix}${ext}`;
259
+ }
260
+
261
+ /** Newest per-provider fetch timestamp (0 when nothing fetched yet). */
262
+ function globalFetchedAt(): number {
263
+ let max = 0;
264
+ for (const t of lastFetchedByProvider.values()) if (t > max) max = t;
265
+ return max;
266
+ }
267
+
268
+ function buildCachedState(): SharedUsageState {
269
+ return {
270
+ reports: cachedReports,
271
+ health: cachedHealth,
272
+ fetchedAt: globalFetchedAt(),
273
+ loadingProviders: pendingProviders,
274
+ };
275
+ }
276
+
277
+ /** Record which providers are covered by a direct fetcher (the fresh source). */
278
+ function setDirectFetcherCoverage(directFetchers: DirectFetcher[]): void {
279
+ for (const f of directFetchers) directFetcherIds.add(canonicalizeProvider(f.provider));
280
+ }
281
+
282
+ /**
283
+ * Disk-loaded reports are UNCONFIRMED for direct-fetcher providers whose
284
+ * health says exhausted: the direct fetcher is authoritative, and disk may hold
285
+ * a stale exhausted snapshot (e.g. an older bulk/discovery write) that a fresh
286
+ * process would otherwise trust silently (fresh timestamps → not TTL-stale →
287
+ * never re-fetched). Flagging them keeps the auto path re-attempting until THIS
288
+ * process confirms the real state — which recovers a healthy PAYG provider
289
+ * (deepseek/zhipu) instead of leaving it "exhausted". Providers already
290
+ * confirmed this process are skipped so a genuinely-exhausted provider is not
291
+ * re-hammered.
292
+ */
293
+ function seedUnconfirmedFromDisk(): void {
294
+ if (directFetcherIds.size === 0) return;
295
+ for (const [p, h] of cachedHealth) {
296
+ if (h.exhausted && directFetcherIds.has(p) && !confirmedThisProcess.has(p)) {
297
+ unconfirmedProviders.add(p);
298
+ }
299
+ }
300
+ }
301
+
302
+ /** Read the realtime cache file, tolerating missing/partial fields. */
303
+ function readRealtimeDisk(): RealtimeFile | null {
304
+ try {
305
+ const file = resolveRealtimeFile();
306
+ if (!existsSync(file)) return null;
307
+ const parsed = JSON.parse(readFileSync(file, 'utf-8')) as Partial<RealtimeFile> & { sessionUsage?: LegacyCacheFile['sessionUsage'] };
308
+ if (typeof parsed !== 'object' || parsed === null) return null;
309
+ // Older realtime shape (per-session entries): aggregate on read — the
310
+ // next locked write re-serializes the file in the global shape.
311
+ if (parsed.sessionUsage !== undefined) {
312
+ return aggregateLegacySessionUsage(parsed.sessionUsage);
313
+ }
314
+ return {
315
+ totals: (parsed.totals ?? {}) as Record<string, UsageBucket>,
316
+ seconds: (parsed.seconds ?? {}) as Record<string, Record<string, number>>,
317
+ };
318
+ } catch {
319
+ return null;
320
+ }
321
+ }
322
+
323
+ /** Read the fetch cache file, tolerating missing/partial fields. */
324
+ function readFetchDisk(): FetchFile | null {
325
+ try {
326
+ const file = resolveFetchFile();
327
+ if (!existsSync(file)) return null;
328
+ const parsed = JSON.parse(readFileSync(file, 'utf-8')) as Partial<FetchFile>;
329
+ return {
330
+ fetchedAt: typeof parsed.fetchedAt === 'number' ? parsed.fetchedAt : 0,
331
+ reports: Array.isArray(parsed.reports) ? parsed.reports : [],
332
+ fetchedByProvider: parsed.fetchedByProvider ?? {},
333
+ };
334
+ } catch {
335
+ return null;
336
+ }
337
+ }
338
+
339
+ /**
340
+ * Per-session snapshot of what THIS process last wrote to the realtime file —
341
+ * the delta baseline. On flush we add `current - lastFlushed` to the file's
342
+ * GLOBAL aggregates under the lock, then advance the snapshot. A failed write
343
+ * leaves the snapshot untouched, so the same delta is re-applied next time
344
+ * (no double count, no loss). Session shutdown just drops the baseline — the
345
+ * already-flushed aggregates stay (totals are permanent cross-session sums).
346
+ */
347
+ const lastFlushedBySession = new Map<string, Record<string, ProviderUsageTotals>>();
348
+ /** Last-seen global totals from `syncFromDisk` — the cross-process consumption signal baseline. */
349
+ let lastSyncedTotals: Map<string, number> = new Map();
350
+
351
+ function ensureDir(dir: string): void {
352
+ if (!existsSync(dir)) mkdirSync(dir, { recursive: true });
353
+ }
354
+
355
+ // ── cross-process write lock ───────────────────────────────────────
356
+ // Every write cycle is a locked read-modify-write: under the exclusive lock we
357
+ // re-read the file, add this process's deltas, serialize, and atomically
358
+ // rename. Concurrent writers queue on the lock (blocking acquire), so two
359
+ // sessions consuming in the same second accumulate instead of clobbering.
360
+ // Readers take NO lock — the tmp+rename swap guarantees a complete snapshot.
361
+ // A lock left by a crashed process is stolen once older than LOCK_STALE_MS.
362
+
363
+
364
+ const LOCK_STALE_MS = 5_000;
365
+ const LOCK_RETRY_MS = 10;
366
+
367
+ function sleepSync(ms: number): void {
368
+ try {
369
+ Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms);
370
+ } catch {
371
+ const until = Date.now() + ms;
372
+ while (Date.now() < until) { /* spin */ }
373
+ }
374
+ }
375
+
376
+ /** Acquire the exclusive write lock for `file`; returns its release fn. */
377
+ function acquireWriteLock(file: string): () => void {
378
+ const lockPath = `${file}.lock`;
379
+ for (;;) {
380
+ try {
381
+ const fd = openSync(lockPath, 'wx');
382
+ writeSync(fd, String(process.pid));
383
+ closeSync(fd);
384
+ return () => {
385
+ try { rmSync(lockPath); } catch { /* already gone */ }
386
+ };
387
+ } catch {
388
+ // Held by another process (or a crashed one). Steal stale locks.
389
+ try {
390
+ const st = statSync(lockPath);
391
+ if (Date.now() - st.mtimeMs > LOCK_STALE_MS) {
392
+ try { rmSync(lockPath); } catch { /* raced */ }
393
+ continue;
394
+ }
395
+ } catch {
396
+ continue; // lock vanished — retry immediately
397
+ }
398
+ sleepSync(LOCK_RETRY_MS);
399
+ }
400
+ }
401
+ }
402
+ let writeSeq = 0;
403
+
404
+ /**
405
+ * Locked atomic replace of `file` with the payload `payload()` produces.
406
+ * `payload()` runs INSIDE the lock so a caller can build the payload from a
407
+ * lock-consistent disk read (read-modify-write); concurrent writers queue on
408
+ * the lock, and the tmp+rename swap guarantees a fresh inode per write so read
409
+ * gates keyed on ino never miss an update (even same-ms, same-size).
410
+ */
411
+ function writePayloadLocked(file: string, payload: () => string): boolean {
412
+ const release = acquireWriteLock(file);
413
+ try {
414
+ ensureDir(dirname(file));
415
+ const body = payload();
416
+ const tmp = `${file}.tmp-${process.pid}-${writeSeq++}`;
417
+ writeFileSync(tmp, body, 'utf-8');
418
+ renameSync(tmp, file);
419
+ return true;
420
+ } catch {
421
+ return false;
422
+ } finally {
423
+ release();
424
+ }
425
+ }
426
+
427
+ /** Locked atomic write of a realtime payload (global totals + per-second sums). */
428
+ function writeRealtimePayload(disk: RealtimeFile): boolean {
429
+ return writePayloadLocked(resolveRealtimeFile(), () => JSON.stringify(disk, null, 2));
430
+ }
431
+
432
+ /** Locked atomic write of a fetch payload (only reports/fetchedByProvider/fetchedAt). */
433
+ function writeFetchPayload(fetch: {
434
+ fetchedAt: number;
435
+ reports: UsageReport[];
436
+ fetchedByProvider?: Record<string, number>;
437
+ }): boolean {
438
+ return writePayloadLocked(resolveFetchFile(), () => JSON.stringify(fetch, null, 2));
439
+ }
440
+
441
+ /**
442
+ * Realtime write path: under the lock, re-read the file (the LATEST global
443
+ * aggregates), add this process's per-provider deltas onto the corresponding
444
+ * absolute-second slots and totals, prune the seconds tail to 120s, and
445
+ * atomically rename. Every writer therefore accumulates on the freshest data —
446
+ * concurrent sessions sum instead of clobbering.
447
+ */
448
+ function writeRealtimeDisk(deltas: Record<string, ProviderUsageTotals>): boolean {
449
+ return writePayloadLocked(resolveRealtimeFile(), () => {
450
+ const disk = readRealtimeDisk() ?? { totals: {}, seconds: {} };
451
+ for (const [provider, delta] of Object.entries(deltas)) {
452
+ const prev = normalizeBucket(disk.totals[provider]);
453
+ disk.totals[provider] = {
454
+ total: prev.total + delta.total,
455
+ input: prev.input + delta.input,
456
+ output: prev.output + delta.output,
457
+ };
458
+ for (const [sec, dv] of Object.entries(delta.seconds ?? {})) {
459
+ if (dv === 0) continue;
460
+ const secMap = (disk.seconds[sec] ??= {});
461
+ secMap[provider] = (secMap[provider] ?? 0) + dv;
462
+ }
463
+ }
464
+ const cutoff = Math.floor(Date.now() / 1000) - 120;
465
+ for (const sec of Object.keys(disk.seconds)) {
466
+ if (Number(sec) < cutoff) delete disk.seconds[sec];
467
+ }
468
+ return JSON.stringify(disk, null, 2);
469
+ });
470
+ }
471
+
472
+ /**
473
+ * Fetch write path: serializes the in-memory fetch mirror under the same lock
474
+ * protocol. Never reads or rewrites the realtime aggregates.
475
+ */
476
+ function writeFetchDisk(): boolean {
477
+ const fetchedByProvider: Record<string, number> = {};
478
+ for (const [p, t] of lastFetchedByProvider) fetchedByProvider[p] = t;
479
+ return writeFetchPayload({
480
+ fetchedAt: globalFetchedAt(),
481
+ reports: cachedReports,
482
+ fetchedByProvider,
483
+ });
484
+ }
485
+
486
+
487
+ /**
488
+ * Aggregate the legacy PER-SESSION structure into the GLOBAL aggregate shape:
489
+ * every session's per-provider buckets sum into `totals`, and its per-second
490
+ * deltas sum into the same absolute-second slots (legacy `number` values are
491
+ * normalized first).
492
+ */
493
+ function aggregateLegacySessionUsage(
494
+ sessionUsage: LegacyCacheFile['sessionUsage'],
495
+ ): RealtimeFile {
496
+ const disk: RealtimeFile = { totals: {}, seconds: {} };
497
+ for (const entry of Object.values(sessionUsage ?? {})) {
498
+ for (const [provider, raw] of Object.entries(entry?.perProvider ?? {})) {
499
+ const bucket = normalizeBucket(raw);
500
+ const prev = normalizeBucket(disk.totals[provider]);
501
+ disk.totals[provider] = {
502
+ total: prev.total + bucket.total,
503
+ input: prev.input + bucket.input,
504
+ output: prev.output + bucket.output,
505
+ };
506
+ const seconds = typeof raw === 'object' && raw !== null ? (raw as ProviderUsageTotals).seconds : undefined;
507
+ for (const [sec, n] of Object.entries(seconds ?? {})) {
508
+ const secMap = (disk.seconds[sec] ??= {});
509
+ secMap[provider] = (secMap[provider] ?? 0) + n;
510
+ }
511
+ }
512
+ }
513
+ return disk;
514
+ }
515
+
516
+ /**
517
+ * One-time migration from the legacy single cache file to the split realtime +
518
+ * fetch files. Runs at poller start. Idempotent: when BOTH new files already
519
+ * exist the legacy file is skipped entirely (newer data is never overwritten);
520
+ * each missing new file is written under the same lock protocol from the
521
+ * legacy payload, then the legacy file is removed once both landed. A crash
522
+ * mid-migration simply re-runs on the next start (writes are atomic).
523
+ */
524
+ function migrateLegacyCache(): void {
525
+ const legacyPath = resolveLegacyCacheFile();
526
+ if (!existsSync(legacyPath)) return;
527
+ const realtimePath = resolveRealtimeFile();
528
+ const fetchPath = resolveFetchFile();
529
+ if (existsSync(realtimePath) && existsSync(fetchPath)) return; // already migrated
530
+ let legacy: LegacyCacheFile | null = null;
531
+ try {
532
+ legacy = JSON.parse(readFileSync(legacyPath, 'utf-8')) as LegacyCacheFile;
533
+ } catch {
534
+ return; // unreadable legacy file — leave it untouched
535
+ }
536
+ if (!legacy) return;
537
+ if (!existsSync(realtimePath)) {
538
+ writeRealtimePayload(aggregateLegacySessionUsage(legacy.sessionUsage));
539
+ }
540
+ if (!existsSync(fetchPath)) {
541
+ writeFetchPayload({
542
+ fetchedAt: typeof legacy.fetchedAt === 'number' ? legacy.fetchedAt : 0,
543
+ reports: Array.isArray(legacy.reports) ? legacy.reports : [],
544
+ fetchedByProvider: legacy.fetchedByProvider,
545
+ });
546
+ }
547
+ if (existsSync(realtimePath) && existsSync(fetchPath)) {
548
+ try { rmSync(legacyPath); } catch { /* best effort */ }
549
+ }
550
+ }
551
+
552
+ /**
553
+ * Merge disk data into memory if any per-provider timestamp on disk is newer
554
+ * than the in-memory one (a cross-process fetch). Falls back to the legacy
555
+ * global `fetchedAt` when the cache file predates per-provider tracking.
556
+ */
557
+ function syncFromDisk(loggedInProviders?: ReadonlySet<string>): void {
558
+ // Consumption signal (→ markDirty) reads the REALTIME file's GLOBAL totals:
559
+ // any provider whose aggregate grew since the last sync marks dirty.
560
+ const realtime = readRealtimeDisk();
561
+ if (realtime) {
562
+ const nowTotals = new Map<string, number>();
563
+ for (const [p, raw] of Object.entries(realtime.totals)) nowTotals.set(p, normalizeBucket(raw).total);
564
+ // Empty baseline = nothing seen yet → any positive total is new consumption.
565
+ for (const [p, total] of nowTotals) {
566
+ if (total > (lastSyncedTotals.get(p) ?? 0)) dirtyProviders.add(canonicalizeProvider(p));
567
+ }
568
+ lastSyncedTotals = nowTotals;
569
+ }
570
+ // ...freshness signal (→ merge reports) reads the FETCH file. The two are
571
+ // independent: a realtime write never re-fires a fetch merge, and a fetch
572
+ // write never re-fires consumption.
573
+ const disk = readFetchDisk();
574
+ if (!disk) {
575
+ // No disk data, but still prune in-memory caches.
576
+ if (loggedInProviders) pruneDisconnected(loggedInProviders);
577
+ return;
578
+ }
579
+ let touched = false;
580
+ if (disk.fetchedByProvider) {
581
+ for (const [p, t] of Object.entries(disk.fetchedByProvider)) {
582
+ const canonical = canonicalizeProvider(p);
583
+ // Skip providers no longer connected when we know the current set.
584
+ if (loggedInProviders && !loggedInProviders.has(canonical)) continue;
585
+ const prev = lastFetchedByProvider.get(canonical) ?? 0;
586
+ if (t > prev) {
587
+ lastFetchedByProvider.set(canonical, t);
588
+ touched = true;
589
+ }
590
+ }
591
+ } else if (disk.fetchedAt > globalFetchedAt()) {
592
+ // Legacy cache file: treat the global timestamp as the freshness floor
593
+ // for every provider present in reports.
594
+ for (const r of disk.reports) {
595
+ if (loggedInProviders && !loggedInProviders.has(r.provider)) continue;
596
+ lastFetchedByProvider.set(r.provider, disk.fetchedAt);
597
+ }
598
+ touched = true;
599
+ }
600
+ if (touched) {
601
+ cachedReports = mergeReports(cachedReports, disk.reports);
602
+ cachedHealth = buildHealthMap(cachedReports);
603
+ }
604
+ // Disk-loaded exhausted reports for direct-fetcher providers are unconfirmed.
605
+ seedUnconfirmedFromDisk();
606
+ // Prune in-memory caches to only connected providers, regardless of
607
+ // whether disk contributed new data — stale entries must go.
608
+ if (loggedInProviders) pruneDisconnected(loggedInProviders);
609
+ }
610
+
611
+ function seedFromDisk(loggedInProviders?: ReadonlySet<string>): void {
612
+ if (lastFetchedByProvider.size > 0) return;
613
+ const disk = readFetchDisk();
614
+ if (!disk) return;
615
+ // Canonicalize disk reports once (legacy files may carry pre-canonical ids).
616
+ let reports = disk.reports.map((r) => ({ ...r, provider: canonicalizeProvider(r.provider) }));
617
+ if (loggedInProviders) {
618
+ reports = reports.filter((r) => loggedInProviders.has(r.provider));
619
+ }
620
+ cachedReports = reports;
621
+ cachedHealth = buildHealthMap(cachedReports);
622
+ seedUnconfirmedFromDisk();
623
+ if (disk.fetchedByProvider) {
624
+ for (const [p, t] of Object.entries(disk.fetchedByProvider)) {
625
+ const canonical = canonicalizeProvider(p);
626
+ if (loggedInProviders && !loggedInProviders.has(canonical)) continue;
627
+ lastFetchedByProvider.set(canonical, t);
628
+ }
629
+ } else if (disk.reports.length > 0) {
630
+ for (const r of reports) lastFetchedByProvider.set(r.provider, disk.fetchedAt);
631
+ }
632
+ }
633
+ /** Remove stale in-memory AND on-disk cached data for any provider NOT in `loggedIn`. */
634
+ function pruneDisconnected(loggedIn: ReadonlySet<string>): void {
635
+ let removed = false;
636
+ for (const key of dirtyProviders) {
637
+ if (!loggedIn.has(key)) { dirtyProviders.delete(key); removed = true; }
638
+ }
639
+ for (const key of lastAttemptByProvider.keys()) {
640
+ if (!loggedIn.has(key)) { lastAttemptByProvider.delete(key); removed = true; }
641
+ }
642
+ for (const key of lastFetchedByProvider.keys()) {
643
+ if (!loggedIn.has(key)) { lastFetchedByProvider.delete(key); removed = true; }
644
+ }
645
+ for (const key of unconfirmedProviders) {
646
+ if (!loggedIn.has(key)) { unconfirmedProviders.delete(key); removed = true; }
647
+ }
648
+ for (const key of confirmedThisProcess) {
649
+ if (!loggedIn.has(key)) { confirmedThisProcess.delete(key); removed = true; }
650
+ }
651
+ const next = cachedReports.filter((r) => loggedIn.has(r.provider));
652
+ removed = removed || next.length !== cachedReports.length;
653
+ cachedReports = next;
654
+ cachedHealth = buildHealthMap(cachedReports);
655
+ // Persist only when something was actually dropped — the 5s sync tick must
656
+ if (removed) writeFetchDisk();
657
+ }
658
+ /** Merge `incoming` into `existing` by canonical provider id (incoming wins). */
659
+ function mergeReports(existing: UsageReport[], incoming: UsageReport[]): UsageReport[] {
660
+ const byId = new Map(existing.map((r) => [r.provider, r]));
661
+ for (const r of incoming) byId.set(r.provider, r);
662
+ return [...byId.values()];
663
+ }
664
+
665
+ /** All canonical provider ids the host considers logged-in. */
666
+ function getLoggedInProviders(
667
+ authStorage: AuthStorageLike | undefined,
668
+ directFetchers: DirectFetcher[],
669
+ ): string[] {
670
+ const detectable = listDetectableProviders(directFetchers);
671
+ if (authStorage?.list) {
672
+ return [...new Set([...authStorage.list().map(canonicalizeProvider), ...detectable])];
673
+ }
674
+ // No authStorage (tests / offline): direct fetchers' providers are the universe,
675
+ // with detectable probes still honored when present.
676
+ return [...new Set([...directFetchers.map((f) => canonicalizeProvider(f.provider)), ...detectable])];
677
+ }
678
+
679
+
680
+ /**
681
+ * True when any limit window for `provider` resets within `RESET_IMMINENT_MS`
682
+ * or has already elapsed. Drives a forced fetch so the widget's "即将重置"
683
+ * countdown flips to the new window promptly instead of waiting out the static
684
+ * TTL. `resetsAt - now < RESET_IMMINENT_MS` covers both near-future and past.
685
+ */
686
+ export function isResetImminent(provider: string): boolean {
687
+ const canonical = canonicalizeProvider(provider);
688
+ const report = cachedReports.find((r) => r.provider === canonical);
689
+ if (!report) return false;
690
+ return report.limits.some(
691
+ (l) => typeof l.window?.resetsAt === "number" && l.window.resetsAt - Date.now() < RESET_IMMINENT_MS,
692
+ );
693
+ }
694
+
695
+ /**
696
+ * Fetch a single provider and merge its report into the cache. Per-provider
697
+ * pending guard makes concurrent calls for the same provider no-op. Toggles the
698
+ * provider's id in the loading set for the duration of the fetch so the render
699
+ * can show a `⟳` hint. Failure leaves the previous report (if any) intact.
700
+ */
701
+ async function fetchProvider(
702
+ provider: string,
703
+ authStorage: AuthStorageLike | undefined,
704
+ directFetchers: DirectFetcher[],
705
+ getApiKey?: ApiKeyResolver,
706
+ ): Promise<void> {
707
+ const canonical = canonicalizeProvider(provider);
708
+ if (pendingProviders.has(canonical)) return; // already fetching this provider
709
+ pendingProviders.add(canonical);
710
+ try {
711
+ const { reports } = await resolveProviderReports(authStorage, directFetchers, undefined, getApiKey, [canonical]);
712
+ const fetchedAt = Date.now();
713
+ lastFetchedByProvider.set(canonical, fetchedAt);
714
+ // Replace (or insert) just this provider's report; keep all others.
715
+ const byId = new Map(cachedReports.map((r) => [r.provider, r]));
716
+ const newReport = reports.find((r) => r.provider === canonical);
717
+ if (newReport) {
718
+ byId.set(canonical, newReport);
719
+ // Fresh report landed → the provider's health is CONFIRMED.
720
+ unconfirmedProviders.delete(canonical);
721
+ confirmedThisProcess.add(canonical);
722
+ } else {
723
+ // Fetch produced nothing for this provider: any existing report is
724
+ // UNCONFIRMED (possibly stale) — keep it for display, but flag it so
725
+ // the auto path re-attempts sooner than TTL.
726
+ if (byId.has(canonical)) unconfirmedProviders.add(canonical);
727
+ }
728
+ // Fetch returned nothing: still advance TTL via lastFetchedByProvider
729
+ // above, but KEEP any prior report (matches the comment / avoids
730
+ // multi-process cache races wiping a good Cursor report).
731
+ cachedReports = [...byId.values()];
732
+ cachedHealth = buildHealthMap(cachedReports);
733
+ writeFetchDisk();
734
+ } catch {
735
+ // Fetch threw: any existing report is unconfirmed until a fetch lands,
736
+ // and the provider re-enters the dirty set so the next floor-eligible
737
+ // tick retries (attempt timestamp already advanced by `fetchDue`).
738
+ if (cachedReports.some((r) => r.provider === canonical)) unconfirmedProviders.add(canonical);
739
+ dirtyProviders.add(canonical);
740
+ /* keep previous data */
741
+ } finally {
742
+ pendingProviders.delete(canonical);
743
+ }
744
+ }
745
+
746
+ /**
747
+ * Fetch every dirty provider whose per-provider minimum-interval floor has
748
+ * elapsed (skipping in-flight ones), consuming the dirty marks on initiation.
749
+ * `force` bypasses the floor for decision paths that await fresh data.
750
+ * `fetchProvider` re-marks a provider dirty when it throws, so a failed fetch
751
+ * retries on the next floor-eligible tick instead of being lost.
752
+ */
753
+ async function fetchDue(
754
+ authStorage: AuthStorageLike | undefined,
755
+ directFetchers: DirectFetcher[],
756
+ getApiKey: ApiKeyResolver | undefined,
757
+ opts?: { force?: boolean },
758
+ ): Promise<void> {
759
+ const now = Date.now();
760
+ const due: string[] = [];
761
+ for (const provider of dirtyProviders) {
762
+ if (pendingProviders.has(provider)) continue;
763
+ const lastAttempt = lastAttemptByProvider.get(provider);
764
+ // No attempt record yet → first fetch always passes the floor.
765
+ if (!opts?.force && lastAttempt !== undefined && now - lastAttempt < minFetchIntervalMs) continue;
766
+ due.push(provider);
767
+ }
768
+ if (due.length === 0) return;
769
+ for (const provider of due) {
770
+ dirtyProviders.delete(provider); // consumed on initiation — failures re-add
771
+ lastAttemptByProvider.set(provider, now);
772
+ }
773
+ await Promise.all(due.map((p) => fetchProvider(p, authStorage, directFetchers, getApiKey)));
774
+ }
775
+
776
+ /**
777
+ * Mark every logged-in provider that should fetch right now: forced triggers
778
+ * (reset-imminent / exhausted-unconfirmed) and the idle fallback (no
779
+ * consumption, last SUCCESSFUL fetch older than `idleDirtyMs`).
780
+ */
781
+ function evaluateDirty(loggedIn: ReadonlySet<string>): void {
782
+ const now = Date.now();
783
+ for (const provider of loggedIn) {
784
+ if (isResetImminent(provider) || isExhaustedUnconfirmed(provider)) {
785
+ dirtyProviders.add(provider);
786
+ } else if (now - (lastFetchedByProvider.get(provider) ?? 0) >= idleDirtyMs) {
787
+ dirtyProviders.add(provider);
788
+ }
789
+ }
790
+ }
791
+
792
+ /**
793
+ * Poke one or more providers: mark them dirty, sync the shared cache from
794
+ * disk, run {@link fetchDue} immediately, and render once at the end. The
795
+ * `turn_end` hook uses the default (floor-honoring) path for the current
796
+ * model's provider; the 429-diagnosis and `/pick-model` decision paths pass
797
+ * `force: true` because they await fresh data before deciding.
798
+ */
799
+ export async function pokeProvider(
800
+ authStorage: AuthStorageLike | undefined,
801
+ directFetchers: DirectFetcher[],
802
+ getApiKey: ApiKeyResolver | undefined,
803
+ opts?: { providers?: readonly string[]; force?: boolean },
804
+ ): Promise<void> {
805
+ setDirectFetcherCoverage(directFetchers);
806
+ const loggedIn = getLoggedInProviders(authStorage, directFetchers);
807
+ const targets =
808
+ opts?.providers && opts.providers.length > 0
809
+ ? opts.providers.map(canonicalizeProvider)
810
+ : loggedIn;
811
+ for (const provider of targets) dirtyProviders.add(provider);
812
+ syncFromDisk(new Set(loggedIn));
813
+ await fetchDue(authStorage, directFetchers, getApiKey, { force: opts?.force });
814
+ onTickRef?.();
815
+ }
816
+ /**
817
+ * Provider whose health says `exhausted` but whose most recent fetch attempt
818
+ * produced no fresh report (its report is unconfirmed and possibly stale).
819
+ * These are re-attempted by the automatic refresh path every cycle so a
820
+ * healthy provider flagged exhausted by old data recovers on its own.
821
+ */
822
+ function isExhaustedUnconfirmed(provider: string): boolean {
823
+ const canonical = canonicalizeProvider(provider);
824
+ return unconfirmedProviders.has(canonical) && (cachedHealth.get(canonical)?.exhausted ?? false);
825
+ }
826
+
827
+ /** Canonical ids of providers whose health is unconfirmed (last fetch produced nothing). */
828
+ export function getUnconfirmedProviders(): string[] {
829
+ return [...unconfirmedProviders];
830
+ }
831
+
832
+ /** Start the poller. Resolves when first usable data is available (disk or fetch). Idempotent. */
833
+ export async function startSharedPoller(
834
+ authStorage: AuthStorageLike | undefined,
835
+ directFetchers: DirectFetcher[],
836
+ getApiKey?: ApiKeyResolver,
837
+ opts?: StartPollerOptions,
838
+ ): Promise<void> {
839
+ // Subagent session_start shares this process; its render closure writes to a no-op UI and would freeze the host widget.
840
+ if (opts?.redis?.enabled && relay === null) {
841
+ relay = new RedisRelay({
842
+ host: opts.redis.host,
843
+ port: opts.redis.port,
844
+ connectTimeoutMs: opts.redis.connectTimeoutMs,
845
+ commandTimeoutMs: opts.redis.commandTimeoutMs,
846
+ probeIntervalMs: opts.redis.probeIntervalMs,
847
+ unhealthyRetryMs: opts.redis.unhealthyRetryMs,
848
+ reconcileIntervalMs: opts.redis.reconcileIntervalMs,
849
+ ttlMs: opts.redis.ttlMs,
850
+ transport: opts.redis.transport,
851
+ });
852
+ reconcileIntervalMs = opts.redis.reconcileIntervalMs;
853
+ }
854
+ if (opts?.onTick && !(opts.ownsUi === false && getTickOwner() !== undefined)) {
855
+ onTickRef = opts.onTick;
856
+ setTickOwner(opts.onTick);
857
+ }
858
+ if (startInFlight) return startInFlight;
859
+ if (renderTimer) return;
860
+
861
+ startInFlight = (async () => {
862
+ const tickMs = opts?.tickMs ?? DEFAULT_TICK_MS;
863
+ const renderTickMs = opts?.renderTickMs ?? DEFAULT_RENDER_TICK_MS;
864
+ if (opts?.minFetchIntervalMs !== undefined) minFetchIntervalMs = opts.minFetchIntervalMs;
865
+ if (opts?.idleDirtyMs !== undefined) idleDirtyMs = opts.idleDirtyMs;
866
+
867
+ // One-time migration from the legacy single cache file (idempotent).
868
+ migrateLegacyCache();
869
+ const loggedInSet = new Set(getLoggedInProviders(authStorage, directFetchers));
870
+ setDirectFetcherCoverage(directFetchers);
871
+ syncFromDisk(loggedInSet);
872
+ seedFromDisk(loggedInSet);
873
+
874
+ const tick = async (): Promise<void> => {
875
+ syncFromDisk(loggedInSet);
876
+ evaluateDirty(loggedInSet);
877
+ await fetchDue(authStorage, directFetchers, getApiKey);
878
+ onTickRef?.();
879
+ };
880
+
881
+ // Redis relay steps run on EVERY render firing (before the data tick /
882
+ // render), per design D6 — all absolute-time gated inside the relay,
883
+ // no extra timer. Failures never escape into the interval.
884
+ const relaySteps = async (): Promise<void> => {
885
+ // Local ref: `_resetForTest` may null `relay` while an in-flight
886
+ // tick is still awaiting — steps must not crash on the dangling read.
887
+ const r = relay;
888
+ if (r === null) return;
889
+ const now = Date.now();
890
+ r.observeMirrorFailure(now);
891
+ await r.maybeProbe(now);
892
+ if (r.state === 'healthy') {
893
+ // Unconditional per-tick prefetch: redis is the LIVE side (its
894
+ // aggregates already sum across sessions at write time), so the
895
+ // window is re-read every second. Local loopback cost is
896
+ // negligible; the file is only the fallback + authority.
897
+ try {
898
+ const snapshot = await r.prefetch(readRealtimeDisk() ?? { totals: {}, seconds: {} }, now);
899
+ const st = statRealtimeForRelay();
900
+ if (st !== null) {
901
+ redisUsageCache = { ino: st.ino, mtimeMs: st.mtimeMs, size: st.size, totals: snapshot.totals, perSecond: snapshot.perSecond };
902
+ }
903
+ } catch {
904
+ // prefetch already flipped the relay unhealthy; fall back to file reads.
905
+ redisUsageCache = null;
906
+ }
907
+ if (r.state === 'healthy' && now - r.lastReconcileAt >= reconcileIntervalMs) {
908
+ try {
909
+ await r.reconcile(readRealtimeDisk() ?? { totals: {}, seconds: {} }, now);
910
+ } catch {
911
+ // reconcile already flipped the relay unhealthy (defensive).
912
+ }
913
+ }
914
+ }
915
+ };
916
+
917
+
918
+ // Single timer. Every renderTickMs we run the relay steps, consume dirty
919
+ // providers and render; when the absolute elapsed time since the
920
+ // previous data tick reaches tickMs we also sync disk + evaluate
921
+ // triggers + fetch. Absolute time (not tick counting) survives machine
922
+ // suspend: the first tick after a closed-lid wake refreshes
923
+ // immediately, and tickInFlight prevents overlapping refreshes if a
924
+ // fetch is slower than the cadence.
925
+ let lastTickAt = Date.now();
926
+ let tickInFlight = false;
927
+ const dataTickPath = (): void => {
928
+ if (!tickInFlight && Date.now() - lastTickAt >= tickMs) {
929
+ lastTickAt = Date.now();
930
+ tickInFlight = true;
931
+ void tick().finally(() => {
932
+ tickInFlight = false;
933
+ });
934
+ } else {
935
+ void fetchDue(authStorage, directFetchers, getApiKey);
936
+ onTickRef?.();
937
+ }
938
+ };
939
+ renderTimer = setInterval(() => {
940
+ // Relay disabled: keep the data tick + render fully synchronous
941
+ // (fake-timer tests and the render loop rely on it). Relay enabled:
942
+ // run the D6 relay steps first, then the data path.
943
+ if (relay === null) {
944
+ dataTickPath();
945
+ return;
946
+ }
947
+ void relaySteps().then(dataTickPath);
948
+ }, renderTickMs);
949
+
950
+ const fresh = [...lastFetchedByProvider.values()].some((t) => Date.now() - t < idleDirtyMs);
951
+ if (fresh) return;
952
+
953
+ evaluateDirty(loggedInSet);
954
+ await fetchDue(authStorage, directFetchers, getApiKey);
955
+ })();
956
+ try {
957
+ await startInFlight;
958
+ } finally {
959
+ startInFlight = null;
960
+ }
961
+ }
962
+
963
+ export function getSharedUsage(): SharedUsageState | null {
964
+ syncFromDisk();
965
+ // Return state once we have data OR an in-flight fetch (so the UI can show
966
+ // a loading hint even before the first report lands).
967
+ if (cachedReports.length === 0 && lastFetchedByProvider.size === 0 && pendingProviders.size === 0) return null;
968
+ return buildCachedState();
969
+ }
970
+
971
+ // ── cross-process session usage (consumption waveform source) ──────
972
+
973
+ /**
974
+ * Publish this session's cumulative per-provider token totals to the shared
975
+ * cache under the cross-process write lock: other processes' entries are
976
+ * re-read and preserved — only the entry for `sessionId` is overwritten.
977
+ * After a successful file write, mirrors the full snapshot to Redis
978
+ * (fire-and-forget HSET + EXPIRE) and DELs any entries reclaimed this cycle.
979
+ */
980
+ /** Whether THIS process still holds a delta baseline for the given session. */
981
+ export function hasManagedSessionUsage(sessionId: string): boolean {
982
+ return !!sessionId && lastFlushedBySession.has(sessionId);
983
+ }
984
+
985
+ export function writeSessionUsage(sessionId: string, perProvider: Record<string, ProviderUsageTotals>): boolean {
986
+ if (!sessionId) return false;
987
+ const prev = lastFlushedBySession.get(sessionId) ?? {};
988
+ const deltas: Record<string, ProviderUsageTotals> = {};
989
+ const snapshot: Record<string, ProviderUsageTotals> = {};
990
+ for (const [provider, raw] of Object.entries(perProvider)) {
991
+ const cur = normalizeBucket(raw);
992
+ const curSeconds = typeof raw === 'object' && raw !== null ? (raw as ProviderUsageTotals).seconds : undefined;
993
+ const beforeRaw = prev[provider];
994
+ const before = normalizeBucket(beforeRaw);
995
+ const beforeSeconds = typeof beforeRaw === 'object' && beforeRaw !== null ? (beforeRaw as ProviderUsageTotals).seconds : undefined;
996
+ const seconds: Record<string, number> = {};
997
+ for (const [sec, n] of Object.entries(curSeconds ?? {})) {
998
+ const dv = n - (beforeSeconds?.[sec] ?? 0);
999
+ if (dv !== 0) seconds[sec] = dv;
1000
+ }
1001
+ // input/output are observational maxima — never regress the aggregate.
1002
+ const dTotal = Math.max(0, cur.total - before.total);
1003
+ const dInput = Math.max(0, cur.input - before.input);
1004
+ const dOutput = Math.max(0, cur.output - before.output);
1005
+ deltas[provider] = { total: dTotal, input: dInput, output: dOutput, seconds };
1006
+ snapshot[provider] = {
1007
+ total: before.total + dTotal,
1008
+ input: before.input + dInput,
1009
+ output: before.output + dOutput,
1010
+ seconds: { ...(curSeconds ?? {}) },
1011
+ };
1012
+ }
1013
+ if (Object.keys(deltas).length === 0) return true;
1014
+ const ok = writeRealtimeDisk(deltas);
1015
+ if (ok) {
1016
+ lastFlushedBySession.set(sessionId, snapshot);
1017
+ // Event-level totals mirror: HINCRBY deltas (no retry — mirrorFailed;
1018
+ // a lost increment is repaired by the next reconcile overwrite).
1019
+ for (const [provider, d] of Object.entries(deltas)) {
1020
+ if (d.total > 0 || d.input > 0 || d.output > 0) relay?.onTotalsDelta(provider, d.total, d.input, d.output);
1021
+ }
1022
+ }
1023
+ return ok;
1024
+ }
1025
+
1026
+ /**
1027
+ * Session shutdown: the aggregates are PERMANENT cross-session sums — nothing
1028
+ * is subtracted. Only this process's delta baseline is dropped so a later
1029
+ * session id never re-applies stale deltas.
1030
+ */
1031
+ export function removeSessionUsage(sessionId: string): void {
1032
+ if (!sessionId) return;
1033
+ lastFlushedBySession.delete(sessionId);
1034
+ }
1035
+
1036
+ /**
1037
+ * Event-level sec-delta mirror entry (wired to `SessionUsageSink.onDelta` in
1038
+ * index.ts): healthy relay → fire-and-forget `HINCRBY` + `EXPIRE`; otherwise
1039
+ * no-op. Never throws.
1040
+ */
1041
+ export function mirrorSessionDelta(sessionId: string, provider: string, secNow: string, deltaTotal: number): void {
1042
+ relay?.onDelta(sessionId, provider, secNow, deltaTotal);
1043
+ }
1044
+
1045
+ /**
1046
+ * Global totals per provider — the file stores cross-session aggregates
1047
+ * directly, so the getter is a plain stat-gated read with NO per-session
1048
+ * traversal. Healthy relay → served from the prefetched Redis cache; stat
1049
+ * miss (prefetch hasn't refreshed yet, e.g. render fired outside the tick
1050
+ * flow) falls through to the file path.
1051
+ */
1052
+ let usageReadCache: { ino: number; mtimeMs: number; size: number; totals: Record<string, UsageBucket> } | null = null;
1053
+
1054
+ export function getSessionUsageTotals(): Record<string, UsageBucket> {
1055
+ const file = resolveRealtimeFile();
1056
+ try {
1057
+ const st = statSync(file);
1058
+ if (relay !== null && relay.state === 'healthy') {
1059
+ if (redisUsageCache && redisUsageCache.ino === st.ino && redisUsageCache.mtimeMs === st.mtimeMs && redisUsageCache.size === st.size) {
1060
+ return redisUsageCache.totals;
1061
+ }
1062
+ }
1063
+ if (usageReadCache && usageReadCache.ino === st.ino && usageReadCache.mtimeMs === st.mtimeMs && usageReadCache.size === st.size) {
1064
+ return usageReadCache.totals;
1065
+ }
1066
+ const disk = JSON.parse(readFileSync(file, 'utf-8')) as RealtimeFile;
1067
+ const totals: Record<string, UsageBucket> = {};
1068
+ for (const [p, raw] of Object.entries(disk.totals ?? {})) totals[p] = normalizeBucket(raw);
1069
+ usageReadCache = { ino: st.ino, mtimeMs: st.mtimeMs, size: st.size, totals };
1070
+ return totals;
1071
+ } catch {
1072
+ // File absent or unreadable: nothing aggregated.
1073
+ usageReadCache = null;
1074
+ return {};
1075
+ }
1076
+ }
1077
+
1078
+ /**
1079
+ * Global per-second token map (unix second → provider → tokens, already summed
1080
+ * across sessions in storage). Same hot-file mtime+ino+size gate as
1081
+ * {@link getSessionUsageTotals}; seconds older than `windowSecs` are dropped.
1082
+ * Healthy relay → served from the prefetched Redis snapshot.
1083
+ */
1084
+ let perSecondReadCache: { ino: number; mtimeMs: number; size: number; perSecond: Record<string, Record<string, number>> } | null = null;
1085
+
1086
+ export function getSessionUsagePerSecond(windowSecs = 60): Record<string, Record<string, number>> {
1087
+ const file = resolveRealtimeFile();
1088
+ const cutoff = Math.floor(Date.now() / 1_000) - windowSecs;
1089
+ try {
1090
+ const st = statSync(file);
1091
+ if (relay !== null && relay.state === 'healthy') {
1092
+ if (redisUsageCache && redisUsageCache.ino === st.ino && redisUsageCache.mtimeMs === st.mtimeMs && redisUsageCache.size === st.size) {
1093
+ return filterSeconds(redisUsageCache.perSecond, cutoff);
1094
+ }
1095
+ }
1096
+ if (perSecondReadCache && perSecondReadCache.ino === st.ino && perSecondReadCache.mtimeMs === st.mtimeMs && perSecondReadCache.size === st.size) {
1097
+ return filterSeconds(perSecondReadCache.perSecond, cutoff);
1098
+ }
1099
+ const disk = JSON.parse(readFileSync(file, 'utf-8')) as RealtimeFile;
1100
+ const perSecond = transposeSeconds((disk.seconds ?? {}) as Record<string, Record<string, number>>, cutoff);
1101
+ perSecondReadCache = { ino: st.ino, mtimeMs: st.mtimeMs, size: st.size, perSecond };
1102
+ return perSecond;
1103
+ } catch {
1104
+ perSecondReadCache = null;
1105
+ return {};
1106
+ }
1107
+ }
1108
+
1109
+ /** Transpose the file's {sec: {provider: tokens}} into {provider: {sec: tokens}}, dropping old seconds. */
1110
+ function transposeSeconds(seconds: Record<string, Record<string, number>>, cutoff: number): Record<string, Record<string, number>> {
1111
+ const out: Record<string, Record<string, number>> = {};
1112
+ for (const [sec, secMap] of Object.entries(seconds)) {
1113
+ if (Number(sec) < cutoff) continue;
1114
+ for (const [provider, n] of Object.entries(secMap)) {
1115
+ (out[provider] ??= {})[sec] = n;
1116
+ }
1117
+ }
1118
+ return out;
1119
+ }
1120
+
1121
+ /** Drop seconds older than `cutoff` from a provider → {sec: tokens} map. */
1122
+ function filterSeconds(seconds: Record<string, Record<string, number>>, cutoff: number): Record<string, Record<string, number>> {
1123
+ const out: Record<string, Record<string, number>> = {};
1124
+ for (const [p, secMap] of Object.entries(seconds)) {
1125
+ const m: Record<string, number> = {};
1126
+ for (const [k, v] of Object.entries(secMap)) {
1127
+ if (Number(k) >= cutoff) m[k] = v;
1128
+ }
1129
+ if (Object.keys(m).length > 0) out[p] = m;
1130
+ }
1131
+ return out;
1132
+ }
1133
+ export function _setCacheFileForTest(path: string | null): void {
1134
+ realtimeFileOverride = path;
1135
+ fetchFileOverride = path === null ? null : deriveSiblingPath(path, 'fetch');
1136
+ legacyFileOverride = path === null ? null : deriveSiblingPath(path, 'legacy');
1137
+ }
1138
+
1139
+ export function _setRealtimeFileForTest(path: string | null): void {
1140
+ realtimeFileOverride = path;
1141
+ }
1142
+
1143
+ export function _setFetchFileForTest(path: string | null): void {
1144
+ fetchFileOverride = path;
1145
+ }
1146
+
1147
+ export function _setLegacyCacheFileForTest(path: string | null): void {
1148
+ legacyFileOverride = path;
1149
+ }
1150
+
1151
+ export function _resetForTest(): void {
1152
+ if (renderTimer) {
1153
+ clearInterval(renderTimer);
1154
+ renderTimer = null;
1155
+ }
1156
+ lastFetchedByProvider.clear();
1157
+ lastAttemptByProvider.clear();
1158
+ dirtyProviders.clear();
1159
+ pendingProviders.clear();
1160
+ unconfirmedProviders.clear();
1161
+ confirmedThisProcess.clear();
1162
+ directFetcherIds.clear();
1163
+ cachedReports = [];
1164
+ cachedHealth = new Map();
1165
+ lastFlushedBySession.clear();
1166
+ lastSyncedTotals = new Map();
1167
+ perSecondReadCache = null;
1168
+ redisUsageCache = null;
1169
+ if (relay !== null) {
1170
+ relay.close();
1171
+ relay = null;
1172
+ }
1173
+ reconcileIntervalMs = 30_000;
1174
+ onTickRef = undefined;
1175
+ setTickOwner(undefined);
1176
+ minFetchIntervalMs = DEFAULT_MIN_FETCH_INTERVAL_MS;
1177
+ idleDirtyMs = DEFAULT_IDLE_DIRTY_MS;
1178
+ if (realtimeFileOverride !== null || fetchFileOverride !== null) {
1179
+ // Drop any lock files this process left behind (e.g. a crashed worker).
1180
+ for (const file of [resolveRealtimeFile(), resolveFetchFile()]) {
1181
+ try { rmSync(`${file}.lock`); } catch { /* absent */ }
1182
+ }
1183
+ }
1184
+ }
1185
+ /** Test seam: current relay instance (null when disabled). */
1186
+ export function _getRelayForTest(): RedisRelay | null {
1187
+ return relay;
1188
+ }
1189
+ /** Test seam: current redis prefetch cache (null when not yet prefetched). */
1190
+ export function _getRedisUsageCacheForTest(): typeof redisUsageCache {
1191
+ return redisUsageCache;
1192
+ }
1193
+ export function clearOnTickRef(): void {
1194
+ onTickRef = undefined;
1195
+ setTickOwner(undefined);
1196
+ }
1197
+ /** True once a session claimed the process-wide tick-render callback. */
1198
+ export function hasTickOwner(): boolean {
1199
+ return getTickOwner() !== undefined;
1200
+ }