@phnx-labs/agents-cli 1.22.64 → 1.22.66

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -54,14 +54,23 @@ export { flagValue, hasHostRoutingFlag } from './routing-flag.js';
54
54
  * `add`/`use`/`list`, and none) and were removed.
55
55
  */
56
56
  export const REMOTE_PASSTHROUGH = {
57
- // inspect
58
- view: {},
59
- inspect: {},
60
- doctor: {},
57
+ // inspect — pure read-only renders: forward over a pipe, never a forced PTY
58
+ // (PHNX-3583), so the drawn output persists instead of vanishing on exit.
59
+ view: {
60
+ render: true,
61
+ // Only `--prune` (without --yes/--dry-run) asks a confirm(); that one
62
+ // invocation needs the PTY. Every other `view` is a pure render.
63
+ interactiveWhen: (f) => f.includes('--prune') &&
64
+ !f.includes('--dry-run') &&
65
+ !f.includes('--yes') &&
66
+ !f.includes('-y'),
67
+ },
68
+ inspect: { render: true },
69
+ doctor: { render: true },
61
70
  check: {},
62
71
  list: {},
63
72
  usage: {},
64
- insights: {},
73
+ insights: { render: true },
65
74
  // config / resources
66
75
  config: {},
67
76
  sync: { nonInteractive: ['--yes'] },
@@ -162,6 +171,41 @@ export function buildPassthroughForwardedArgs(command, allArgs, interactive) {
162
171
  }
163
172
  return forwarded;
164
173
  }
174
+ /**
175
+ * Decide whether a `--device` passthrough should forward over a PLAIN PIPE
176
+ * instead of a PTY, and what color/geometry env to inject when it does.
177
+ *
178
+ * A pure read-only render ({@link RemoteSpec.render}) drawn under a forced
179
+ * `ssh -tt` PTY vanishes on clean exit: the PTY teardown + local-terminal
180
+ * restore (`restoreLocalTerminal` in ssh-exec.ts) wipes the output the command
181
+ * just drew (PHNX-3583). The non-TTY (piped) path never had this problem, so a
182
+ * render command takes it too — but only when a human is actually at a real
183
+ * local terminal (`isTTY` and not `--no-tty`); a genuinely piped local run is
184
+ * already on the pipe path and wants neither color nor forced geometry.
185
+ *
186
+ * A render command's narrow interactive sub-path ({@link RemoteSpec.interactiveWhen},
187
+ * e.g. `view --prune`'s confirm) keeps the PTY. When forwarding over a pipe the
188
+ * remote sees `isTTY=false`, so chalk goes colorless and `terminalWidth()` has no
189
+ * `$COLUMNS` to read; `FORCE_COLOR`/`COLUMNS`/`LINES` restore both. `FORCE_COLOR`
190
+ * is withheld under `--json` so it can never taint machine-readable output.
191
+ */
192
+ export function renderForwardDecision(command, allArgs, io) {
193
+ const spec = REMOTE_PASSTHROUGH[command];
194
+ const localTty = io.isTTY && !io.noTty;
195
+ if (!spec?.render || !localTty)
196
+ return { noPty: false };
197
+ const forwarded = stripRoutingFlags(allArgs, STRIP_SPECS);
198
+ if (spec.interactiveWhen?.(forwarded))
199
+ return { noPty: false };
200
+ const env = {};
201
+ if (!allArgs.includes('--json'))
202
+ env.FORCE_COLOR = '1';
203
+ if (io.columns && io.columns > 0)
204
+ env.COLUMNS = String(io.columns);
205
+ if (io.rows && io.rows > 0)
206
+ env.LINES = String(io.rows);
207
+ return { noPty: true, env: Object.keys(env).length ? env : undefined };
208
+ }
165
209
  /** Synthesize a `Host` for a raw `user@host` / bare-alias target (not enrolled). */
166
210
  function syntheticHost(target) {
167
211
  const at = target.indexOf('@');
@@ -554,10 +598,21 @@ export async function maybeRunOnHost(command, allArgs, opts) {
554
598
  return true;
555
599
  }
556
600
  const target = sshTargetFor(host);
601
+ // A read-only render (view/inspect/insights/doctor) must forward over a plain
602
+ // pipe even from a real terminal — a forced `ssh -tt` PTY wipes its output on
603
+ // clean exit (PHNX-3583). renderForwardDecision returns that choice plus the
604
+ // color/geometry env that keeps a piped render colored and correctly wrapped.
605
+ const { noPty: renderNoPty, env: renderForwardEnv } = renderForwardDecision(command, allArgs, {
606
+ isTTY: !!process.stdout.isTTY,
607
+ noTty: allArgs.includes('--no-tty'),
608
+ columns: process.stdout.columns,
609
+ rows: process.stdout.rows,
610
+ });
557
611
  // Interactive only when our own stdout is a terminal and the caller didn't opt
558
612
  // out — otherwise force the command's non-interactive path so no half-drawn
559
- // picker is piped into a file or another program.
560
- const interactive = !!process.stdout.isTTY && !allArgs.includes('--no-tty');
613
+ // picker is piped into a file or another program. A no-PTY render is
614
+ // non-interactive by construction.
615
+ const interactive = !!process.stdout.isTTY && !allArgs.includes('--no-tty') && !renderNoPty;
561
616
  const forwarded = buildPassthroughForwardedArgs(command, allArgs, interactive);
562
617
  // The one long-running case: keep the remote team supervisor alive past a
563
618
  // disconnect by dispatching it detached (nohup), still streaming live.
@@ -583,10 +638,13 @@ export async function maybeRunOnHost(command, allArgs, opts) {
583
638
  const doctorPath = isDoctorCommand && !/^win/i.test((remoteOs ?? '').trim())
584
639
  ? { PATH: '$HOME/.agents/.cache/shims:$HOME/.local/bin:$PATH' }
585
640
  : undefined;
641
+ // Merge the doctor PATH bootstrap with the render color/geometry env (doctor is
642
+ // itself a render command, so both can apply). undefined when neither is needed.
643
+ const extraEnv = doctorPath || renderForwardEnv ? { ...doctorPath, ...renderForwardEnv } : undefined;
586
644
  process.exitCode = streamAgentsOnHost(host, forwarded, {
587
645
  remoteCwd,
588
646
  interactive,
589
- extraEnv: doctorPath,
647
+ extraEnv,
590
648
  remoteOs,
591
649
  target,
592
650
  });
@@ -17,7 +17,7 @@ import { AUTH_BUNDLE_NAME, inspectReservedAuthBundle } from './bundles.js';
17
17
  import { pushBundleToHost } from './push.js';
18
18
  import { loadDevicesSync } from '../devices/registry.js';
19
19
  import { sshTargetFor } from '../devices/connect.js';
20
- import { isHostPinned, managedKnownHostsPath } from '../devices/known-hosts.js';
20
+ import { isHostPinned, isDevicePinned, managedKnownHostsPath } from '../devices/known-hosts.js';
21
21
  import { machineId, normalizeHost } from '../session/sync/config.js';
22
22
  import { probeDevice } from '../fleet/apply.js';
23
23
  export function planAuthBundlePush(localAuthOk, devices) {
@@ -81,9 +81,9 @@ export function syncReservedAuthBundle(deps = {}) {
81
81
  // No live probe until we know a push is even possible: missing local auth
82
82
  // skips every device, and an unpinned host is refused before we SSH.
83
83
  if (!localOk) {
84
- return { name: d.name, reachable: true, pinned: pinned(d.name), remoteHasAuth: false };
84
+ return { name: d.name, reachable: true, pinned: isDevicePinned(d, pinned), remoteHasAuth: false };
85
85
  }
86
- const isPinnedDev = pinned(d.name);
86
+ const isPinnedDev = isDevicePinned(d, pinned);
87
87
  if (!isPinnedDev) {
88
88
  return { name: d.name, reachable: true, pinned: false, remoteHasAuth: false };
89
89
  }
@@ -18,6 +18,56 @@ export declare const REFRESH_BURN_DIVISOR = 4;
18
18
  export declare const HOURLY_CALL_CAP = 12;
19
19
  /** How often the daemon wakes to *consider* a refresh pass (due accounts only). */
20
20
  export declare const USAGE_REFRESH_TICK_MS: number;
21
+ /**
22
+ * Minimum wall-clock spacing between two live usage fetches to ONE network
23
+ * provider, across all of its accounts. This is the pacing primitive: refreshes
24
+ * are issued round-robin (stalest account first) no faster than one per spacing,
25
+ * so aggregate endpoint load is a smooth, fixed rate — never the synchronized
26
+ * burst-then-stall a plain rolling-hour cap produces when every account falls
27
+ * due on the same tick. Set to two daemon ticks so the floor-based pacing lands
28
+ * on exact tick boundaries (no drift): one refresh every other tick ⇒ 30/hr.
29
+ */
30
+ export declare const PROVIDER_MIN_REFRESH_SPACING_MS: number;
31
+ /**
32
+ * Aggregate live fetches this daemon may spend on ONE network provider's usage
33
+ * endpoint per rolling hour, across ALL of that provider's local accounts —
34
+ * derived from {@link PROVIDER_MIN_REFRESH_SPACING_MS} so the two are always
35
+ * consistent (HOUR / 120s = 30).
36
+ *
37
+ * The per-account {@link HOURLY_CALL_CAP} alone scales linearly with account
38
+ * count — 8 Claude accounts × 12/hr = ~96 usage calls/hr from one box — and
39
+ * Anthropic's `/api/oauth/usage` rate-limits around ~100/hr (see the
40
+ * `usage-backoff.ts` header). That tripped the endpoint into per-account 429s
41
+ * with Retry-After penalties up to an hour: measured live on `zion`, 7 of 8
42
+ * Claude accounts sat parked, never refreshed inside their 5h window, so
43
+ * `agents view` showed `S: unavailable` and balanced routing read stale/absent
44
+ * usage. It got WORSE with every account added.
45
+ *
46
+ * 30/hr is a fixed rate that does NOT grow with account count, and leaves ample
47
+ * headroom under the ~100/hr ceiling for the auth probe (same endpoint, ~3/hr
48
+ * per account, RUSH-2998) and foreground `agents view` bursts. Because refreshes
49
+ * are paced round-robin (stalest first), each account's worst-case proactive
50
+ * cadence is bounded at N × spacing (8 accounts ⇒ 16 min; 16 ⇒ 32 min) — kept
51
+ * deliberately under the {@link USAGE_STALE_REFUSAL_MAX_AGE_MS} routing window so
52
+ * a budget-paced account never reads as "genuinely stale". A slightly
53
+ * older-but-present reading beats a 45-minute 429 park. Network providers only;
54
+ * grok/codex read local logs and have no rate-limited endpoint.
55
+ */
56
+ export declare const PROVIDER_HOURLY_BUDGET: number;
57
+ /**
58
+ * Most refreshes a single tick may catch up after the daemon has been idle/down
59
+ * (elapsed ≫ spacing). Without this clamp a long gap would grant many tokens at
60
+ * once and re-synchronize every account into the very burst the spacing exists
61
+ * to prevent. A small catch-up keeps the load smooth even after a restart.
62
+ */
63
+ export declare const PROVIDER_CATCHUP_MAX = 2;
64
+ /**
65
+ * A usage row this recently captured (by the free statusline ingest of a live
66
+ * `agents run`, or any writer) is already fresh — do not spend an API call to
67
+ * re-refresh it. Actively-used accounts stay current at zero endpoint cost, so
68
+ * the proactive budget is reserved for genuinely idle accounts.
69
+ */
70
+ export declare const STATUSLINE_FRESH_MS: number;
21
71
  /** Consecutive failed live reads before one broken account is quarantined. */
22
72
  export declare const FAILURE_QUARANTINE_THRESHOLD = 3;
23
73
  /** A chronic offender waits this long while healthy siblings keep their cadence. */
@@ -78,6 +128,24 @@ export declare function shouldRefreshAccount(entry: HeadroomEntry | null | undef
78
128
  * projection, and record this call for the hourly cap.
79
129
  */
80
130
  export declare function nextHeadroomEntry(prev: HeadroomEntry | null | undefined, snapshot: UsageSnapshot | null, now: number): HeadroomEntry;
131
+ /**
132
+ * Most-recent live-fetch time per network provider (the max call timestamp
133
+ * across its accounts, 0 when none), which the smooth per-provider pacing spaces
134
+ * the next refresh from. Non-network providers are omitted — they have no
135
+ * rate-limited endpoint to pace.
136
+ */
137
+ export declare function providerLastCall(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>): Map<AgentId, number>;
138
+ /**
139
+ * How many live fetches the smooth pacing permits a provider THIS tick: one per
140
+ * elapsed {@link PROVIDER_MIN_REFRESH_SPACING_MS} since its last fetch, clamped
141
+ * to {@link PROVIDER_CATCHUP_MAX} so a long idle gap (or a cold provider with no
142
+ * prior fetch) cannot re-burst the whole due set at once. At the daemon's 60 s
143
+ * tick this yields at most one fetch every other tick in steady state (⇒ the
144
+ * hourly budget), while a small fleet whose total demand fits under budget is
145
+ * never throttled — the {@link PROVIDER_HOURLY_BUDGET} rolling cap is the only
146
+ * gate that binds it.
147
+ */
148
+ export declare function providerSpacingTokens(lastCallMs: number, now: number): number;
81
149
  /** An account whose credentials live on the publisher host. */
82
150
  export interface LocalUsageAccount {
83
151
  usageKey: string;
@@ -85,8 +153,27 @@ export interface LocalUsageAccount {
85
153
  /** Live-fetch this account's usage; the daemon passes the real network fetch. */
86
154
  fetch: () => Promise<UsageInfo>;
87
155
  }
88
- /** Cold accounts lead each pass; both cold and cached groups rotate every tick. */
156
+ /**
157
+ * Order a pass STALEST-FIRST so a scarce per-provider budget
158
+ * ({@link PROVIDER_HOURLY_BUDGET}) is always spent on the accounts most in need
159
+ * of a fresh reading, and no account is starved indefinitely.
160
+ *
161
+ * - **Cold accounts** (never refreshed → no cache entry) are maximally stale
162
+ * and lead the pass. They rotate by `tick` so, when the budget can't cover
163
+ * them all in one tick, a different cold account leads each tick.
164
+ * - **Cached accounts** follow, oldest `capturedAt` first (a null capture time
165
+ * counts as maximally stale). As accounts refresh their `capturedAt` advances,
166
+ * so the next pass naturally rotates to whoever is now most out of date.
167
+ */
89
168
  export declare function orderUsageAccounts(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>, tick: number): LocalUsageAccount[];
169
+ /**
170
+ * Aggregate live calls a network provider has already spent in the trailing hour,
171
+ * summed across the accounts in this pass. Seeds the per-provider budget counter
172
+ * so {@link PROVIDER_HOURLY_BUDGET} bounds the rolling-hour total, not just this
173
+ * one tick. Non-network providers (grok/codex, local logs) are excluded — they
174
+ * have no rate-limited endpoint to budget.
175
+ */
176
+ export declare function providerRecentCalls(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>, now: number): Map<AgentId, number>;
90
177
  /**
91
178
  * Enumerate the usage accounts whose credentials live on THIS host — one
92
179
  * per unique usage key, deduped to the most-recently-active version (the same
@@ -111,12 +198,24 @@ export interface UsageRefreshDeps {
111
198
  * this loop's fixed iteration order.
112
199
  */
113
200
  backoffUntil: (agentId: AgentId, usageKey?: string) => number | null;
201
+ /**
202
+ * The account's current usage row from the shared cache (the row the routing
203
+ * hot path reads), or null when absent. Lets the refresher see the FREE
204
+ * statusline ingest of a live `agents run` and, when that row is recent, skip a
205
+ * redundant API refresh while still re-deriving headroom from it — instead of
206
+ * spending scarce provider budget re-fetching an already-current account.
207
+ */
208
+ readCachedSnapshot?: (usageKey: string) => UsageSnapshot | null;
114
209
  }
115
210
  export interface UsageRefreshResult {
116
211
  refreshed: number;
117
212
  skippedNotDue: number;
118
213
  skippedBackoff: number;
119
214
  skippedCap: number;
215
+ /** Skipped because the provider's rolling-hour budget was already spent. */
216
+ skippedBudget: number;
217
+ /** Skipped because a free statusline ingest already captured it recently. */
218
+ skippedFresh: number;
120
219
  failed: number;
121
220
  }
122
221
  /**
@@ -50,7 +50,7 @@ import * as fs from 'fs';
50
50
  import * as path from 'path';
51
51
  import { getCacheDir } from './state.js';
52
52
  import { atomicWriteFileSync, ensureLockTarget, withFileLock } from './fs-atomic.js';
53
- import { deriveUsageHeadroom, buildCanonicalUsageContext, USAGE_SOURCE_AGENT_IDS, } from './accounting/usage.js';
53
+ import { deriveUsageHeadroom, buildCanonicalUsageContext, agentUsesNetworkUsage, USAGE_SOURCE_AGENT_IDS, } from './accounting/usage.js';
54
54
  import { getAccountInfo } from './agents.js';
55
55
  import { listInstalledVersions, getVersionHomePath } from './installations/versions.js';
56
56
  /**
@@ -71,13 +71,63 @@ export const REFRESH_BURN_DIVISOR = 4;
71
71
  export const HOURLY_CALL_CAP = 12;
72
72
  /** How often the daemon wakes to *consider* a refresh pass (due accounts only). */
73
73
  export const USAGE_REFRESH_TICK_MS = 60 * 1000;
74
+ const HOUR_MS = 60 * 60 * 1000;
75
+ /**
76
+ * Minimum wall-clock spacing between two live usage fetches to ONE network
77
+ * provider, across all of its accounts. This is the pacing primitive: refreshes
78
+ * are issued round-robin (stalest account first) no faster than one per spacing,
79
+ * so aggregate endpoint load is a smooth, fixed rate — never the synchronized
80
+ * burst-then-stall a plain rolling-hour cap produces when every account falls
81
+ * due on the same tick. Set to two daemon ticks so the floor-based pacing lands
82
+ * on exact tick boundaries (no drift): one refresh every other tick ⇒ 30/hr.
83
+ */
84
+ export const PROVIDER_MIN_REFRESH_SPACING_MS = 2 * USAGE_REFRESH_TICK_MS;
85
+ /**
86
+ * Aggregate live fetches this daemon may spend on ONE network provider's usage
87
+ * endpoint per rolling hour, across ALL of that provider's local accounts —
88
+ * derived from {@link PROVIDER_MIN_REFRESH_SPACING_MS} so the two are always
89
+ * consistent (HOUR / 120s = 30).
90
+ *
91
+ * The per-account {@link HOURLY_CALL_CAP} alone scales linearly with account
92
+ * count — 8 Claude accounts × 12/hr = ~96 usage calls/hr from one box — and
93
+ * Anthropic's `/api/oauth/usage` rate-limits around ~100/hr (see the
94
+ * `usage-backoff.ts` header). That tripped the endpoint into per-account 429s
95
+ * with Retry-After penalties up to an hour: measured live on `zion`, 7 of 8
96
+ * Claude accounts sat parked, never refreshed inside their 5h window, so
97
+ * `agents view` showed `S: unavailable` and balanced routing read stale/absent
98
+ * usage. It got WORSE with every account added.
99
+ *
100
+ * 30/hr is a fixed rate that does NOT grow with account count, and leaves ample
101
+ * headroom under the ~100/hr ceiling for the auth probe (same endpoint, ~3/hr
102
+ * per account, RUSH-2998) and foreground `agents view` bursts. Because refreshes
103
+ * are paced round-robin (stalest first), each account's worst-case proactive
104
+ * cadence is bounded at N × spacing (8 accounts ⇒ 16 min; 16 ⇒ 32 min) — kept
105
+ * deliberately under the {@link USAGE_STALE_REFUSAL_MAX_AGE_MS} routing window so
106
+ * a budget-paced account never reads as "genuinely stale". A slightly
107
+ * older-but-present reading beats a 45-minute 429 park. Network providers only;
108
+ * grok/codex read local logs and have no rate-limited endpoint.
109
+ */
110
+ export const PROVIDER_HOURLY_BUDGET = HOUR_MS / PROVIDER_MIN_REFRESH_SPACING_MS;
111
+ /**
112
+ * Most refreshes a single tick may catch up after the daemon has been idle/down
113
+ * (elapsed ≫ spacing). Without this clamp a long gap would grant many tokens at
114
+ * once and re-synchronize every account into the very burst the spacing exists
115
+ * to prevent. A small catch-up keeps the load smooth even after a restart.
116
+ */
117
+ export const PROVIDER_CATCHUP_MAX = 2;
118
+ /**
119
+ * A usage row this recently captured (by the free statusline ingest of a live
120
+ * `agents run`, or any writer) is already fresh — do not spend an API call to
121
+ * re-refresh it. Actively-used accounts stay current at zero endpoint cost, so
122
+ * the proactive budget is reserved for genuinely idle accounts.
123
+ */
124
+ export const STATUSLINE_FRESH_MS = REFRESH_INTERVAL_MS;
74
125
  /** Consecutive failed live reads before one broken account is quarantined. */
75
126
  export const FAILURE_QUARANTINE_THRESHOLD = 3;
76
127
  /** A chronic offender waits this long while healthy siblings keep their cadence. */
77
128
  export const FAILURE_QUARANTINE_MS = 30 * 60 * 1000;
78
129
  const SKIP_JITTER_MIN_MS = 2_000;
79
130
  const SKIP_JITTER_RANGE_MS = 3_001;
80
- const HOUR_MS = 60 * 60 * 1000;
81
131
  /** Test seam for the headroom cache path (see usage.ts `setClaudeUsageCachePathForTest`). */
82
132
  let headroomCachePathOverride = null;
83
133
  export function setHeadroomCachePathForTest(cachePath) {
@@ -197,6 +247,71 @@ function skippedHeadroomEntry(prev, usageKey, now, index) {
197
247
  consecutiveFailures: prev?.consecutiveFailures ?? 0,
198
248
  };
199
249
  }
250
+ /**
251
+ * Reschedule an account we skipped because a free statusline ingest already
252
+ * captured it inside {@link STATUSLINE_FRESH_MS}. The statusline row IS a real,
253
+ * live sample, so RE-DERIVE headroom (status / minutesToLimit) from it against
254
+ * the prior sample — otherwise `status`/`minutesToLimit` would freeze at their
255
+ * last API-refresh value forever for exactly the actively-used accounts that
256
+ * stay statusline-fresh, and `capacityWeight` reads `minutesToLimit`. No call
257
+ * timestamp is recorded (this cost zero API budget); the next proactive attempt
258
+ * is pushed to one interval past the free capture.
259
+ */
260
+ function freshHeadroomEntry(prev, snapshot, now, capturedAtMs) {
261
+ const headroom = deriveUsageHeadroom(snapshot, prev && prev.capturedAt !== null && prev.sessionUsedPercent !== null
262
+ ? { capturedAt: prev.capturedAt, usedPercent: prev.sessionUsedPercent }
263
+ : null);
264
+ const session = snapshot.windows.find((window) => window.key === 'session') ?? null;
265
+ return {
266
+ status: headroom.status,
267
+ minutesToLimit: headroom.minutesToLimit,
268
+ sessionUsedPercent: session?.usedPercent ?? prev?.sessionUsedPercent ?? null,
269
+ capturedAt: snapshot.capturedAt?.getTime() ?? prev?.capturedAt ?? null,
270
+ nextRefreshAt: capturedAtMs + REFRESH_INTERVAL_MS,
271
+ // Not an API call — do NOT record a timestamp (would wrongly spend budget).
272
+ callTimestamps: pruneCallTimestamps(prev?.callTimestamps ?? [], now),
273
+ computedAt: now,
274
+ consecutiveFailures: 0,
275
+ };
276
+ }
277
+ /**
278
+ * Most-recent live-fetch time per network provider (the max call timestamp
279
+ * across its accounts, 0 when none), which the smooth per-provider pacing spaces
280
+ * the next refresh from. Non-network providers are omitted — they have no
281
+ * rate-limited endpoint to pace.
282
+ */
283
+ export function providerLastCall(accounts, cache) {
284
+ const last = new Map();
285
+ for (const account of accounts) {
286
+ if (!agentUsesNetworkUsage(account.agentId))
287
+ continue;
288
+ if (!last.has(account.agentId))
289
+ last.set(account.agentId, 0);
290
+ for (const ts of cache[account.usageKey]?.callTimestamps ?? []) {
291
+ if (ts > (last.get(account.agentId) ?? 0))
292
+ last.set(account.agentId, ts);
293
+ }
294
+ }
295
+ return last;
296
+ }
297
+ /**
298
+ * How many live fetches the smooth pacing permits a provider THIS tick: one per
299
+ * elapsed {@link PROVIDER_MIN_REFRESH_SPACING_MS} since its last fetch, clamped
300
+ * to {@link PROVIDER_CATCHUP_MAX} so a long idle gap (or a cold provider with no
301
+ * prior fetch) cannot re-burst the whole due set at once. At the daemon's 60 s
302
+ * tick this yields at most one fetch every other tick in steady state (⇒ the
303
+ * hourly budget), while a small fleet whose total demand fits under budget is
304
+ * never throttled — the {@link PROVIDER_HOURLY_BUDGET} rolling cap is the only
305
+ * gate that binds it.
306
+ */
307
+ export function providerSpacingTokens(lastCallMs, now) {
308
+ // A cold provider (never fetched) is treated as maximally idle: grant the
309
+ // catch-up ceiling so a couple of accounts warm immediately without bursting.
310
+ const elapsed = lastCallMs <= 0 ? Infinity : now - lastCallMs;
311
+ if (elapsed < PROVIDER_MIN_REFRESH_SPACING_MS)
312
+ return 0;
313
+ return Math.min(PROVIDER_CATCHUP_MAX, Math.floor(elapsed / PROVIDER_MIN_REFRESH_SPACING_MS));
314
+ }
200
315
  function failedHeadroomEntry(prev, now) {
201
316
  const next = nextHeadroomEntry(prev, null, now);
202
317
  if ((next.consecutiveFailures ?? 0) >= FAILURE_QUARANTINE_THRESHOLD) {
@@ -204,7 +319,18 @@ function failedHeadroomEntry(prev, now) {
204
319
  }
205
320
  return next;
206
321
  }
207
- /** Cold accounts lead each pass; both cold and cached groups rotate every tick. */
322
+ /**
323
+ * Order a pass STALEST-FIRST so a scarce per-provider budget
324
+ * ({@link PROVIDER_HOURLY_BUDGET}) is always spent on the accounts most in need
325
+ * of a fresh reading, and no account is starved indefinitely.
326
+ *
327
+ * - **Cold accounts** (never refreshed → no cache entry) are maximally stale
328
+ * and lead the pass. They rotate by `tick` so, when the budget can't cover
329
+ * them all in one tick, a different cold account leads each tick.
330
+ * - **Cached accounts** follow, oldest `capturedAt` first (a null capture time
331
+ * counts as maximally stale). As accounts refresh their `capturedAt` advances,
332
+ * so the next pass naturally rotates to whoever is now most out of date.
333
+ */
208
334
  export function orderUsageAccounts(accounts, cache, tick) {
209
335
  const rotate = (group) => {
210
336
  if (group.length < 2)
@@ -214,7 +340,26 @@ export function orderUsageAccounts(accounts, cache, tick) {
214
340
  };
215
341
  const cold = accounts.filter((account) => cache[account.usageKey] == null);
216
342
  const cached = accounts.filter((account) => cache[account.usageKey] != null);
217
- return [...rotate(cold), ...rotate(cached)];
343
+ const staleness = (account) => cache[account.usageKey]?.capturedAt ?? 0;
344
+ const byStalest = [...cached].sort((a, b) => staleness(a) - staleness(b));
345
+ return [...rotate(cold), ...byStalest];
346
+ }
347
+ /**
348
+ * Aggregate live calls a network provider has already spent in the trailing hour,
349
+ * summed across the accounts in this pass. Seeds the per-provider budget counter
350
+ * so {@link PROVIDER_HOURLY_BUDGET} bounds the rolling-hour total, not just this
351
+ * one tick. Non-network providers (grok/codex, local logs) are excluded — they
352
+ * have no rate-limited endpoint to budget.
353
+ */
354
+ export function providerRecentCalls(accounts, cache, now) {
355
+ const counts = new Map();
356
+ for (const account of accounts) {
357
+ if (!agentUsesNetworkUsage(account.agentId))
358
+ continue;
359
+ const recent = pruneCallTimestamps(cache[account.usageKey]?.callTimestamps ?? [], now);
360
+ counts.set(account.agentId, (counts.get(account.agentId) ?? 0) + recent.length);
361
+ }
362
+ return counts;
218
363
  }
219
364
  /**
220
365
  * Enumerate the usage accounts whose credentials live on THIS host — one
@@ -274,13 +419,31 @@ export async function runUsageRefresh(deps) {
274
419
  skippedNotDue: 0,
275
420
  skippedBackoff: 0,
276
421
  skippedCap: 0,
422
+ skippedBudget: 0,
423
+ skippedFresh: 0,
277
424
  failed: 0,
278
425
  };
279
426
  const cache = readHeadroomCache();
280
427
  const accounts = orderUsageAccounts(await deps.listAccounts(), cache, Math.floor(now / USAGE_REFRESH_TICK_MS));
428
+ // Per-provider pacing. Two gates keep aggregate endpoint load smooth and bounded:
429
+ // - a rolling-hour ceiling (PROVIDER_HOURLY_BUDGET) — the hard cap, seeded
430
+ // with calls already spent in the trailing hour;
431
+ // - a min-spacing token count (PROVIDER_MIN_REFRESH_SPACING_MS) — the smoother,
432
+ // which issues refreshes round-robin at a fixed rate instead of the
433
+ // synchronized burst-then-stall a plain rolling cap produces when every
434
+ // account falls due on the same tick.
435
+ // Both are per-provider and network-only; accounts are ordered stalest-first, so
436
+ // the scarce budget always serves the account most in need and none is starved.
437
+ const budgetSpent = providerRecentCalls(accounts, cache, now);
438
+ const lastCall = providerLastCall(accounts, cache);
439
+ const spacingTokens = new Map();
440
+ for (const [agent, last] of lastCall)
441
+ spacingTokens.set(agent, providerSpacingTokens(last, now));
442
+ const spacingUsed = new Map();
281
443
  const updates = {};
282
444
  for (const [index, account] of accounts.entries()) {
283
445
  const entry = cache[account.usageKey] ?? null;
446
+ const network = agentUsesNetworkUsage(account.agentId);
284
447
  // A penalized account/provider is off-limits — poking it re-arms the
285
448
  // penalty (the whole reason usage-backoff exists).
286
449
  if ((deps.backoffUntil(account.agentId, account.usageKey) ?? 0) > now) {
@@ -295,6 +458,34 @@ export async function runUsageRefresh(deps) {
295
458
  result.skippedCap += 1;
296
459
  continue;
297
460
  }
461
+ // A live `agents run` already refreshed this account's usage row for free via
462
+ // the statusline ingest — re-derive headroom from that row and skip the API
463
+ // call. Network providers only: a local-log provider's cache is always its
464
+ // own last write, so this must not suppress its refresh (grok/codex).
465
+ if (network) {
466
+ const cached = deps.readCachedSnapshot?.(account.usageKey) ?? null;
467
+ const capturedAtMs = cached?.capturedAt?.getTime() ?? null;
468
+ if (cached && capturedAtMs !== null && now - capturedAtMs < STATUSLINE_FRESH_MS) {
469
+ updates[account.usageKey] = freshHeadroomEntry(entry, cached, now, capturedAtMs);
470
+ result.skippedFresh += 1;
471
+ continue;
472
+ }
473
+ }
474
+ // Global per-provider budget: cap aggregate endpoint traffic so it does not
475
+ // scale linearly with account count and trip the ~100/hr rate limit, and pace
476
+ // it smoothly. Non-network providers (local logs) have no endpoint to protect.
477
+ if (network) {
478
+ const overHourly = (budgetSpent.get(account.agentId) ?? 0) >= PROVIDER_HOURLY_BUDGET;
479
+ const overSpacing = (spacingUsed.get(account.agentId) ?? 0) >= (spacingTokens.get(account.agentId) ?? 0);
480
+ if (overHourly || overSpacing) {
481
+ // Leave the entry untouched so this still-due account competes again next
482
+ // tick, when budget/spacing frees — never starved (stalest-first serves it).
483
+ result.skippedBudget += 1;
484
+ continue;
485
+ }
486
+ budgetSpent.set(account.agentId, (budgetSpent.get(account.agentId) ?? 0) + 1);
487
+ spacingUsed.set(account.agentId, (spacingUsed.get(account.agentId) ?? 0) + 1);
488
+ }
298
489
  try {
299
490
  const usage = await account.fetch();
300
491
  if (usage.snapshot) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@phnx-labs/agents-cli",
3
- "version": "1.22.64",
3
+ "version": "1.22.66",
4
4
  "description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",