@phnx-labs/agents-cli 1.22.63 → 1.22.65
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +64 -0
- package/dist/commands/accounts.d.ts +8 -7
- package/dist/commands/accounts.js +40 -13
- package/dist/commands/exec.js +40 -0
- package/dist/commands/feed.js +2 -0
- package/dist/commands/resume.d.ts +8 -0
- package/dist/commands/resume.js +66 -4
- package/dist/lib/accounting/account-pool-collect.js +7 -2
- package/dist/lib/accounting/account-pool.d.ts +12 -0
- package/dist/lib/accounting/account-pool.js +1 -0
- package/dist/lib/accounting/rotate.d.ts +57 -8
- package/dist/lib/accounting/rotate.js +57 -10
- package/dist/lib/accounting/usage.d.ts +12 -0
- package/dist/lib/accounting/usage.js +119 -15
- package/dist/lib/agent-spec/agents.js +21 -6
- package/dist/lib/claude-account-token.d.ts +26 -0
- package/dist/lib/claude-account-token.js +65 -11
- package/dist/lib/daemon-ticks.js +6 -2
- package/dist/lib/event-families.js +1 -1
- package/dist/lib/exec.d.ts +66 -1
- package/dist/lib/exec.js +104 -8
- package/dist/lib/feed/events.d.ts +1 -1
- package/dist/lib/feed/events.js +5 -0
- package/dist/lib/feed-broadcast.d.ts +2 -0
- package/dist/lib/feed-broadcast.js +31 -4
- package/dist/lib/harness/adapters/claude.js +33 -27
- package/dist/lib/hosts/passthrough.d.ts +44 -0
- package/dist/lib/hosts/passthrough.js +66 -8
- package/dist/lib/identity/client.d.ts +5 -4
- package/dist/lib/identity/client.js +6 -5
- package/dist/lib/session/active.d.ts +10 -1
- package/dist/lib/session/active.js +8 -0
- package/dist/lib/session/digest.js +8 -2
- package/dist/lib/session/parse.d.ts +4 -0
- package/dist/lib/session/parse.js +16 -11
- package/dist/lib/session/state.d.ts +5 -0
- package/dist/lib/session/state.js +6 -0
- package/dist/lib/usage-refresh.d.ts +100 -1
- package/dist/lib/usage-refresh.js +195 -4
- package/package.json +1 -1
|
@@ -70,6 +70,10 @@ export declare function parseCodex(filePath: string): SessionEvent[];
|
|
|
70
70
|
* Returns every path in order (a multi-file patch emits multiple paths so
|
|
71
71
|
* artifact discovery sees each file — RUSH-1410). Empty when unparseable.
|
|
72
72
|
*/
|
|
73
|
+
export declare function applyPatchTargets(input: string): Array<{
|
|
74
|
+
path: string;
|
|
75
|
+
op: 'Add' | 'Update' | 'Delete';
|
|
76
|
+
}>;
|
|
73
77
|
export declare function applyPatchTargetPaths(input: string): string[];
|
|
74
78
|
/**
|
|
75
79
|
* Parse Codex JSONL *content* (already read into a string) into normalized
|
|
@@ -555,15 +555,18 @@ export function parseCodex(filePath) {
|
|
|
555
555
|
* Returns every path in order (a multi-file patch emits multiple paths so
|
|
556
556
|
* artifact discovery sees each file — RUSH-1410). Empty when unparseable.
|
|
557
557
|
*/
|
|
558
|
-
export function
|
|
559
|
-
const
|
|
560
|
-
const re = /^\*\*\* (
|
|
558
|
+
export function applyPatchTargets(input) {
|
|
559
|
+
const targets = [];
|
|
560
|
+
const re = /^\*\*\* (Update|Add|Delete) File: (.+)$/gm;
|
|
561
561
|
for (const m of input.matchAll(re)) {
|
|
562
|
-
const p = m[
|
|
562
|
+
const p = m[2].trim();
|
|
563
563
|
if (p)
|
|
564
|
-
|
|
564
|
+
targets.push({ path: p, op: m[1] });
|
|
565
565
|
}
|
|
566
|
-
return
|
|
566
|
+
return targets;
|
|
567
|
+
}
|
|
568
|
+
export function applyPatchTargetPaths(input) {
|
|
569
|
+
return applyPatchTargets(input).map((target) => target.path);
|
|
567
570
|
}
|
|
568
571
|
/** @deprecated Prefer applyPatchTargetPaths — kept for single-file call sites. */
|
|
569
572
|
function applyPatchTargetPath(input) {
|
|
@@ -724,13 +727,15 @@ export function parseCodexContent(content) {
|
|
|
724
727
|
const execCommand = rawName === 'exec' ? codexExecCommand(input) : undefined;
|
|
725
728
|
// Multi-file patches: one tool_use per file so artifact discovery sees
|
|
726
729
|
// every path (RUSH-1410). Single-file / non-patch keep one event.
|
|
727
|
-
const
|
|
730
|
+
const patchTargets = isApplyPatch ? applyPatchTargets(input) : [];
|
|
728
731
|
const tool = isApplyPatch ? 'Edit' : rawName;
|
|
729
732
|
const truncatedInput = input.length > 500 ? input.slice(0, 497) + '...' : input;
|
|
730
|
-
const emitOne = (patchPath) => {
|
|
733
|
+
const emitOne = (patchPath, patchOp) => {
|
|
731
734
|
const args = { input: truncatedInput };
|
|
732
735
|
if (patchPath)
|
|
733
736
|
args.file_path = patchPath;
|
|
737
|
+
if (patchOp)
|
|
738
|
+
args.patch_op = patchOp;
|
|
734
739
|
if (execCommand)
|
|
735
740
|
args.command = execCommand;
|
|
736
741
|
const callId = payload.call_id || payload.id;
|
|
@@ -747,9 +752,9 @@ export function parseCodexContent(content) {
|
|
|
747
752
|
path: patchPath,
|
|
748
753
|
});
|
|
749
754
|
};
|
|
750
|
-
if (isApplyPatch &&
|
|
751
|
-
for (const
|
|
752
|
-
emitOne(
|
|
755
|
+
if (isApplyPatch && patchTargets.length > 0) {
|
|
756
|
+
for (const target of patchTargets)
|
|
757
|
+
emitOne(target.path, target.op);
|
|
753
758
|
}
|
|
754
759
|
else {
|
|
755
760
|
emitOne(undefined);
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
* shape + mtime — same function, driven off the normalized events.
|
|
16
16
|
*/
|
|
17
17
|
import type { SessionAttachment, SessionEvent, TodoItem, TodoProgress } from './types.js';
|
|
18
|
+
import { type ProducedArtifact } from './highlights.js';
|
|
18
19
|
export type { TodoItem, TodoProgress };
|
|
19
20
|
export type SessionActivity = 'working' | 'waiting_input' | 'idle';
|
|
20
21
|
export type AwaitingReason = 'question' | 'plan_review' | 'permission';
|
|
@@ -95,6 +96,10 @@ export interface SessionState {
|
|
|
95
96
|
* checklist, notably for remote/device-dispatched agents with no local stream.
|
|
96
97
|
*/
|
|
97
98
|
todos?: TodoProgress;
|
|
99
|
+
/** Durable documents created in the bounded transcript slice. */
|
|
100
|
+
artifacts?: ProducedArtifact[];
|
|
101
|
+
/** First created plan document, for consumers that give plans special treatment. */
|
|
102
|
+
planFile?: string;
|
|
98
103
|
/** Last few assistant turns (most-recent last), one line each — panel context. */
|
|
99
104
|
tail?: string[];
|
|
100
105
|
lastActivityMs?: number;
|
|
@@ -17,6 +17,8 @@
|
|
|
17
17
|
import * as path from 'path';
|
|
18
18
|
import { isCompletedTodoStatus, SNAPSHOT_TODO_TOOLS, summarizeToolUse } from './parse.js';
|
|
19
19
|
import { isShellExecTool } from './shell-programs.js';
|
|
20
|
+
import { classifyFileChanges } from './digest.js';
|
|
21
|
+
import { extractArtifacts } from './highlights.js';
|
|
20
22
|
/**
|
|
21
23
|
* Detect per-session rate-limit / usage-limit signals in assistant or error
|
|
22
24
|
* text (RUSH-1523). Matches the same shapes the ext's prewarm detectBlockingPrompt
|
|
@@ -626,6 +628,8 @@ export function inferSessionState(events, ctx = {}) {
|
|
|
626
628
|
const state = inferActivity(events, ctx);
|
|
627
629
|
const { pr, ticket, createdTickets, spawnedTeam, attachments } = detectDurableSignals(events);
|
|
628
630
|
const worktree = detectWorktree(ctx.cwd, ctx.gitBranch);
|
|
631
|
+
const artifacts = extractArtifacts(classifyFileChanges(events));
|
|
632
|
+
const planFile = artifacts.find((artifact) => artifact.bucket === 'plans')?.path;
|
|
629
633
|
// Rate-limit: scan the most recent assistant messages + tool errors (tail-first).
|
|
630
634
|
let rateLimited = false;
|
|
631
635
|
for (let i = events.length - 1; i >= 0 && i >= events.length - 12; i--) {
|
|
@@ -651,6 +655,8 @@ export function inferSessionState(events, ctx = {}) {
|
|
|
651
655
|
createdTickets,
|
|
652
656
|
spawnedTeam,
|
|
653
657
|
attachments,
|
|
658
|
+
artifacts: artifacts.length > 0 ? artifacts : undefined,
|
|
659
|
+
planFile,
|
|
654
660
|
rateLimited: rateLimited || undefined,
|
|
655
661
|
};
|
|
656
662
|
}
|
|
@@ -18,6 +18,56 @@ export declare const REFRESH_BURN_DIVISOR = 4;
|
|
|
18
18
|
export declare const HOURLY_CALL_CAP = 12;
|
|
19
19
|
/** How often the daemon wakes to *consider* a refresh pass (due accounts only). */
|
|
20
20
|
export declare const USAGE_REFRESH_TICK_MS: number;
|
|
21
|
+
/**
|
|
22
|
+
* Minimum wall-clock spacing between two live usage fetches to ONE network
|
|
23
|
+
* provider, across all of its accounts. This is the pacing primitive: refreshes
|
|
24
|
+
* are issued round-robin (stalest account first) no faster than one per spacing,
|
|
25
|
+
* so aggregate endpoint load is a smooth, fixed rate — never the synchronized
|
|
26
|
+
* burst-then-stall a plain rolling-hour cap produces when every account falls
|
|
27
|
+
* due on the same tick. Set to two daemon ticks so the floor-based pacing lands
|
|
28
|
+
* on exact tick boundaries (no drift): one refresh every other tick ⇒ 30/hr.
|
|
29
|
+
*/
|
|
30
|
+
export declare const PROVIDER_MIN_REFRESH_SPACING_MS: number;
|
|
31
|
+
/**
|
|
32
|
+
* Aggregate live fetches this daemon may spend on ONE network provider's usage
|
|
33
|
+
* endpoint per rolling hour, across ALL of that provider's local accounts —
|
|
34
|
+
* derived from {@link PROVIDER_MIN_REFRESH_SPACING_MS} so the two are always
|
|
35
|
+
* consistent (HOUR / 120s = 30).
|
|
36
|
+
*
|
|
37
|
+
* The per-account {@link HOURLY_CALL_CAP} alone scales linearly with account
|
|
38
|
+
* count — 8 Claude accounts × 12/hr = ~96 usage calls/hr from one box — and
|
|
39
|
+
* Anthropic's `/api/oauth/usage` rate-limits around ~100/hr (see the
|
|
40
|
+
* `usage-backoff.ts` header). That tripped the endpoint into per-account 429s
|
|
41
|
+
* with Retry-After penalties up to an hour: measured live on `zion`, 7 of 8
|
|
42
|
+
* Claude accounts sat parked, never refreshed inside their 5h window, so
|
|
43
|
+
* `agents view` showed `S: unavailable` and balanced routing read stale/absent
|
|
44
|
+
* usage. It got WORSE with every account added.
|
|
45
|
+
*
|
|
46
|
+
* 30/hr is a fixed rate that does NOT grow with account count, and leaves ample
|
|
47
|
+
* headroom under the ~100/hr ceiling for the auth probe (same endpoint, ~3/hr
|
|
48
|
+
* per account, RUSH-2998) and foreground `agents view` bursts. Because refreshes
|
|
49
|
+
* are paced round-robin (stalest first), each account's worst-case proactive
|
|
50
|
+
* cadence is bounded at N × spacing (8 accounts ⇒ 16 min; 16 ⇒ 32 min) — kept
|
|
51
|
+
* deliberately under the {@link USAGE_STALE_REFUSAL_MAX_AGE_MS} routing window so
|
|
52
|
+
* a budget-paced account never reads as "genuinely stale". A slightly
|
|
53
|
+
* older-but-present reading beats a 45-minute 429 park. Network providers only;
|
|
54
|
+
* grok/codex read local logs and have no rate-limited endpoint.
|
|
55
|
+
*/
|
|
56
|
+
export declare const PROVIDER_HOURLY_BUDGET: number;
|
|
57
|
+
/**
|
|
58
|
+
* Most refreshes a single tick may catch up after the daemon has been idle/down
|
|
59
|
+
* (elapsed ≫ spacing). Without this clamp a long gap would grant many tokens at
|
|
60
|
+
* once and re-synchronize every account into the very burst the spacing exists
|
|
61
|
+
* to prevent. A small catch-up keeps the load smooth even after a restart.
|
|
62
|
+
*/
|
|
63
|
+
export declare const PROVIDER_CATCHUP_MAX = 2;
|
|
64
|
+
/**
|
|
65
|
+
* A usage row this recently captured (by the free statusline ingest of a live
|
|
66
|
+
* `agents run`, or any writer) is already fresh — do not spend an API call to
|
|
67
|
+
* re-refresh it. Actively-used accounts stay current at zero endpoint cost, so
|
|
68
|
+
* the proactive budget is reserved for genuinely idle accounts.
|
|
69
|
+
*/
|
|
70
|
+
export declare const STATUSLINE_FRESH_MS: number;
|
|
21
71
|
/** Consecutive failed live reads before one broken account is quarantined. */
|
|
22
72
|
export declare const FAILURE_QUARANTINE_THRESHOLD = 3;
|
|
23
73
|
/** A chronic offender waits this long while healthy siblings keep their cadence. */
|
|
@@ -78,6 +128,24 @@ export declare function shouldRefreshAccount(entry: HeadroomEntry | null | undef
|
|
|
78
128
|
* projection, and record this call for the hourly cap.
|
|
79
129
|
*/
|
|
80
130
|
export declare function nextHeadroomEntry(prev: HeadroomEntry | null | undefined, snapshot: UsageSnapshot | null, now: number): HeadroomEntry;
|
|
131
|
+
/**
|
|
132
|
+
* Most-recent live-fetch time per network provider (the max call timestamp
|
|
133
|
+
* across its accounts, 0 when none), which the smooth per-provider pacing spaces
|
|
134
|
+
* the next refresh from. Non-network providers are omitted — they have no
|
|
135
|
+
* rate-limited endpoint to pace.
|
|
136
|
+
*/
|
|
137
|
+
export declare function providerLastCall(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>): Map<AgentId, number>;
|
|
138
|
+
/**
|
|
139
|
+
* How many live fetches the smooth pacing permits a provider THIS tick: one per
|
|
140
|
+
* elapsed {@link PROVIDER_MIN_REFRESH_SPACING_MS} since its last fetch, clamped
|
|
141
|
+
* to {@link PROVIDER_CATCHUP_MAX} so a long idle gap (or a cold provider with no
|
|
142
|
+
* prior fetch) cannot re-burst the whole due set at once. At the daemon's 60 s
|
|
143
|
+
* tick this yields at most one fetch every other tick in steady state (⇒ the
|
|
144
|
+
* hourly budget), while a small fleet whose total demand fits under budget is
|
|
145
|
+
* never throttled — the {@link PROVIDER_HOURLY_BUDGET} rolling cap is the only
|
|
146
|
+
* gate that binds it.
|
|
147
|
+
*/
|
|
148
|
+
export declare function providerSpacingTokens(lastCallMs: number, now: number): number;
|
|
81
149
|
/** An account whose credentials live on the publisher host. */
|
|
82
150
|
export interface LocalUsageAccount {
|
|
83
151
|
usageKey: string;
|
|
@@ -85,8 +153,27 @@ export interface LocalUsageAccount {
|
|
|
85
153
|
/** Live-fetch this account's usage; the daemon passes the real network fetch. */
|
|
86
154
|
fetch: () => Promise<UsageInfo>;
|
|
87
155
|
}
|
|
88
|
-
/**
|
|
156
|
+
/**
|
|
157
|
+
* Order a pass STALEST-FIRST so a scarce per-provider budget
|
|
158
|
+
* ({@link PROVIDER_HOURLY_BUDGET}) is always spent on the accounts most in need
|
|
159
|
+
* of a fresh reading, and no account is starved indefinitely.
|
|
160
|
+
*
|
|
161
|
+
* - **Cold accounts** (never refreshed → no cache entry) are maximally stale
|
|
162
|
+
* and lead the pass. They rotate by `tick` so, when the budget can't cover
|
|
163
|
+
* them all in one tick, a different cold account leads each tick.
|
|
164
|
+
* - **Cached accounts** follow, oldest `capturedAt` first (a null capture time
|
|
165
|
+
* counts as maximally stale). As accounts refresh their `capturedAt` advances,
|
|
166
|
+
* so the next pass naturally rotates to whoever is now most out of date.
|
|
167
|
+
*/
|
|
89
168
|
export declare function orderUsageAccounts(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>, tick: number): LocalUsageAccount[];
|
|
169
|
+
/**
|
|
170
|
+
* Aggregate live calls a network provider has already spent in the trailing hour,
|
|
171
|
+
* summed across the accounts in this pass. Seeds the per-provider budget counter
|
|
172
|
+
* so {@link PROVIDER_HOURLY_BUDGET} bounds the rolling-hour total, not just this
|
|
173
|
+
* one tick. Non-network providers (grok/codex, local logs) are excluded — they
|
|
174
|
+
* have no rate-limited endpoint to budget.
|
|
175
|
+
*/
|
|
176
|
+
export declare function providerRecentCalls(accounts: LocalUsageAccount[], cache: Record<string, HeadroomEntry>, now: number): Map<AgentId, number>;
|
|
90
177
|
/**
|
|
91
178
|
* Enumerate the usage accounts whose credentials live on THIS host — one
|
|
92
179
|
* per unique usage key, deduped to the most-recently-active version (the same
|
|
@@ -111,12 +198,24 @@ export interface UsageRefreshDeps {
|
|
|
111
198
|
* this loop's fixed iteration order.
|
|
112
199
|
*/
|
|
113
200
|
backoffUntil: (agentId: AgentId, usageKey?: string) => number | null;
|
|
201
|
+
/**
|
|
202
|
+
* The account's current usage row from the shared cache (the row the routing
|
|
203
|
+
* hot path reads), or null when absent. Lets the refresher see the FREE
|
|
204
|
+
* statusline ingest of a live `agents run` and, when that row is recent, skip a
|
|
205
|
+
* redundant API refresh while still re-deriving headroom from it — instead of
|
|
206
|
+
* spending scarce provider budget re-fetching an already-current account.
|
|
207
|
+
*/
|
|
208
|
+
readCachedSnapshot?: (usageKey: string) => UsageSnapshot | null;
|
|
114
209
|
}
|
|
115
210
|
export interface UsageRefreshResult {
|
|
116
211
|
refreshed: number;
|
|
117
212
|
skippedNotDue: number;
|
|
118
213
|
skippedBackoff: number;
|
|
119
214
|
skippedCap: number;
|
|
215
|
+
/** Skipped because the provider's rolling-hour budget was already spent. */
|
|
216
|
+
skippedBudget: number;
|
|
217
|
+
/** Skipped because a free statusline ingest already captured it recently. */
|
|
218
|
+
skippedFresh: number;
|
|
120
219
|
failed: number;
|
|
121
220
|
}
|
|
122
221
|
/**
|
|
@@ -50,7 +50,7 @@ import * as fs from 'fs';
|
|
|
50
50
|
import * as path from 'path';
|
|
51
51
|
import { getCacheDir } from './state.js';
|
|
52
52
|
import { atomicWriteFileSync, ensureLockTarget, withFileLock } from './fs-atomic.js';
|
|
53
|
-
import { deriveUsageHeadroom, buildCanonicalUsageContext, USAGE_SOURCE_AGENT_IDS, } from './accounting/usage.js';
|
|
53
|
+
import { deriveUsageHeadroom, buildCanonicalUsageContext, agentUsesNetworkUsage, USAGE_SOURCE_AGENT_IDS, } from './accounting/usage.js';
|
|
54
54
|
import { getAccountInfo } from './agents.js';
|
|
55
55
|
import { listInstalledVersions, getVersionHomePath } from './installations/versions.js';
|
|
56
56
|
/**
|
|
@@ -71,13 +71,63 @@ export const REFRESH_BURN_DIVISOR = 4;
|
|
|
71
71
|
export const HOURLY_CALL_CAP = 12;
|
|
72
72
|
/** How often the daemon wakes to *consider* a refresh pass (due accounts only). */
|
|
73
73
|
export const USAGE_REFRESH_TICK_MS = 60 * 1000;
|
|
74
|
+
const HOUR_MS = 60 * 60 * 1000;
|
|
75
|
+
/**
|
|
76
|
+
* Minimum wall-clock spacing between two live usage fetches to ONE network
|
|
77
|
+
* provider, across all of its accounts. This is the pacing primitive: refreshes
|
|
78
|
+
* are issued round-robin (stalest account first) no faster than one per spacing,
|
|
79
|
+
* so aggregate endpoint load is a smooth, fixed rate — never the synchronized
|
|
80
|
+
* burst-then-stall a plain rolling-hour cap produces when every account falls
|
|
81
|
+
* due on the same tick. Set to two daemon ticks so the floor-based pacing lands
|
|
82
|
+
* on exact tick boundaries (no drift): one refresh every other tick ⇒ 30/hr.
|
|
83
|
+
*/
|
|
84
|
+
export const PROVIDER_MIN_REFRESH_SPACING_MS = 2 * USAGE_REFRESH_TICK_MS;
|
|
85
|
+
/**
|
|
86
|
+
* Aggregate live fetches this daemon may spend on ONE network provider's usage
|
|
87
|
+
* endpoint per rolling hour, across ALL of that provider's local accounts —
|
|
88
|
+
* derived from {@link PROVIDER_MIN_REFRESH_SPACING_MS} so the two are always
|
|
89
|
+
* consistent (HOUR / 120s = 30).
|
|
90
|
+
*
|
|
91
|
+
* The per-account {@link HOURLY_CALL_CAP} alone scales linearly with account
|
|
92
|
+
* count — 8 Claude accounts × 12/hr = ~96 usage calls/hr from one box — and
|
|
93
|
+
* Anthropic's `/api/oauth/usage` rate-limits around ~100/hr (see the
|
|
94
|
+
* `usage-backoff.ts` header). That tripped the endpoint into per-account 429s
|
|
95
|
+
* with Retry-After penalties up to an hour: measured live on `zion`, 7 of 8
|
|
96
|
+
* Claude accounts sat parked, never refreshed inside their 5h window, so
|
|
97
|
+
* `agents view` showed `S: unavailable` and balanced routing read stale/absent
|
|
98
|
+
* usage. It got WORSE with every account added.
|
|
99
|
+
*
|
|
100
|
+
* 30/hr is a fixed rate that does NOT grow with account count, and leaves ample
|
|
101
|
+
* headroom under the ~100/hr ceiling for the auth probe (same endpoint, ~3/hr
|
|
102
|
+
* per account, RUSH-2998) and foreground `agents view` bursts. Because refreshes
|
|
103
|
+
* are paced round-robin (stalest first), each account's worst-case proactive
|
|
104
|
+
* cadence is bounded at N × spacing (8 accounts ⇒ 16 min; 16 ⇒ 32 min) — kept
|
|
105
|
+
* deliberately under the {@link USAGE_STALE_REFUSAL_MAX_AGE_MS} routing window so
|
|
106
|
+
* a budget-paced account never reads as "genuinely stale". A slightly
|
|
107
|
+
* older-but-present reading beats a 45-minute 429 park. Network providers only;
|
|
108
|
+
* grok/codex read local logs and have no rate-limited endpoint.
|
|
109
|
+
*/
|
|
110
|
+
export const PROVIDER_HOURLY_BUDGET = HOUR_MS / PROVIDER_MIN_REFRESH_SPACING_MS;
|
|
111
|
+
/**
|
|
112
|
+
* Most refreshes a single tick may catch up after the daemon has been idle/down
|
|
113
|
+
* (elapsed ≫ spacing). Without this clamp a long gap would grant many tokens at
|
|
114
|
+
* once and re-synchronize every account into the very burst the spacing exists
|
|
115
|
+
* to prevent. A small catch-up keeps the load smooth even after a restart.
|
|
116
|
+
*/
|
|
117
|
+
export const PROVIDER_CATCHUP_MAX = 2;
|
|
118
|
+
/**
|
|
119
|
+
* A usage row this recently captured (by the free statusline ingest of a live
|
|
120
|
+
* `agents run`, or any writer) is already fresh — do not spend an API call to
|
|
121
|
+
* re-refresh it. Actively-used accounts stay current at zero endpoint cost, so
|
|
122
|
+
* the proactive budget is reserved for genuinely idle accounts.
|
|
123
|
+
*/
|
|
124
|
+
export const STATUSLINE_FRESH_MS = REFRESH_INTERVAL_MS;
|
|
74
125
|
/** Consecutive failed live reads before one broken account is quarantined. */
|
|
75
126
|
export const FAILURE_QUARANTINE_THRESHOLD = 3;
|
|
76
127
|
/** A chronic offender waits this long while healthy siblings keep their cadence. */
|
|
77
128
|
export const FAILURE_QUARANTINE_MS = 30 * 60 * 1000;
|
|
78
129
|
const SKIP_JITTER_MIN_MS = 2_000;
|
|
79
130
|
const SKIP_JITTER_RANGE_MS = 3_001;
|
|
80
|
-
const HOUR_MS = 60 * 60 * 1000;
|
|
81
131
|
/** Test seam for the headroom cache path (see usage.ts `setClaudeUsageCachePathForTest`). */
|
|
82
132
|
let headroomCachePathOverride = null;
|
|
83
133
|
export function setHeadroomCachePathForTest(cachePath) {
|
|
@@ -197,6 +247,71 @@ function skippedHeadroomEntry(prev, usageKey, now, index) {
|
|
|
197
247
|
consecutiveFailures: prev?.consecutiveFailures ?? 0,
|
|
198
248
|
};
|
|
199
249
|
}
|
|
250
|
+
/**
|
|
251
|
+
* Reschedule an account we skipped because a free statusline ingest already
|
|
252
|
+
* captured it inside {@link STATUSLINE_FRESH_MS}. The statusline row IS a real,
|
|
253
|
+
* live sample, so RE-DERIVE headroom (status / minutesToLimit) from it against
|
|
254
|
+
* the prior sample — otherwise `status`/`minutesToLimit` would freeze at their
|
|
255
|
+
* last API-refresh value forever for exactly the actively-used accounts that
|
|
256
|
+
* stay statusline-fresh, and `capacityWeight` reads `minutesToLimit`. No call
|
|
257
|
+
* timestamp is recorded (this cost zero API budget); the next proactive attempt
|
|
258
|
+
* is pushed to one interval past the free capture.
|
|
259
|
+
*/
|
|
260
|
+
function freshHeadroomEntry(prev, snapshot, now, capturedAtMs) {
|
|
261
|
+
const headroom = deriveUsageHeadroom(snapshot, prev && prev.capturedAt !== null && prev.sessionUsedPercent !== null
|
|
262
|
+
? { capturedAt: prev.capturedAt, usedPercent: prev.sessionUsedPercent }
|
|
263
|
+
: null);
|
|
264
|
+
const session = snapshot.windows.find((window) => window.key === 'session') ?? null;
|
|
265
|
+
return {
|
|
266
|
+
status: headroom.status,
|
|
267
|
+
minutesToLimit: headroom.minutesToLimit,
|
|
268
|
+
sessionUsedPercent: session?.usedPercent ?? prev?.sessionUsedPercent ?? null,
|
|
269
|
+
capturedAt: snapshot.capturedAt?.getTime() ?? prev?.capturedAt ?? null,
|
|
270
|
+
nextRefreshAt: capturedAtMs + REFRESH_INTERVAL_MS,
|
|
271
|
+
// Not an API call — do NOT record a timestamp (would wrongly spend budget).
|
|
272
|
+
callTimestamps: pruneCallTimestamps(prev?.callTimestamps ?? [], now),
|
|
273
|
+
computedAt: now,
|
|
274
|
+
consecutiveFailures: 0,
|
|
275
|
+
};
|
|
276
|
+
}
|
|
277
|
+
/**
|
|
278
|
+
* Most-recent live-fetch time per network provider (the max call timestamp
|
|
279
|
+
* across its accounts, 0 when none), which the smooth per-provider pacing spaces
|
|
280
|
+
* the next refresh from. Non-network providers are omitted — they have no
|
|
281
|
+
* rate-limited endpoint to pace.
|
|
282
|
+
*/
|
|
283
|
+
export function providerLastCall(accounts, cache) {
|
|
284
|
+
const last = new Map();
|
|
285
|
+
for (const account of accounts) {
|
|
286
|
+
if (!agentUsesNetworkUsage(account.agentId))
|
|
287
|
+
continue;
|
|
288
|
+
if (!last.has(account.agentId))
|
|
289
|
+
last.set(account.agentId, 0);
|
|
290
|
+
for (const ts of cache[account.usageKey]?.callTimestamps ?? []) {
|
|
291
|
+
if (ts > (last.get(account.agentId) ?? 0))
|
|
292
|
+
last.set(account.agentId, ts);
|
|
293
|
+
}
|
|
294
|
+
}
|
|
295
|
+
return last;
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* How many live fetches the smooth pacing permits a provider THIS tick: one per
|
|
299
|
+
* elapsed {@link PROVIDER_MIN_REFRESH_SPACING_MS} since its last fetch, clamped
|
|
300
|
+
* to {@link PROVIDER_CATCHUP_MAX} so a long idle gap (or a cold provider with no
|
|
301
|
+
* prior fetch) cannot re-burst the whole due set at once. At the daemon's 60 s
|
|
302
|
+
* tick this yields at most one fetch every other tick in steady state (⇒ the
|
|
303
|
+
* hourly budget), while a small fleet whose total demand fits under budget is
|
|
304
|
+
* never throttled — the {@link PROVIDER_HOURLY_BUDGET} rolling cap is the only
|
|
305
|
+
* gate that binds it.
|
|
306
|
+
*/
|
|
307
|
+
export function providerSpacingTokens(lastCallMs, now) {
|
|
308
|
+
// A cold provider (never fetched) is treated as maximally idle: grant the
|
|
309
|
+
// catch-up ceiling so a couple of accounts warm immediately without bursting.
|
|
310
|
+
const elapsed = lastCallMs <= 0 ? Infinity : now - lastCallMs;
|
|
311
|
+
if (elapsed < PROVIDER_MIN_REFRESH_SPACING_MS)
|
|
312
|
+
return 0;
|
|
313
|
+
return Math.min(PROVIDER_CATCHUP_MAX, Math.floor(elapsed / PROVIDER_MIN_REFRESH_SPACING_MS));
|
|
314
|
+
}
|
|
200
315
|
function failedHeadroomEntry(prev, now) {
|
|
201
316
|
const next = nextHeadroomEntry(prev, null, now);
|
|
202
317
|
if ((next.consecutiveFailures ?? 0) >= FAILURE_QUARANTINE_THRESHOLD) {
|
|
@@ -204,7 +319,18 @@ function failedHeadroomEntry(prev, now) {
|
|
|
204
319
|
}
|
|
205
320
|
return next;
|
|
206
321
|
}
|
|
207
|
-
/**
|
|
322
|
+
/**
|
|
323
|
+
* Order a pass STALEST-FIRST so a scarce per-provider budget
|
|
324
|
+
* ({@link PROVIDER_HOURLY_BUDGET}) is always spent on the accounts most in need
|
|
325
|
+
* of a fresh reading, and no account is starved indefinitely.
|
|
326
|
+
*
|
|
327
|
+
* - **Cold accounts** (never refreshed → no cache entry) are maximally stale
|
|
328
|
+
* and lead the pass. They rotate by `tick` so, when the budget can't cover
|
|
329
|
+
* them all in one tick, a different cold account leads each tick.
|
|
330
|
+
* - **Cached accounts** follow, oldest `capturedAt` first (a null capture time
|
|
331
|
+
* counts as maximally stale). As accounts refresh their `capturedAt` advances,
|
|
332
|
+
* so the next pass naturally rotates to whoever is now most out of date.
|
|
333
|
+
*/
|
|
208
334
|
export function orderUsageAccounts(accounts, cache, tick) {
|
|
209
335
|
const rotate = (group) => {
|
|
210
336
|
if (group.length < 2)
|
|
@@ -214,7 +340,26 @@ export function orderUsageAccounts(accounts, cache, tick) {
|
|
|
214
340
|
};
|
|
215
341
|
const cold = accounts.filter((account) => cache[account.usageKey] == null);
|
|
216
342
|
const cached = accounts.filter((account) => cache[account.usageKey] != null);
|
|
217
|
-
|
|
343
|
+
const staleness = (account) => cache[account.usageKey]?.capturedAt ?? 0;
|
|
344
|
+
const byStalest = [...cached].sort((a, b) => staleness(a) - staleness(b));
|
|
345
|
+
return [...rotate(cold), ...byStalest];
|
|
346
|
+
}
|
|
347
|
+
/**
|
|
348
|
+
* Aggregate live calls a network provider has already spent in the trailing hour,
|
|
349
|
+
* summed across the accounts in this pass. Seeds the per-provider budget counter
|
|
350
|
+
* so {@link PROVIDER_HOURLY_BUDGET} bounds the rolling-hour total, not just this
|
|
351
|
+
* one tick. Non-network providers (grok/codex, local logs) are excluded — they
|
|
352
|
+
* have no rate-limited endpoint to budget.
|
|
353
|
+
*/
|
|
354
|
+
export function providerRecentCalls(accounts, cache, now) {
|
|
355
|
+
const counts = new Map();
|
|
356
|
+
for (const account of accounts) {
|
|
357
|
+
if (!agentUsesNetworkUsage(account.agentId))
|
|
358
|
+
continue;
|
|
359
|
+
const recent = pruneCallTimestamps(cache[account.usageKey]?.callTimestamps ?? [], now);
|
|
360
|
+
counts.set(account.agentId, (counts.get(account.agentId) ?? 0) + recent.length);
|
|
361
|
+
}
|
|
362
|
+
return counts;
|
|
218
363
|
}
|
|
219
364
|
/**
|
|
220
365
|
* Enumerate the usage accounts whose credentials live on THIS host — one
|
|
@@ -274,13 +419,31 @@ export async function runUsageRefresh(deps) {
|
|
|
274
419
|
skippedNotDue: 0,
|
|
275
420
|
skippedBackoff: 0,
|
|
276
421
|
skippedCap: 0,
|
|
422
|
+
skippedBudget: 0,
|
|
423
|
+
skippedFresh: 0,
|
|
277
424
|
failed: 0,
|
|
278
425
|
};
|
|
279
426
|
const cache = readHeadroomCache();
|
|
280
427
|
const accounts = orderUsageAccounts(await deps.listAccounts(), cache, Math.floor(now / USAGE_REFRESH_TICK_MS));
|
|
428
|
+
// Per-provider pacing. Two gates keep aggregate endpoint load smooth and bounded:
|
|
429
|
+
// - a rolling-hour ceiling (PROVIDER_HOURLY_BUDGET) — the hard cap, seeded
|
|
430
|
+
// with calls already spent in the trailing hour;
|
|
431
|
+
// - a min-spacing token count (PROVIDER_MIN_REFRESH_SPACING_MS) — the smoother,
|
|
432
|
+
// which issues refreshes round-robin at a fixed rate instead of the
|
|
433
|
+
// synchronized burst-then-stall a plain rolling cap produces when every
|
|
434
|
+
// account falls due on the same tick.
|
|
435
|
+
// Both are per-provider and network-only; accounts are ordered stalest-first, so
|
|
436
|
+
// the scarce budget always serves the account most in need and none is starved.
|
|
437
|
+
const budgetSpent = providerRecentCalls(accounts, cache, now);
|
|
438
|
+
const lastCall = providerLastCall(accounts, cache);
|
|
439
|
+
const spacingTokens = new Map();
|
|
440
|
+
for (const [agent, last] of lastCall)
|
|
441
|
+
spacingTokens.set(agent, providerSpacingTokens(last, now));
|
|
442
|
+
const spacingUsed = new Map();
|
|
281
443
|
const updates = {};
|
|
282
444
|
for (const [index, account] of accounts.entries()) {
|
|
283
445
|
const entry = cache[account.usageKey] ?? null;
|
|
446
|
+
const network = agentUsesNetworkUsage(account.agentId);
|
|
284
447
|
// A penalized account/provider is off-limits — poking it re-arms the
|
|
285
448
|
// penalty (the whole reason usage-backoff exists).
|
|
286
449
|
if ((deps.backoffUntil(account.agentId, account.usageKey) ?? 0) > now) {
|
|
@@ -295,6 +458,34 @@ export async function runUsageRefresh(deps) {
|
|
|
295
458
|
result.skippedCap += 1;
|
|
296
459
|
continue;
|
|
297
460
|
}
|
|
461
|
+
// A live `agents run` already refreshed this account's usage row for free via
|
|
462
|
+
// the statusline ingest — re-derive headroom from that row and skip the API
|
|
463
|
+
// call. Network providers only: a local-log provider's cache is always its
|
|
464
|
+
// own last write, so this must not suppress its refresh (grok/codex).
|
|
465
|
+
if (network) {
|
|
466
|
+
const cached = deps.readCachedSnapshot?.(account.usageKey) ?? null;
|
|
467
|
+
const capturedAtMs = cached?.capturedAt?.getTime() ?? null;
|
|
468
|
+
if (cached && capturedAtMs !== null && now - capturedAtMs < STATUSLINE_FRESH_MS) {
|
|
469
|
+
updates[account.usageKey] = freshHeadroomEntry(entry, cached, now, capturedAtMs);
|
|
470
|
+
result.skippedFresh += 1;
|
|
471
|
+
continue;
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
// Global per-provider budget: cap aggregate endpoint traffic so it does not
|
|
475
|
+
// scale linearly with account count and trip the ~100/hr rate limit, and pace
|
|
476
|
+
// it smoothly. Non-network providers (local logs) have no endpoint to protect.
|
|
477
|
+
if (network) {
|
|
478
|
+
const overHourly = (budgetSpent.get(account.agentId) ?? 0) >= PROVIDER_HOURLY_BUDGET;
|
|
479
|
+
const overSpacing = (spacingUsed.get(account.agentId) ?? 0) >= (spacingTokens.get(account.agentId) ?? 0);
|
|
480
|
+
if (overHourly || overSpacing) {
|
|
481
|
+
// Leave the entry untouched so this still-due account competes again next
|
|
482
|
+
// tick, when budget/spacing frees — never starved (stalest-first serves it).
|
|
483
|
+
result.skippedBudget += 1;
|
|
484
|
+
continue;
|
|
485
|
+
}
|
|
486
|
+
budgetSpent.set(account.agentId, (budgetSpent.get(account.agentId) ?? 0) + 1);
|
|
487
|
+
spacingUsed.set(account.agentId, (spacingUsed.get(account.agentId) ?? 0) + 1);
|
|
488
|
+
}
|
|
298
489
|
try {
|
|
299
490
|
const usage = await account.fetch();
|
|
300
491
|
if (usage.snapshot) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@phnx-labs/agents-cli",
|
|
3
|
-
"version": "1.22.
|
|
3
|
+
"version": "1.22.65",
|
|
4
4
|
"description": "One CLI for all your AI coding agents - versions, config, cloud dispatch, sessions, and teams (now with first-class Grok Build CLI support)",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|