@phnx-labs/agents-cli 1.22.57 → 1.22.58
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +56 -0
- package/dist/bootstrap.js +8 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +11 -0
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions.js +1 -0
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/doctor-diff.js +77 -7
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +26 -133
- package/dist/lib/installations/versions.js +41 -204
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +23 -0
- package/dist/lib/self-update.js +50 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/db.d.ts +20 -1
- package/dist/lib/session/db.js +139 -9
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +15 -0
- package/dist/lib/traces/sync.js +104 -19
- package/dist/lib/traces/worker-template.js +154 -1
- package/package.json +1 -1
|
@@ -20,6 +20,7 @@ import * as fs from 'node:fs';
|
|
|
20
20
|
import * as path from 'node:path';
|
|
21
21
|
import * as crypto from 'node:crypto';
|
|
22
22
|
import * as yaml from 'yaml';
|
|
23
|
+
import chalk from 'chalk';
|
|
23
24
|
import { atomicWriteFileSync } from './fs-atomic.js';
|
|
24
25
|
import { getUserAgentsDir, readMeta, updateMeta } from './state.js';
|
|
25
26
|
import { deleteKeychainToken, getKeychainToken, hasKeychainToken } from './secrets/index.js';
|
|
@@ -307,18 +308,19 @@ export function accountBindings(accountId, meta) {
|
|
|
307
308
|
/** Explicit selection wins over a configured per-harness default. */
|
|
308
309
|
export function resolveAccountSelection(explicit, agent, meta, opts = {}) {
|
|
309
310
|
if (explicit)
|
|
310
|
-
return explicit;
|
|
311
|
+
return { id: explicit, source: 'explicit' };
|
|
311
312
|
// This box's device-doc bindings win over the fleet-shared central bindings
|
|
312
313
|
// (PHNX-3315), so a per-box account attachment resolves without touching the
|
|
313
314
|
// shared file. Defaults are genuinely fleet-shared and stay central.
|
|
314
315
|
const bindings = { ...meta.accounts?.bindings, ...meta.deviceAccounts?.bindings };
|
|
315
316
|
const bound = opts.target ? bindings[opts.target] : undefined;
|
|
316
317
|
if (bound)
|
|
317
|
-
return bound;
|
|
318
|
+
return { id: bound, source: 'binding' };
|
|
318
319
|
const deviceScoped = bindings[agent];
|
|
319
320
|
if (deviceScoped)
|
|
320
|
-
return deviceScoped;
|
|
321
|
-
|
|
321
|
+
return { id: deviceScoped, source: 'binding' };
|
|
322
|
+
const defaulted = opts.useDefault === false ? undefined : meta.accounts?.defaults?.[agent];
|
|
323
|
+
return defaulted ? { id: defaulted, source: 'default' } : undefined;
|
|
322
324
|
}
|
|
323
325
|
function profileConsumers(name, base) {
|
|
324
326
|
const dir = path.join(base, 'profiles');
|
|
@@ -371,19 +373,30 @@ export function renameAccount(oldName, newName, base = getUserAgentsDir()) {
|
|
|
371
373
|
const rows = nativeRowsForNameOrId(meta, oldName);
|
|
372
374
|
if (rows.length) {
|
|
373
375
|
assertUniqueUnifiedName(newName, meta, doc, new Set(rows.map(account => account.id)));
|
|
374
|
-
// Sweep every row for the identity (PHNX-3206) in its owning store (PHNX-3315)
|
|
376
|
+
// Sweep every row for the identity (PHNX-3206) in its owning store (PHNX-3315)
|
|
377
|
+
// and any per-harness default that points to the old name or row ids.
|
|
375
378
|
const rowScope = rows[0].scope;
|
|
379
|
+
const idsToRename = new Set([oldName, ...rows.map(row => row.id)]);
|
|
376
380
|
updateMeta(current => {
|
|
381
|
+
const defaults = { ...current.accounts?.defaults };
|
|
382
|
+
for (const [agent, value] of Object.entries(defaults)) {
|
|
383
|
+
if (idsToRename.has(value))
|
|
384
|
+
defaults[agent] = newName;
|
|
385
|
+
}
|
|
386
|
+
const next = { ...current, accounts: { ...current.accounts, defaults } };
|
|
377
387
|
if (rowScope === 'device') {
|
|
378
388
|
const native = { ...current.deviceAccounts?.native };
|
|
379
389
|
for (const row of rows)
|
|
380
390
|
native[row.id] = { ...native[row.id], name: newName };
|
|
381
|
-
|
|
391
|
+
next.deviceAccounts = { ...current.deviceAccounts, native };
|
|
382
392
|
}
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
393
|
+
else {
|
|
394
|
+
const native = { ...current.accounts?.native };
|
|
395
|
+
for (const row of rows)
|
|
396
|
+
native[row.id] = { ...native[row.id], name: newName };
|
|
397
|
+
next.accounts = { ...next.accounts, native };
|
|
398
|
+
}
|
|
399
|
+
return next;
|
|
387
400
|
});
|
|
388
401
|
return;
|
|
389
402
|
}
|
|
@@ -391,6 +404,15 @@ export function renameAccount(oldName, newName, base = getUserAgentsDir()) {
|
|
|
391
404
|
if (!account)
|
|
392
405
|
throw new Error(`Unknown account '${oldName}'.`);
|
|
393
406
|
assertUniqueUnifiedName(newName, meta, doc);
|
|
407
|
+
const idsToRename = new Set([account.name, account.id]);
|
|
408
|
+
updateMeta(current => {
|
|
409
|
+
const defaults = { ...current.accounts?.defaults };
|
|
410
|
+
for (const [agent, value] of Object.entries(defaults)) {
|
|
411
|
+
if (idsToRename.has(value))
|
|
412
|
+
defaults[agent] = newName;
|
|
413
|
+
}
|
|
414
|
+
return { ...current, accounts: { ...current.accounts, defaults } };
|
|
415
|
+
});
|
|
394
416
|
renameBundle(account.name, newName); // moves metadata + secret, preserves ACCOUNT_ID
|
|
395
417
|
renameProfileConsumers(account.name, newName, base);
|
|
396
418
|
}
|
|
@@ -422,7 +444,7 @@ export function removeAccount(name, base = getUserAgentsDir()) {
|
|
|
422
444
|
if (!account)
|
|
423
445
|
throw new Error(`Unknown account '${name}'.`);
|
|
424
446
|
const bindings = accountBindings(account.id, meta);
|
|
425
|
-
const defaults = Object.entries(meta.accounts?.defaults ?? {}).filter(([,
|
|
447
|
+
const defaults = Object.entries(meta.accounts?.defaults ?? {}).filter(([, value]) => value === account.id || value === account.name).map(([agent]) => agent);
|
|
426
448
|
if (bindings.length || defaults.length) {
|
|
427
449
|
const refs = [...bindings.map(target => `binding ${target}`), ...defaults.map(agent => `default ${agent}`)];
|
|
428
450
|
throw new Error(`Account '${account.name}' is still referenced by: ${refs.join(', ')}. Detach or clear those references before removing it.`);
|
|
@@ -505,9 +527,20 @@ export function resolveSpawnAccount(explicit, agent, version, meta, opts = {}) {
|
|
|
505
527
|
const selection = resolveAccountSelection(explicit, agent, meta, { useDefault: opts.useDefault, target });
|
|
506
528
|
if (!selection)
|
|
507
529
|
return null;
|
|
508
|
-
const unified = findUnifiedAccount(selection, meta);
|
|
509
|
-
if (!unified)
|
|
510
|
-
|
|
530
|
+
const unified = findUnifiedAccount(selection.id, meta);
|
|
531
|
+
if (!unified) {
|
|
532
|
+
// A stale per-harness default is a preference, not a hard requirement: the
|
|
533
|
+
// machine stays runnable by falling back to balanced rotation. Bindings and
|
|
534
|
+
// explicit --account are intentional, so they still fail loud.
|
|
535
|
+
if (selection.source === 'default') {
|
|
536
|
+
process.stderr.write(chalk.yellow(`[agents] default account '${selection.id}' for ${agent} no longer exists on this machine; falling back to balanced selection. Clear the stale default with: agents accounts clear-default ${agent}\n`));
|
|
537
|
+
return null;
|
|
538
|
+
}
|
|
539
|
+
const remedy = selection.source === 'binding'
|
|
540
|
+
? `Detach the stale binding with: agents accounts detach ${selection.id} ${target}`
|
|
541
|
+
: `Add an account named '${selection.id}' or remove the --account override.`;
|
|
542
|
+
throw new Error(`Unknown account '${selection.id}' for ${agent} harness. ${remedy}`);
|
|
543
|
+
}
|
|
511
544
|
if (unified.kind === 'native') {
|
|
512
545
|
if (unified.agent !== agent) {
|
|
513
546
|
throw new Error(`Account '${unified.name}' is a ${unified.agent} login and cannot authenticate the ${agent} harness.`);
|
|
@@ -10,15 +10,26 @@
|
|
|
10
10
|
* floor, so an account racing toward its 5h cap loses priority before it maxes.
|
|
11
11
|
*/
|
|
12
12
|
export declare const PROJECTION_HORIZON_MIN = 30;
|
|
13
|
+
/**
|
|
14
|
+
* The weight an account with NO usage snapshot draws. Absence of a usage signal
|
|
15
|
+
* is NOT capacity (specifications.md GWT-E5c, SING-1a): a null snapshot means
|
|
16
|
+
* "unverifiable", not "empty" — on a worker box every account reads null
|
|
17
|
+
* (setup-token lacks the `user:profile` scope, RUSH-2392), so scoring null as
|
|
18
|
+
* full capacity made the blind pool's top pick an account that was actually
|
|
19
|
+
* weekly-exhausted (PHNX-3392). Floored at 1, never 0: an all-unverified pool
|
|
20
|
+
* must still draw a pick rather than strand the launch.
|
|
21
|
+
*/
|
|
22
|
+
export declare const UNVERIFIED_WEIGHT = 1;
|
|
13
23
|
/**
|
|
14
24
|
* Weight one candidate by remaining routing capacity, deprioritized by how soon
|
|
15
25
|
* it is projected to cap. The base is weekly headroom (`max(1, 100 - used)`);
|
|
16
|
-
* an account with no live snapshot is
|
|
17
|
-
* is
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
26
|
+
* an account with no live snapshot is NOT full-capacity — absence of a usage
|
|
27
|
+
* signal is not evidence of headroom, so it draws `UNVERIFIED_WEIGHT` and any
|
|
28
|
+
* verified-healthy account outranks it (GWT-E5c). `minutesToLimit` (the daemon's
|
|
29
|
+
* burn-rate projection on the 5h session window) then scales that base: >=
|
|
30
|
+
* horizon (or unknown) keeps full weight, and closer-to-cap scales toward the
|
|
31
|
+
* floor of 1 — so a launch avoids an account projected to cap soon, not just a
|
|
32
|
+
* 100%-maxed one. Pure + exported so the deprioritization is unit-tested
|
|
33
|
+
* directly (a weighted-random draw is not).
|
|
23
34
|
*/
|
|
24
35
|
export declare function capacityWeight(usedPercent: number | null, minutesToLimit: number | null): number;
|
|
@@ -10,19 +10,30 @@
|
|
|
10
10
|
* floor, so an account racing toward its 5h cap loses priority before it maxes.
|
|
11
11
|
*/
|
|
12
12
|
export const PROJECTION_HORIZON_MIN = 30;
|
|
13
|
+
/**
|
|
14
|
+
* The weight an account with NO usage snapshot draws. Absence of a usage signal
|
|
15
|
+
* is NOT capacity (specifications.md GWT-E5c, SING-1a): a null snapshot means
|
|
16
|
+
* "unverifiable", not "empty" — on a worker box every account reads null
|
|
17
|
+
* (setup-token lacks the `user:profile` scope, RUSH-2392), so scoring null as
|
|
18
|
+
* full capacity made the blind pool's top pick an account that was actually
|
|
19
|
+
* weekly-exhausted (PHNX-3392). Floored at 1, never 0: an all-unverified pool
|
|
20
|
+
* must still draw a pick rather than strand the launch.
|
|
21
|
+
*/
|
|
22
|
+
export const UNVERIFIED_WEIGHT = 1;
|
|
13
23
|
/**
|
|
14
24
|
* Weight one candidate by remaining routing capacity, deprioritized by how soon
|
|
15
25
|
* it is projected to cap. The base is weekly headroom (`max(1, 100 - used)`);
|
|
16
|
-
* an account with no live snapshot is
|
|
17
|
-
* is
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
26
|
+
* an account with no live snapshot is NOT full-capacity — absence of a usage
|
|
27
|
+
* signal is not evidence of headroom, so it draws `UNVERIFIED_WEIGHT` and any
|
|
28
|
+
* verified-healthy account outranks it (GWT-E5c). `minutesToLimit` (the daemon's
|
|
29
|
+
* burn-rate projection on the 5h session window) then scales that base: >=
|
|
30
|
+
* horizon (or unknown) keeps full weight, and closer-to-cap scales toward the
|
|
31
|
+
* floor of 1 — so a launch avoids an account projected to cap soon, not just a
|
|
32
|
+
* 100%-maxed one. Pure + exported so the deprioritization is unit-tested
|
|
33
|
+
* directly (a weighted-random draw is not).
|
|
23
34
|
*/
|
|
24
35
|
export function capacityWeight(usedPercent, minutesToLimit) {
|
|
25
|
-
const base = usedPercent === null ?
|
|
36
|
+
const base = usedPercent === null ? UNVERIFIED_WEIGHT : Math.max(1, 100 - usedPercent);
|
|
26
37
|
if (minutesToLimit === null || !Number.isFinite(minutesToLimit))
|
|
27
38
|
return base;
|
|
28
39
|
const factor = Math.max(0, Math.min(1, minutesToLimit / PROJECTION_HORIZON_MIN));
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import { type DeviceProfile } from '../devices/registry.js';
|
|
2
2
|
import { type ConfiguredDeviceRole } from '../device-config.js';
|
|
3
|
-
import { type CachedUsageSnapshot } from './usage.js';
|
|
3
|
+
import { type CachedUsageSnapshot, type UsageSnapshot } from './usage.js';
|
|
4
4
|
/** How long a single peer push may take before it is abandoned for this tick. */
|
|
5
5
|
export declare const USAGE_PUSH_DEADLINE_MS = 20000;
|
|
6
|
+
export declare const USAGE_PULL_DEADLINE_MS = 20000;
|
|
6
7
|
/** The stdin envelope the `__usage-ingest` verb reads. `v` guards the shape. */
|
|
7
8
|
export interface UsageSyncPayload {
|
|
8
9
|
v: 1;
|
|
@@ -62,6 +63,33 @@ export interface UsageSyncDeps {
|
|
|
62
63
|
message?: string;
|
|
63
64
|
};
|
|
64
65
|
}
|
|
66
|
+
export interface UsagePullResult {
|
|
67
|
+
pulledFrom: string | null;
|
|
68
|
+
merged: number;
|
|
69
|
+
skipped: string | null;
|
|
70
|
+
error: string | null;
|
|
71
|
+
}
|
|
72
|
+
export interface UsagePullDeps {
|
|
73
|
+
selfRole?: () => ConfiguredDeviceRole | undefined;
|
|
74
|
+
listDevices?: () => DeviceProfile[];
|
|
75
|
+
listRoles?: () => Record<string, ConfiguredDeviceRole>;
|
|
76
|
+
isPinned?: (name: string) => boolean;
|
|
77
|
+
exportRows?: () => Record<string, CachedUsageSnapshot>;
|
|
78
|
+
readRow?: (usageKey: string) => Pick<UsageSnapshot, 'windows'> | null;
|
|
79
|
+
ingestRows?: (rows: Record<string, CachedUsageSnapshot>) => number;
|
|
80
|
+
/** Read the versioned payload from the primary. Default: ssh `__usage-export`. */
|
|
81
|
+
pull?: (device: DeviceProfile) => {
|
|
82
|
+
ok: boolean;
|
|
83
|
+
stdout?: string;
|
|
84
|
+
message?: string;
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Pull usage from the fleet's primary headed device when this worker's local
|
|
89
|
+
* cache is empty or contains an expired row. `personal` is the primary role;
|
|
90
|
+
* a `desktop` is used only when the fleet has no personal device.
|
|
91
|
+
*/
|
|
92
|
+
export declare function pullUsageFromPrimary(deps?: UsagePullDeps): UsagePullResult;
|
|
65
93
|
/**
|
|
66
94
|
* Push the local identity-keyed usage snapshot to every reachable, pinned,
|
|
67
95
|
* non-headed peer. A no-op on a non-headed box or when the local cache is empty.
|
|
@@ -25,16 +25,17 @@
|
|
|
25
25
|
* The planner is pure so tests cover every skip/push branch with no SSH.
|
|
26
26
|
*/
|
|
27
27
|
import { sshExec } from '../ssh-exec.js';
|
|
28
|
-
import { buildRemoteAgentsInvocation, buildWindowsStdinAgentsCommand, remoteShellFor } from '../hosts/remote-cmd.js';
|
|
28
|
+
import { buildRemoteAgentsInvocation, buildWindowsStdinAgentsCommand, remoteShellFor, stripClixml } from '../hosts/remote-cmd.js';
|
|
29
29
|
import { resolveRemoteOsSync } from '../hosts/remote-os.js';
|
|
30
30
|
import { loadDevicesSync } from '../devices/registry.js';
|
|
31
31
|
import { sshTargetFor } from '../devices/connect.js';
|
|
32
32
|
import { isHostPinned } from '../devices/known-hosts.js';
|
|
33
33
|
import { machineId, normalizeHost } from '../session/sync/config.js';
|
|
34
34
|
import { isHeadedDeviceRole, listConfiguredDeviceRoles, selfConfiguredDeviceRole, } from '../device-config.js';
|
|
35
|
-
import { exportClaudeUsageCacheRows } from './usage.js';
|
|
35
|
+
import { exportClaudeUsageCacheRows, ingestPeerClaudeUsageRows, readClaudeUsageCache, } from './usage.js';
|
|
36
36
|
/** How long a single peer push may take before it is abandoned for this tick. */
|
|
37
37
|
export const USAGE_PUSH_DEADLINE_MS = 20_000;
|
|
38
|
+
export const USAGE_PULL_DEADLINE_MS = 20_000;
|
|
38
39
|
/**
|
|
39
40
|
* Decide, per peer, whether to push the local usage snapshot. Pure.
|
|
40
41
|
*
|
|
@@ -84,6 +85,79 @@ function defaultUsagePush(device, payload) {
|
|
|
84
85
|
return { ok: false, message: res.stderr.trim() || `remote exit ${res.code}` };
|
|
85
86
|
return { ok: true };
|
|
86
87
|
}
|
|
88
|
+
/** Production pull: read the primary's cache through its hidden export verb. */
|
|
89
|
+
function defaultUsagePull(device) {
|
|
90
|
+
const target = sshTargetFor(device);
|
|
91
|
+
const os = resolveRemoteOsSync(device.name);
|
|
92
|
+
const remoteCmd = buildRemoteAgentsInvocation(['__usage-export'], undefined, os);
|
|
93
|
+
const res = sshExec(target, remoteCmd, { timeoutMs: USAGE_PULL_DEADLINE_MS });
|
|
94
|
+
if (res.timedOut)
|
|
95
|
+
return { ok: false, message: 'timed out' };
|
|
96
|
+
if (res.code !== 0)
|
|
97
|
+
return { ok: false, message: res.stderr.trim() || `remote exit ${res.code}` };
|
|
98
|
+
return { ok: true, stdout: res.stdout };
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Pull usage from the fleet's primary headed device when this worker's local
|
|
102
|
+
* cache is empty or contains an expired row. `personal` is the primary role;
|
|
103
|
+
* a `desktop` is used only when the fleet has no personal device.
|
|
104
|
+
*/
|
|
105
|
+
export function pullUsageFromPrimary(deps = {}) {
|
|
106
|
+
const result = { pulledFrom: null, merged: 0, skipped: null, error: null };
|
|
107
|
+
if ((deps.selfRole ?? selfConfiguredDeviceRole)() !== 'worker') {
|
|
108
|
+
result.skipped = 'this device is not a worker';
|
|
109
|
+
return result;
|
|
110
|
+
}
|
|
111
|
+
const localRows = (deps.exportRows ?? exportClaudeUsageCacheRows)();
|
|
112
|
+
const readRow = deps.readRow ?? readClaudeUsageCache;
|
|
113
|
+
const localEntries = Object.entries(localRows);
|
|
114
|
+
if (localEntries.length > 0 && localEntries.every(([key, row]) => {
|
|
115
|
+
const fresh = readRow(key);
|
|
116
|
+
return fresh !== null && fresh.windows.length === row.windows.length;
|
|
117
|
+
})) {
|
|
118
|
+
result.skipped = 'local usage cache is fresh';
|
|
119
|
+
return result;
|
|
120
|
+
}
|
|
121
|
+
const devices = deps.listDevices?.() ?? Object.values(loadDevicesSync());
|
|
122
|
+
const roles = deps.listRoles?.() ?? (deps.listDevices
|
|
123
|
+
? listConfiguredDeviceRoles(devices.map((device) => device.name))
|
|
124
|
+
: listConfiguredDeviceRoles());
|
|
125
|
+
const isPinned = deps.isPinned ?? isHostPinned;
|
|
126
|
+
const primary = devices
|
|
127
|
+
.filter((device) => roles[device.name] === 'personal')
|
|
128
|
+
.concat(devices.filter((device) => roles[device.name] === 'desktop'))
|
|
129
|
+
.find((device) => device.tailscale?.online !== false && isPinned(device.name));
|
|
130
|
+
if (!primary) {
|
|
131
|
+
result.error = 'no reachable, pinned personal or desktop device is configured as the usage primary';
|
|
132
|
+
return result;
|
|
133
|
+
}
|
|
134
|
+
const outcome = (deps.pull ?? defaultUsagePull)(primary);
|
|
135
|
+
if (!outcome.ok) {
|
|
136
|
+
result.error = outcome.message ?? 'pull failed';
|
|
137
|
+
return result;
|
|
138
|
+
}
|
|
139
|
+
let payload;
|
|
140
|
+
try {
|
|
141
|
+
// A Windows usage primary's `agents.ps1` shim prepends a PowerShell CLIXML
|
|
142
|
+
// progress banner to stdout, exactly as every other remote-JSON boundary
|
|
143
|
+
// strips (remote-cmd.ts:325). Without this, a headed Windows box (win-mini
|
|
144
|
+
// is a live fleet device) makes every pull fail as "malformed JSON", the
|
|
145
|
+
// worker's cache stays null, and the PHNX-3392 capacity floor silently
|
|
146
|
+
// becomes the only thing standing between a blind pool and an exhausted pick.
|
|
147
|
+
payload = JSON.parse(stripClixml(outcome.stdout ?? ''));
|
|
148
|
+
}
|
|
149
|
+
catch {
|
|
150
|
+
result.error = 'primary returned malformed JSON';
|
|
151
|
+
return result;
|
|
152
|
+
}
|
|
153
|
+
if (!payload || payload.v !== 1 || typeof payload.rows !== 'object' || payload.rows === null || Array.isArray(payload.rows)) {
|
|
154
|
+
result.error = 'primary returned an unrecognized usage-sync payload shape';
|
|
155
|
+
return result;
|
|
156
|
+
}
|
|
157
|
+
result.pulledFrom = primary.name;
|
|
158
|
+
result.merged = (deps.ingestRows ?? ingestPeerClaudeUsageRows)(payload.rows);
|
|
159
|
+
return result;
|
|
160
|
+
}
|
|
87
161
|
/**
|
|
88
162
|
* Push the local identity-keyed usage snapshot to every reachable, pinned,
|
|
89
163
|
* non-headed peer. A no-op on a non-headed box or when the local cache is empty.
|
|
@@ -284,6 +284,12 @@ const USAGE_BAR_LEN = 10;
|
|
|
284
284
|
const FULL = '\u2588';
|
|
285
285
|
const EMPTY = '\u2591';
|
|
286
286
|
const PARTIAL_BLOCKS = ['', '\u258F', '\u258E', '\u258D', '\u258C', '\u258B', '\u258A', '\u2589'];
|
|
287
|
+
// A window we EXPECTED but have no reading for \u2014 e.g. Claude's 5h "session"
|
|
288
|
+
// window when the account has no usage in the current rolling window, so the
|
|
289
|
+
// usage API returns five_hour.utilization = null and no session bar is written.
|
|
290
|
+
// It must read as neither 0% (EMPTY '\u2591') nor 100% (FULL '\u2588') \u2014 a full block was
|
|
291
|
+
// alarming and looked maxed-out \u2014 so use a dashed row that says "no data".
|
|
292
|
+
const NO_DATA = '\u2504';
|
|
287
293
|
/** Construct the benign no-local-log result without overloading `error`. */
|
|
288
294
|
export function usageNoRecentUsageInfo() {
|
|
289
295
|
return { snapshot: null, error: null, [USAGE_BENIGN_STATE]: 'no-recent-usage' };
|
|
@@ -590,7 +596,7 @@ export function formatUsageSummary(plan, snapshot, planWidth = 3, opts) {
|
|
|
590
596
|
: selected.map((window) => ({ window, shortLabel: window.shortLabel }));
|
|
591
597
|
const windowParts = windowsToRender.map(({ window, shortLabel }, index) => {
|
|
592
598
|
if (!window) {
|
|
593
|
-
const missing = chalk.
|
|
599
|
+
const missing = chalk.dim(`${shortLabel}: ${NO_DATA.repeat(COMPACT_BAR_LEN)} unavailable`);
|
|
594
600
|
return index < windowsToRender.length - 1 ? padToWidth(missing, 20) : missing;
|
|
595
601
|
}
|
|
596
602
|
const bar = renderCompactUsageBar(window.usedPercent);
|
package/dist/lib/auth-mint.d.ts
CHANGED
|
@@ -143,7 +143,17 @@ export interface MintAndSeedResult {
|
|
|
143
143
|
*/
|
|
144
144
|
export declare function mintAndSeed(input: MintAndSeedInput): Promise<MintAndSeedResult>;
|
|
145
145
|
export declare function resolveSyncTargets(fleet: boolean, devices: string[]): Promise<string[]>;
|
|
146
|
-
/**
|
|
146
|
+
/**
|
|
147
|
+
* True when a Claude setup-token is already seeded on this box (setup status).
|
|
148
|
+
*
|
|
149
|
+
* Read-only status probe (`agents setup`, `agents doctor`) — the reserved
|
|
150
|
+
* auth-bundle check below MUST NOT crash the whole command because the
|
|
151
|
+
* keychain could not be reached (e.g. the macOS Keychain helper source is
|
|
152
|
+
* unavailable — PHNX-3385); `bundleExists()`/`hasKeychainToken()` document
|
|
153
|
+
* themselves as failing loud, but that contract is for destructive-write
|
|
154
|
+
* guards, not a diagnostic. An unreachable keychain here just means "cannot
|
|
155
|
+
* confirm the reserved bundle," reported honestly rather than propagated.
|
|
156
|
+
*/
|
|
147
157
|
export declare function hasMintedSetupToken(): {
|
|
148
158
|
ready: boolean;
|
|
149
159
|
detail: string;
|
package/dist/lib/auth-mint.js
CHANGED
|
@@ -414,18 +414,33 @@ async function syncMintedBundles(accountName, device) {
|
|
|
414
414
|
return { device, ok: false, message: err instanceof Error ? err.message : String(err) };
|
|
415
415
|
}
|
|
416
416
|
}
|
|
417
|
-
/**
|
|
417
|
+
/**
|
|
418
|
+
* True when a Claude setup-token is already seeded on this box (setup status).
|
|
419
|
+
*
|
|
420
|
+
* Read-only status probe (`agents setup`, `agents doctor`) — the reserved
|
|
421
|
+
* auth-bundle check below MUST NOT crash the whole command because the
|
|
422
|
+
* keychain could not be reached (e.g. the macOS Keychain helper source is
|
|
423
|
+
* unavailable — PHNX-3385); `bundleExists()`/`hasKeychainToken()` document
|
|
424
|
+
* themselves as failing loud, but that contract is for destructive-write
|
|
425
|
+
* guards, not a diagnostic. An unreachable keychain here just means "cannot
|
|
426
|
+
* confirm the reserved bundle," reported honestly rather than propagated.
|
|
427
|
+
*/
|
|
418
428
|
export function hasMintedSetupToken() {
|
|
419
429
|
const records = Object.values(readAccountRegistry().accounts);
|
|
420
430
|
const setup = records.filter((a) => a.auth === 'setup-token' && a.provider === 'anthropic');
|
|
421
431
|
if (setup.length) {
|
|
422
432
|
return { ready: true, detail: `${setup.length} Claude setup-token account${setup.length === 1 ? '' : 's'}` };
|
|
423
433
|
}
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
434
|
+
try {
|
|
435
|
+
if (bundleExists(AUTH_BUNDLE) && bundleBackend(AUTH_BUNDLE) === 'file') {
|
|
436
|
+
const bundle = readBundle(AUTH_BUNDLE);
|
|
437
|
+
const keys = Object.keys(bundle.vars).filter((k) => k.startsWith('CLAUDE_CODE_OAUTH_TOKEN_'));
|
|
438
|
+
if (keys.length)
|
|
439
|
+
return { ready: true, detail: `reserved auth bundle (${keys.length} account key${keys.length === 1 ? '' : 's'})` };
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
catch (err) {
|
|
443
|
+
return { ready: false, detail: `could not check the reserved auth bundle: ${err.message}` };
|
|
429
444
|
}
|
|
430
445
|
return { ready: false, detail: 'no Claude setup-token minted — agents accounts mint claude' };
|
|
431
446
|
}
|
|
@@ -12,6 +12,14 @@ export declare function getSocketPath(): string;
|
|
|
12
12
|
/** Is the daemon reachable? A real connect probe on every platform — a socket
|
|
13
13
|
* file existing on disk is not proof a daemon is listening on it. */
|
|
14
14
|
export declare function isDaemonReachable(): Promise<boolean>;
|
|
15
|
+
/**
|
|
16
|
+
* Is a *reachable* daemon actually responsive, or is its event loop wedged
|
|
17
|
+
* (PHNX-3411)? Retries {@link probeDaemonResponsive} up to
|
|
18
|
+
* {@link RESPONSIVENESS_PROBE_ATTEMPTS} times so a single transient miss (a GC
|
|
19
|
+
* pause) never condemns a healthy daemon; returns true as soon as any attempt
|
|
20
|
+
* gets a reply, and false only when every attempt fails.
|
|
21
|
+
*/
|
|
22
|
+
export declare function isDaemonResponsive(endpoint?: string, attempts?: number): Promise<boolean>;
|
|
15
23
|
/**
|
|
16
24
|
* Wait until the browser daemon is genuinely reachable, or throw.
|
|
17
25
|
*
|
package/dist/lib/browser/ipc.js
CHANGED
|
@@ -92,6 +92,31 @@ const SOCKET_WAIT_TIMEOUT_MS = 15_000;
|
|
|
92
92
|
* doesn't race a restart it happened to probe mid-flight.
|
|
93
93
|
*/
|
|
94
94
|
const SOCKET_WAIT_STABLE_PROBES = 2;
|
|
95
|
+
/**
|
|
96
|
+
* How long a single {@link probeDaemonResponsive} attempt waits for the daemon to
|
|
97
|
+
* answer a trivial `version` request before giving up (PHNX-3411).
|
|
98
|
+
*
|
|
99
|
+
* A unix-socket `connect` succeeds at the KERNEL level the moment the connection
|
|
100
|
+
* is queued in the listen backlog — it does NOT require the server's event loop
|
|
101
|
+
* to run. So a daemon whose event loop is blocked (e.g. a long synchronous burst
|
|
102
|
+
* on the shared loop) still passes {@link isDaemonReachable}: the socket accepts,
|
|
103
|
+
* but the request is never serviced. The live symptom on zion was exactly this —
|
|
104
|
+
* `browser.sock` accepted every connection while the daemon re-indexed sessions,
|
|
105
|
+
* then never replied, so cross-device browser drives hung or surfaced a confusing
|
|
106
|
+
* bare socket timeout. Requiring an actual reply is the only probe that tells a
|
|
107
|
+
* healthy daemon apart from a wedged one.
|
|
108
|
+
*/
|
|
109
|
+
const RESPONSIVENESS_PROBE_TIMEOUT_MS = 1_500;
|
|
110
|
+
/**
|
|
111
|
+
* Consecutive missed {@link probeDaemonResponsive} attempts before a *reachable*
|
|
112
|
+
* daemon is declared wedged. One missed reply can be a transient GC pause on an
|
|
113
|
+
* otherwise-healthy loop, so a single failure never condemns the daemon; a daemon
|
|
114
|
+
* that cannot answer a trivial `version` request across this whole window has a
|
|
115
|
+
* genuinely blocked event loop. The added latency (attempts × timeout) is paid
|
|
116
|
+
* ONLY on the wedged path — where the alternative was an indefinite hang — while
|
|
117
|
+
* a healthy daemon answers the first attempt in ~1ms.
|
|
118
|
+
*/
|
|
119
|
+
const RESPONSIVENESS_PROBE_ATTEMPTS = 3;
|
|
95
120
|
export class BrowserDaemonNotRunningError extends Error {
|
|
96
121
|
constructor() {
|
|
97
122
|
super(formatBrowserDaemonNotRunningError());
|
|
@@ -140,6 +165,57 @@ function probeDaemon(endpoint, timeoutMs = 500) {
|
|
|
140
165
|
export async function isDaemonReachable() {
|
|
141
166
|
return probeDaemon(getIpcEndpoint());
|
|
142
167
|
}
|
|
168
|
+
/**
|
|
169
|
+
* Can the daemon actually REPLY right now? Opens a connection and requires a
|
|
170
|
+
* parseable response to a `version` request within `timeoutMs`. Unlike
|
|
171
|
+
* {@link probeDaemon} (which resolves on the kernel-level `connect`), this only
|
|
172
|
+
* succeeds when the daemon's event loop is running and services the request — so
|
|
173
|
+
* it is the one probe that distinguishes a healthy daemon from a wedged one whose
|
|
174
|
+
* loop is blocked (PHNX-3411). Resolves false on connect error, timeout, an early
|
|
175
|
+
* close, or an unparseable reply. `version` is the right probe: its handler is
|
|
176
|
+
* trivial and synchronous (never touches the browser), so a slow reply means the
|
|
177
|
+
* loop is blocked, not that a real action is in flight.
|
|
178
|
+
*/
|
|
179
|
+
function probeDaemonResponsive(endpoint, timeoutMs = RESPONSIVENESS_PROBE_TIMEOUT_MS) {
|
|
180
|
+
return new Promise((resolve) => {
|
|
181
|
+
const sock = net.createConnection(endpoint);
|
|
182
|
+
let buffer = '';
|
|
183
|
+
let done = false;
|
|
184
|
+
const finish = (ok) => { if (done)
|
|
185
|
+
return; done = true; clearTimeout(timer); sock.destroy(); resolve(ok); };
|
|
186
|
+
const timer = setTimeout(() => finish(false), timeoutMs);
|
|
187
|
+
sock.on('connect', () => { sock.write(JSON.stringify({ action: 'version' }) + '\n'); });
|
|
188
|
+
sock.on('data', (data) => {
|
|
189
|
+
buffer += data.toString();
|
|
190
|
+
const idx = buffer.indexOf('\n');
|
|
191
|
+
if (idx === -1)
|
|
192
|
+
return;
|
|
193
|
+
try {
|
|
194
|
+
const resp = JSON.parse(buffer.slice(0, idx));
|
|
195
|
+
finish(resp.ok === true);
|
|
196
|
+
}
|
|
197
|
+
catch {
|
|
198
|
+
finish(false);
|
|
199
|
+
}
|
|
200
|
+
});
|
|
201
|
+
sock.on('error', () => finish(false));
|
|
202
|
+
sock.on('close', () => finish(false));
|
|
203
|
+
});
|
|
204
|
+
}
|
|
205
|
+
/**
|
|
206
|
+
* Is a *reachable* daemon actually responsive, or is its event loop wedged
|
|
207
|
+
* (PHNX-3411)? Retries {@link probeDaemonResponsive} up to
|
|
208
|
+
* {@link RESPONSIVENESS_PROBE_ATTEMPTS} times so a single transient miss (a GC
|
|
209
|
+
* pause) never condemns a healthy daemon; returns true as soon as any attempt
|
|
210
|
+
* gets a reply, and false only when every attempt fails.
|
|
211
|
+
*/
|
|
212
|
+
export async function isDaemonResponsive(endpoint = getIpcEndpoint(), attempts = RESPONSIVENESS_PROBE_ATTEMPTS) {
|
|
213
|
+
for (let i = 0; i < attempts; i++) {
|
|
214
|
+
if (await probeDaemonResponsive(endpoint))
|
|
215
|
+
return true;
|
|
216
|
+
}
|
|
217
|
+
return false;
|
|
218
|
+
}
|
|
143
219
|
/**
|
|
144
220
|
* Wait until the browser daemon is genuinely reachable, or throw.
|
|
145
221
|
*
|
|
@@ -1087,6 +1163,17 @@ async function prepareIPC(action, opts) {
|
|
|
1087
1163
|
}
|
|
1088
1164
|
await new Promise((r) => setTimeout(r, 300));
|
|
1089
1165
|
}
|
|
1166
|
+
// The socket accepts connections — but a unix-socket `connect` succeeds at the
|
|
1167
|
+
// kernel level even when the daemon's event loop is blocked and never services
|
|
1168
|
+
// the request (PHNX-3411). Require a real reply before treating the daemon as
|
|
1169
|
+
// usable, so a wedged daemon fails LOUD with an accurate, actionable error
|
|
1170
|
+
// instead of the confusing bare socket timeout / indefinite hang the browser
|
|
1171
|
+
// verb would otherwise hit (including inside reconcileDaemonVersion's own
|
|
1172
|
+
// version probe, which has no response timeout). Skipped for callers that opt
|
|
1173
|
+
// out of auto-start — they want the clean BrowserDaemonNotRunningError instead.
|
|
1174
|
+
if (autoStartDaemon && !(await isDaemonResponsive(getIpcEndpoint()))) {
|
|
1175
|
+
throw new Error(actionable('Browser daemon is running but unresponsive — its event loop is blocked, so it accepts the connection but never replies.', `Endpoint: ${getIpcEndpoint()}`, `Log: ${getDaemonLogPath()}`, 'Next: agents browser stop --daemon (resets the wedged daemon; the next browser command restarts it)'));
|
|
1176
|
+
}
|
|
1090
1177
|
// Before serving a real request, make sure the daemon isn't running stale
|
|
1091
1178
|
// code. Skips the internal `version` probe (avoids recursion) and callers
|
|
1092
1179
|
// that opt out of auto-start. No-ops once reconciled or when versions match.
|
|
@@ -534,6 +534,25 @@ export declare class BrowserService {
|
|
|
534
534
|
* tasks.json entries. Used before identity-based resolution so a daemon
|
|
535
535
|
* restart does not make the caller's tasks invisible.
|
|
536
536
|
*/
|
|
537
|
+
/**
|
|
538
|
+
* Register a freshly-rehydrated connection, unless a concurrent rehydrate won
|
|
539
|
+
* the race and already registered one for this key. `attachRunningProfile`
|
|
540
|
+
* awaits an ssh-tunnel spawn / CDP connect, so two commands landing right
|
|
541
|
+
* after a daemon restart can each build a connection for the same key; the
|
|
542
|
+
* loser must be released instead of silently overwriting the winner (whose
|
|
543
|
+
* `cleanup` would then never run).
|
|
544
|
+
*
|
|
545
|
+
* The subtlety is the SSH tunnel: when the loser's `connectSSH` landed after
|
|
546
|
+
* the winner's tunnel had already bound the local port, `isOwnTunnel` makes it
|
|
547
|
+
* **reuse** that same OS process (`drivers/ssh.ts`) — so both connections carry
|
|
548
|
+
* the same `pid`/`port`. Calling the loser's `cleanup()` then does
|
|
549
|
+
* `tunnel.kill()` on the SHARED process and breaks the winner's live CDP. So
|
|
550
|
+
* when the loser shares the winner's tunnel we close ONLY the loser's own CDP
|
|
551
|
+
* socket and leave the tunnel to the winner; only a loser that spawned its own
|
|
552
|
+
* distinct tunnel gets the full `cleanup()` (else it would leak). Returns the
|
|
553
|
+
* winning connection to use.
|
|
554
|
+
*/
|
|
555
|
+
private registerRehydratedConnection;
|
|
537
556
|
private rehydrateAllFromDisk;
|
|
538
557
|
/**
|
|
539
558
|
* On a RAM miss, scan profile runtime dirs for a tasks.json entry matching
|