@phnx-labs/agents-cli 1.21.3 → 1.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +136 -0
- package/README.md +32 -3
- package/dist/bin/agents +0 -0
- package/dist/commands/computer-actions.js +1 -0
- package/dist/commands/doctor.js +34 -2
- package/dist/commands/exec.d.ts +27 -0
- package/dist/commands/exec.js +123 -6
- package/dist/commands/models.js +36 -1
- package/dist/commands/projects.js +120 -62
- package/dist/commands/sessions-backfill.d.ts +32 -0
- package/dist/commands/sessions-backfill.js +186 -0
- package/dist/commands/sessions.d.ts +17 -1
- package/dist/commands/sessions.js +317 -18
- package/dist/commands/teams.js +1 -1
- package/dist/commands/worktree.d.ts +3 -3
- package/dist/commands/worktree.js +35 -4
- package/dist/lib/app-bundle-install.d.ts +17 -0
- package/dist/lib/app-bundle-install.js +94 -0
- package/dist/lib/daemon.d.ts +5 -1
- package/dist/lib/daemon.js +63 -14
- package/dist/lib/devices/doctor-findings.js +12 -4
- package/dist/lib/devices/doctor-overview-cache.d.ts +45 -0
- package/dist/lib/devices/doctor-overview-cache.js +168 -0
- package/dist/lib/devices/fleet.js +7 -2
- package/dist/lib/devices/resolve-target.d.ts +6 -0
- package/dist/lib/devices/resolve-target.js +9 -3
- package/dist/lib/devices/self-host.d.ts +9 -0
- package/dist/lib/devices/self-host.js +61 -0
- package/dist/lib/exec.js +39 -8
- package/dist/lib/hosts/dispatch.d.ts +12 -0
- package/dist/lib/hosts/dispatch.js +23 -6
- package/dist/lib/hosts/passthrough.js +8 -6
- package/dist/lib/hosts/reconnect.d.ts +38 -0
- package/dist/lib/hosts/reconnect.js +85 -4
- package/dist/lib/hosts/run-target.js +14 -2
- package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
- package/dist/lib/menubar/install-menubar.js +27 -23
- package/dist/lib/model-tiers.d.ts +54 -0
- package/dist/lib/model-tiers.js +229 -0
- package/dist/lib/models.d.ts +3 -0
- package/dist/lib/models.js +44 -7
- package/dist/lib/pricing/prices.json +16 -1
- package/dist/lib/project-focus.d.ts +42 -0
- package/dist/lib/project-focus.js +80 -0
- package/dist/lib/project-schedule.d.ts +75 -0
- package/dist/lib/project-schedule.js +110 -0
- package/dist/lib/project-status.d.ts +7 -0
- package/dist/lib/project-status.js +9 -0
- package/dist/lib/projects.d.ts +11 -2
- package/dist/lib/projects.js +57 -9
- package/dist/lib/redact.d.ts +2 -0
- package/dist/lib/redact.js +22 -0
- package/dist/lib/remote-agents-json.d.ts +2 -0
- package/dist/lib/remote-agents-json.js +3 -3
- package/dist/lib/rotate.d.ts +84 -1
- package/dist/lib/rotate.js +155 -5
- package/dist/lib/runner.d.ts +4 -2
- package/dist/lib/runner.js +13 -4
- package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
- package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
- package/dist/lib/secrets/install-helper.js +28 -31
- package/dist/lib/session/bash-command.js +60 -9
- package/dist/lib/session/db.d.ts +7 -1
- package/dist/lib/session/db.js +301 -32
- package/dist/lib/session/discover.d.ts +40 -7
- package/dist/lib/session/discover.js +144 -83
- package/dist/lib/session/parse.d.ts +8 -1
- package/dist/lib/session/parse.js +83 -32
- package/dist/lib/session/remote-list.d.ts +71 -0
- package/dist/lib/session/remote-list.js +410 -2
- package/dist/lib/session/shell-programs.d.ts +15 -0
- package/dist/lib/session/shell-programs.js +359 -0
- package/dist/lib/session/tool-calls.d.ts +88 -0
- package/dist/lib/session/tool-calls.js +612 -0
- package/dist/lib/session/tool-index.d.ts +100 -0
- package/dist/lib/session/tool-index.js +773 -0
- package/dist/lib/session/tool-store.d.ts +15 -0
- package/dist/lib/session/tool-store.js +198 -0
- package/dist/lib/session/types.d.ts +7 -0
- package/dist/lib/state.d.ts +10 -1
- package/dist/lib/state.js +11 -2
- package/dist/lib/teams/remoteWorktree.d.ts +3 -4
- package/dist/lib/teams/remoteWorktree.js +3 -4
- package/dist/lib/teams/worktree.d.ts +11 -1
- package/dist/lib/teams/worktree.js +42 -4
- package/dist/lib/types.d.ts +17 -0
- package/dist/lib/types.js +17 -0
- package/dist/lib/versions.js +69 -22
- package/package.json +3 -1
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Atomic, serialized install of a macOS `.app` bundle to a stable user path.
|
|
3
|
+
*
|
|
4
|
+
* Shared by the two helpers agents-cli installs on darwin — the secrets keychain
|
|
5
|
+
* helper (`lib/secrets/install-helper.ts`) and the menu-bar helper
|
|
6
|
+
* (`lib/menubar/install-menubar.ts`) — both of which are (re)installed on the hot
|
|
7
|
+
* path of ordinary `agents` invocations. Both previously did a non-atomic
|
|
8
|
+
* `rm -rf dest` + `cp -R src dest` straight onto the live bundle. That copy takes
|
|
9
|
+
* long enough that a concurrent reader (Gatekeeper, or an exec of the bundle) sees
|
|
10
|
+
* a half-written `.app` — a truncated Mach-O / mismatched `_CodeSignature` hash —
|
|
11
|
+
* which macOS reports as **"is damaged and can't be opened."** On a busy box dozens
|
|
12
|
+
* of concurrent invocations raced the same path, so the dialog fired intermittently.
|
|
13
|
+
*
|
|
14
|
+
* {@link copyAppBundle} stages the copy in a sibling directory and swaps it into
|
|
15
|
+
* place with renames, so a reader sees either the old or the new complete bundle,
|
|
16
|
+
* never a half-written one — the only moment `dest` is briefly absent is the
|
|
17
|
+
* sub-millisecond gap between the two renames (vs the seconds-long `cp`), and a
|
|
18
|
+
* failed copy never touches the live bundle. {@link withInstallLock} serializes concurrent
|
|
19
|
+
* installers (via the shared `withFileLock`) so a burst of invocations installs
|
|
20
|
+
* once instead of stampeding.
|
|
21
|
+
*/
|
|
22
|
+
import * as fs from 'fs';
|
|
23
|
+
import * as path from 'path';
|
|
24
|
+
import { spawnSync } from 'child_process';
|
|
25
|
+
import { withFileLock, ensureLockTarget } from './fs-atomic.js';
|
|
26
|
+
// A helper `cp -R` under load can take a few seconds; the lock must outlast it so
|
|
27
|
+
// a peer never treats a live installer as crashed and interleaves a second swap.
|
|
28
|
+
// (fs-atomic's 5s default is tuned for sub-second read-modify-writes.)
|
|
29
|
+
const INSTALL_LOCK_STALE_MS = 60_000;
|
|
30
|
+
const INSTALL_LOCK_ACQUIRE_TIMEOUT_MS = 60_000;
|
|
31
|
+
/**
|
|
32
|
+
* Copy an `.app` bundle to `dest` atomically. Stages into a sibling dir, then
|
|
33
|
+
* swaps with renames — the window where `dest` is absent shrinks from the
|
|
34
|
+
* seconds-long `cp` to a single microsecond rename, and a failed copy leaves the
|
|
35
|
+
* existing bundle untouched. Serialize concurrent callers with {@link withInstallLock}
|
|
36
|
+
* so the two-step swap never races another swap.
|
|
37
|
+
*/
|
|
38
|
+
export function copyAppBundle(src, dest, io = {}) {
|
|
39
|
+
const rename = io.renameSync ?? fs.renameSync;
|
|
40
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
41
|
+
const staging = `${dest}.installing.${process.pid}`;
|
|
42
|
+
const backup = `${dest}.replaced.${process.pid}`;
|
|
43
|
+
fs.rmSync(staging, { recursive: true, force: true });
|
|
44
|
+
fs.rmSync(backup, { recursive: true, force: true });
|
|
45
|
+
// `cp -R` preserves the bundle's signature, symlinks, and resource forks;
|
|
46
|
+
// `fs.cpSync({recursive:true})` has historically mishandled xattrs on `.app`
|
|
47
|
+
// bundles, breaking codesign.
|
|
48
|
+
const r = spawnSync('cp', ['-R', src, staging], { stdio: ['ignore', 'pipe', 'pipe'], encoding: 'utf-8' });
|
|
49
|
+
if (r.status !== 0) {
|
|
50
|
+
fs.rmSync(staging, { recursive: true, force: true });
|
|
51
|
+
const msg = (r.stderr || r.stdout || '').toString().trim();
|
|
52
|
+
throw new Error(`Failed to copy ${src} -> ${staging}: ${msg || 'unknown error'}`);
|
|
53
|
+
}
|
|
54
|
+
// rename(2) cannot replace a non-empty directory, so move the current bundle
|
|
55
|
+
// aside, then move staging into place. If the second rename fails, restore the
|
|
56
|
+
// backup so `dest` is never left missing.
|
|
57
|
+
try {
|
|
58
|
+
if (fs.existsSync(dest))
|
|
59
|
+
rename(dest, backup);
|
|
60
|
+
rename(staging, dest);
|
|
61
|
+
}
|
|
62
|
+
catch (err) {
|
|
63
|
+
if (!fs.existsSync(dest) && fs.existsSync(backup)) {
|
|
64
|
+
try {
|
|
65
|
+
rename(backup, dest);
|
|
66
|
+
}
|
|
67
|
+
catch {
|
|
68
|
+
/* best-effort restore */
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
fs.rmSync(staging, { recursive: true, force: true });
|
|
72
|
+
throw new Error(`Failed to install ${src} -> ${dest}: ${err.message}`);
|
|
73
|
+
}
|
|
74
|
+
fs.rmSync(backup, { recursive: true, force: true });
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Serialize installs across the many concurrent `agents` invocations that pass
|
|
78
|
+
* through the helper-install path, so a burst copies once instead of stampeding
|
|
79
|
+
* the atomic swap. Locks a sentinel file beside the bundle (the bundle itself may
|
|
80
|
+
* not exist yet on first install) via the shared {@link withFileLock}.
|
|
81
|
+
*/
|
|
82
|
+
export function withInstallLock(dest, fn) {
|
|
83
|
+
const lockTarget = `${dest}.install-lock`;
|
|
84
|
+
ensureLockTarget(lockTarget);
|
|
85
|
+
// Pass proper-lockfile's `heartbeat` straight through: the install body is a
|
|
86
|
+
// fully SYNCHRONOUS chain of blocking `spawnSync`s (`cp -R`, then codesign /
|
|
87
|
+
// spctl), so the event loop never turns and proper-lockfile's own async mtime
|
|
88
|
+
// refresh can't fire. Callers invoke heartbeat() between those steps to keep a
|
|
89
|
+
// long hold from ageing past staleMs and being broken by a contending peer.
|
|
90
|
+
withFileLock(lockTarget, (heartbeat) => fn(heartbeat), {
|
|
91
|
+
staleMs: INSTALL_LOCK_STALE_MS,
|
|
92
|
+
acquireTimeoutMs: INSTALL_LOCK_ACQUIRE_TIMEOUT_MS,
|
|
93
|
+
});
|
|
94
|
+
}
|
package/dist/lib/daemon.d.ts
CHANGED
|
@@ -43,7 +43,11 @@ export declare function writeHeartbeat(pid?: number): void;
|
|
|
43
43
|
export declare function readHeartbeat(): DaemonHeartbeat | null;
|
|
44
44
|
export declare function removeHeartbeat(): void;
|
|
45
45
|
export declare function isDaemonWedged(): boolean;
|
|
46
|
-
/**
|
|
46
|
+
/**
|
|
47
|
+
* Check whether a daemon is alive — via the pid file, or a fresh heartbeat when
|
|
48
|
+
* the pid file has been lost (see resolveLiveDaemonPid). Heals the pid file as a
|
|
49
|
+
* side effect so a subsequent read is consistent.
|
|
50
|
+
*/
|
|
47
51
|
export declare function isDaemonRunning(): boolean;
|
|
48
52
|
/**
|
|
49
53
|
* Single-instance claim for the daemon foreground entrypoint.
|
package/dist/lib/daemon.js
CHANGED
|
@@ -172,6 +172,16 @@ export function removeHeartbeat() {
|
|
|
172
172
|
}
|
|
173
173
|
catch { /* already removed */ }
|
|
174
174
|
}
|
|
175
|
+
/**
|
|
176
|
+
* A heartbeat is "fresh" when its last tick falls inside the wedge window — the
|
|
177
|
+
* same threshold isDaemonWedged() uses to decide a still-present daemon has gone
|
|
178
|
+
* unresponsive. A fresh heartbeat whose pid is alive is proof of a live, ticking
|
|
179
|
+
* daemon even when the pid file has been lost.
|
|
180
|
+
*/
|
|
181
|
+
function isHeartbeatFresh(hb) {
|
|
182
|
+
const elapsed = Date.now() - Date.parse(hb.lastTick);
|
|
183
|
+
return elapsed <= WEDGE_THRESHOLD_TICKS * MONITOR_TICK_MS;
|
|
184
|
+
}
|
|
175
185
|
export function isDaemonWedged() {
|
|
176
186
|
const pid = readDaemonPid();
|
|
177
187
|
if (!pid)
|
|
@@ -183,22 +193,49 @@ export function isDaemonWedged() {
|
|
|
183
193
|
return false;
|
|
184
194
|
if (hb.pid !== pid)
|
|
185
195
|
return false;
|
|
186
|
-
|
|
187
|
-
return elapsed > WEDGE_THRESHOLD_TICKS * MONITOR_TICK_MS;
|
|
196
|
+
return !isHeartbeatFresh(hb);
|
|
188
197
|
}
|
|
189
198
|
/** How long stopDaemon waits for a SIGTERMed daemon to exit before escalating. */
|
|
190
199
|
const STOP_GRACE_MS = 5000;
|
|
191
200
|
/** How long it waits after the hard tree-kill before giving up. */
|
|
192
201
|
const STOP_KILL_GRACE_MS = 2000;
|
|
193
|
-
/**
|
|
194
|
-
|
|
202
|
+
/**
|
|
203
|
+
* Resolve the PID of the live daemon, tolerant of a pid-file/heartbeat desync.
|
|
204
|
+
*
|
|
205
|
+
* The daemon writes the pid file once (on claim/start) but rewrites the
|
|
206
|
+
* heartbeat every tick. If the pid file is lost while the daemon keeps ticking
|
|
207
|
+
* — e.g. an earlier isDaemonRunning() found a stale/reused/dead pid and cleared
|
|
208
|
+
* the file, or it was removed out from under a live daemon — the pid file reads
|
|
209
|
+
* empty even though a daemon is genuinely alive and firing jobs. Reading only
|
|
210
|
+
* the pid file then reports "stopped" for a running scheduler, and (worse) lets
|
|
211
|
+
* claimDaemonInstance() start a SECOND daemon that double-fires every routine.
|
|
212
|
+
*
|
|
213
|
+
* So: trust the pid file when its pid is alive; otherwise trust a FRESH
|
|
214
|
+
* heartbeat whose pid is alive, and re-adopt the pid file so the desync heals.
|
|
215
|
+
* Returns null only when neither points at a live process (clearing a stale pid
|
|
216
|
+
* file on the way out).
|
|
217
|
+
*/
|
|
218
|
+
function resolveLiveDaemonPid() {
|
|
195
219
|
const pid = readDaemonPid();
|
|
196
|
-
if (
|
|
197
|
-
return
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
220
|
+
if (pid !== null && isAlive(pid))
|
|
221
|
+
return pid;
|
|
222
|
+
const hb = readHeartbeat();
|
|
223
|
+
if (hb && isAlive(hb.pid) && isHeartbeatFresh(hb)) {
|
|
224
|
+
if (pid !== hb.pid)
|
|
225
|
+
writeDaemonPid(hb.pid); // heal the pid-file/heartbeat desync
|
|
226
|
+
return hb.pid;
|
|
227
|
+
}
|
|
228
|
+
if (pid !== null)
|
|
229
|
+
removeDaemonPid();
|
|
230
|
+
return null;
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* Check whether a daemon is alive — via the pid file, or a fresh heartbeat when
|
|
234
|
+
* the pid file has been lost (see resolveLiveDaemonPid). Heals the pid file as a
|
|
235
|
+
* side effect so a subsequent read is consistent.
|
|
236
|
+
*/
|
|
237
|
+
export function isDaemonRunning() {
|
|
238
|
+
return resolveLiveDaemonPid() !== null;
|
|
202
239
|
}
|
|
203
240
|
/**
|
|
204
241
|
* Single-instance claim for the daemon foreground entrypoint.
|
|
@@ -217,16 +254,28 @@ export function isDaemonRunning() {
|
|
|
217
254
|
*/
|
|
218
255
|
export function claimDaemonInstance() {
|
|
219
256
|
const release = acquireStartLock();
|
|
257
|
+
// acquireStartLock() returns null only when another __daemon-run currently
|
|
258
|
+
// holds the O_EXCL lock — a dead holder's lock is reclaimed and retried inside
|
|
259
|
+
// acquireStartLock, so null means a *live* claimer is mid-claim. Bail rather
|
|
260
|
+
// than run the read-decide-write unlocked: otherwise two first-start processes
|
|
261
|
+
// could each see no pid file (before either writes one) and both claim,
|
|
262
|
+
// running the concurrent JobScheduler this guard exists to prevent.
|
|
263
|
+
if (!release)
|
|
264
|
+
return false;
|
|
220
265
|
try {
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
266
|
+
// resolveLiveDaemonPid() also consults a fresh heartbeat, so a live daemon
|
|
267
|
+
// whose pid file was lost still blocks a second claim — otherwise a missing
|
|
268
|
+
// pid file would let this instance start a concurrent JobScheduler and
|
|
269
|
+
// double-fire every routine.
|
|
270
|
+
const existing = resolveLiveDaemonPid();
|
|
271
|
+
if (existing !== null && existing !== process.pid) {
|
|
272
|
+
return false; // another live daemon already owns the instance
|
|
224
273
|
}
|
|
225
274
|
writeDaemonPid(process.pid);
|
|
226
275
|
return true;
|
|
227
276
|
}
|
|
228
277
|
finally {
|
|
229
|
-
release
|
|
278
|
+
release();
|
|
230
279
|
}
|
|
231
280
|
}
|
|
232
281
|
/**
|
|
@@ -304,8 +304,12 @@ export function buildLocalFindings(input) {
|
|
|
304
304
|
}
|
|
305
305
|
}
|
|
306
306
|
if (neverSynced) {
|
|
307
|
-
// Everything is "missing" because it was never synced —
|
|
308
|
-
//
|
|
307
|
+
// Everything is "missing" because it was never synced — collapse to ONE
|
|
308
|
+
// line. A never-synced version is almost always an old/unused install (you
|
|
309
|
+
// don't run a version you never synced), so this is a WARNING, not a "needs
|
|
310
|
+
// you now" critical: it isn't hurting anything until you actually launch it.
|
|
311
|
+
// The real criticals are a logged-out account or a hook/plugin missing from
|
|
312
|
+
// a version you DO keep synced (the `else` branch below).
|
|
309
313
|
const total = missingHooks.length + missingPlugins.length + missingOther.length;
|
|
310
314
|
if (total > 0) {
|
|
311
315
|
const breakdown = [
|
|
@@ -313,7 +317,7 @@ export function buildLocalFindings(input) {
|
|
|
313
317
|
missingPlugins.length ? `${missingPlugins.length} plugin${missingPlugins.length === 1 ? '' : 's'}` : '',
|
|
314
318
|
].filter(Boolean).join(', ');
|
|
315
319
|
out.push(finding({
|
|
316
|
-
severity: '
|
|
320
|
+
severity: 'warning', kind: 'never-synced', device, agent, version,
|
|
317
321
|
message: `never synced — ${total} resource${total === 1 ? '' : 's'}${breakdown ? ` (incl. ${breakdown})` : ''} not installed`,
|
|
318
322
|
}));
|
|
319
323
|
}
|
|
@@ -464,7 +468,11 @@ function duplicateHookFindings(device, dups) {
|
|
|
464
468
|
`${versions.length} version${versions.length === 1 ? '' : 's'} ` +
|
|
465
469
|
`(incl. ${group.slice(0, 2).map((d) => `'${d.name}'`).join(', ')}) — ${authority}`;
|
|
466
470
|
out.push({
|
|
467
|
-
|
|
471
|
+
// A hook that DIFFERS across versions is still installed and firing — it's
|
|
472
|
+
// stale/sync drift, not a missing hook. Resolvable by one sync; a WARNING,
|
|
473
|
+
// not a "needs you now" critical. (A genuinely MISSING hook stays critical
|
|
474
|
+
// via the missing-hook path.)
|
|
475
|
+
severity: 'warning',
|
|
468
476
|
kind: drift ? 'duplicate-hook-drift' : 'duplicate-hook',
|
|
469
477
|
device, agent, versions, message,
|
|
470
478
|
remediation: `agents sync ${agent}@all --yes`,
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/** Serve a cached snapshot without recomputing while it is younger than this. */
|
|
2
|
+
export declare const DOCTOR_OVERVIEW_FRESH_MS = 90000;
|
|
3
|
+
/** Injectable IO + clock so tests exercise the real fs at a temp dir, no mocks. */
|
|
4
|
+
export interface DoctorOverviewCacheDeps {
|
|
5
|
+
/** Cache directory (default: the real `~/.agents/.cache`). */
|
|
6
|
+
dir?: string;
|
|
7
|
+
/** Clock (default: {@link Date.now}). */
|
|
8
|
+
now?: () => number;
|
|
9
|
+
}
|
|
10
|
+
/** Read the last snapshot (best-effort; missing/corrupt/wrong-version → null). */
|
|
11
|
+
export declare function readDoctorOverviewCache(deps?: DoctorOverviewCacheDeps): {
|
|
12
|
+
fetchedAt: number;
|
|
13
|
+
payload: unknown;
|
|
14
|
+
} | null;
|
|
15
|
+
/** Persist a fresh overview payload (best-effort; tmp+rename so reads are atomic). */
|
|
16
|
+
export declare function writeDoctorOverviewCache(payload: unknown, deps?: DoctorOverviewCacheDeps): void;
|
|
17
|
+
/**
|
|
18
|
+
* Result of {@link enterDoctorOverviewGate}.
|
|
19
|
+
* - `cached` non-null → the caller MUST print this string and return; no compute.
|
|
20
|
+
* - `cached` null → the caller holds the singleflight lock: compute the
|
|
21
|
+
* overview, call {@link writeDoctorOverviewCache}, and invoke `release()` on
|
|
22
|
+
* the way out. Call `release()` in a `finally` so a compute that throws still
|
|
23
|
+
* frees the lock promptly (idempotent).
|
|
24
|
+
*/
|
|
25
|
+
export interface OverviewGate {
|
|
26
|
+
cached: string | null;
|
|
27
|
+
release?: () => void;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Enter the doctor-overview singleflight gate. Returns a cached string to print,
|
|
31
|
+
* or a lock token telling the caller to compute (and then write + release).
|
|
32
|
+
*
|
|
33
|
+
* Contract:
|
|
34
|
+
* - Fresh snapshot present (and not `forceRefresh`) → `{ cached }`, no lock.
|
|
35
|
+
* - Otherwise exactly one caller holds the lock and gets `{ cached: null,
|
|
36
|
+
* release }`; everyone else blocks on the lock, then (on acquiring it)
|
|
37
|
+
* double-checks and serves the winner's fresh write — or, if the winner runs
|
|
38
|
+
* past the wait budget, serves the last snapshot — rather than recomputing.
|
|
39
|
+
* - Never throws: any IO/lock failure degrades to a compute token or a served
|
|
40
|
+
* snapshot.
|
|
41
|
+
*/
|
|
42
|
+
export declare function enterDoctorOverviewGate(opts?: {
|
|
43
|
+
forceRefresh?: boolean;
|
|
44
|
+
freshMs?: number;
|
|
45
|
+
}, deps?: DoctorOverviewCacheDeps): Promise<OverviewGate>;
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Singleflight + short-TTL disk cache for the `agents doctor --json` OVERVIEW
|
|
3
|
+
* payload (the bare, no-target form the menu-bar helper and other pollers read).
|
|
4
|
+
*
|
|
5
|
+
* Why this exists (RUSH-2153): the bare `doctor --json` overview is expensive —
|
|
6
|
+
* it probes every host CLI, spawns every installed agent CLI for its sign-in,
|
|
7
|
+
* and diffs every agent×version against its source. On an idle box that is a few
|
|
8
|
+
* seconds; on a loaded one it is minutes. The menu-bar helper polls it on a 60s
|
|
9
|
+
* timer with a per-*process* in-flight guard, so nothing coalesces ACROSS
|
|
10
|
+
* processes: a helper relaunch (or any second poller) each launches its own
|
|
11
|
+
* live compute, and a helper killed mid-run orphans its `doctor --json` child,
|
|
12
|
+
* which keeps spinning. In steady state this stacked to dozens of concurrent
|
|
13
|
+
* `doctor --json` processes pinning ~14 cores and driving load to ~300.
|
|
14
|
+
*
|
|
15
|
+
* The fix mirrors the {@link readStatsCache}/`writeStatsCache` mirror-file
|
|
16
|
+
* convention: reads are cache-first, and when a live compute IS needed exactly
|
|
17
|
+
* ONE runs at a time. The singleflight is the shared `proper-lockfile` lock (via
|
|
18
|
+
* `ensureLockTarget` + `lockfile.lock`) — the SAME battle-tested lock the rest of
|
|
19
|
+
* the CLI uses (fs-atomic.ts) — which owns two things a hand-rolled lock got
|
|
20
|
+
* wrong: (1) it auto-refreshes the lock's mtime on a timer while held, so a live
|
|
21
|
+
* computer whose compute runs for minutes is never mistaken for a crashed one
|
|
22
|
+
* and stolen; (2) `release()` only ever releases the lock THIS caller acquired,
|
|
23
|
+
* so a slow computer can't delete a successor's lock. Waiters block on the lock
|
|
24
|
+
* up to a bounded budget, then serve the last snapshot rather than pile on.
|
|
25
|
+
*
|
|
26
|
+
* The cache write is tmp+rename so a concurrent reader never sees a partial file.
|
|
27
|
+
* All IO is best-effort: a failure degrades to a live compute, never a throw.
|
|
28
|
+
*/
|
|
29
|
+
import * as fs from 'fs';
|
|
30
|
+
import * as path from 'path';
|
|
31
|
+
import lockfile from 'proper-lockfile';
|
|
32
|
+
import { getCacheDir } from '../state.js';
|
|
33
|
+
import { ensureLockTarget } from '../fs-atomic.js';
|
|
34
|
+
const CACHE_FILE = '.doctor-overview.json';
|
|
35
|
+
const LOCK_TARGET_FILE = '.doctor-overview.lock-target';
|
|
36
|
+
/** Serve a cached snapshot without recomputing while it is younger than this. */
|
|
37
|
+
export const DOCTOR_OVERVIEW_FRESH_MS = 90_000;
|
|
38
|
+
/**
|
|
39
|
+
* A held lock older than this is treated as a crashed computer and broken. The
|
|
40
|
+
* lock's mtime is auto-refreshed by proper-lockfile every `stale/2` while a live
|
|
41
|
+
* computer holds it (the event loop turns during the compute's `await`ed
|
|
42
|
+
* subprocess spawns), so this only ever breaks a genuinely dead holder.
|
|
43
|
+
*/
|
|
44
|
+
const LOCK_STALE_MS = 60_000;
|
|
45
|
+
/**
|
|
46
|
+
* How long a waiter blocks on the lock before giving up and serving the last
|
|
47
|
+
* snapshot. Sized to comfortably exceed a slow (multi-second-to-minutes) compute
|
|
48
|
+
* so a waiter normally gets the winner's fresh write; capped so a truly wedged
|
|
49
|
+
* holder never hangs the CLI (it serves stale instead).
|
|
50
|
+
*/
|
|
51
|
+
const LOCK_RETRIES = { retries: 240, factor: 1, minTimeout: 500, maxTimeout: 500 };
|
|
52
|
+
function cachePath(dir) {
|
|
53
|
+
return path.join(dir, CACHE_FILE);
|
|
54
|
+
}
|
|
55
|
+
/** Read the last snapshot (best-effort; missing/corrupt/wrong-version → null). */
|
|
56
|
+
export function readDoctorOverviewCache(deps = {}) {
|
|
57
|
+
const dir = deps.dir ?? getCacheDir();
|
|
58
|
+
try {
|
|
59
|
+
const parsed = JSON.parse(fs.readFileSync(cachePath(dir), 'utf-8'));
|
|
60
|
+
if (parsed && parsed.version === 1 && typeof parsed.fetchedAt === 'number') {
|
|
61
|
+
return { fetchedAt: parsed.fetchedAt, payload: parsed.payload };
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
catch {
|
|
65
|
+
// missing or corrupt — treat as no snapshot
|
|
66
|
+
}
|
|
67
|
+
return null;
|
|
68
|
+
}
|
|
69
|
+
/** Persist a fresh overview payload (best-effort; tmp+rename so reads are atomic). */
|
|
70
|
+
export function writeDoctorOverviewCache(payload, deps = {}) {
|
|
71
|
+
const dir = deps.dir ?? getCacheDir();
|
|
72
|
+
const now = deps.now ?? Date.now;
|
|
73
|
+
try {
|
|
74
|
+
if (!fs.existsSync(dir))
|
|
75
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
76
|
+
const body = { version: 1, fetchedAt: now(), payload };
|
|
77
|
+
const tmp = `${cachePath(dir)}.tmp.${process.pid}`;
|
|
78
|
+
fs.writeFileSync(tmp, JSON.stringify(body, null, 2));
|
|
79
|
+
fs.renameSync(tmp, cachePath(dir));
|
|
80
|
+
}
|
|
81
|
+
catch {
|
|
82
|
+
// best-effort; a failed write just means the next read falls back to live
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Enter the doctor-overview singleflight gate. Returns a cached string to print,
|
|
87
|
+
* or a lock token telling the caller to compute (and then write + release).
|
|
88
|
+
*
|
|
89
|
+
* Contract:
|
|
90
|
+
* - Fresh snapshot present (and not `forceRefresh`) → `{ cached }`, no lock.
|
|
91
|
+
* - Otherwise exactly one caller holds the lock and gets `{ cached: null,
|
|
92
|
+
* release }`; everyone else blocks on the lock, then (on acquiring it)
|
|
93
|
+
* double-checks and serves the winner's fresh write — or, if the winner runs
|
|
94
|
+
* past the wait budget, serves the last snapshot — rather than recomputing.
|
|
95
|
+
* - Never throws: any IO/lock failure degrades to a compute token or a served
|
|
96
|
+
* snapshot.
|
|
97
|
+
*/
|
|
98
|
+
export async function enterDoctorOverviewGate(opts = {}, deps = {}) {
|
|
99
|
+
const dir = deps.dir ?? getCacheDir();
|
|
100
|
+
const now = deps.now ?? Date.now;
|
|
101
|
+
const freshMs = opts.freshMs ?? DOCTOR_OVERVIEW_FRESH_MS;
|
|
102
|
+
const serveFresh = () => {
|
|
103
|
+
if (opts.forceRefresh)
|
|
104
|
+
return null;
|
|
105
|
+
const c = readDoctorOverviewCache({ dir });
|
|
106
|
+
if (c && now() - c.fetchedAt < freshMs)
|
|
107
|
+
return JSON.stringify(c.payload, null, 2);
|
|
108
|
+
return null;
|
|
109
|
+
};
|
|
110
|
+
// 1. Fast path: a fresh snapshot serves without any compute or lock.
|
|
111
|
+
const fast = serveFresh();
|
|
112
|
+
if (fast !== null)
|
|
113
|
+
return { cached: fast };
|
|
114
|
+
try {
|
|
115
|
+
if (!fs.existsSync(dir))
|
|
116
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
117
|
+
}
|
|
118
|
+
catch {
|
|
119
|
+
// Can't even make the cache dir — fall back to an unguarded compute.
|
|
120
|
+
return { cached: null, release: () => { } };
|
|
121
|
+
}
|
|
122
|
+
const lockTarget = path.join(dir, LOCK_TARGET_FILE);
|
|
123
|
+
ensureLockTarget(lockTarget);
|
|
124
|
+
// 2. Singleflight via proper-lockfile: one caller holds the lock and computes;
|
|
125
|
+
// the rest block here until it releases.
|
|
126
|
+
let release = null;
|
|
127
|
+
try {
|
|
128
|
+
release = await lockfile.lock(lockTarget, {
|
|
129
|
+
stale: LOCK_STALE_MS,
|
|
130
|
+
retries: LOCK_RETRIES,
|
|
131
|
+
// A peer broke our lock (only possible if we somehow went stale). Don't
|
|
132
|
+
// crash on the async callback; we re-check the cache and serve/recompute.
|
|
133
|
+
onCompromised: () => { },
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
catch {
|
|
137
|
+
// 3. Winner held the lock past our wait budget. Serve the last snapshot
|
|
138
|
+
// (even if stale) rather than pile on; only if there is genuinely none do
|
|
139
|
+
// we compute unguarded (rare cold-start under sustained load).
|
|
140
|
+
const c = readDoctorOverviewCache({ dir });
|
|
141
|
+
if (c)
|
|
142
|
+
return { cached: JSON.stringify(c.payload, null, 2) };
|
|
143
|
+
return { cached: null, release: () => { } };
|
|
144
|
+
}
|
|
145
|
+
// 4. Acquired. The winner may have written a fresh snapshot while we waited —
|
|
146
|
+
// serve it and release, instead of recomputing.
|
|
147
|
+
const afterWait = serveFresh();
|
|
148
|
+
if (afterWait !== null) {
|
|
149
|
+
// AWAIT, don't fire-and-forget: returning while the lockfile is still on
|
|
150
|
+
// disk makes the next caller retry against a lock that is logically free —
|
|
151
|
+
// the same pile-up this gate exists to prevent, just narrowed to the window
|
|
152
|
+
// between return and unlink. We are already in an async function, so the
|
|
153
|
+
// wait costs one unlink.
|
|
154
|
+
await release().catch(() => { });
|
|
155
|
+
return { cached: afterWait };
|
|
156
|
+
}
|
|
157
|
+
const rel = release;
|
|
158
|
+
let released = false;
|
|
159
|
+
return {
|
|
160
|
+
cached: null,
|
|
161
|
+
release: () => {
|
|
162
|
+
if (released)
|
|
163
|
+
return;
|
|
164
|
+
released = true;
|
|
165
|
+
void rel();
|
|
166
|
+
},
|
|
167
|
+
};
|
|
168
|
+
}
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
*/
|
|
10
10
|
import { spawnSync } from 'child_process';
|
|
11
11
|
import { isControlDevice } from './registry.js';
|
|
12
|
+
import { isSelfHost } from './self-host.js';
|
|
12
13
|
import { buildSshInvocation, sshTargetFor, writeAskpassShim } from './connect.js';
|
|
13
14
|
/** npm dist-tags / semver pins only — rejects shell metacharacters. */
|
|
14
15
|
export const FLEET_VERSION_RE = /^[A-Za-z0-9._-]+$/;
|
|
@@ -52,7 +53,11 @@ export function planFleetTargets(reg) {
|
|
|
52
53
|
* surface — so their `skip` reason still flows through as an `unreachable` row.
|
|
53
54
|
*/
|
|
54
55
|
export function remoteFleetTargets(planned, self) {
|
|
55
|
-
|
|
56
|
+
// Exclude self by name AND by full identity (tailscale dnsName, loopback): a
|
|
57
|
+
// device referenced by its dnsName slipped past the bare name check and got a
|
|
58
|
+
// remote version+doctor dial back to THIS box, which orphaned on timeout and
|
|
59
|
+
// piled up (RUSH-2114). `isSelfHost` matches every alias the box answers to.
|
|
60
|
+
return planned.filter((t) => t.device.name !== self && !isSelfHost(t.device.name) && t.skip !== 'control');
|
|
56
61
|
}
|
|
57
62
|
/**
|
|
58
63
|
* Decide whether a fleet-health target should skip the expensive version+doctor
|
|
@@ -189,7 +194,7 @@ export function runFleet(targets, cmd, opts = {}) {
|
|
|
189
194
|
continue;
|
|
190
195
|
}
|
|
191
196
|
try {
|
|
192
|
-
const isSelf = opts.self !== undefined && t.device.name === opts.self;
|
|
197
|
+
const isSelf = (opts.self !== undefined && t.device.name === opts.self) || isSelfHost(t.device.name);
|
|
193
198
|
const res = isSelf ? localRunner(cmd) : runner(t.device, cmd);
|
|
194
199
|
const ok = res.code === 0;
|
|
195
200
|
const detail = (res.stderr || res.stdout).trim().slice(0, 200);
|
|
@@ -9,6 +9,10 @@ export interface ResolvedSshTarget {
|
|
|
9
9
|
name: string;
|
|
10
10
|
os?: string;
|
|
11
11
|
}
|
|
12
|
+
export interface ResolvedExplicitTargetSet {
|
|
13
|
+
targets: ResolvedSshTarget[];
|
|
14
|
+
unresolved: string[];
|
|
15
|
+
}
|
|
12
16
|
/**
|
|
13
17
|
* Resolve a target token to a full {@link DeviceProfile} for `agents ssh`. Same
|
|
14
18
|
* grammar as the fan-out, but returns the whole profile (auth, shell, tailscale
|
|
@@ -27,3 +31,5 @@ export declare function resolveDeviceTarget(token: string): Promise<DeviceProfil
|
|
|
27
31
|
* cross-machine fan-out so they can never diverge onto two routes.
|
|
28
32
|
*/
|
|
29
33
|
export declare function resolveExplicitTargets(hosts: string[]): Promise<ResolvedSshTarget[]>;
|
|
34
|
+
/** Resolve explicit tokens while retaining failures for coverage-sensitive callers. */
|
|
35
|
+
export declare function resolveExplicitTargetSet(hosts: string[]): Promise<ResolvedExplicitTargetSet>;
|
|
@@ -96,14 +96,20 @@ export async function resolveDeviceTarget(token) {
|
|
|
96
96
|
* cross-machine fan-out so they can never diverge onto two routes.
|
|
97
97
|
*/
|
|
98
98
|
export async function resolveExplicitTargets(hosts) {
|
|
99
|
-
|
|
99
|
+
return (await resolveExplicitTargetSet(hosts)).targets;
|
|
100
|
+
}
|
|
101
|
+
/** Resolve explicit tokens while retaining failures for coverage-sensitive callers. */
|
|
102
|
+
export async function resolveExplicitTargetSet(hosts) {
|
|
103
|
+
const targets = [];
|
|
104
|
+
const unresolved = [];
|
|
100
105
|
for (const h of hosts) {
|
|
101
106
|
const resolved = await toResolvedTarget(h);
|
|
102
107
|
if (!resolved) {
|
|
103
108
|
process.stderr.write(chalk.gray(` ${h}: not a resolvable ssh target — skipped\n`));
|
|
109
|
+
unresolved.push(h);
|
|
104
110
|
continue;
|
|
105
111
|
}
|
|
106
|
-
|
|
112
|
+
targets.push(resolved);
|
|
107
113
|
}
|
|
108
|
-
return
|
|
114
|
+
return { targets, unresolved };
|
|
109
115
|
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* True when `name` refers to the local machine. Case-insensitive and
|
|
3
|
+
* trailing-dot-tolerant. Use this everywhere a `--host`/fleet target is compared
|
|
4
|
+
* against "self" so a tailscale-name reference short-circuits to a local run
|
|
5
|
+
* instead of self-SSHing.
|
|
6
|
+
*/
|
|
7
|
+
export declare function isSelfHost(name: string | undefined | null): boolean;
|
|
8
|
+
/** Test hook: drop the memoized alias set so a fresh registry/env is re-read. */
|
|
9
|
+
export declare function resetSelfHostCache(): void;
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* "Is this hostname the local machine?" — matched against every identity the box
|
|
3
|
+
* answers to, not just its short id.
|
|
4
|
+
*
|
|
5
|
+
* The self-checks that gate `--host` dispatch and the fleet-health fan-out used to
|
|
6
|
+
* compare only against {@link machineId} (the lowercased short hostname, e.g.
|
|
7
|
+
* `zion`). A caller that referenced the box by its **tailscale MagicDNS name**
|
|
8
|
+
* (`zion.tail1a85a1.ts.net`) — which is exactly what `fleetDialTarget` and the
|
|
9
|
+
* Factory floor's `--host` probes use — slipped past the check and SSH'd to the
|
|
10
|
+
* LOCAL box over its own tailscale name. On a loaded machine that self-SSH'd
|
|
11
|
+
* `doctor --json` orphaned on timeout and piled up until the host was crushed
|
|
12
|
+
* (RUSH-2114). Matching the full identity set closes that gap at the source.
|
|
13
|
+
*/
|
|
14
|
+
import { machineId } from '../machine-id.js';
|
|
15
|
+
import { loadDevicesSync } from './registry.js';
|
|
16
|
+
const LOOPBACK = ['localhost', '127.0.0.1', '::1'];
|
|
17
|
+
/** Lowercase + strip a trailing dot (FQDNs are equivalent with or without it). */
|
|
18
|
+
function normalize(name) {
|
|
19
|
+
return name.trim().toLowerCase().replace(/\.$/, '');
|
|
20
|
+
}
|
|
21
|
+
let cached = null;
|
|
22
|
+
/**
|
|
23
|
+
* Every name that resolves to THIS machine: the short id, loopback, and the self
|
|
24
|
+
* device's tailscale dnsName plus its short form. Computed once per process — the
|
|
25
|
+
* self identity does not change under a running CLI — and reads the registry
|
|
26
|
+
* best-effort (an unreadable registry still leaves the short id + loopback).
|
|
27
|
+
*/
|
|
28
|
+
function selfAliases() {
|
|
29
|
+
if (cached)
|
|
30
|
+
return cached;
|
|
31
|
+
const aliases = new Set([machineId(), ...LOOPBACK]);
|
|
32
|
+
try {
|
|
33
|
+
const dns = loadDevicesSync()[machineId()]?.address?.dnsName;
|
|
34
|
+
if (dns) {
|
|
35
|
+
const d = normalize(dns);
|
|
36
|
+
aliases.add(d);
|
|
37
|
+
aliases.add(d.split('.')[0]); // the short form of the FQDN
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
catch {
|
|
41
|
+
/* registry unreadable — the short id + loopback aliases still hold */
|
|
42
|
+
}
|
|
43
|
+
cached = aliases;
|
|
44
|
+
return cached;
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* True when `name` refers to the local machine. Case-insensitive and
|
|
48
|
+
* trailing-dot-tolerant. Use this everywhere a `--host`/fleet target is compared
|
|
49
|
+
* against "self" so a tailscale-name reference short-circuits to a local run
|
|
50
|
+
* instead of self-SSHing.
|
|
51
|
+
*/
|
|
52
|
+
export function isSelfHost(name) {
|
|
53
|
+
if (!name)
|
|
54
|
+
return false;
|
|
55
|
+
const n = normalize(name);
|
|
56
|
+
return n.length > 0 && selfAliases().has(n);
|
|
57
|
+
}
|
|
58
|
+
/** Test hook: drop the memoized alias set so a fresh registry/env is re-read. */
|
|
59
|
+
export function resetSelfHostCache() {
|
|
60
|
+
cached = null;
|
|
61
|
+
}
|