@phnx-labs/agents-cli 1.22.56 → 1.22.58

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/README.md +4 -4
  3. package/dist/bootstrap.js +11 -2
  4. package/dist/cli/command-registry.d.ts +0 -1
  5. package/dist/cli/command-registry.js +0 -3
  6. package/dist/commands/accounts.js +7 -3
  7. package/dist/commands/apply.js +10 -2
  8. package/dist/commands/exec.js +1 -1
  9. package/dist/commands/fork.d.ts +23 -10
  10. package/dist/commands/fork.js +115 -58
  11. package/dist/commands/hooks.js +4 -4
  12. package/dist/commands/insights.d.ts +7 -5
  13. package/dist/commands/insights.js +16 -9
  14. package/dist/commands/monitors.js +11 -0
  15. package/dist/commands/perf.d.ts +16 -7
  16. package/dist/commands/perf.js +29 -20
  17. package/dist/commands/prune.js +5 -3
  18. package/dist/commands/routines.d.ts +8 -0
  19. package/dist/commands/routines.js +57 -3
  20. package/dist/commands/rules.js +1 -1
  21. package/dist/commands/sessions-picker.d.ts +11 -0
  22. package/dist/commands/sessions-picker.js +16 -0
  23. package/dist/commands/sessions.js +1 -0
  24. package/dist/commands/share.d.ts +14 -0
  25. package/dist/commands/share.js +43 -2
  26. package/dist/commands/ssh.js +24 -14
  27. package/dist/commands/status.js +1 -1
  28. package/dist/commands/sync.js +83 -7
  29. package/dist/commands/traces.js +7 -0
  30. package/dist/commands/trash.d.ts +2 -2
  31. package/dist/commands/trash.js +2 -6
  32. package/dist/commands/versions.d.ts +2 -2
  33. package/dist/commands/versions.js +1 -10
  34. package/dist/commands/view.d.ts +2 -2
  35. package/dist/commands/view.js +7 -6
  36. package/dist/index.d.ts +1 -0
  37. package/dist/index.js +14 -0
  38. package/dist/lib/account-registry.d.ts +5 -1
  39. package/dist/lib/account-registry.js +47 -14
  40. package/dist/lib/accounting/capacity.d.ts +18 -7
  41. package/dist/lib/accounting/capacity.js +19 -8
  42. package/dist/lib/accounting/usage-ingest.d.ts +1 -0
  43. package/dist/lib/accounting/usage-ingest.js +75 -0
  44. package/dist/lib/accounting/usage-sync.d.ts +97 -0
  45. package/dist/lib/accounting/usage-sync.js +203 -0
  46. package/dist/lib/accounting/usage.d.ts +48 -2
  47. package/dist/lib/accounting/usage.js +79 -2
  48. package/dist/lib/agent-spec/agents.js +1 -1
  49. package/dist/lib/analytics/mix-commands.d.ts +8 -7
  50. package/dist/lib/analytics/mix-commands.js +50 -73
  51. package/dist/lib/auth-mint.d.ts +11 -1
  52. package/dist/lib/auth-mint.js +21 -6
  53. package/dist/lib/browser/ipc.d.ts +8 -0
  54. package/dist/lib/browser/ipc.js +87 -0
  55. package/dist/lib/browser/service.d.ts +19 -0
  56. package/dist/lib/browser/service.js +96 -11
  57. package/dist/lib/browser/sessions-list.js +10 -1
  58. package/dist/lib/daemon/daemon.js +5 -0
  59. package/dist/lib/daemon/runner.d.ts +3 -0
  60. package/dist/lib/daemon/runner.js +95 -53
  61. package/dist/lib/daemon/usage-sync-service.d.ts +21 -0
  62. package/dist/lib/daemon/usage-sync-service.js +42 -0
  63. package/dist/lib/daemon-services.d.ts +1 -1
  64. package/dist/lib/daemon-services.js +5 -0
  65. package/dist/lib/device-config.d.ts +17 -6
  66. package/dist/lib/device-config.js +25 -11
  67. package/dist/lib/devices/connect.d.ts +17 -8
  68. package/dist/lib/devices/connect.js +31 -14
  69. package/dist/lib/devices/pool.d.ts +4 -3
  70. package/dist/lib/devices/pool.js +13 -5
  71. package/dist/lib/doctor-diff.js +77 -7
  72. package/dist/lib/exec.d.ts +6 -41
  73. package/dist/lib/exec.js +6 -41
  74. package/dist/lib/fleet/manifest.d.ts +17 -0
  75. package/dist/lib/fleet/manifest.js +26 -0
  76. package/dist/lib/git.d.ts +13 -1
  77. package/dist/lib/git.js +36 -7
  78. package/dist/lib/harness/adapter.d.ts +7 -7
  79. package/dist/lib/harness/adapters/claude.js +3 -2
  80. package/dist/lib/hooks/install.d.ts +27 -11
  81. package/dist/lib/hooks/install.js +42 -17
  82. package/dist/lib/hosts/reconnect.d.ts +52 -203
  83. package/dist/lib/hosts/reconnect.js +64 -284
  84. package/dist/lib/hosts/remote-cmd.d.ts +9 -0
  85. package/dist/lib/hosts/remote-cmd.js +22 -0
  86. package/dist/lib/installations/migrate.d.ts +6 -120
  87. package/dist/lib/installations/migrate.js +27 -259
  88. package/dist/lib/installations/shims.d.ts +13 -95
  89. package/dist/lib/installations/shims.js +22 -139
  90. package/dist/lib/installations/store.js +1 -1
  91. package/dist/lib/installations/versions.d.ts +26 -133
  92. package/dist/lib/installations/versions.js +41 -204
  93. package/dist/lib/perf/db.d.ts +1 -1
  94. package/dist/lib/perf/db.js +1 -1
  95. package/dist/lib/plugins/skills.d.ts +8 -1
  96. package/dist/lib/plugins/skills.js +18 -2
  97. package/dist/lib/refresh.d.ts +9 -0
  98. package/dist/lib/refresh.js +3 -1
  99. package/dist/lib/routine-readiness.d.ts +15 -1
  100. package/dist/lib/routine-readiness.js +41 -0
  101. package/dist/lib/sandbox.d.ts +4 -1
  102. package/dist/lib/sandbox.js +30 -1
  103. package/dist/lib/secrets/agent.d.ts +80 -225
  104. package/dist/lib/secrets/agent.js +139 -401
  105. package/dist/lib/secrets/bundles.d.ts +73 -222
  106. package/dist/lib/secrets/bundles.js +168 -467
  107. package/dist/lib/secrets/reaper.d.ts +28 -70
  108. package/dist/lib/secrets/reaper.js +30 -85
  109. package/dist/lib/secrets/remote.d.ts +42 -129
  110. package/dist/lib/secrets/remote.js +55 -173
  111. package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
  112. package/dist/lib/self-heal/checks/install-staging.js +96 -0
  113. package/dist/lib/self-heal/registry.js +2 -0
  114. package/dist/lib/self-heal/types.d.ts +1 -1
  115. package/dist/lib/self-update.d.ts +23 -0
  116. package/dist/lib/self-update.js +50 -0
  117. package/dist/lib/session/active.d.ts +16 -32
  118. package/dist/lib/session/active.js +10 -68
  119. package/dist/lib/session/db.d.ts +24 -36
  120. package/dist/lib/session/db.js +143 -44
  121. package/dist/lib/session/discover.d.ts +6 -58
  122. package/dist/lib/session/discover.js +5 -43
  123. package/dist/lib/session/fork.d.ts +45 -26
  124. package/dist/lib/session/fork.js +32 -95
  125. package/dist/lib/session/parse.d.ts +1 -19
  126. package/dist/lib/session/parse.js +2 -15
  127. package/dist/lib/session/tool-calls.d.ts +43 -1
  128. package/dist/lib/session/tool-calls.js +74 -44
  129. package/dist/lib/session/tool-store.d.ts +33 -2
  130. package/dist/lib/session/tool-store.js +56 -3
  131. package/dist/lib/staleness/writers/sources.d.ts +5 -0
  132. package/dist/lib/staleness/writers/sources.js +2 -1
  133. package/dist/lib/startup/command-registry.d.ts +8 -2
  134. package/dist/lib/startup/command-registry.js +12 -4
  135. package/dist/lib/sync-status.d.ts +22 -0
  136. package/dist/lib/sync-status.js +27 -0
  137. package/dist/lib/sync-umbrella.d.ts +9 -0
  138. package/dist/lib/sync-umbrella.js +21 -2
  139. package/dist/lib/traces/insights.d.ts +47 -14
  140. package/dist/lib/traces/insights.js +92 -21
  141. package/dist/lib/traces/phenotype.d.ts +23 -3
  142. package/dist/lib/traces/phenotype.js +72 -24
  143. package/dist/lib/traces/sync.d.ts +15 -0
  144. package/dist/lib/traces/sync.js +104 -19
  145. package/dist/lib/traces/worker-template.js +154 -1
  146. package/package.json +1 -1
package/CHANGELOG.md CHANGED
@@ -1,11 +1,81 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.22.58
4
+
5
+ - **A remote browser task survives a browser-daemon restart on the driving box (PHNX-2663).** When the local daemon restarted, its in-memory task map was rebuilt from disk for local-CDP profiles but **skipped `ssh://` tunnelled tasks** — so an agent driving a browser host over `--device` got "Unknown browser task" / "Tab not found" for a tab that was still alive on the far side, and gave up. `attachRunningProfile` now re-establishes the SSH tunnel and reconnects CDP for a tunnelled task during rehydrate (reusing `connectSSH`, which attaches to the already-running remote browser via `isOwnTunnel` — it never launches one), carrying the tunnel teardown so it can't leak across the next restart. A failed reconnect falls through to disk reconcile exactly like the local-CDP path. Source: `cli/src/lib/browser/service.ts`.
6
+
7
+ - **`agents prune cleanup` no longer over-reports source-present hooks as orphans (PHNX-2693).** Orphan detection diffed the hook files in a version home against the *registered-hook manifest*, but sync copies helper / test / benchmark scripts into every version home alongside registered hooks without registering them — so each one (e.g. `permission-handler`, `verify-work-state`, `*_test`, `benchmark_*`) read as an orphan, and `agents prune cleanup [hooks | --all]` offered to trash ~1000 in-use files across version homes. It now diffs against the resolved **source** set (user + system + enabled extras hook dirs), so a hook present in any configured source is never an orphan, exactly as the command's own help promised — while a genuinely dead file (present in a home, absent from every source) is still flagged. Same detector backs `agents doctor`'s `orphan` warning. Source: `cli/src/lib/hooks/install.ts`.
8
+
9
+ - **`agents prune cleanup` no longer offers to delete plugin-provided skills (PHNX-3185).** The skill orphan detector compared a version home's skills against `skills/` + `.system/skills/` + extras only, and never credited plugin-bundled skills (`plugins/<plugin>/skills`, `.system/plugins/<plugin>/skills`) — so every skill a plugin legitimately installed (`design`, `plan`, `review`, `blog`, …) read as an orphan. On a real machine that was ~95% false positives, and `agents doctor` framed the same set as `⚠ orphans … → agents prune cleanup --all`, so following the tool's own guidance destroyed the working skill library. The detector now credits plugin skill sources when deciding orphans, and the skill-source resolver finds them too so a plugin skill reads as up-to-date instead of drifted. A genuinely dead skill (no central, extra, or plugin source) is still flagged. Source: `cli/src/lib/plugins/skills.ts`, `cli/src/lib/staleness/writers/sources.ts`.
10
+
11
+ - **`agents sync` stops reporting `✓ reconciled` while the drift it was asked to fix stays put (PHNX-3186, PHNX-2955).** `agents sync status` over-reported four classes of phantom drift that the sync writer never creates and can never clear — so re-running sync forever printed success while a phantom "N drifted / N missing" stuck (42–46 drifted per version observed fleet-wide). `diffVersionResources` now mirrors what the writer actually installs: (1) directory docs (`README`/`AGENTS`/`CLAUDE`/`GEMINI`) in `commands/` are documentation, not commands, so they are no longer counted as missing/drifted/orphan commands; (2) on command-as-skill agents (Codex ≥ 0.117, Kimi) a command whose name collides with a real same-named skill is reported present (the skill wins the shared `skills/<name>/` slot and the wrapper is deliberately never written), not the phantom "missing"; (3) version-gated presence-only kinds (e.g. subagents below the harness floor) are capability-gated so a version that structurally cannot hold the resource is not reported missing it; (4) `diffRules` now finds an agent's non-`.md` instructions file (Cursor's `.cursorrules`), which an `.md`-only scan never saw, so the rules row stops false-reporting `missing`. On top of that, every `agents sync` now runs a **post-reconcile verification**: it re-reads each version it wrote into and, if any drift the reconcile was asked to fix survives, names the exact residual resources instead of printing a bare `✓` (the umbrella line reads `⚠ sync: reconcile INCOMPLETE`, and `--json` carries `residualDrift` with `ok:false`; the exit code stays 0, matching the declined-write precedent so the fleet fan-out never discards the residual payload). Source: `cli/src/lib/doctor-diff.ts`, `cli/src/lib/sync-status.ts`, `cli/src/lib/refresh.ts`, `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/sync.ts`.
12
+
13
+ - **A stale per-version plugin marketplace copy is now reported as drift instead of `everywhere`/`ok` (PHNX-2955).** `describePluginDrift` (the content-aware plugin diff behind `agents sync status` / `agents doctor` / the post-reconcile verification) compared only the *presence* of a mirror's skills/commands, so a skill whose bytes went stale — an edit pulled into `~/.agents/.system` but never refreshed into a version's `marketplaces/…` mirror — read as fully synced while agents ran the OLD skill text. It now content-compares each skill/command present in both central and the mirror and reports `stale skill: <name>` / `stale command: <name>`, so the drift is visible and a `--yes` reconcile (which rewrites the mirror) clears it. (Out of scope, left as follow-ups: the interactive `Already in sync` short-circuit does not yet consult this signal, and `agents plugins update` for marketplace-resolved plugins is unchanged.) Source: `cli/src/lib/doctor-diff.ts`.
14
+
15
+ - **`agents sync` on a box where `~/.agents` is not a git repo adopts it in place instead of silently no-op'ing the pull (PHNX-3239).** A partial install — runtime state present but no `.git` — made the umbrella repo-pull stage fail quietly while `agents sync` still reported success, so fleet dotfiles and resources never propagated and nothing said so. The umbrella now runs the same self-heal `agents sync user` does (`adoptUserRepoIfNeeded`): it materializes the tracked resource set from origin and continues, or fails loud with the reason (and the `agents repo pull user <git-url>` hint) when there is no recorded remote to adopt from — never a silent no-op that looks like success. Source: `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/status.ts`.
16
+
17
+ - **`agents browser sessions --profile <name>` surfaces the profile's real task store instead of an empty one (PHNX-3317).** A browser's live tasks and captures live under a composite runtime dir (`<name>@<device>`), but the sessions reader used the `--profile` value as a raw directory name and read the bare legacy `<name>/` dir — which is empty on a machine where the daemon writes the composite key — so a driving agent saw "no captures" / "Unknown browser task" while 13 real tasks sat one dir over. The reader now resolves a profile to every cache dir that belongs to it via `listProfileCacheDirs`/`keyBelongsToProfile` — the same single rule `status()` and `findTask` already use (RUSH-2709) — so the bare name and its `@<device>` stores are unified in the listing. Source: `cli/src/lib/browser/sessions-list.ts`.
18
+
19
+ - **Failure phenotype folded into the traces insight-engine fingerprint via a persisted per-session cache (PHNX-3327).** `computeInsights()` grouped cross-session failure patterns by `(tool, cause, normalized-error)` but did not carry a `phenotype` (false-termination / premature-completion / out-of-order / failure-to-act) — that classification needs the full derived trajectory, which flat `tool_calls` rows don't have. It is now computed per-session once (`classifyPhenotype`) and persisted in a new mtime+size-keyed `session_phenotypes` cache (same shape as `session_topics` / `session_insights`), then read for the **whole corpus** on every sync and folded in as a fourth grouping dimension. Reading the cache for all rows — never just this run's freshly-parsed batch — is what keeps two identically-signatured sessions in ONE failure cluster regardless of which incremental sync first saw each (the fragmentation an in-memory, batch-only map would cause). The `signature` output (`{ tool, cause, key }`) is unchanged; `FailurePattern` gains a `phenotype` field and the pattern id incorporates it, so the same signature under two phenotypes is two deep-linkable clusters. Source: `cli/src/lib/traces/insights.ts`, `cli/src/lib/traces/sync.ts`, `cli/src/lib/session/db.ts`.
20
+
21
+ - **`agents doctor` and `agents setup` no longer crash on macOS when the Keychain helper source is unavailable (PHNX-3385).** Both read the reserved `auth` secrets bundle as an optional status check; on a macOS box where the bundled/local Keychain helper couldn't be found (a source checkout, or a fresh npm install before `agents setup secrets` downloads the signed helper — RUSH-3100 dropped it from the tarball), that check threw and took the whole command down with it instead of reporting the finding as unavailable. Source: `cli/src/lib/secrets/bundles.ts` (`inspectReservedAuthBundle`), `cli/src/lib/auth-mint.ts` (`hasMintedSetupToken`).
22
+
23
+ - **`agents traces sync` reports a truthful run outcome — recover-then-succeed is `completed`, not `errored` (PHNX-3387).** `SessionDetail.meta.outcome` was derived as `errorCount > 0 ? 'errored' : 'completed'`, so a run that hit a tool failure but *recovered* and finished the task was mislabeled `errored` — and the Evals console's "green run with hidden tool failures" callout (the LangSmith trap) could never fire honestly because `surfacedToolFailures` only existed when the run was already `errored`. Outcome is now derived from a causal-recovery predicate (`recoveredAfterErrors`, shared with the false-termination phenotype in `phenotype.ts`): a run with tool errors is `completed` only when a substantive, non-human-facing tool step succeeded strictly after the last error **and resolved the failed work** — its work signature (the effective program for a shell step, the tool identity otherwise) matches an errored step's. A run that ended in error, punted to a human (`AskUserQuestion`), or whose only post-error success is unrelated work (a failed `bun test` followed by an incidental `ls`) stays `errored` — no regression. `surfacedToolFailures` is retained on `completed` runs so the console can honestly show "recovered from these failures." Source: `cli/src/lib/traces/sync.ts`, `cli/src/lib/traces/phenotype.ts`.
24
+
25
+ - **`--strategy balanced` no longer routes to a weekly-exhausted Claude account on worker boxes; a missing usage snapshot is treated as unknown (fails closed) instead of full capacity (PHNX-3392).** A worker's usage cache is empty (the setup-token lacks the `user:profile` scope, so `/api/oauth/usage` 403s — RUSH-2392), and `capacityWeight(null, …)` scored that absence as 100 — max weight — making a blind, actually-exhausted account the top pick. The null arm now draws `UNVERIFIED_WEIGHT` (1): any verified-healthy account outranks an unverifiable one, while an all-unverified pool still draws a pick. Pinned by spec GWT-E5c, a pure unit test (`capacity.test.ts`), and a real-cache-path regression test proving a persisted 7d-100% `week` window makes the account ineligible. Source: `cli/src/lib/accounting/capacity.ts`, `cli/src/lib/accounting/{capacity.test.ts,rotate.test.ts}`.
26
+
27
+ - **A worker pulling Claude usage from a Windows primary no longer silently blanks its cache (PHNX-3392 follow-up).** The `usage-sync` pull path (`pullUsageFromPrimary`) now wraps the primary's JSON in `stripClixml` before parsing, exactly as every other remote-JSON boundary does. Without it, a headed Windows box serving `__usage-export` prepends a PowerShell CLIXML progress banner to stdout, every pull fails as malformed JSON, the worker's usage cache stays null, and the capacity floor becomes the only thing keeping `--strategy balanced` from picking a blind, exhausted account. Source: `cli/src/lib/accounting/usage-sync.ts`.
28
+
29
+ - **`agents upgrade` self-heals the orphaned npm reify staging dir that dead-ended it forever (PHNX-3393).** npm stages an upgrade in a hidden sibling directory next to the running install, deterministically named from the install's path, then renames it into place; a crash mid-upgrade left that directory behind non-empty, and every subsequent upgrade's rename onto the same path failed `ENOTEMPTY` — recurring across boxes with no re-run ever clearing it. `agents upgrade` now sweeps any such orphan before reifying, and `agents doctor --fix` / the daemon's periodic self-heal do the same fleet-wide (with a 10-minute age guard so a concurrent in-flight upgrade is never touched). Source: `cli/src/lib/self-update.ts` (`sweepStaleInstallStaging`), `cli/src/bootstrap.ts`, `cli/src/lib/self-heal/checks/install-staging.ts`.
30
+
31
+ - **The Evals console's "all agents" view now renders real fleet data (PHNX-3397).** The CLI writes traces per-device (`<userId>/<hostname>/…`), but the console asks for the fleet-wide `<userId>/all/…` view, which was never a stored key — so every request 404'd and the console fell back to sample data. The traces Worker now synthesizes `/all` on read: it lists the owner's device prefixes (one delimited list, not every session object) and merges their shards for `all/index.json`, and resolves `all/sessions/<id>.json` to whichever device holds that session. A single device is an exact passthrough; across devices, counts sum and failure patterns fold by id (median/latency are session-weighted approximations, exact when filtered to one device). No CLI change and no extra upload — the per-device shards already in R2 are the source. Owner auth is unchanged: `/all` still requires a Phoenix bearer whose id matches the path prefix. Source: `cli/src/lib/traces/worker-template.ts`.
32
+
33
+ - **`agents traces sync` no longer silently ships a stale Evals dashboard when the sessions DB is busy (PHNX-3401).** `buildIndexShard` writes freshly-classified topics/insights back into the local `sessions.db` while building the console index. If another process (e.g. the Rush app) holds that DB, the write waits out the 30s `busy_timeout` and throws `SQLITE_BUSY` — which used to escape the build and get swallowed by a bare `catch {}`, so per-session shards uploaded but the aggregated index (wasted-time, failure clusters, latency) never refreshed and the console sat stale while `sync` reported a clean run. The cache write-backs are now best-effort (they only warm a cache the shard doesn't read from), and a genuine index build/upload failure is surfaced as `SyncResult.indexError` plus a `⚠ console index not refreshed` warning instead of a silent success. Source: `cli/src/lib/traces/sync.ts`, `cli/src/commands/traces.ts`.
34
+
35
+ - **A release now redeploys the managed share OG-cover Worker so prod can't drift from the shipped template (PHNX-3403).** The Worker that renders share preview cards was deployed only by a manual `agents artifacts share update` run, decoupled from the release train — so a release could publish a CLI whose `worker-template.ts` changed while the deployed Worker stayed stale (the gap that made PHNX-2835 look shipped while every new share still 404'd its cover). `release.sh` now takes `--deploy-worker <auto|on|off>` (default `auto`): after publish, on the home base, it redeploys the Worker from the just-published CLI when the shipped template differs from what the endpoint deployed (`auto`), always (`on`), or never (`off`). The decision is a pure local render+hash via the new `agents artifacts share update --check` (needs no Cloudflare credentials), so an unchanged template is a cheap no-op; the promote preflight verifies the share endpoint and `cloudflare.com` token on the home base before publishing, and a deploy failure fails the release loud with the manual fallback. Source: `cli/scripts/release.sh`, `cli/scripts/promote-home-base-probe.sh`, `cli/src/commands/share.ts`.
36
+
37
+ - **A browser task created over `agents ssh <host> 'agents browser …'` is owned by the real caller, not `UNRESOLVED@<host>` (PHNX-3405).** The `agents ssh` browser-drive passthrough stamped the fleet-remote consent marker (`AGENTS_FLEET_REMOTE=1`, PHNX-3065) but never forwarded the caller's actor, so the peer's daemon re-resolved identity from its own empty env and collapsed every such task to `UNRESOLVED@<host>` — making the browser look contended by phantom owners. `markFleetRemote` (the single seam behind `buildSshInvocation`, the fleet passthrough, and `leasedBoxRemoteCmd`) now forwards `AGENTS_ACTOR*`/`GIT_*` alongside the marker, in both POSIX and PowerShell dialects, marker-first so the consent gate and idempotency guard are unchanged. This closes the same RUSH-2028 actor-provenance gap on the `agents ssh` seam that the `--device` dispatch already closed via `withActorEnv`. Source: `cli/src/lib/devices/connect.ts`.
38
+
39
+ - **Monitor `--run` actions now require an authenticated account and resolve routine hooks (PHNX-3406).** Claude routine selection verifies the balanced account with Claude's own local `auth status` instead of trusting stale identity metadata, then launches with the operator HOME plus that version's `CLAUDE_CONFIG_DIR`, matching ordinary `agents run` so both the native login and portable `~/...` hooks resolve. `agents monitors list` and its JSON output also surface the latest asynchronous action failure, and the docs clarify that `--run` creates a new headless conversation rather than resuming the caller. Source: `cli/src/lib/{sandbox,daemon/runner}.ts`, `cli/src/commands/monitors.ts`, `cli/docs/monitors.md`.
40
+
41
+ - **`agents sessions fork` now works cross-device and cross-harness (PHNX-3409).** Forking a session that lived on another box used to dead-end with `transcript not found` — fork resolved the source with a local-only lookup and copied the transcript off local disk, and it only worked for Claude. It now resolves the source across the fleet (the same path `agents sessions preview` uses) and launches a NEW same-harness session, load-balanced, seeded with a recap of the source (label, cwd, ticket, last state, changed-files, and the source id for `/continue`) — so a session on any device, in any REPL harness, forks and picks up where it left off. Add `--device <host>` to place the sibling on the fleet or `--terminal` to open it in a fresh tab; `--name` still labels it. Source: `cli/src/commands/fork.ts`, `cli/src/lib/session/fork.ts`.
42
+
43
+ - **`agents routines` now says WHY a routine failed, not just that it did (PHNX-3410).** The list, the interactive detail's Recent runs, and `agents routines logs` showed only the status enum (`failed`/`missed`/`skipped`) — so diagnosing a routine that keeps failing on `auth_failed: Please run /login`, a wedged active-run lock, a readiness block, or a missed slot meant drilling into the run dir. A new `runFailureReason(run)` maps the `RunMeta` already on disk (`errorMessage` / `readiness` / `skipReason` / `missed`) to one compact line, rendered inline in the `Last Status` cell, the detail's Recent runs, and a `reason:` line in the `logs` header. Visibility only — no scheduling behavior changes. Source: `cli/src/commands/routines.ts`.
44
+
45
+ - **The daemon warm-tick session indexer no longer re-derives an active non-Claude/Codex session's whole tool history every tick — the root cause behind the wedged event loop and browser-IPC stalls (PHNX-3411).** The symptom fix (above) made a wedged browser daemon fail loud; this removes what wedged it. The warm-tick indexer parsed AND re-redacted every tool call in a changed session on every tick, in `toolIndexMode: 'replace'`. claude/codex resumed from a stored parse offset; the other 11 harnesses (kimi, grok, opencode, antigravity, cursor, droid, …) did not, so an active large session on the interactive hub re-sanitized its entire tool history several times a second — synchronous work that blocked the daemon event loop, so `browser.sock` accepted connections it never answered. The full-file harness path is incremental now too: the tool ledger records how many events were folded plus the collector snapshot, and a later scan of the same append-only stream re-derives only the newly appended tool calls (`mode: 'append'`), falling back to a full re-scan for a first scan, a truncated/rewritten transcript, or an extractor-version bump. Measured on a 400-tool-call Grok session indexed once per tick, re-derive+redact work drops from 80,200 tool calls to 400 over 400 ticks (~201× less), with the index content byte-for-byte identical to a full re-parse. Source: `cli/src/lib/session/{db,tool-calls,tool-store}.ts`.
46
+
47
+ - **`agents browser` detects a wedged daemon instead of hanging on it (PHNX-3411).** A unix-socket `connect` succeeds at the kernel level even when the browser daemon's shared event loop is blocked, so `isDaemonReachable` (a bare accept probe) reported a busy-but-unresponsive daemon as healthy — `browser.sock` accepted every connection while the daemon never replied, and cross-device browser drives to a busy hub (e.g. zion) hung or surfaced a confusing `Timeout waiting for browser daemon socket`. Readiness is now a *reply*: before serving a verb the CLI requires the daemon to actually answer a `version` probe (`isDaemonResponsive`, retried so a transient GC pause never condemns a healthy daemon), and a reachable-but-wedged daemon fails loud — naming the blocked event loop, the daemon log, and the `agents browser stop --daemon` reset — instead of hanging (which also fixes `reconcileDaemonVersion`'s own untimed version probe). Also adds a critical-path regression guard proving the daemon's session-index warm tick resumes an actively-growing Claude session incrementally (per-tick cost tracks appended bytes, never a full O(session) re-parse) — the invariant whose violation would re-starve browser IPC. Source: `cli/src/lib/browser/ipc.ts`, `cli/src/lib/session/__tests__/active-session-warmtick-incremental.test.ts`.
48
+
49
+ - **A routine whose dispatch account is signed out now records a `blocked` run instead of a doomed `failed` one (PHNX-3415).** When the daemon fires a routine, it preflights the rotation-resolved account against the auth-health cache; if that account is provably signed out (`revoked` or `unconfigured` — the "OAuth revoked" / "Please run /login" cases), it records a terminal `blocked` / `agent_auth_failed` run carrying the exact re-login repair (`agents run <agent>@<v> -- login`) and spawns nothing — instead of launching a run that 401s, lands as `failed`, and burns a session (surfaced from the 2026-08-28 reliability investigation, where signed-out accounts were the top cause of routine failures). The check is cache-only and fails **open** on any still-usable or indeterminate verdict (`rate_limited`/`unverified`/`expired`/`error`/absent), so a stale probe never wedges a routine. Broadens and corrects the prior preflight, which only caught `revoked` and mislabeled it `failed`. Source: `cli/src/lib/routine-readiness.ts` (`fireTimeAuthReadiness`), `cli/src/lib/daemon/runner.ts`; spec RT-12.
50
+
51
+ - **`agents fleet apply` on an empty roster now tells you how to fix it (PHNX-3422).** When `fleet.devices` is `{}` (the default on a fresh box), `apply`/`apply --plan` used to print a gray `No target devices — nothing to apply.` that read like the command was broken. It now prints an actionable `fleet.devices is empty — nothing to converge.` plus the two ways to declare a roster (`fleet: { devices: all }` or a named map), and still exits 0. Other zero-target cases (a `devices: all` fleet with no online peers, or a named roster whose every device was unresolved — off-tailnet/ignored/typo) keep the plain note, since their reason was already surfaced above. Source: `cli/src/lib/fleet/manifest.ts` (`emptyTargetsMessage`), `cli/src/commands/apply.ts`.
52
+
53
+ - **Traces now attributes a failed tool call's OWN blocking duration to wasted time, not just the bounded gap to the next call (PHNX-3437).** `tool_calls` gains an `end_timestamp` column (the call's result-record time, already available at ingestion) so the insight engine books `end - start` — the time a call actually blocked — as its wasted time, independent of whether another call follows. Before, a call that hung for minutes and then failed registered as ~0 waste whenever it was the last call in its session or was followed quickly by an unrelated call. This is what makes a fail-fast fix measurable on the Evals console: a channel that stops hanging ~5.5 minutes on stdin and instead fails in under a second (PHNX-3407) drops from ~5.5 minutes of attributed waste to ~0. The existing 30-minute-bounded inter-call gap heuristic is kept as the fallback for rows an older extractor stored (NULL end), so nothing crashes or yields NaN, and — when the end time is known — the post-call retry/stall gap is measured from the call's END so the blocking duration is never double-counted. `wastedMs`/`wastedMsTotal`/`failurePatterns` shard shapes are unchanged. Source: `cli/src/lib/session/{db.ts,tool-calls.ts,tool-store.ts}`, `cli/src/lib/traces/{sync.ts,insights.ts}`.
54
+
55
+ - **`agents view`: an idle Claude session window no longer renders as a full red "unavailable" block (PHNX-3453).** When a Claude account has no usage in the current rolling 5-hour window, the usage API returns `five_hour.utilization = null`, so no `S:` session bar exists — and `agents view` drew that missing window as a solid red `█████ unavailable`, which reads as "100% used / maxed-out", the opposite of what it means. It now renders a neutral dim dashed row `┄┄┄┄┄ unavailable`, visually distinct from both a real `0%` (`░░░░░`) and a full `100%` (`█████`). The reading itself is unchanged and honest — an idle account genuinely has no current 5-hour number; the bar fills in the moment that account runs Claude, or on `agents view claude --refresh` when the box isn't being usage-endpoint rate-limited. Source: `cli/src/lib/accounting/usage.ts` (`formatUsageSummary`, `NO_DATA`).
56
+
57
+ ## 1.22.57
58
+
59
+ - **Removed two duplicate commands: `agents list` and `agents trash restore` (PHNX-3391).** `agents list` was a long-deprecated full duplicate of `agents view` (it already printed "agents list is now agents view" before delegating) — `agents view` is now the one version-listing surface. `agents trash restore <target>` was an exact duplicate of the top-level `agents restore <target>` (both call `restoreVersion`); `agents restore` stays, and `agents trash` keeps `trash list`. Both removed names now error as retired top-level commands rather than auto-correcting. Source: `cli/src/commands/versions.ts`, `cli/src/commands/trash.ts`.
60
+
61
+ - **`agents perf` moved to `agents insights perf` (PHNX-3391).** Performance metrics are an insight, not a top-level noun — latency now sits beside `agents insights cost` / `output` under the one `insights` group. Every subcommand is unchanged (`agents insights perf [hooks|commands|run|friction]`), the disposable warehouse at `~/.agents/.cache/perf/perf.db` is untouched, and bare `agents perf` now errors (retired top-level name) rather than auto-correcting. Source: `cli/src/commands/perf.ts`, `cli/src/commands/insights.ts`, `cli/src/cli/command-registry.ts`.
62
+
63
+ - **`agents insights mix` is now the one counter surface — the eight per-recipe shortcut commands, `recipes`, and `trends` are gone (PHNX-3391).** `agents insights mix` already printed the whole recipe board, yet each section was *also* a standalone command (`agents insights harness-mix`, `model-mix`, …), *also* reachable as `mix <recipe>`, *also* listed by `agents insights recipes`, and the board was *also* aliased by `agents insights trends` — one query engine wearing five surfaces (each registered twice, under `insights` and `sessions insights`). It collapses to one: `agents insights mix` (board), `agents insights mix <recipe>` (one section, e.g. `mix harness-mix`), and `agents insights mix --list` (the recipe ids). The removed spellings now error and point at `mix`. This is surface removal, not hiding — 20 command registrations deleted. Source: `cli/src/lib/analytics/mix-commands.ts`.
64
+
65
+ - **Worker boxes now show real Claude usage (S:/W:) bars in `agents view` (PHNX-3392).** A rate limit is metered per account, so the number is identical on every box — but only a headed device (`personal`/`desktop`) can read it, because the usage endpoint needs the `user:profile` scope only the interactive login carries. Headless workers, which have just the `user:inference` setup-token, got a 403 and showed blank bars. A new `usage-sync` daemon service closes the gap: each personal/desktop daemon pushes its per-account usage snapshot to worker peers, which merge it newest-wins (via the hidden `agents __usage-ingest` verb). Role-gated (personal/desktop publish, workers consume), single-executor per destination, and idempotent. A synced bar reads as last-seen, never a live fetch. Source: `cli/src/lib/accounting/usage-sync.ts`, `cli/src/lib/daemon/usage-sync-service.ts`.
66
+
67
+ - **`agents devices` gains a `desktop` role and stops double-labeling your interactive box (PHNX-3392).** `agents devices role <name>` now accepts a third value — `desktop` — for a headed always-on box (the release/credential home, e.g. a Mac mini): it is excluded from `--device auto` like `personal` (agents never auto-land there), but authenticates from its own interactive login like `personal` rather than a worker setup-token. And `agents devices list` no longer prints both `★ interactive` and `personal` on the same row — a `personal` box IS your interactive seat, so the star is folded into the role (it still shows for a non-personal box pinned as `interactive.host`). Source: `cli/src/lib/device-config.ts`, `cli/src/commands/ssh.ts`.
68
+
69
+ - **The system DotAgents repo `phnx-labs/.agents-system` was renamed to `phnx-labs/.agents` on GitHub, and both names are now recognized as the system origin (PHNX-3394).** The `-system` suffix was redundant once the layering itself is the role. The clone target stays `gh:phnx-labs/.agents-system` (GitHub's rename redirect makes that resolve fine, so there is no forced cutover), but a checkout under either name is recognized as the system origin across all transport forms (ssh / https / scp-style), via a `RENAMED_REMOTE_ALIASES` fold in `canonicalGitRemote` and the pure `isSystemRepoRemote` behind `isSystemRepoOrigin`. The companion extras repo `phnx-labs/.agents-extras` keeps its name but is now private and opt-in (`agents repo add`, not cloned by default). Source: `cli/src/lib/git.ts`.
70
+
3
71
  ## 1.22.56
4
72
 
5
73
  - **Scheduled routines claim a durable per-occurrence slot, so a duplicate delivery or a catch-up of the same slot can't double-launch (PHNX-3215).** The forward-timer path keyed its single-fire claim on croner's `currentRun()`, which is the jittered wall-clock fire instant, not the aligned schedule boundary — so two timer callbacks for one occurrence minted different run ids and both launched, and a live fire never collided with its catch-up twin. The scheduler now floors the fire to the aligned `(routine, scheduledFor)` boundary (`fireSlot`/`alignedSlotForFire`), the same derivation catch-up uses, making the atomic slot claim a structural guarantee (SING-15/16, self-overlap SING-13). Source: `cli/src/lib/scheduler.ts`, `cli/src/lib/scheduling/routines.ts`, `cli/src/lib/overdue.ts`.
6
74
 
7
75
  - **Managed-share Open Graph cards render in Cloudflare instead of failing on every new publish (PHNX-2835).** The Worker upload now carries Yoga and resvg as compiled WebAssembly modules alongside the bundled JavaScript and fonts. The previous single-file bundle decoded the WASM into byte arrays and compiled it at request time; Node accepted that in tests, but Cloudflare workerd forbids runtime WASM code generation, so a new share's lazy `<slug>.png` render failed. A real-workerd regression test now renders and validates the PNG, and a genuine renderer failure returns a diagnostic `500` instead of falling through as a missing cover. Source: `cli/src/lib/share/{worker-template,provision}.ts`.
8
76
 
77
+ - **A dangling per-harness default account no longer hard-fails every launch (PHNX-3326).** When `agents.yaml` `accounts.defaults.<agent>` pointed at an account id absent from the local registry, `resolveSpawnAccount` threw `Unknown account '<uuid>'` and the launch exited before doing any work. The default is now a preference, not a requirement: a stale pointer emits a warning naming the harness and the remedy (`agents accounts clear-default <agent>`), then falls back to the existing balanced-selection path. `accounts set-default`/`switch` now store the portable account name instead of the per-device uuid; legacy uuid entries continue to resolve, and `accounts remove`/`switch` correctly treat both name- and id-stored defaults as references. Source: `cli/src/lib/account-registry.ts`, `cli/src/commands/accounts.ts`.
78
+
9
79
  ## 1.22.55
10
80
 
11
81
  - **Managed shares generate their Open Graph cover server-side (PHNX-2835).** Publishing HTML to `share.agents-cli.sh` no longer launches local Chromium. The Worker lazily renders a deterministic 1200×630 AGI card with Satori and resvg-wasm, bundles Inter and JetBrains Mono, inherits the canonical page's visibility gate, and caches the PNG in R2. BYO endpoints keep their local screenshot fallback. An explicit missing `AGENTS_SHARE_BROWSER` or `PUPPETEER_EXECUTABLE_PATH` now fails loudly instead of silently falling through. Source: `cli/src/lib/share/{capture,publish,worker-template}.ts`.
package/README.md CHANGED
@@ -11,7 +11,7 @@
11
11
  <a href="https://github.com/phnx-labs/agi-cli"><img src="https://img.shields.io/badge/github-phnx--labs%2Fagi--cli-blue?style=flat-square" alt="github" /></a>
12
12
  </p>
13
13
 
14
- **A framework for running a distributed agent factory.** Dispatch Claude, Codex, Antigravity, Grok, and more across your own machines, in parallel, on your existing subscriptions. Measure every run with `agents perf` / `agents insights`, fold what you learn back into `AGENTS.md` and skills, then put the loop on a schedule with routines and monitors. Spawn parallel teams in isolated terminals or dispatch to the cloud for a PR. Watch live state across the fleet, nudge stalled runs, and message agents mid-flight. Store secrets behind Touch ID, drive real browsers and Electron apps, and steer the whole fleet from a menu bar — all from one CLI.
14
+ **A framework for running a distributed agent factory.** Dispatch Claude, Codex, Antigravity, Grok, and more across your own machines, in parallel, on your existing subscriptions. Measure every run with `agents insights` (latency lives at `agents insights perf`), fold what you learn back into `AGENTS.md` and skills, then put the loop on a schedule with routines and monitors. Spawn parallel teams in isolated terminals or dispatch to the cloud for a PR. Watch live state across the fleet, nudge stalled runs, and message agents mid-flight. Store secrets behind Touch ID, drive real browsers and Electron apps, and steer the whole fleet from a menu bar — all from one CLI.
15
15
 
16
16
  <p align="center">
17
17
  <a href="https://github.com/anthropics/claude-code" title="Claude Code"><img src="assets/harnesses/anthropic.svg" height="32" alt="Claude Code" /></a>
@@ -110,7 +110,7 @@ agents teams add checkout codex "Write tests for the new code" --name qa --afte
110
110
  agents teams start checkout --watch
111
111
 
112
112
  # Measure what happened -- latency, friction, dead-weight skills
113
- agents perf commands --days 7 # slowest CLI entrypoints
113
+ agents insights perf commands --days 7 # slowest CLI entrypoints
114
114
  agents insights --since 30d # friction, harness comparison, ranked actions
115
115
 
116
116
  # Fold the lesson back into the harness -- every agent picks it up next run
@@ -125,7 +125,7 @@ agents routines add nightly-payments-audit \
125
125
  agents menubar setup
126
126
  ```
127
127
 
128
- `agents perf` reads a disposable warehouse at `~/.agents/.cache/perf/perf.db` -- hook, command, and run timing rollups, deletable any time. `agents insights` (alias `agents sessions insights`) is deterministic and offline: it caches per-session facets, compares harnesses, and ranks actions by evidence count -- no model call unless you pass `--narrative`. Routines put any of this on a cron ([Routines](#routines)); monitors fire it on a change instead of a clock ([Monitors](#monitors)); the menu bar is the always-on control surface for the fleet these commands drive ([Menu bar](#menu-bar)).
128
+ `agents insights perf` reads a disposable warehouse at `~/.agents/.cache/perf/perf.db` -- hook, command, and run timing rollups, deletable any time. `agents insights` (alias `agents sessions insights`) is deterministic and offline: it caches per-session facets, compares harnesses, and ranks actions by evidence count -- no model call unless you pass `--narrative`. Routines put any of this on a cron ([Routines](#routines)); monitors fire it on a change instead of a clock ([Monitors](#monitors)); the menu bar is the always-on control surface for the fleet these commands drive ([Menu bar](#menu-bar)).
129
129
 
130
130
  ---
131
131
 
@@ -1407,7 +1407,7 @@ Two repos with the same shape, different roles:
1407
1407
 
1408
1408
  | Repo | Role | Owner |
1409
1409
  |---|---|---|
1410
- | `~/.agents-system/` | **System repo** — core/built-in skills, commands, hooks, rules, MCP configs, permissions, and profiles that ship with `agi-cli`. The defaults every install gets. | Maintained upstream at [phnx-labs/.agents-system](https://github.com/phnx-labs/.agents-system) |
1410
+ | `~/.agents-system/` | **System repo** — core/built-in skills, commands, hooks, rules, MCP configs, permissions, and profiles that ship with `agi-cli`. The defaults every install gets. | Maintained upstream at [phnx-labs/.agents](https://github.com/phnx-labs/.agents) (formerly `.agents-system`) |
1411
1411
  | `~/.agents/` | **User repo** — your personal additions and overrides. This is what `agents repo push`/`pull` syncs. | You |
1412
1412
 
1413
1413
  **Version pinning:** `agents.yaml` at project root pins which agent version to use (like `.nvmrc` for Node).
package/dist/bootstrap.js CHANGED
@@ -29,7 +29,7 @@ const __dirname = path.dirname(fileURLToPath(import.meta.url));
29
29
  const packageJsonPath = path.join(__dirname, '..', 'package.json');
30
30
  const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf-8'));
31
31
  const VERSION = packageJson.version;
32
- import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, } from './lib/self-update.js';
32
+ import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, sweepStaleInstallStaging, } from './lib/self-update.js';
33
33
  import { registerUpgradeCommand } from './commands/upgrade.js';
34
34
  // Detect dev/working-tree builds and default the noisy startup steps off.
35
35
  // Three cases trip this:
@@ -156,7 +156,9 @@ program.hook('postAction', (_thisCommand, actionCommand) => {
156
156
  }).catch(() => { });
157
157
  }
158
158
  // Disposable perf warehouse — fail-soft spool append (no SQLite on this path).
159
- if (durationMs !== undefined && parts[0] !== 'perf') {
159
+ // Skip the perf reader itself (now `agents insights perf`, PHNX-3391) so it
160
+ // never records its own latency into the board it prints.
161
+ if (durationMs !== undefined && !(parts[0] === 'insights' && parts[1] === 'perf')) {
160
162
  // sessionId/agent are resolvable here the same way emit() resolves them
161
163
  // for command.start/command.end above (the shared provenance floor,
162
164
  // event-provenance.ts) — without this, every command.end perf sample
@@ -319,6 +321,13 @@ async function installResolvedPackage(metadata) {
319
321
  // trusted .tgz. A mismatch throws and nothing below runs — fail closed.
320
322
  const tarball = await downloadVerifiedTarball(metadata.tarball, metadata.integrity);
321
323
  try {
324
+ // Clear any orphaned npm reify staging dir from a prior crashed upgrade
325
+ // BEFORE the package manager stages the new one (PHNX-3393) — otherwise
326
+ // npm's rename onto that exact, deterministic path fails ENOTEMPTY and
327
+ // every subsequent upgrade dead-ends there forever. bun does not use
328
+ // npm's retire-path staging scheme, so this only needs to run once, ahead
329
+ // of both package-manager branches below.
330
+ sweepStaleInstallStaging(packageRoot);
322
331
  // Upgrade with the package manager that owns this install. A bun global
323
332
  // install lives at <bunGlobalDir>/node_modules/... (no `lib` segment), so an
324
333
  // `npm install --prefix` would write to <bunGlobalDir>/lib/node_modules and
@@ -65,7 +65,6 @@ export declare const loadRefreshRules: ModuleLoader;
65
65
  export declare const loadFactory: ModuleLoader;
66
66
  export declare const loadInsights: ModuleLoader;
67
67
  export declare const loadTrace: ModuleLoader;
68
- export declare const loadPerf: ModuleLoader;
69
68
  export declare const loadPty: ModuleLoader;
70
69
  export declare const loadTmux: ModuleLoader;
71
70
  export declare const loadWatchdog: ModuleLoader;
@@ -67,7 +67,6 @@ export const loadRefreshRules = async () => (await import('../commands/refresh-r
67
67
  export const loadFactory = async () => (await import('../commands/factory.js')).registerFactoryCommands;
68
68
  export const loadInsights = async () => (await import('../commands/insights.js')).registerInsightsCommand;
69
69
  export const loadTrace = async () => (await import('../commands/sessions-trace.js')).registerTraceCommand;
70
- export const loadPerf = async () => (await import('../commands/perf.js')).registerPerfCommand;
71
70
  export const loadPty = async () => (await import('../commands/pty.js')).registerPtyCommands;
72
71
  export const loadTmux = async () => (await import('../commands/tmux.js')).registerTmuxCommands;
73
72
  export const loadWatchdog = async () => (await import('../commands/watchdog.js')).registerWatchdogCommand;
@@ -142,7 +141,6 @@ export const COMMAND_LOADERS = {
142
141
  workflows: [loadWorkflows],
143
142
  add: [loadVersions],
144
143
  use: [loadVersions],
145
- list: [loadVersions],
146
144
  remove: [loadVersions],
147
145
  rm: [loadVersions],
148
146
  purge: [loadVersions],
@@ -177,7 +175,6 @@ export const COMMAND_LOADERS = {
177
175
  // `agents trace` is a top-level alias of `agents sessions trace` (precedent:
178
176
  // `agents insights` aliases `agents sessions insights`). One implementation.
179
177
  trace: [loadTrace],
180
- perf: [loadPerf],
181
178
  pty: [loadPty],
182
179
  tmux: [loadTmux],
183
180
  watchdog: [loadWatchdog],
@@ -342,13 +342,17 @@ export function setDefaultAccount(agentRaw, name) {
342
342
  }
343
343
  assertNativeAccountNameable(account.agent);
344
344
  }
345
- updateMeta(meta => ({ ...meta, accounts: { ...meta.accounts, defaults: { ...meta.accounts?.defaults, [agent]: account.id } } }));
345
+ // Reference by NAME, not id: defaults sync fleet-wide with `agents repo push/pull`
346
+ // while account ids are minted per-device, so an id ref breaks on every other
347
+ // machine ("Unknown account '<uuid>'"). Names are the portable handle — the
348
+ // registry resolves both, and existing uuid entries still resolve.
349
+ updateMeta(meta => ({ ...meta, accounts: { ...meta.accounts, defaults: { ...meta.accounts?.defaults, [agent]: account.name } } }));
346
350
  return { agent, account };
347
351
  }
348
352
  async function switchAccountRows(agent) {
349
353
  const accounts = await listSwitchableAccounts(agent);
350
354
  const candidates = await collectRunCandidates(agent);
351
- const defaultId = readMeta().accounts?.defaults?.[agent];
355
+ const defaultValue = readMeta().accounts?.defaults?.[agent];
352
356
  return accounts.map(account => {
353
357
  const candidate = account.kind === 'native'
354
358
  ? candidates.find(row => row.accountKey === account.identityKey) ?? null
@@ -357,7 +361,7 @@ async function switchAccountRows(agent) {
357
361
  accountName: account.name,
358
362
  kind: account.kind,
359
363
  detail: account.kind === 'provider' ? account.provider : (account.identityLabel ?? account.identityKey),
360
- current: account.id === defaultId,
364
+ current: account.id === defaultValue || account.name === defaultValue,
361
365
  candidate,
362
366
  };
363
367
  });
@@ -16,7 +16,7 @@ import { machineId } from '../lib/session/sync/config.js';
16
16
  import { loadDevices } from '../lib/devices/registry.js';
17
17
  import { isHostPinned, managedKnownHostsPath } from '../lib/devices/known-hosts.js';
18
18
  import { ensureDevicesRegistered } from '../lib/devices/sync.js';
19
- import { readFleetFile, resolveDesired } from '../lib/fleet/manifest.js';
19
+ import { readFleetFile, resolveDesired, emptyTargetsMessage } from '../lib/fleet/manifest.js';
20
20
  import { snapshotAuth } from '../lib/fleet/auth-sync.js';
21
21
  import { AUTH_BUNDLE_NAME } from '../lib/secrets/bundles.js';
22
22
  import { agentIdOf, diffFleet, probeDevice, runFleetApply, pool, sourceHome, expandAllSpecs, rosterNeedsVersions, fleetSecretsBundles, } from '../lib/fleet/apply.js';
@@ -188,7 +188,15 @@ async function runApply(opts) {
188
188
  if (opts.login === false)
189
189
  desired = desired.map((d) => ({ ...d, login: 'skip' }));
190
190
  if (desired.length === 0) {
191
- console.log(chalk.gray('No target devices — nothing to apply.'));
191
+ const msg = emptyTargetsMessage(manifest);
192
+ if (msg.style === 'hint') {
193
+ console.log(chalk.yellow(msg.lines[0]));
194
+ for (const line of msg.lines.slice(1))
195
+ console.log(chalk.gray(` ${line}`));
196
+ }
197
+ else {
198
+ console.log(chalk.gray(msg.lines[0]));
199
+ }
192
200
  return;
193
201
  }
194
202
  // Snapshot source auth once for every agent named anywhere in the profile.
@@ -431,7 +431,7 @@ async function handleTerminalHandoff(agentSpec, options, prompt) {
431
431
  if (!knownAgent) {
432
432
  const probeCwd = options.cwd ?? process.cwd();
433
433
  if (!hasProfile && !resolveWorkflowRef(rawTarget, probeCwd)) {
434
- console.error(chalk.red(`Unknown agent, profile, or workflow: ${rawTarget}. See \`agents list\` for the installed harnesses.`));
434
+ console.error(chalk.red(`Unknown agent, profile, or workflow: ${rawTarget}. See \`agents view\` for the installed harnesses.`));
435
435
  process.exit(1);
436
436
  }
437
437
  }
@@ -1,19 +1,32 @@
1
- /**
2
- * `agents sessions fork <session>` — branch an existing conversation into a new,
3
- * independent session you can continue separately. The original is untouched.
4
- * Also exposed as the hidden top-level alias `agents fork` (back-compat).
5
- *
6
- * Thin command layer; the copy/register logic lives in `lib/session/fork.ts`.
7
- */
8
1
  import type { Command } from 'commander';
9
2
  interface ForkOptions {
10
3
  name?: string;
4
+ device?: string;
5
+ /** Open the sibling in a real terminal tab instead of in-place; optional backend. */
6
+ terminal?: string | boolean;
7
+ }
8
+ /**
9
+ * The two process boundaries fork crosses — a preview subprocess (cross-fleet
10
+ * resolve + digest) and the sibling launch. Injectable so the resolve→recap→run
11
+ * argv logic is unit-tested without spawning real CLIs.
12
+ */
13
+ export interface ForkDeps {
14
+ /** Run `agents sessions preview <sub…>` and capture stdout + exit status. */
15
+ runPreview: (sub: string[]) => {
16
+ status: number | null;
17
+ stdout: string;
18
+ };
19
+ /** Launch `agents <sub…>` inheriting stdio; returns its exit status. */
20
+ launch: (sub: string[]) => {
21
+ status: number | null;
22
+ };
11
23
  }
12
24
  /**
13
- * Resolve the source session, copy it under a fresh id, and print how to
14
- * continue the fork. Shared by `agents sessions fork` and the `agents fork` alias.
25
+ * Resolve the source cross-fleet, build a recap from its preview digest, and
26
+ * launch a same-harness sibling seeded with that recap. Shared by
27
+ * `agents sessions fork` and the `agents fork` alias.
15
28
  */
16
- export declare function runFork(sessionArg: string, options: ForkOptions): Promise<void>;
29
+ export declare function runFork(sessionArg: string, options: ForkOptions, deps?: ForkDeps): Promise<void>;
17
30
  /**
18
31
  * Register `agents sessions fork <session>` — the canonical surface (fork is a
19
32
  * session operation, so it lives under the `sessions` group).
@@ -1,79 +1,132 @@
1
+ /**
2
+ * `agents sessions fork <session>` — branch an existing conversation into a new,
3
+ * independent same-harness sibling, seeded with a recap so it picks up where the
4
+ * original left off. The original is untouched. Also exposed as the hidden
5
+ * top-level alias `agents fork` (back-compat).
6
+ *
7
+ * The source is resolved CROSS-FLEET (the same path `agents sessions preview`
8
+ * uses), so a session that lives on another device forks fine — the sibling is
9
+ * handed a plain-text recap as its opening input and never has to reach the
10
+ * source transcript. Because the seed is text, any REPL harness can be forked,
11
+ * not just Claude.
12
+ *
13
+ * Thin command layer; the pure recap text lives in `lib/session/fork.ts`.
14
+ */
15
+ import { spawnSync } from 'child_process';
1
16
  import chalk from 'chalk';
2
17
  import { setHelpSections } from '../lib/help.js';
3
- import { findSessionsById } from '../lib/session/db.js';
4
- import { discoverSessions } from '../lib/session/discover.js';
5
- import { forkSession, isForkableAgent, FORKABLE_AGENTS } from '../lib/session/fork.js';
18
+ import { getCliLaunch } from '../lib/cli-entry.js';
19
+ import { buildForkRecap, forkLabelFor } from '../lib/session/fork.js';
20
+ function defaultDeps() {
21
+ return {
22
+ runPreview: (sub) => {
23
+ const p = getCliLaunch(['sessions', 'preview', ...sub]);
24
+ // stderr inherited so preview's own resolution errors reach the user verbatim.
25
+ const r = spawnSync(p.command, p.args, { encoding: 'utf-8', stdio: ['ignore', 'pipe', 'inherit'] });
26
+ return { status: r.status, stdout: r.stdout ?? '' };
27
+ },
28
+ launch: (sub) => {
29
+ const l = getCliLaunch(sub);
30
+ // In-place stdio so the sibling takes over this terminal.
31
+ const r = spawnSync(l.command, l.args, { stdio: 'inherit' });
32
+ return { status: r.status };
33
+ },
34
+ };
35
+ }
6
36
  const FORK_HELP = {
7
37
  examples: `
8
- # Fork a session by (partial) id, then continue the fork
38
+ # Fork a session by (partial) id — launches a same-harness sibling seeded with a recap
9
39
  agents sessions fork 4f3a9c21
10
- agents sessions resume <new-id>
11
40
 
12
- # Give the fork a name
41
+ # Name the fork's session label
13
42
  agents sessions fork 4f3a9c21 --name "try redis instead"
43
+
44
+ # Place the sibling on a fleet worker instead of here
45
+ agents sessions fork 4f3a9c21 --device auto
46
+
47
+ # Open the sibling in a fresh terminal tab where you work
48
+ agents sessions fork 4f3a9c21 --terminal
14
49
  `,
15
50
  notes: `
16
- - 'resume' continues the SAME conversation; 'fork' copies it under a new id so the two diverge.
17
- - The fork is a full copy of the conversation so far; continuing it never touches the original.
18
- - Resolve the session the same way as resume: an exact or prefix id fragment.
19
- - Native copy currently supports: ${FORKABLE_AGENTS.join(', ')}. For other harnesses, branch by
20
- starting a fresh agent and seeding it with '/continue <id>' — the source stays put.
51
+ - 'resume' continues the SAME conversation; 'fork' launches a NEW same-harness
52
+ session seeded with a recap of the source, so the two diverge.
53
+ - Works cross-device and cross-harness: the source is resolved across the fleet
54
+ and the sibling gets a plain-text recap, so it never reaches the source transcript.
55
+ - The recap carries the source id — the sibling can run '/continue <id>' for the
56
+ full history if it needs more than the recap.
57
+ - Resolve the source the same way as resume: an exact or prefix id fragment.
21
58
  `,
22
59
  };
23
60
  /**
24
- * Resolve the source session, copy it under a fresh id, and print how to
25
- * continue the fork. Shared by `agents sessions fork` and the `agents fork` alias.
61
+ * Resolve the source cross-fleet, build a recap from its preview digest, and
62
+ * launch a same-harness sibling seeded with that recap. Shared by
63
+ * `agents sessions fork` and the `agents fork` alias.
26
64
  */
27
- export async function runFork(sessionArg, options) {
28
- // Resolve the source. Try the index first; only pay for a rescan if the id
29
- // isn't found yet (mirrors the resume path's freshen-then-lookup).
30
- let matches = findSessionsById(sessionArg, {});
31
- if (matches.length === 0) {
32
- await discoverSessions({});
33
- matches = findSessionsById(sessionArg, {});
34
- }
35
- if (matches.length === 0) {
36
- // Errors go to stderr and set a non-zero exit code so a script chaining on
37
- // `agents sessions fork <id> && agents sessions resume <new>` doesn't proceed
38
- // on a failed fork.
39
- console.error(chalk.red(`No session matching "${sessionArg}".`));
40
- console.error(chalk.gray('List candidates with: agents sessions'));
65
+ export async function runFork(sessionArg, options, deps = defaultDeps()) {
66
+ // Resolve + digest in one cross-fleet hop by shelling the existing preview
67
+ // verb: it resolves the id across the fleet (SSH fan-out + peer hop), computes
68
+ // the digest on the OWNING device, and prints it as JSON — so a remote source
69
+ // resolves fine and we never re-implement resolution or digesting here.
70
+ // --terminal opens a tab on THIS machine; --device dispatches over SSH. `agents
71
+ // run` refuses the combination, so reject it here with a fork-specific message
72
+ // before resolving anything, rather than after an overpromising progress line.
73
+ if (options.device && options.terminal !== undefined) {
74
+ console.error(chalk.red('Pick one placement: --terminal opens a tab here; --device places the sibling on another box. They cannot combine.'));
41
75
  process.exitCode = 1;
42
76
  return;
43
77
  }
44
- if (matches.length > 1) {
45
- console.error(chalk.yellow(`"${sessionArg}" is ambiguous — ${matches.length} sessions match. Use a longer id:`));
46
- for (const m of matches.slice(0, 8)) {
47
- console.error(chalk.gray(` ${m.shortId} ${m.agent} ${m.label || m.topic || ''}`));
48
- }
49
- process.exitCode = 1;
78
+ const res = deps.runPreview([sessionArg, '--json']);
79
+ if (res.status !== 0) {
80
+ // preview already explained why on stderr; propagate its exit code.
81
+ process.exitCode = res.status ?? 1;
50
82
  return;
51
83
  }
52
- const source = matches[0];
53
- if (!isForkableAgent(source.agent)) {
54
- // A native copy needs the agent's transcript to be resumable by id; harnesses
55
- // without that can still be branched by hand. Fail loud with the manual path
56
- // rather than a silent no-op or a fake copy.
57
- console.error(chalk.yellow(`A native fork copy isn't supported for ${source.agent} sessions yet (supported: ${FORKABLE_AGENTS.join(', ')}).`));
58
- console.error(chalk.gray(` Branch it by hand — start a fresh ${source.agent} and seed it with the source's context:`));
59
- console.error(chalk.gray(` agents run ${source.agent} --terminal # then, in the new session:`));
60
- console.error(chalk.gray(` /continue ${source.shortId}`));
84
+ let data;
85
+ try {
86
+ data = JSON.parse(res.stdout);
87
+ }
88
+ catch {
89
+ console.error(chalk.red(`Could not read the source session for "${sessionArg}".`));
61
90
  process.exitCode = 1;
62
91
  return;
63
92
  }
64
- let result;
65
- try {
66
- result = forkSession(source, { name: options.name });
67
- }
68
- catch (err) {
69
- console.error(chalk.red(`Could not fork ${source.shortId}: ${err.message}`));
93
+ const source = data?.session;
94
+ if (!source?.id || !source?.agent) {
95
+ console.error(chalk.red(`Could not resolve a forkable source for "${sessionArg}".`));
70
96
  process.exitCode = 1;
71
97
  return;
72
98
  }
73
- console.log(chalk.green(`Forked ${source.shortId} -> ${result.shortId}`));
74
- console.log(chalk.gray(` Label: ${result.label}`));
75
- console.log(chalk.gray(` Continue: agents sessions resume ${result.shortId}`));
76
- console.log(chalk.gray(` Original ${source.shortId} is untouched.`));
99
+ const digest = data?.preview ?? undefined;
100
+ // Most sessions have no explicit --name label; fall back to the auto-derived
101
+ // topic the rest of the CLI shows, not the raw short id (forkLabelFor is the
102
+ // shared 3-tier resolver, and preview's --json now carries `topic`).
103
+ const label = forkLabelFor({ label: source.label, topic: source.topic, shortId: source.shortId });
104
+ const recap = buildForkRecap({
105
+ agent: source.agent,
106
+ label,
107
+ cwd: source.cwd,
108
+ ticketId: source.ticketId,
109
+ machine: source.machine,
110
+ shortId: source.shortId,
111
+ id: source.id,
112
+ lastAssistant: digest?.lastAssistant,
113
+ changes: digest?.changes,
114
+ });
115
+ // Launch a NEW same-harness session, load-balanced across accounts (balanced),
116
+ // seeded with the recap as its opening input. Runs here by default; --device
117
+ // places it on the fleet; --terminal opens it in a fresh tab where the user works.
118
+ const runArgs = ['run', source.agent, recap, '-i', '--strategy', 'balanced', '--name', options.name || `fork of ${label}`];
119
+ if (options.device)
120
+ runArgs.push('--device', options.device);
121
+ if (options.terminal !== undefined) {
122
+ runArgs.push('--terminal');
123
+ if (typeof options.terminal === 'string')
124
+ runArgs.push(options.terminal);
125
+ }
126
+ const where = options.device ? ` on ${options.device}` : options.terminal !== undefined ? ' in a new terminal' : '';
127
+ console.error(chalk.gray(`Forking ${source.shortId} → new ${source.agent} session${where}, seeded with a recap…`));
128
+ const child = deps.launch(runArgs);
129
+ process.exitCode = child.status ?? 0;
77
130
  }
78
131
  /**
79
132
  * Register `agents sessions fork <session>` — the canonical surface (fork is a
@@ -82,10 +135,12 @@ export async function runFork(sessionArg, options) {
82
135
  export function registerSessionsForkCommand(sessionsCmd) {
83
136
  const cmd = sessionsCmd
84
137
  .command('fork <session>')
85
- .description('Branch a session into a new, independent copy you can continue separately. The original is untouched.')
86
- .option('--name <label>', 'Label for the fork (default: "fork of <original>")');
138
+ .description('Branch a session into a new same-harness sibling, seeded with a recap so it continues the work. The original is untouched.')
139
+ .option('--name <label>', 'Session label for the fork (default: "fork of <original>")')
140
+ .option('--device <host>', 'Place the sibling on a fleet device (name or "auto"); defaults to here')
141
+ .option('--terminal [backend]', 'Open the sibling in a real terminal tab (iterm | ghostty | terminal | tmux | vscodium-agent) instead of in-place');
87
142
  setHelpSections(cmd, FORK_HELP);
88
- cmd.action(runFork);
143
+ cmd.action((session, options) => runFork(session, options));
89
144
  }
90
145
  /**
91
146
  * Register the hidden top-level `agents fork` alias. Kept working for back-compat
@@ -94,7 +149,9 @@ export function registerSessionsForkCommand(sessionsCmd) {
94
149
  export function registerForkCommand(program) {
95
150
  const cmd = program
96
151
  .command('fork <session>', { hidden: true })
97
- .description('Alias for `agents sessions fork` — branch a session into a new, independent copy.')
98
- .option('--name <label>', 'Label for the fork (default: "fork of <original>")');
99
- cmd.action(runFork);
152
+ .description('Alias for `agents sessions fork` — branch a session into a new same-harness sibling.')
153
+ .option('--name <label>', 'Session label for the fork (default: "fork of <original>")')
154
+ .option('--device <host>', 'Place the sibling on a fleet device (name or "auto"); defaults to here')
155
+ .option('--terminal [backend]', 'Open the sibling in a real terminal tab (iterm | ghostty | terminal | tmux | vscodium-agent) instead of in-place');
156
+ cmd.action((session, options) => runFork(session, options));
100
157
  }
@@ -594,22 +594,22 @@ Examples:
594
594
  .addHelpText('after', `
595
595
  Shows aggregated stats for every hook that fired through a generated shim.
596
596
  Primary source: disposable SQLite warehouse ~/.agents/.cache/perf/perf.db
597
- (same data as \`agents perf hooks\`). Falls back to the legacy daily JSONL under
597
+ (same data as \`agents insights perf hooks\`). Falls back to the legacy daily JSONL under
598
598
  ~/.agents/.cache/logs/ when the warehouse is empty.
599
599
 
600
600
  Examples:
601
601
  agents hooks profile # last 7 days, table form
602
602
  agents hooks profile --days 30 # roll up the full month
603
603
  agents hooks profile --json | jq # pipe somewhere
604
- agents perf hooks # same rollup under the perf surface
604
+ agents insights perf hooks # same rollup under the perf surface
605
605
 
606
606
  A hook whose p99 exceeds --warn-ms gets flagged in the cache column. Add
607
607
  'cache: 5m' or 'cache: 5m-bg' to its hooks.yaml entry to fix it.
608
608
  `)
609
- .option('--project <key>', 'Scope to one project (see agents perf --help)')
609
+ .option('--project <key>', 'Scope to one project (see agents insights perf --help)')
610
610
  .action(async (options) => {
611
611
  const { DEFAULT_SLOW_HOOK_WARN_MS } = await import('../lib/hooks/profile.js');
612
- // Same rollup as `agents perf hooks` — this command predates the `perf`
612
+ // Same rollup as `agents insights perf hooks` — this command predates the `perf`
613
613
  // surface and is kept as a documented alias; delegate instead of
614
614
  // duplicating the SQLite-vs-legacy-JSONL fallback and table rendering.
615
615
  const { loadHookProfile, renderHookTable } = await import('./perf.js');