@phnx-labs/agents-cli 1.22.56 → 1.22.58
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -0
- package/README.md +4 -4
- package/dist/bootstrap.js +11 -2
- package/dist/cli/command-registry.d.ts +0 -1
- package/dist/cli/command-registry.js +0 -3
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/exec.js +1 -1
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/hooks.js +4 -4
- package/dist/commands/insights.d.ts +7 -5
- package/dist/commands/insights.js +16 -9
- package/dist/commands/monitors.js +11 -0
- package/dist/commands/perf.d.ts +16 -7
- package/dist/commands/perf.js +29 -20
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/rules.js +1 -1
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions.js +1 -0
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +24 -14
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/trash.d.ts +2 -2
- package/dist/commands/trash.js +2 -6
- package/dist/commands/versions.d.ts +2 -2
- package/dist/commands/versions.js +1 -10
- package/dist/commands/view.d.ts +2 -2
- package/dist/commands/view.js +7 -6
- package/dist/index.d.ts +1 -0
- package/dist/index.js +14 -0
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-ingest.d.ts +1 -0
- package/dist/lib/accounting/usage-ingest.js +75 -0
- package/dist/lib/accounting/usage-sync.d.ts +97 -0
- package/dist/lib/accounting/usage-sync.js +203 -0
- package/dist/lib/accounting/usage.d.ts +48 -2
- package/dist/lib/accounting/usage.js +79 -2
- package/dist/lib/agent-spec/agents.js +1 -1
- package/dist/lib/analytics/mix-commands.d.ts +8 -7
- package/dist/lib/analytics/mix-commands.js +50 -73
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/daemon/daemon.js +5 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +95 -53
- package/dist/lib/daemon/usage-sync-service.d.ts +21 -0
- package/dist/lib/daemon/usage-sync-service.js +42 -0
- package/dist/lib/daemon-services.d.ts +1 -1
- package/dist/lib/daemon-services.js +5 -0
- package/dist/lib/device-config.d.ts +17 -6
- package/dist/lib/device-config.js +25 -11
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/pool.d.ts +4 -3
- package/dist/lib/devices/pool.js +13 -5
- package/dist/lib/doctor-diff.js +77 -7
- package/dist/lib/exec.d.ts +6 -41
- package/dist/lib/exec.js +6 -41
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +13 -1
- package/dist/lib/git.js +36 -7
- package/dist/lib/harness/adapter.d.ts +7 -7
- package/dist/lib/harness/adapters/claude.js +3 -2
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/hosts/remote-cmd.d.ts +9 -0
- package/dist/lib/hosts/remote-cmd.js +22 -0
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +26 -133
- package/dist/lib/installations/versions.js +41 -204
- package/dist/lib/perf/db.d.ts +1 -1
- package/dist/lib/perf/db.js +1 -1
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +23 -0
- package/dist/lib/self-update.js +50 -0
- package/dist/lib/session/active.d.ts +16 -32
- package/dist/lib/session/active.js +10 -68
- package/dist/lib/session/db.d.ts +24 -36
- package/dist/lib/session/db.js +143 -44
- package/dist/lib/session/discover.d.ts +6 -58
- package/dist/lib/session/discover.js +5 -43
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/parse.d.ts +1 -19
- package/dist/lib/session/parse.js +2 -15
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/startup/command-registry.d.ts +8 -2
- package/dist/lib/startup/command-registry.js +12 -4
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +15 -0
- package/dist/lib/traces/sync.js +104 -19
- package/dist/lib/traces/worker-template.js +154 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,11 +1,81 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.22.58
|
|
4
|
+
|
|
5
|
+
- **A remote browser task survives a browser-daemon restart on the driving box (PHNX-2663).** When the local daemon restarted, its in-memory task map was rebuilt from disk for local-CDP profiles but **skipped `ssh://` tunnelled tasks** — so an agent driving a browser host over `--device` got "Unknown browser task" / "Tab not found" for a tab that was still alive on the far side, and gave up. `attachRunningProfile` now re-establishes the SSH tunnel and reconnects CDP for a tunnelled task during rehydrate (reusing `connectSSH`, which attaches to the already-running remote browser via `isOwnTunnel` — it never launches one), carrying the tunnel teardown so it can't leak across the next restart. A failed reconnect falls through to disk reconcile exactly like the local-CDP path. Source: `cli/src/lib/browser/service.ts`.
|
|
6
|
+
|
|
7
|
+
- **`agents prune cleanup` no longer over-reports source-present hooks as orphans (PHNX-2693).** Orphan detection diffed the hook files in a version home against the *registered-hook manifest*, but sync copies helper / test / benchmark scripts into every version home alongside registered hooks without registering them — so each one (e.g. `permission-handler`, `verify-work-state`, `*_test`, `benchmark_*`) read as an orphan, and `agents prune cleanup [hooks | --all]` offered to trash ~1000 in-use files across version homes. It now diffs against the resolved **source** set (user + system + enabled extras hook dirs), so a hook present in any configured source is never an orphan, exactly as the command's own help promised — while a genuinely dead file (present in a home, absent from every source) is still flagged. Same detector backs `agents doctor`'s `orphan` warning. Source: `cli/src/lib/hooks/install.ts`.
|
|
8
|
+
|
|
9
|
+
- **`agents prune cleanup` no longer offers to delete plugin-provided skills (PHNX-3185).** The skill orphan detector compared a version home's skills against `skills/` + `.system/skills/` + extras only, and never credited plugin-bundled skills (`plugins/<plugin>/skills`, `.system/plugins/<plugin>/skills`) — so every skill a plugin legitimately installed (`design`, `plan`, `review`, `blog`, …) read as an orphan. On a real machine that was ~95% false positives, and `agents doctor` framed the same set as `⚠ orphans … → agents prune cleanup --all`, so following the tool's own guidance destroyed the working skill library. The detector now credits plugin skill sources when deciding orphans, and the skill-source resolver finds them too so a plugin skill reads as up-to-date instead of drifted. A genuinely dead skill (no central, extra, or plugin source) is still flagged. Source: `cli/src/lib/plugins/skills.ts`, `cli/src/lib/staleness/writers/sources.ts`.
|
|
10
|
+
|
|
11
|
+
- **`agents sync` stops reporting `✓ reconciled` while the drift it was asked to fix stays put (PHNX-3186, PHNX-2955).** `agents sync status` over-reported four classes of phantom drift that the sync writer never creates and can never clear — so re-running sync forever printed success while a phantom "N drifted / N missing" stuck (42–46 drifted per version observed fleet-wide). `diffVersionResources` now mirrors what the writer actually installs: (1) directory docs (`README`/`AGENTS`/`CLAUDE`/`GEMINI`) in `commands/` are documentation, not commands, so they are no longer counted as missing/drifted/orphan commands; (2) on command-as-skill agents (Codex ≥ 0.117, Kimi) a command whose name collides with a real same-named skill is reported present (the skill wins the shared `skills/<name>/` slot and the wrapper is deliberately never written), not the phantom "missing"; (3) version-gated presence-only kinds (e.g. subagents below the harness floor) are capability-gated so a version that structurally cannot hold the resource is not reported missing it; (4) `diffRules` now finds an agent's non-`.md` instructions file (Cursor's `.cursorrules`), which an `.md`-only scan never saw, so the rules row stops false-reporting `missing`. On top of that, every `agents sync` now runs a **post-reconcile verification**: it re-reads each version it wrote into and, if any drift the reconcile was asked to fix survives, names the exact residual resources instead of printing a bare `✓` (the umbrella line reads `⚠ sync: reconcile INCOMPLETE`, and `--json` carries `residualDrift` with `ok:false`; the exit code stays 0, matching the declined-write precedent so the fleet fan-out never discards the residual payload). Source: `cli/src/lib/doctor-diff.ts`, `cli/src/lib/sync-status.ts`, `cli/src/lib/refresh.ts`, `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/sync.ts`.
|
|
12
|
+
|
|
13
|
+
- **A stale per-version plugin marketplace copy is now reported as drift instead of `everywhere`/`ok` (PHNX-2955).** `describePluginDrift` (the content-aware plugin diff behind `agents sync status` / `agents doctor` / the post-reconcile verification) compared only the *presence* of a mirror's skills/commands, so a skill whose bytes went stale — an edit pulled into `~/.agents/.system` but never refreshed into a version's `marketplaces/…` mirror — read as fully synced while agents ran the OLD skill text. It now content-compares each skill/command present in both central and the mirror and reports `stale skill: <name>` / `stale command: <name>`, so the drift is visible and a `--yes` reconcile (which rewrites the mirror) clears it. (Out of scope, left as follow-ups: the interactive `Already in sync` short-circuit does not yet consult this signal, and `agents plugins update` for marketplace-resolved plugins is unchanged.) Source: `cli/src/lib/doctor-diff.ts`.
|
|
14
|
+
|
|
15
|
+
- **`agents sync` on a box where `~/.agents` is not a git repo adopts it in place instead of silently no-op'ing the pull (PHNX-3239).** A partial install — runtime state present but no `.git` — made the umbrella repo-pull stage fail quietly while `agents sync` still reported success, so fleet dotfiles and resources never propagated and nothing said so. The umbrella now runs the same self-heal `agents sync user` does (`adoptUserRepoIfNeeded`): it materializes the tracked resource set from origin and continues, or fails loud with the reason (and the `agents repo pull user <git-url>` hint) when there is no recorded remote to adopt from — never a silent no-op that looks like success. Source: `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/status.ts`.
|
|
16
|
+
|
|
17
|
+
- **`agents browser sessions --profile <name>` surfaces the profile's real task store instead of an empty one (PHNX-3317).** A browser's live tasks and captures live under a composite runtime dir (`<name>@<device>`), but the sessions reader used the `--profile` value as a raw directory name and read the bare legacy `<name>/` dir — which is empty on a machine where the daemon writes the composite key — so a driving agent saw "no captures" / "Unknown browser task" while 13 real tasks sat one dir over. The reader now resolves a profile to every cache dir that belongs to it via `listProfileCacheDirs`/`keyBelongsToProfile` — the same single rule `status()` and `findTask` already use (RUSH-2709) — so the bare name and its `@<device>` stores are unified in the listing. Source: `cli/src/lib/browser/sessions-list.ts`.
|
|
18
|
+
|
|
19
|
+
- **Failure phenotype folded into the traces insight-engine fingerprint via a persisted per-session cache (PHNX-3327).** `computeInsights()` grouped cross-session failure patterns by `(tool, cause, normalized-error)` but did not carry a `phenotype` (false-termination / premature-completion / out-of-order / failure-to-act) — that classification needs the full derived trajectory, which flat `tool_calls` rows don't have. It is now computed per-session once (`classifyPhenotype`) and persisted in a new mtime+size-keyed `session_phenotypes` cache (same shape as `session_topics` / `session_insights`), then read for the **whole corpus** on every sync and folded in as a fourth grouping dimension. Reading the cache for all rows — never just this run's freshly-parsed batch — is what keeps two identically-signatured sessions in ONE failure cluster regardless of which incremental sync first saw each (the fragmentation an in-memory, batch-only map would cause). The `signature` output (`{ tool, cause, key }`) is unchanged; `FailurePattern` gains a `phenotype` field and the pattern id incorporates it, so the same signature under two phenotypes is two deep-linkable clusters. Source: `cli/src/lib/traces/insights.ts`, `cli/src/lib/traces/sync.ts`, `cli/src/lib/session/db.ts`.
|
|
20
|
+
|
|
21
|
+
- **`agents doctor` and `agents setup` no longer crash on macOS when the Keychain helper source is unavailable (PHNX-3385).** Both read the reserved `auth` secrets bundle as an optional status check; on a macOS box where the bundled/local Keychain helper couldn't be found (a source checkout, or a fresh npm install before `agents setup secrets` downloads the signed helper — RUSH-3100 dropped it from the tarball), that check threw and took the whole command down with it instead of reporting the finding as unavailable. Source: `cli/src/lib/secrets/bundles.ts` (`inspectReservedAuthBundle`), `cli/src/lib/auth-mint.ts` (`hasMintedSetupToken`).
|
|
22
|
+
|
|
23
|
+
- **`agents traces sync` reports a truthful run outcome — recover-then-succeed is `completed`, not `errored` (PHNX-3387).** `SessionDetail.meta.outcome` was derived as `errorCount > 0 ? 'errored' : 'completed'`, so a run that hit a tool failure but *recovered* and finished the task was mislabeled `errored` — and the Evals console's "green run with hidden tool failures" callout (the LangSmith trap) could never fire honestly because `surfacedToolFailures` only existed when the run was already `errored`. Outcome is now derived from a causal-recovery predicate (`recoveredAfterErrors`, shared with the false-termination phenotype in `phenotype.ts`): a run with tool errors is `completed` only when a substantive, non-human-facing tool step succeeded strictly after the last error **and resolved the failed work** — its work signature (the effective program for a shell step, the tool identity otherwise) matches an errored step's. A run that ended in error, punted to a human (`AskUserQuestion`), or whose only post-error success is unrelated work (a failed `bun test` followed by an incidental `ls`) stays `errored` — no regression. `surfacedToolFailures` is retained on `completed` runs so the console can honestly show "recovered from these failures." Source: `cli/src/lib/traces/sync.ts`, `cli/src/lib/traces/phenotype.ts`.
|
|
24
|
+
|
|
25
|
+
- **`--strategy balanced` no longer routes to a weekly-exhausted Claude account on worker boxes; a missing usage snapshot is treated as unknown (fails closed) instead of full capacity (PHNX-3392).** A worker's usage cache is empty (the setup-token lacks the `user:profile` scope, so `/api/oauth/usage` 403s — RUSH-2392), and `capacityWeight(null, …)` scored that absence as 100 — max weight — making a blind, actually-exhausted account the top pick. The null arm now draws `UNVERIFIED_WEIGHT` (1): any verified-healthy account outranks an unverifiable one, while an all-unverified pool still draws a pick. Pinned by spec GWT-E5c, a pure unit test (`capacity.test.ts`), and a real-cache-path regression test proving a persisted 7d-100% `week` window makes the account ineligible. Source: `cli/src/lib/accounting/capacity.ts`, `cli/src/lib/accounting/{capacity.test.ts,rotate.test.ts}`.
|
|
26
|
+
|
|
27
|
+
- **A worker pulling Claude usage from a Windows primary no longer silently blanks its cache (PHNX-3392 follow-up).** The `usage-sync` pull path (`pullUsageFromPrimary`) now wraps the primary's JSON in `stripClixml` before parsing, exactly as every other remote-JSON boundary does. Without it, a headed Windows box serving `__usage-export` prepends a PowerShell CLIXML progress banner to stdout, every pull fails as malformed JSON, the worker's usage cache stays null, and the capacity floor becomes the only thing keeping `--strategy balanced` from picking a blind, exhausted account. Source: `cli/src/lib/accounting/usage-sync.ts`.
|
|
28
|
+
|
|
29
|
+
- **`agents upgrade` self-heals the orphaned npm reify staging dir that dead-ended it forever (PHNX-3393).** npm stages an upgrade in a hidden sibling directory next to the running install, deterministically named from the install's path, then renames it into place; a crash mid-upgrade left that directory behind non-empty, and every subsequent upgrade's rename onto the same path failed `ENOTEMPTY` — recurring across boxes with no re-run ever clearing it. `agents upgrade` now sweeps any such orphan before reifying, and `agents doctor --fix` / the daemon's periodic self-heal do the same fleet-wide (with a 10-minute age guard so a concurrent in-flight upgrade is never touched). Source: `cli/src/lib/self-update.ts` (`sweepStaleInstallStaging`), `cli/src/bootstrap.ts`, `cli/src/lib/self-heal/checks/install-staging.ts`.
|
|
30
|
+
|
|
31
|
+
- **The Evals console's "all agents" view now renders real fleet data (PHNX-3397).** The CLI writes traces per-device (`<userId>/<hostname>/…`), but the console asks for the fleet-wide `<userId>/all/…` view, which was never a stored key — so every request 404'd and the console fell back to sample data. The traces Worker now synthesizes `/all` on read: it lists the owner's device prefixes (one delimited list, not every session object) and merges their shards for `all/index.json`, and resolves `all/sessions/<id>.json` to whichever device holds that session. A single device is an exact passthrough; across devices, counts sum and failure patterns fold by id (median/latency are session-weighted approximations, exact when filtered to one device). No CLI change and no extra upload — the per-device shards already in R2 are the source. Owner auth is unchanged: `/all` still requires a Phoenix bearer whose id matches the path prefix. Source: `cli/src/lib/traces/worker-template.ts`.
|
|
32
|
+
|
|
33
|
+
- **`agents traces sync` no longer silently ships a stale Evals dashboard when the sessions DB is busy (PHNX-3401).** `buildIndexShard` writes freshly-classified topics/insights back into the local `sessions.db` while building the console index. If another process (e.g. the Rush app) holds that DB, the write waits out the 30s `busy_timeout` and throws `SQLITE_BUSY` — which used to escape the build and get swallowed by a bare `catch {}`, so per-session shards uploaded but the aggregated index (wasted-time, failure clusters, latency) never refreshed and the console sat stale while `sync` reported a clean run. The cache write-backs are now best-effort (they only warm a cache the shard doesn't read from), and a genuine index build/upload failure is surfaced as `SyncResult.indexError` plus a `⚠ console index not refreshed` warning instead of a silent success. Source: `cli/src/lib/traces/sync.ts`, `cli/src/commands/traces.ts`.
|
|
34
|
+
|
|
35
|
+
- **A release now redeploys the managed share OG-cover Worker so prod can't drift from the shipped template (PHNX-3403).** The Worker that renders share preview cards was deployed only by a manual `agents artifacts share update` run, decoupled from the release train — so a release could publish a CLI whose `worker-template.ts` changed while the deployed Worker stayed stale (the gap that made PHNX-2835 look shipped while every new share still 404'd its cover). `release.sh` now takes `--deploy-worker <auto|on|off>` (default `auto`): after publish, on the home base, it redeploys the Worker from the just-published CLI when the shipped template differs from what the endpoint deployed (`auto`), always (`on`), or never (`off`). The decision is a pure local render+hash via the new `agents artifacts share update --check` (needs no Cloudflare credentials), so an unchanged template is a cheap no-op; the promote preflight verifies the share endpoint and `cloudflare.com` token on the home base before publishing, and a deploy failure fails the release loud with the manual fallback. Source: `cli/scripts/release.sh`, `cli/scripts/promote-home-base-probe.sh`, `cli/src/commands/share.ts`.
|
|
36
|
+
|
|
37
|
+
- **A browser task created over `agents ssh <host> 'agents browser …'` is owned by the real caller, not `UNRESOLVED@<host>` (PHNX-3405).** The `agents ssh` browser-drive passthrough stamped the fleet-remote consent marker (`AGENTS_FLEET_REMOTE=1`, PHNX-3065) but never forwarded the caller's actor, so the peer's daemon re-resolved identity from its own empty env and collapsed every such task to `UNRESOLVED@<host>` — making the browser look contended by phantom owners. `markFleetRemote` (the single seam behind `buildSshInvocation`, the fleet passthrough, and `leasedBoxRemoteCmd`) now forwards `AGENTS_ACTOR*`/`GIT_*` alongside the marker, in both POSIX and PowerShell dialects, marker-first so the consent gate and idempotency guard are unchanged. This closes the same RUSH-2028 actor-provenance gap on the `agents ssh` seam that the `--device` dispatch already closed via `withActorEnv`. Source: `cli/src/lib/devices/connect.ts`.
|
|
38
|
+
|
|
39
|
+
- **Monitor `--run` actions now require an authenticated account and resolve routine hooks (PHNX-3406).** Claude routine selection verifies the balanced account with Claude's own local `auth status` instead of trusting stale identity metadata, then launches with the operator HOME plus that version's `CLAUDE_CONFIG_DIR`, matching ordinary `agents run` so both the native login and portable `~/...` hooks resolve. `agents monitors list` and its JSON output also surface the latest asynchronous action failure, and the docs clarify that `--run` creates a new headless conversation rather than resuming the caller. Source: `cli/src/lib/{sandbox,daemon/runner}.ts`, `cli/src/commands/monitors.ts`, `cli/docs/monitors.md`.
|
|
40
|
+
|
|
41
|
+
- **`agents sessions fork` now works cross-device and cross-harness (PHNX-3409).** Forking a session that lived on another box used to dead-end with `transcript not found` — fork resolved the source with a local-only lookup and copied the transcript off local disk, and it only worked for Claude. It now resolves the source across the fleet (the same path `agents sessions preview` uses) and launches a NEW same-harness session, load-balanced, seeded with a recap of the source (label, cwd, ticket, last state, changed-files, and the source id for `/continue`) — so a session on any device, in any REPL harness, forks and picks up where it left off. Add `--device <host>` to place the sibling on the fleet or `--terminal` to open it in a fresh tab; `--name` still labels it. Source: `cli/src/commands/fork.ts`, `cli/src/lib/session/fork.ts`.
|
|
42
|
+
|
|
43
|
+
- **`agents routines` now says WHY a routine failed, not just that it did (PHNX-3410).** The list, the interactive detail's Recent runs, and `agents routines logs` showed only the status enum (`failed`/`missed`/`skipped`) — so diagnosing a routine that keeps failing on `auth_failed: Please run /login`, a wedged active-run lock, a readiness block, or a missed slot meant drilling into the run dir. A new `runFailureReason(run)` maps the `RunMeta` already on disk (`errorMessage` / `readiness` / `skipReason` / `missed`) to one compact line, rendered inline in the `Last Status` cell, the detail's Recent runs, and a `reason:` line in the `logs` header. Visibility only — no scheduling behavior changes. Source: `cli/src/commands/routines.ts`.
|
|
44
|
+
|
|
45
|
+
- **The daemon warm-tick session indexer no longer re-derives an active non-Claude/Codex session's whole tool history every tick — the root cause behind the wedged event loop and browser-IPC stalls (PHNX-3411).** The symptom fix (above) made a wedged browser daemon fail loud; this removes what wedged it. The warm-tick indexer parsed AND re-redacted every tool call in a changed session on every tick, in `toolIndexMode: 'replace'`. claude/codex resumed from a stored parse offset; the other 11 harnesses (kimi, grok, opencode, antigravity, cursor, droid, …) did not, so an active large session on the interactive hub re-sanitized its entire tool history several times a second — synchronous work that blocked the daemon event loop, so `browser.sock` accepted connections it never answered. The full-file harness path is incremental now too: the tool ledger records how many events were folded plus the collector snapshot, and a later scan of the same append-only stream re-derives only the newly appended tool calls (`mode: 'append'`), falling back to a full re-scan for a first scan, a truncated/rewritten transcript, or an extractor-version bump. Measured on a 400-tool-call Grok session indexed once per tick, re-derive+redact work drops from 80,200 tool calls to 400 over 400 ticks (~201× less), with the index content byte-for-byte identical to a full re-parse. Source: `cli/src/lib/session/{db,tool-calls,tool-store}.ts`.
|
|
46
|
+
|
|
47
|
+
- **`agents browser` detects a wedged daemon instead of hanging on it (PHNX-3411).** A unix-socket `connect` succeeds at the kernel level even when the browser daemon's shared event loop is blocked, so `isDaemonReachable` (a bare accept probe) reported a busy-but-unresponsive daemon as healthy — `browser.sock` accepted every connection while the daemon never replied, and cross-device browser drives to a busy hub (e.g. zion) hung or surfaced a confusing `Timeout waiting for browser daemon socket`. Readiness is now a *reply*: before serving a verb the CLI requires the daemon to actually answer a `version` probe (`isDaemonResponsive`, retried so a transient GC pause never condemns a healthy daemon), and a reachable-but-wedged daemon fails loud — naming the blocked event loop, the daemon log, and the `agents browser stop --daemon` reset — instead of hanging (which also fixes `reconcileDaemonVersion`'s own untimed version probe). Also adds a critical-path regression guard proving the daemon's session-index warm tick resumes an actively-growing Claude session incrementally (per-tick cost tracks appended bytes, never a full O(session) re-parse) — the invariant whose violation would re-starve browser IPC. Source: `cli/src/lib/browser/ipc.ts`, `cli/src/lib/session/__tests__/active-session-warmtick-incremental.test.ts`.
|
|
48
|
+
|
|
49
|
+
- **A routine whose dispatch account is signed out now records a `blocked` run instead of a doomed `failed` one (PHNX-3415).** When the daemon fires a routine, it preflights the rotation-resolved account against the auth-health cache; if that account is provably signed out (`revoked` or `unconfigured` — the "OAuth revoked" / "Please run /login" cases), it records a terminal `blocked` / `agent_auth_failed` run carrying the exact re-login repair (`agents run <agent>@<v> -- login`) and spawns nothing — instead of launching a run that 401s, lands as `failed`, and burns a session (surfaced from the 2026-08-28 reliability investigation, where signed-out accounts were the top cause of routine failures). The check is cache-only and fails **open** on any still-usable or indeterminate verdict (`rate_limited`/`unverified`/`expired`/`error`/absent), so a stale probe never wedges a routine. Broadens and corrects the prior preflight, which only caught `revoked` and mislabeled it `failed`. Source: `cli/src/lib/routine-readiness.ts` (`fireTimeAuthReadiness`), `cli/src/lib/daemon/runner.ts`; spec RT-12.
|
|
50
|
+
|
|
51
|
+
- **`agents fleet apply` on an empty roster now tells you how to fix it (PHNX-3422).** When `fleet.devices` is `{}` (the default on a fresh box), `apply`/`apply --plan` used to print a gray `No target devices — nothing to apply.` that read like the command was broken. It now prints an actionable `fleet.devices is empty — nothing to converge.` plus the two ways to declare a roster (`fleet: { devices: all }` or a named map), and still exits 0. Other zero-target cases (a `devices: all` fleet with no online peers, or a named roster whose every device was unresolved — off-tailnet/ignored/typo) keep the plain note, since their reason was already surfaced above. Source: `cli/src/lib/fleet/manifest.ts` (`emptyTargetsMessage`), `cli/src/commands/apply.ts`.
|
|
52
|
+
|
|
53
|
+
- **Traces now attributes a failed tool call's OWN blocking duration to wasted time, not just the bounded gap to the next call (PHNX-3437).** `tool_calls` gains an `end_timestamp` column (the call's result-record time, already available at ingestion) so the insight engine books `end - start` — the time a call actually blocked — as its wasted time, independent of whether another call follows. Before, a call that hung for minutes and then failed registered as ~0 waste whenever it was the last call in its session or was followed quickly by an unrelated call. This is what makes a fail-fast fix measurable on the Evals console: a channel that stops hanging ~5.5 minutes on stdin and instead fails in under a second (PHNX-3407) drops from ~5.5 minutes of attributed waste to ~0. The existing 30-minute-bounded inter-call gap heuristic is kept as the fallback for rows an older extractor stored (NULL end), so nothing crashes or yields NaN, and — when the end time is known — the post-call retry/stall gap is measured from the call's END so the blocking duration is never double-counted. `wastedMs`/`wastedMsTotal`/`failurePatterns` shard shapes are unchanged. Source: `cli/src/lib/session/{db.ts,tool-calls.ts,tool-store.ts}`, `cli/src/lib/traces/{sync.ts,insights.ts}`.
|
|
54
|
+
|
|
55
|
+
- **`agents view`: an idle Claude session window no longer renders as a full red "unavailable" block (PHNX-3453).** When a Claude account has no usage in the current rolling 5-hour window, the usage API returns `five_hour.utilization = null`, so no `S:` session bar exists — and `agents view` drew that missing window as a solid red `█████ unavailable`, which reads as "100% used / maxed-out", the opposite of what it means. It now renders a neutral dim dashed row `┄┄┄┄┄ unavailable`, visually distinct from both a real `0%` (`░░░░░`) and a full `100%` (`█████`). The reading itself is unchanged and honest — an idle account genuinely has no current 5-hour number; the bar fills in the moment that account runs Claude, or on `agents view claude --refresh` when the box isn't being usage-endpoint rate-limited. Source: `cli/src/lib/accounting/usage.ts` (`formatUsageSummary`, `NO_DATA`).
|
|
56
|
+
|
|
57
|
+
## 1.22.57
|
|
58
|
+
|
|
59
|
+
- **Removed two duplicate commands: `agents list` and `agents trash restore` (PHNX-3391).** `agents list` was a long-deprecated full duplicate of `agents view` (it already printed "agents list is now agents view" before delegating) — `agents view` is now the one version-listing surface. `agents trash restore <target>` was an exact duplicate of the top-level `agents restore <target>` (both call `restoreVersion`); `agents restore` stays, and `agents trash` keeps `trash list`. Both removed names now error as retired top-level commands rather than auto-correcting. Source: `cli/src/commands/versions.ts`, `cli/src/commands/trash.ts`.
|
|
60
|
+
|
|
61
|
+
- **`agents perf` moved to `agents insights perf` (PHNX-3391).** Performance metrics are an insight, not a top-level noun — latency now sits beside `agents insights cost` / `output` under the one `insights` group. Every subcommand is unchanged (`agents insights perf [hooks|commands|run|friction]`), the disposable warehouse at `~/.agents/.cache/perf/perf.db` is untouched, and bare `agents perf` now errors (retired top-level name) rather than auto-correcting. Source: `cli/src/commands/perf.ts`, `cli/src/commands/insights.ts`, `cli/src/cli/command-registry.ts`.
|
|
62
|
+
|
|
63
|
+
- **`agents insights mix` is now the one counter surface — the eight per-recipe shortcut commands, `recipes`, and `trends` are gone (PHNX-3391).** `agents insights mix` already printed the whole recipe board, yet each section was *also* a standalone command (`agents insights harness-mix`, `model-mix`, …), *also* reachable as `mix <recipe>`, *also* listed by `agents insights recipes`, and the board was *also* aliased by `agents insights trends` — one query engine wearing five surfaces (each registered twice, under `insights` and `sessions insights`). It collapses to one: `agents insights mix` (board), `agents insights mix <recipe>` (one section, e.g. `mix harness-mix`), and `agents insights mix --list` (the recipe ids). The removed spellings now error and point at `mix`. This is surface removal, not hiding — 20 command registrations deleted. Source: `cli/src/lib/analytics/mix-commands.ts`.
|
|
64
|
+
|
|
65
|
+
- **Worker boxes now show real Claude usage (S:/W:) bars in `agents view` (PHNX-3392).** A rate limit is metered per account, so the number is identical on every box — but only a headed device (`personal`/`desktop`) can read it, because the usage endpoint needs the `user:profile` scope only the interactive login carries. Headless workers, which have just the `user:inference` setup-token, got a 403 and showed blank bars. A new `usage-sync` daemon service closes the gap: each personal/desktop daemon pushes its per-account usage snapshot to worker peers, which merge it newest-wins (via the hidden `agents __usage-ingest` verb). Role-gated (personal/desktop publish, workers consume), single-executor per destination, and idempotent. A synced bar reads as last-seen, never a live fetch. Source: `cli/src/lib/accounting/usage-sync.ts`, `cli/src/lib/daemon/usage-sync-service.ts`.
|
|
66
|
+
|
|
67
|
+
- **`agents devices` gains a `desktop` role and stops double-labeling your interactive box (PHNX-3392).** `agents devices role <name>` now accepts a third value — `desktop` — for a headed always-on box (the release/credential home, e.g. a Mac mini): it is excluded from `--device auto` like `personal` (agents never auto-land there), but authenticates from its own interactive login like `personal` rather than a worker setup-token. And `agents devices list` no longer prints both `★ interactive` and `personal` on the same row — a `personal` box IS your interactive seat, so the star is folded into the role (it still shows for a non-personal box pinned as `interactive.host`). Source: `cli/src/lib/device-config.ts`, `cli/src/commands/ssh.ts`.
|
|
68
|
+
|
|
69
|
+
- **The system DotAgents repo `phnx-labs/.agents-system` was renamed to `phnx-labs/.agents` on GitHub, and both names are now recognized as the system origin (PHNX-3394).** The `-system` suffix was redundant once the layering itself is the role. The clone target stays `gh:phnx-labs/.agents-system` (GitHub's rename redirect makes that resolve fine, so there is no forced cutover), but a checkout under either name is recognized as the system origin across all transport forms (ssh / https / scp-style), via a `RENAMED_REMOTE_ALIASES` fold in `canonicalGitRemote` and the pure `isSystemRepoRemote` behind `isSystemRepoOrigin`. The companion extras repo `phnx-labs/.agents-extras` keeps its name but is now private and opt-in (`agents repo add`, not cloned by default). Source: `cli/src/lib/git.ts`.
|
|
70
|
+
|
|
3
71
|
## 1.22.56
|
|
4
72
|
|
|
5
73
|
- **Scheduled routines claim a durable per-occurrence slot, so a duplicate delivery or a catch-up of the same slot can't double-launch (PHNX-3215).** The forward-timer path keyed its single-fire claim on croner's `currentRun()`, which is the jittered wall-clock fire instant, not the aligned schedule boundary — so two timer callbacks for one occurrence minted different run ids and both launched, and a live fire never collided with its catch-up twin. The scheduler now floors the fire to the aligned `(routine, scheduledFor)` boundary (`fireSlot`/`alignedSlotForFire`), the same derivation catch-up uses, making the atomic slot claim a structural guarantee (SING-15/16, self-overlap SING-13). Source: `cli/src/lib/scheduler.ts`, `cli/src/lib/scheduling/routines.ts`, `cli/src/lib/overdue.ts`.
|
|
6
74
|
|
|
7
75
|
- **Managed-share Open Graph cards render in Cloudflare instead of failing on every new publish (PHNX-2835).** The Worker upload now carries Yoga and resvg as compiled WebAssembly modules alongside the bundled JavaScript and fonts. The previous single-file bundle decoded the WASM into byte arrays and compiled it at request time; Node accepted that in tests, but Cloudflare workerd forbids runtime WASM code generation, so a new share's lazy `<slug>.png` render failed. A real-workerd regression test now renders and validates the PNG, and a genuine renderer failure returns a diagnostic `500` instead of falling through as a missing cover. Source: `cli/src/lib/share/{worker-template,provision}.ts`.
|
|
8
76
|
|
|
77
|
+
- **A dangling per-harness default account no longer hard-fails every launch (PHNX-3326).** When `agents.yaml` `accounts.defaults.<agent>` pointed at an account id absent from the local registry, `resolveSpawnAccount` threw `Unknown account '<uuid>'` and the launch exited before doing any work. The default is now a preference, not a requirement: a stale pointer emits a warning naming the harness and the remedy (`agents accounts clear-default <agent>`), then falls back to the existing balanced-selection path. `accounts set-default`/`switch` now store the portable account name instead of the per-device uuid; legacy uuid entries continue to resolve, and `accounts remove`/`switch` correctly treat both name- and id-stored defaults as references. Source: `cli/src/lib/account-registry.ts`, `cli/src/commands/accounts.ts`.
|
|
78
|
+
|
|
9
79
|
## 1.22.55
|
|
10
80
|
|
|
11
81
|
- **Managed shares generate their Open Graph cover server-side (PHNX-2835).** Publishing HTML to `share.agents-cli.sh` no longer launches local Chromium. The Worker lazily renders a deterministic 1200×630 AGI card with Satori and resvg-wasm, bundles Inter and JetBrains Mono, inherits the canonical page's visibility gate, and caches the PNG in R2. BYO endpoints keep their local screenshot fallback. An explicit missing `AGENTS_SHARE_BROWSER` or `PUPPETEER_EXECUTABLE_PATH` now fails loudly instead of silently falling through. Source: `cli/src/lib/share/{capture,publish,worker-template}.ts`.
|
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
<a href="https://github.com/phnx-labs/agi-cli"><img src="https://img.shields.io/badge/github-phnx--labs%2Fagi--cli-blue?style=flat-square" alt="github" /></a>
|
|
12
12
|
</p>
|
|
13
13
|
|
|
14
|
-
**A framework for running a distributed agent factory.** Dispatch Claude, Codex, Antigravity, Grok, and more across your own machines, in parallel, on your existing subscriptions. Measure every run with `agents
|
|
14
|
+
**A framework for running a distributed agent factory.** Dispatch Claude, Codex, Antigravity, Grok, and more across your own machines, in parallel, on your existing subscriptions. Measure every run with `agents insights` (latency lives at `agents insights perf`), fold what you learn back into `AGENTS.md` and skills, then put the loop on a schedule with routines and monitors. Spawn parallel teams in isolated terminals or dispatch to the cloud for a PR. Watch live state across the fleet, nudge stalled runs, and message agents mid-flight. Store secrets behind Touch ID, drive real browsers and Electron apps, and steer the whole fleet from a menu bar — all from one CLI.
|
|
15
15
|
|
|
16
16
|
<p align="center">
|
|
17
17
|
<a href="https://github.com/anthropics/claude-code" title="Claude Code"><img src="assets/harnesses/anthropic.svg" height="32" alt="Claude Code" /></a>
|
|
@@ -110,7 +110,7 @@ agents teams add checkout codex "Write tests for the new code" --name qa --afte
|
|
|
110
110
|
agents teams start checkout --watch
|
|
111
111
|
|
|
112
112
|
# Measure what happened -- latency, friction, dead-weight skills
|
|
113
|
-
agents perf commands --days 7
|
|
113
|
+
agents insights perf commands --days 7 # slowest CLI entrypoints
|
|
114
114
|
agents insights --since 30d # friction, harness comparison, ranked actions
|
|
115
115
|
|
|
116
116
|
# Fold the lesson back into the harness -- every agent picks it up next run
|
|
@@ -125,7 +125,7 @@ agents routines add nightly-payments-audit \
|
|
|
125
125
|
agents menubar setup
|
|
126
126
|
```
|
|
127
127
|
|
|
128
|
-
`agents perf` reads a disposable warehouse at `~/.agents/.cache/perf/perf.db` -- hook, command, and run timing rollups, deletable any time. `agents insights` (alias `agents sessions insights`) is deterministic and offline: it caches per-session facets, compares harnesses, and ranks actions by evidence count -- no model call unless you pass `--narrative`. Routines put any of this on a cron ([Routines](#routines)); monitors fire it on a change instead of a clock ([Monitors](#monitors)); the menu bar is the always-on control surface for the fleet these commands drive ([Menu bar](#menu-bar)).
|
|
128
|
+
`agents insights perf` reads a disposable warehouse at `~/.agents/.cache/perf/perf.db` -- hook, command, and run timing rollups, deletable any time. `agents insights` (alias `agents sessions insights`) is deterministic and offline: it caches per-session facets, compares harnesses, and ranks actions by evidence count -- no model call unless you pass `--narrative`. Routines put any of this on a cron ([Routines](#routines)); monitors fire it on a change instead of a clock ([Monitors](#monitors)); the menu bar is the always-on control surface for the fleet these commands drive ([Menu bar](#menu-bar)).
|
|
129
129
|
|
|
130
130
|
---
|
|
131
131
|
|
|
@@ -1407,7 +1407,7 @@ Two repos with the same shape, different roles:
|
|
|
1407
1407
|
|
|
1408
1408
|
| Repo | Role | Owner |
|
|
1409
1409
|
|---|---|---|
|
|
1410
|
-
| `~/.agents-system/` | **System repo** — core/built-in skills, commands, hooks, rules, MCP configs, permissions, and profiles that ship with `agi-cli`. The defaults every install gets. | Maintained upstream at [phnx-labs/.agents
|
|
1410
|
+
| `~/.agents-system/` | **System repo** — core/built-in skills, commands, hooks, rules, MCP configs, permissions, and profiles that ship with `agi-cli`. The defaults every install gets. | Maintained upstream at [phnx-labs/.agents](https://github.com/phnx-labs/.agents) (formerly `.agents-system`) |
|
|
1411
1411
|
| `~/.agents/` | **User repo** — your personal additions and overrides. This is what `agents repo push`/`pull` syncs. | You |
|
|
1412
1412
|
|
|
1413
1413
|
**Version pinning:** `agents.yaml` at project root pins which agent version to use (like `.nvmrc` for Node).
|
package/dist/bootstrap.js
CHANGED
|
@@ -29,7 +29,7 @@ const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
|
29
29
|
const packageJsonPath = path.join(__dirname, '..', 'package.json');
|
|
30
30
|
const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf-8'));
|
|
31
31
|
const VERSION = packageJson.version;
|
|
32
|
-
import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, } from './lib/self-update.js';
|
|
32
|
+
import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, sweepStaleInstallStaging, } from './lib/self-update.js';
|
|
33
33
|
import { registerUpgradeCommand } from './commands/upgrade.js';
|
|
34
34
|
// Detect dev/working-tree builds and default the noisy startup steps off.
|
|
35
35
|
// Three cases trip this:
|
|
@@ -156,7 +156,9 @@ program.hook('postAction', (_thisCommand, actionCommand) => {
|
|
|
156
156
|
}).catch(() => { });
|
|
157
157
|
}
|
|
158
158
|
// Disposable perf warehouse — fail-soft spool append (no SQLite on this path).
|
|
159
|
-
|
|
159
|
+
// Skip the perf reader itself (now `agents insights perf`, PHNX-3391) so it
|
|
160
|
+
// never records its own latency into the board it prints.
|
|
161
|
+
if (durationMs !== undefined && !(parts[0] === 'insights' && parts[1] === 'perf')) {
|
|
160
162
|
// sessionId/agent are resolvable here the same way emit() resolves them
|
|
161
163
|
// for command.start/command.end above (the shared provenance floor,
|
|
162
164
|
// event-provenance.ts) — without this, every command.end perf sample
|
|
@@ -319,6 +321,13 @@ async function installResolvedPackage(metadata) {
|
|
|
319
321
|
// trusted .tgz. A mismatch throws and nothing below runs — fail closed.
|
|
320
322
|
const tarball = await downloadVerifiedTarball(metadata.tarball, metadata.integrity);
|
|
321
323
|
try {
|
|
324
|
+
// Clear any orphaned npm reify staging dir from a prior crashed upgrade
|
|
325
|
+
// BEFORE the package manager stages the new one (PHNX-3393) — otherwise
|
|
326
|
+
// npm's rename onto that exact, deterministic path fails ENOTEMPTY and
|
|
327
|
+
// every subsequent upgrade dead-ends there forever. bun does not use
|
|
328
|
+
// npm's retire-path staging scheme, so this only needs to run once, ahead
|
|
329
|
+
// of both package-manager branches below.
|
|
330
|
+
sweepStaleInstallStaging(packageRoot);
|
|
322
331
|
// Upgrade with the package manager that owns this install. A bun global
|
|
323
332
|
// install lives at <bunGlobalDir>/node_modules/... (no `lib` segment), so an
|
|
324
333
|
// `npm install --prefix` would write to <bunGlobalDir>/lib/node_modules and
|
|
@@ -65,7 +65,6 @@ export declare const loadRefreshRules: ModuleLoader;
|
|
|
65
65
|
export declare const loadFactory: ModuleLoader;
|
|
66
66
|
export declare const loadInsights: ModuleLoader;
|
|
67
67
|
export declare const loadTrace: ModuleLoader;
|
|
68
|
-
export declare const loadPerf: ModuleLoader;
|
|
69
68
|
export declare const loadPty: ModuleLoader;
|
|
70
69
|
export declare const loadTmux: ModuleLoader;
|
|
71
70
|
export declare const loadWatchdog: ModuleLoader;
|
|
@@ -67,7 +67,6 @@ export const loadRefreshRules = async () => (await import('../commands/refresh-r
|
|
|
67
67
|
export const loadFactory = async () => (await import('../commands/factory.js')).registerFactoryCommands;
|
|
68
68
|
export const loadInsights = async () => (await import('../commands/insights.js')).registerInsightsCommand;
|
|
69
69
|
export const loadTrace = async () => (await import('../commands/sessions-trace.js')).registerTraceCommand;
|
|
70
|
-
export const loadPerf = async () => (await import('../commands/perf.js')).registerPerfCommand;
|
|
71
70
|
export const loadPty = async () => (await import('../commands/pty.js')).registerPtyCommands;
|
|
72
71
|
export const loadTmux = async () => (await import('../commands/tmux.js')).registerTmuxCommands;
|
|
73
72
|
export const loadWatchdog = async () => (await import('../commands/watchdog.js')).registerWatchdogCommand;
|
|
@@ -142,7 +141,6 @@ export const COMMAND_LOADERS = {
|
|
|
142
141
|
workflows: [loadWorkflows],
|
|
143
142
|
add: [loadVersions],
|
|
144
143
|
use: [loadVersions],
|
|
145
|
-
list: [loadVersions],
|
|
146
144
|
remove: [loadVersions],
|
|
147
145
|
rm: [loadVersions],
|
|
148
146
|
purge: [loadVersions],
|
|
@@ -177,7 +175,6 @@ export const COMMAND_LOADERS = {
|
|
|
177
175
|
// `agents trace` is a top-level alias of `agents sessions trace` (precedent:
|
|
178
176
|
// `agents insights` aliases `agents sessions insights`). One implementation.
|
|
179
177
|
trace: [loadTrace],
|
|
180
|
-
perf: [loadPerf],
|
|
181
178
|
pty: [loadPty],
|
|
182
179
|
tmux: [loadTmux],
|
|
183
180
|
watchdog: [loadWatchdog],
|
|
@@ -342,13 +342,17 @@ export function setDefaultAccount(agentRaw, name) {
|
|
|
342
342
|
}
|
|
343
343
|
assertNativeAccountNameable(account.agent);
|
|
344
344
|
}
|
|
345
|
-
|
|
345
|
+
// Reference by NAME, not id: defaults sync fleet-wide with `agents repo push/pull`
|
|
346
|
+
// while account ids are minted per-device, so an id ref breaks on every other
|
|
347
|
+
// machine ("Unknown account '<uuid>'"). Names are the portable handle — the
|
|
348
|
+
// registry resolves both, and existing uuid entries still resolve.
|
|
349
|
+
updateMeta(meta => ({ ...meta, accounts: { ...meta.accounts, defaults: { ...meta.accounts?.defaults, [agent]: account.name } } }));
|
|
346
350
|
return { agent, account };
|
|
347
351
|
}
|
|
348
352
|
async function switchAccountRows(agent) {
|
|
349
353
|
const accounts = await listSwitchableAccounts(agent);
|
|
350
354
|
const candidates = await collectRunCandidates(agent);
|
|
351
|
-
const
|
|
355
|
+
const defaultValue = readMeta().accounts?.defaults?.[agent];
|
|
352
356
|
return accounts.map(account => {
|
|
353
357
|
const candidate = account.kind === 'native'
|
|
354
358
|
? candidates.find(row => row.accountKey === account.identityKey) ?? null
|
|
@@ -357,7 +361,7 @@ async function switchAccountRows(agent) {
|
|
|
357
361
|
accountName: account.name,
|
|
358
362
|
kind: account.kind,
|
|
359
363
|
detail: account.kind === 'provider' ? account.provider : (account.identityLabel ?? account.identityKey),
|
|
360
|
-
current: account.id ===
|
|
364
|
+
current: account.id === defaultValue || account.name === defaultValue,
|
|
361
365
|
candidate,
|
|
362
366
|
};
|
|
363
367
|
});
|
package/dist/commands/apply.js
CHANGED
|
@@ -16,7 +16,7 @@ import { machineId } from '../lib/session/sync/config.js';
|
|
|
16
16
|
import { loadDevices } from '../lib/devices/registry.js';
|
|
17
17
|
import { isHostPinned, managedKnownHostsPath } from '../lib/devices/known-hosts.js';
|
|
18
18
|
import { ensureDevicesRegistered } from '../lib/devices/sync.js';
|
|
19
|
-
import { readFleetFile, resolveDesired } from '../lib/fleet/manifest.js';
|
|
19
|
+
import { readFleetFile, resolveDesired, emptyTargetsMessage } from '../lib/fleet/manifest.js';
|
|
20
20
|
import { snapshotAuth } from '../lib/fleet/auth-sync.js';
|
|
21
21
|
import { AUTH_BUNDLE_NAME } from '../lib/secrets/bundles.js';
|
|
22
22
|
import { agentIdOf, diffFleet, probeDevice, runFleetApply, pool, sourceHome, expandAllSpecs, rosterNeedsVersions, fleetSecretsBundles, } from '../lib/fleet/apply.js';
|
|
@@ -188,7 +188,15 @@ async function runApply(opts) {
|
|
|
188
188
|
if (opts.login === false)
|
|
189
189
|
desired = desired.map((d) => ({ ...d, login: 'skip' }));
|
|
190
190
|
if (desired.length === 0) {
|
|
191
|
-
|
|
191
|
+
const msg = emptyTargetsMessage(manifest);
|
|
192
|
+
if (msg.style === 'hint') {
|
|
193
|
+
console.log(chalk.yellow(msg.lines[0]));
|
|
194
|
+
for (const line of msg.lines.slice(1))
|
|
195
|
+
console.log(chalk.gray(` ${line}`));
|
|
196
|
+
}
|
|
197
|
+
else {
|
|
198
|
+
console.log(chalk.gray(msg.lines[0]));
|
|
199
|
+
}
|
|
192
200
|
return;
|
|
193
201
|
}
|
|
194
202
|
// Snapshot source auth once for every agent named anywhere in the profile.
|
package/dist/commands/exec.js
CHANGED
|
@@ -431,7 +431,7 @@ async function handleTerminalHandoff(agentSpec, options, prompt) {
|
|
|
431
431
|
if (!knownAgent) {
|
|
432
432
|
const probeCwd = options.cwd ?? process.cwd();
|
|
433
433
|
if (!hasProfile && !resolveWorkflowRef(rawTarget, probeCwd)) {
|
|
434
|
-
console.error(chalk.red(`Unknown agent, profile, or workflow: ${rawTarget}. See \`agents
|
|
434
|
+
console.error(chalk.red(`Unknown agent, profile, or workflow: ${rawTarget}. See \`agents view\` for the installed harnesses.`));
|
|
435
435
|
process.exit(1);
|
|
436
436
|
}
|
|
437
437
|
}
|
package/dist/commands/fork.d.ts
CHANGED
|
@@ -1,19 +1,32 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* `agents sessions fork <session>` — branch an existing conversation into a new,
|
|
3
|
-
* independent session you can continue separately. The original is untouched.
|
|
4
|
-
* Also exposed as the hidden top-level alias `agents fork` (back-compat).
|
|
5
|
-
*
|
|
6
|
-
* Thin command layer; the copy/register logic lives in `lib/session/fork.ts`.
|
|
7
|
-
*/
|
|
8
1
|
import type { Command } from 'commander';
|
|
9
2
|
interface ForkOptions {
|
|
10
3
|
name?: string;
|
|
4
|
+
device?: string;
|
|
5
|
+
/** Open the sibling in a real terminal tab instead of in-place; optional backend. */
|
|
6
|
+
terminal?: string | boolean;
|
|
7
|
+
}
|
|
8
|
+
/**
|
|
9
|
+
* The two process boundaries fork crosses — a preview subprocess (cross-fleet
|
|
10
|
+
* resolve + digest) and the sibling launch. Injectable so the resolve→recap→run
|
|
11
|
+
* argv logic is unit-tested without spawning real CLIs.
|
|
12
|
+
*/
|
|
13
|
+
export interface ForkDeps {
|
|
14
|
+
/** Run `agents sessions preview <sub…>` and capture stdout + exit status. */
|
|
15
|
+
runPreview: (sub: string[]) => {
|
|
16
|
+
status: number | null;
|
|
17
|
+
stdout: string;
|
|
18
|
+
};
|
|
19
|
+
/** Launch `agents <sub…>` inheriting stdio; returns its exit status. */
|
|
20
|
+
launch: (sub: string[]) => {
|
|
21
|
+
status: number | null;
|
|
22
|
+
};
|
|
11
23
|
}
|
|
12
24
|
/**
|
|
13
|
-
* Resolve the source
|
|
14
|
-
*
|
|
25
|
+
* Resolve the source cross-fleet, build a recap from its preview digest, and
|
|
26
|
+
* launch a same-harness sibling seeded with that recap. Shared by
|
|
27
|
+
* `agents sessions fork` and the `agents fork` alias.
|
|
15
28
|
*/
|
|
16
|
-
export declare function runFork(sessionArg: string, options: ForkOptions): Promise<void>;
|
|
29
|
+
export declare function runFork(sessionArg: string, options: ForkOptions, deps?: ForkDeps): Promise<void>;
|
|
17
30
|
/**
|
|
18
31
|
* Register `agents sessions fork <session>` — the canonical surface (fork is a
|
|
19
32
|
* session operation, so it lives under the `sessions` group).
|
package/dist/commands/fork.js
CHANGED
|
@@ -1,79 +1,132 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `agents sessions fork <session>` — branch an existing conversation into a new,
|
|
3
|
+
* independent same-harness sibling, seeded with a recap so it picks up where the
|
|
4
|
+
* original left off. The original is untouched. Also exposed as the hidden
|
|
5
|
+
* top-level alias `agents fork` (back-compat).
|
|
6
|
+
*
|
|
7
|
+
* The source is resolved CROSS-FLEET (the same path `agents sessions preview`
|
|
8
|
+
* uses), so a session that lives on another device forks fine — the sibling is
|
|
9
|
+
* handed a plain-text recap as its opening input and never has to reach the
|
|
10
|
+
* source transcript. Because the seed is text, any REPL harness can be forked,
|
|
11
|
+
* not just Claude.
|
|
12
|
+
*
|
|
13
|
+
* Thin command layer; the pure recap text lives in `lib/session/fork.ts`.
|
|
14
|
+
*/
|
|
15
|
+
import { spawnSync } from 'child_process';
|
|
1
16
|
import chalk from 'chalk';
|
|
2
17
|
import { setHelpSections } from '../lib/help.js';
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
|
|
18
|
+
import { getCliLaunch } from '../lib/cli-entry.js';
|
|
19
|
+
import { buildForkRecap, forkLabelFor } from '../lib/session/fork.js';
|
|
20
|
+
function defaultDeps() {
|
|
21
|
+
return {
|
|
22
|
+
runPreview: (sub) => {
|
|
23
|
+
const p = getCliLaunch(['sessions', 'preview', ...sub]);
|
|
24
|
+
// stderr inherited so preview's own resolution errors reach the user verbatim.
|
|
25
|
+
const r = spawnSync(p.command, p.args, { encoding: 'utf-8', stdio: ['ignore', 'pipe', 'inherit'] });
|
|
26
|
+
return { status: r.status, stdout: r.stdout ?? '' };
|
|
27
|
+
},
|
|
28
|
+
launch: (sub) => {
|
|
29
|
+
const l = getCliLaunch(sub);
|
|
30
|
+
// In-place stdio so the sibling takes over this terminal.
|
|
31
|
+
const r = spawnSync(l.command, l.args, { stdio: 'inherit' });
|
|
32
|
+
return { status: r.status };
|
|
33
|
+
},
|
|
34
|
+
};
|
|
35
|
+
}
|
|
6
36
|
const FORK_HELP = {
|
|
7
37
|
examples: `
|
|
8
|
-
# Fork a session by (partial) id
|
|
38
|
+
# Fork a session by (partial) id — launches a same-harness sibling seeded with a recap
|
|
9
39
|
agents sessions fork 4f3a9c21
|
|
10
|
-
agents sessions resume <new-id>
|
|
11
40
|
|
|
12
|
-
#
|
|
41
|
+
# Name the fork's session label
|
|
13
42
|
agents sessions fork 4f3a9c21 --name "try redis instead"
|
|
43
|
+
|
|
44
|
+
# Place the sibling on a fleet worker instead of here
|
|
45
|
+
agents sessions fork 4f3a9c21 --device auto
|
|
46
|
+
|
|
47
|
+
# Open the sibling in a fresh terminal tab where you work
|
|
48
|
+
agents sessions fork 4f3a9c21 --terminal
|
|
14
49
|
`,
|
|
15
50
|
notes: `
|
|
16
|
-
- 'resume' continues the SAME conversation; 'fork'
|
|
17
|
-
|
|
18
|
-
-
|
|
19
|
-
|
|
20
|
-
|
|
51
|
+
- 'resume' continues the SAME conversation; 'fork' launches a NEW same-harness
|
|
52
|
+
session seeded with a recap of the source, so the two diverge.
|
|
53
|
+
- Works cross-device and cross-harness: the source is resolved across the fleet
|
|
54
|
+
and the sibling gets a plain-text recap, so it never reaches the source transcript.
|
|
55
|
+
- The recap carries the source id — the sibling can run '/continue <id>' for the
|
|
56
|
+
full history if it needs more than the recap.
|
|
57
|
+
- Resolve the source the same way as resume: an exact or prefix id fragment.
|
|
21
58
|
`,
|
|
22
59
|
};
|
|
23
60
|
/**
|
|
24
|
-
* Resolve the source
|
|
25
|
-
*
|
|
61
|
+
* Resolve the source cross-fleet, build a recap from its preview digest, and
|
|
62
|
+
* launch a same-harness sibling seeded with that recap. Shared by
|
|
63
|
+
* `agents sessions fork` and the `agents fork` alias.
|
|
26
64
|
*/
|
|
27
|
-
export async function runFork(sessionArg, options) {
|
|
28
|
-
// Resolve
|
|
29
|
-
//
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
if (
|
|
36
|
-
|
|
37
|
-
// `agents sessions fork <id> && agents sessions resume <new>` doesn't proceed
|
|
38
|
-
// on a failed fork.
|
|
39
|
-
console.error(chalk.red(`No session matching "${sessionArg}".`));
|
|
40
|
-
console.error(chalk.gray('List candidates with: agents sessions'));
|
|
65
|
+
export async function runFork(sessionArg, options, deps = defaultDeps()) {
|
|
66
|
+
// Resolve + digest in one cross-fleet hop by shelling the existing preview
|
|
67
|
+
// verb: it resolves the id across the fleet (SSH fan-out + peer hop), computes
|
|
68
|
+
// the digest on the OWNING device, and prints it as JSON — so a remote source
|
|
69
|
+
// resolves fine and we never re-implement resolution or digesting here.
|
|
70
|
+
// --terminal opens a tab on THIS machine; --device dispatches over SSH. `agents
|
|
71
|
+
// run` refuses the combination, so reject it here with a fork-specific message
|
|
72
|
+
// before resolving anything, rather than after an overpromising progress line.
|
|
73
|
+
if (options.device && options.terminal !== undefined) {
|
|
74
|
+
console.error(chalk.red('Pick one placement: --terminal opens a tab here; --device places the sibling on another box. They cannot combine.'));
|
|
41
75
|
process.exitCode = 1;
|
|
42
76
|
return;
|
|
43
77
|
}
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
}
|
|
49
|
-
process.exitCode = 1;
|
|
78
|
+
const res = deps.runPreview([sessionArg, '--json']);
|
|
79
|
+
if (res.status !== 0) {
|
|
80
|
+
// preview already explained why on stderr; propagate its exit code.
|
|
81
|
+
process.exitCode = res.status ?? 1;
|
|
50
82
|
return;
|
|
51
83
|
}
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
console.error(chalk.
|
|
58
|
-
console.error(chalk.gray(` Branch it by hand — start a fresh ${source.agent} and seed it with the source's context:`));
|
|
59
|
-
console.error(chalk.gray(` agents run ${source.agent} --terminal # then, in the new session:`));
|
|
60
|
-
console.error(chalk.gray(` /continue ${source.shortId}`));
|
|
84
|
+
let data;
|
|
85
|
+
try {
|
|
86
|
+
data = JSON.parse(res.stdout);
|
|
87
|
+
}
|
|
88
|
+
catch {
|
|
89
|
+
console.error(chalk.red(`Could not read the source session for "${sessionArg}".`));
|
|
61
90
|
process.exitCode = 1;
|
|
62
91
|
return;
|
|
63
92
|
}
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
}
|
|
68
|
-
catch (err) {
|
|
69
|
-
console.error(chalk.red(`Could not fork ${source.shortId}: ${err.message}`));
|
|
93
|
+
const source = data?.session;
|
|
94
|
+
if (!source?.id || !source?.agent) {
|
|
95
|
+
console.error(chalk.red(`Could not resolve a forkable source for "${sessionArg}".`));
|
|
70
96
|
process.exitCode = 1;
|
|
71
97
|
return;
|
|
72
98
|
}
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
99
|
+
const digest = data?.preview ?? undefined;
|
|
100
|
+
// Most sessions have no explicit --name label; fall back to the auto-derived
|
|
101
|
+
// topic the rest of the CLI shows, not the raw short id (forkLabelFor is the
|
|
102
|
+
// shared 3-tier resolver, and preview's --json now carries `topic`).
|
|
103
|
+
const label = forkLabelFor({ label: source.label, topic: source.topic, shortId: source.shortId });
|
|
104
|
+
const recap = buildForkRecap({
|
|
105
|
+
agent: source.agent,
|
|
106
|
+
label,
|
|
107
|
+
cwd: source.cwd,
|
|
108
|
+
ticketId: source.ticketId,
|
|
109
|
+
machine: source.machine,
|
|
110
|
+
shortId: source.shortId,
|
|
111
|
+
id: source.id,
|
|
112
|
+
lastAssistant: digest?.lastAssistant,
|
|
113
|
+
changes: digest?.changes,
|
|
114
|
+
});
|
|
115
|
+
// Launch a NEW same-harness session, load-balanced across accounts (balanced),
|
|
116
|
+
// seeded with the recap as its opening input. Runs here by default; --device
|
|
117
|
+
// places it on the fleet; --terminal opens it in a fresh tab where the user works.
|
|
118
|
+
const runArgs = ['run', source.agent, recap, '-i', '--strategy', 'balanced', '--name', options.name || `fork of ${label}`];
|
|
119
|
+
if (options.device)
|
|
120
|
+
runArgs.push('--device', options.device);
|
|
121
|
+
if (options.terminal !== undefined) {
|
|
122
|
+
runArgs.push('--terminal');
|
|
123
|
+
if (typeof options.terminal === 'string')
|
|
124
|
+
runArgs.push(options.terminal);
|
|
125
|
+
}
|
|
126
|
+
const where = options.device ? ` on ${options.device}` : options.terminal !== undefined ? ' in a new terminal' : '';
|
|
127
|
+
console.error(chalk.gray(`Forking ${source.shortId} → new ${source.agent} session${where}, seeded with a recap…`));
|
|
128
|
+
const child = deps.launch(runArgs);
|
|
129
|
+
process.exitCode = child.status ?? 0;
|
|
77
130
|
}
|
|
78
131
|
/**
|
|
79
132
|
* Register `agents sessions fork <session>` — the canonical surface (fork is a
|
|
@@ -82,10 +135,12 @@ export async function runFork(sessionArg, options) {
|
|
|
82
135
|
export function registerSessionsForkCommand(sessionsCmd) {
|
|
83
136
|
const cmd = sessionsCmd
|
|
84
137
|
.command('fork <session>')
|
|
85
|
-
.description('Branch a session into a new,
|
|
86
|
-
.option('--name <label>', '
|
|
138
|
+
.description('Branch a session into a new same-harness sibling, seeded with a recap so it continues the work. The original is untouched.')
|
|
139
|
+
.option('--name <label>', 'Session label for the fork (default: "fork of <original>")')
|
|
140
|
+
.option('--device <host>', 'Place the sibling on a fleet device (name or "auto"); defaults to here')
|
|
141
|
+
.option('--terminal [backend]', 'Open the sibling in a real terminal tab (iterm | ghostty | terminal | tmux | vscodium-agent) instead of in-place');
|
|
87
142
|
setHelpSections(cmd, FORK_HELP);
|
|
88
|
-
cmd.action(runFork);
|
|
143
|
+
cmd.action((session, options) => runFork(session, options));
|
|
89
144
|
}
|
|
90
145
|
/**
|
|
91
146
|
* Register the hidden top-level `agents fork` alias. Kept working for back-compat
|
|
@@ -94,7 +149,9 @@ export function registerSessionsForkCommand(sessionsCmd) {
|
|
|
94
149
|
export function registerForkCommand(program) {
|
|
95
150
|
const cmd = program
|
|
96
151
|
.command('fork <session>', { hidden: true })
|
|
97
|
-
.description('Alias for `agents sessions fork` — branch a session into a new
|
|
98
|
-
.option('--name <label>', '
|
|
99
|
-
|
|
152
|
+
.description('Alias for `agents sessions fork` — branch a session into a new same-harness sibling.')
|
|
153
|
+
.option('--name <label>', 'Session label for the fork (default: "fork of <original>")')
|
|
154
|
+
.option('--device <host>', 'Place the sibling on a fleet device (name or "auto"); defaults to here')
|
|
155
|
+
.option('--terminal [backend]', 'Open the sibling in a real terminal tab (iterm | ghostty | terminal | tmux | vscodium-agent) instead of in-place');
|
|
156
|
+
cmd.action((session, options) => runFork(session, options));
|
|
100
157
|
}
|
package/dist/commands/hooks.js
CHANGED
|
@@ -594,22 +594,22 @@ Examples:
|
|
|
594
594
|
.addHelpText('after', `
|
|
595
595
|
Shows aggregated stats for every hook that fired through a generated shim.
|
|
596
596
|
Primary source: disposable SQLite warehouse ~/.agents/.cache/perf/perf.db
|
|
597
|
-
(same data as \`agents perf hooks\`). Falls back to the legacy daily JSONL under
|
|
597
|
+
(same data as \`agents insights perf hooks\`). Falls back to the legacy daily JSONL under
|
|
598
598
|
~/.agents/.cache/logs/ when the warehouse is empty.
|
|
599
599
|
|
|
600
600
|
Examples:
|
|
601
601
|
agents hooks profile # last 7 days, table form
|
|
602
602
|
agents hooks profile --days 30 # roll up the full month
|
|
603
603
|
agents hooks profile --json | jq # pipe somewhere
|
|
604
|
-
agents perf hooks # same rollup under the perf surface
|
|
604
|
+
agents insights perf hooks # same rollup under the perf surface
|
|
605
605
|
|
|
606
606
|
A hook whose p99 exceeds --warn-ms gets flagged in the cache column. Add
|
|
607
607
|
'cache: 5m' or 'cache: 5m-bg' to its hooks.yaml entry to fix it.
|
|
608
608
|
`)
|
|
609
|
-
.option('--project <key>', 'Scope to one project (see agents perf --help)')
|
|
609
|
+
.option('--project <key>', 'Scope to one project (see agents insights perf --help)')
|
|
610
610
|
.action(async (options) => {
|
|
611
611
|
const { DEFAULT_SLOW_HOOK_WARN_MS } = await import('../lib/hooks/profile.js');
|
|
612
|
-
// Same rollup as `agents perf hooks` — this command predates the `perf`
|
|
612
|
+
// Same rollup as `agents insights perf hooks` — this command predates the `perf`
|
|
613
613
|
// surface and is kept as a documented alias; delegate instead of
|
|
614
614
|
// duplicating the SQLite-vs-legacy-JSONL fallback and table rendering.
|
|
615
615
|
const { loadHookProfile, renderHookTable } = await import('./perf.js');
|