@phnx-labs/agents-cli 1.22.57 → 1.22.59
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +39 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +198 -23
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +40 -5
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +120 -9
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +43 -133
- package/dist/lib/installations/versions.js +94 -206
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +65 -0
- package/dist/lib/self-update.js +138 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +51 -6
- package/dist/lib/session/db.js +266 -20
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +128 -6
- package/dist/lib/traces/sync.js +294 -35
- package/dist/lib/traces/worker-template.js +154 -1
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,297 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.22.59
|
|
4
|
+
|
|
5
|
+
- **`agents devices disable/prefer` now actually change `--device auto` placement (PHNX-2092).**
|
|
6
|
+
The per-device `auto-launch.enabled` / `auto-launch.preferred` flags (written by
|
|
7
|
+
`agents devices disable/enable` and `prefer/unprefer`, and by `agents devices config
|
|
8
|
+
<name> auto-launch.*`) were stored and synced but never consulted by the CLI's one
|
|
9
|
+
placement path, so a disabled box was still picked and a preferred box got no boost.
|
|
10
|
+
They now feed the single automatic-placement rule: `filterAutoPool` drops any device
|
|
11
|
+
with `auto-launch.enabled` = false from EVERY auto path (`run`, `teams`, `ssh auto`,
|
|
12
|
+
the AGI EXT launch commands, which resolve placement through the CLI) exactly as a
|
|
13
|
+
`personal`/`desktop` role does, and `pickBestDevice` ranks an `auto-launch.preferred`
|
|
14
|
+
device ahead of its load-equal peers — after the signed-in tier, before load, so an
|
|
15
|
+
operator boost overrides load-based ordering without overriding hard health. A
|
|
16
|
+
fleet-wide default (`--fleet`) reaches doc-less devices via the candidate roster. No
|
|
17
|
+
second placement path was added — the existing `filterAutoPool`/`pickBestDevice`
|
|
18
|
+
rule was extended. Source: `cli/src/lib/devices/pool.ts`,
|
|
19
|
+
`cli/src/lib/teams/scheduler.ts`, `cli/src/lib/smart-launch.ts`.
|
|
20
|
+
|
|
21
|
+
- **`agents sessions stats` coverage now reports scan coverage, not with-usage,
|
|
22
|
+
so a completed backfill clears the "run the backfill" hint (PHNX-2301).** The
|
|
23
|
+
coverage line read `sessionsWithUsage / sessionsIndexed` — a ratio that stays
|
|
24
|
+
near-zero (~1.2% on a real fleet) even after `agents sessions backfill
|
|
25
|
+
resources` has fully run, because most sessions genuinely invoke no
|
|
26
|
+
skill/slash-command and a non-recording harness contributes none by
|
|
27
|
+
construction. The nag therefore never cleared and the low number read as a
|
|
28
|
+
coverage bug when it was not. `resourceUsageCoverage()` now also returns
|
|
29
|
+
`scanned` — the count of sessions carrying a `resource_scan_ledger` row at the
|
|
30
|
+
current `RESOURCE_INDEX_VERSION`, the honest "has the backfill folded in
|
|
31
|
+
history?" signal (the ledger is stamped for EVERY scanned session, including
|
|
32
|
+
ones that invoked nothing). The human line shows both — `N/T sessions scanned ·
|
|
33
|
+
M carry an explicit invocation` — and the backfill hint keys on
|
|
34
|
+
`scanned/total`, so it appears only when history really is un-scanned. The
|
|
35
|
+
`--json` `coverage` object gains `sessionsScanned` alongside the unchanged
|
|
36
|
+
`sessionsWithUsage`/`sessionsIndexed` (SES-IF-4b envelope preserved;
|
|
37
|
+
`schemaVersion` bumped to 2), and `signal.recording` now names the recorded set
|
|
38
|
+
(`skills: claude+kimi`, `commands: claude`) so a machine consumer can tell a
|
|
39
|
+
zero from a non-recording harness apart from genuine non-use. The zero-invoked
|
|
40
|
+
heading reads "never explicitly invoked" to match. Recording coverage itself is
|
|
41
|
+
unchanged — extending the `Skill`-tool set to another harness needs a verified
|
|
42
|
+
transcript tool-name, not a blind capability-table edit. Source:
|
|
43
|
+
`cli/src/lib/session/db.ts` (`resourceUsageCoverage`),
|
|
44
|
+
`cli/src/commands/sessions-stats.ts`.
|
|
45
|
+
|
|
46
|
+
- **Menu bar: favorite (pin) devices below the current Mac (PHNX-2376).** Each
|
|
47
|
+
device's submenu in the menu bar's DEVICES section now has a **★ Favorite /
|
|
48
|
+
Unfavorite** toggle. A favorited device renders with a ★ and sorts immediately
|
|
49
|
+
below the current machine (`<name> (this Mac)`), above all other devices, with a
|
|
50
|
+
divider between the favorites and the rest. "Favorite" reuses the EXISTING
|
|
51
|
+
per-device `auto-launch.preferred` config that automatic placement already
|
|
52
|
+
honors (PHNX-2092) — the toggle writes `agents devices config <name>
|
|
53
|
+
auto-launch.preferred on|off`, not a new store — so a box favorited from the
|
|
54
|
+
menu bar is also boosted in `--device auto` placement, and vice-versa. Source:
|
|
55
|
+
`cli/menubar/Sources/MenubarHelper/StatusItemController.swift`,
|
|
56
|
+
`cli/menubar/Sources/MenubarHelper/AgentsCLI.swift`.
|
|
57
|
+
|
|
58
|
+
- **Built-in monitors ship visible + enabled, `monitors list` is fleet-aware, and built-ins are tagged `(built-in)` (PHNX-2506).** A monitor shipped in the system mirror (`~/.agents/.system/monitors/`) used to be the lone system-layer resource that shipped **disabled and untagged** — `readMonitorFile` special-cased system scope to `enabled: false`, so a shipped built-in read `off` on every install and nobody could tell it came from the system layer. It now defaults to enabled like every other system resource (rules, hooks, commands, skills), on for every install unless the user shadows it with `enabled: false` (via `agents monitors pause`, which materializes a user copy — the system mirror stays pull-only; there is deliberately no `enable`/`disable` verb). Each monitor now carries its read-time `scope` (`user`/`system`), so `agents monitors list` tags a built-in `(built-in)` and `--json` emits `scope` + `builtin`, mirroring routines. And `agents monitors list` now **fans out across the fleet** (the same `gatherRemoteAgentsJson` sweep `sessions --active` and the add-time duplicate guard use), so every monitor on every device is visible with the box it lives on — the original "monitors shouldn't be device-local" complaint was a **visibility** gap, not a git-sync one (monitors are per-work-item watchers and are deliberately NOT synced through the DotAgents repo). `--local` pins the listing to this device; a peer answering the fan-out reports itself only, so the duplicate guard's bare-array contract is unchanged. **Enabled-by-default is not, on its own, permission to fire on every daemon (SING-9a).** A shared-input built-in — one whose source polls a fleet-shared queue such as `gh pr list --author @me`, `pr-merge-on-green` being the canonical case — is placed on a single owner **in code**, not just by a `device:` pin in the shipped YAML: an unpinned `scope: system` monitor is treated as shared-input unless it sets `sharedInput: false`, and fires only on the resolved owner (`interactive.host`, else the sole box on a single-device fleet, else nowhere) via `requiresSingleOwner` / `monitorRunsOnThisDevice`. So even a built-in whose YAML forgot the pin can never fan out across the fleet and race to merge the same PR. A device-local built-in (input = the firing box's own state) opts back into fleet-wide firing with `sharedInput: false`; a user monitor keeps its fleet-wide default and opts into owner-only with `sharedInput: true`. Source: `cli/src/lib/monitors/config.ts`, `cli/src/lib/monitors/remote.ts`, `cli/src/commands/monitors.ts`, `cli/src/lib/state.ts`.
|
|
59
|
+
|
|
60
|
+
- **A version-home plugin left behind by a marketplace move is now reconciled
|
|
61
|
+
away instead of shadowing the current copy (PHNX-2618).** Plugin orphan
|
|
62
|
+
detection keyed on plugin **name** alone, but a version home installs plugins
|
|
63
|
+
per **marketplace** (`marketplaces/<name>/plugins/<plugin>`). When a plugin
|
|
64
|
+
moved marketplaces — e.g. `code` once shipped from the user repo (the
|
|
65
|
+
`agents-cli` marketplace) and later from the system repo (`agents-system`) —
|
|
66
|
+
the stale `agents-cli` copy was never cleaned, because the name `code` was
|
|
67
|
+
still active via `agents-system`. Fleet boxes then carried **two** `code`
|
|
68
|
+
plugins: the current one, plus a shadow serving skills the repo deleted on
|
|
69
|
+
purpose (`code:quality` / `code:ship` / `code:verify`), and resolution order
|
|
70
|
+
decided which `code:review` an agent got. `cleanOrphanedPluginSkills` /
|
|
71
|
+
`diffVersionPlugins` now key on the `(marketplace, name)` pair: an installed
|
|
72
|
+
copy whose marketplace source repo is present but no longer ships that plugin
|
|
73
|
+
is trashed (soft-deleted to `~/.agents/.trash/plugins/`), even when another
|
|
74
|
+
marketplace still ships the same name — so a plain `agents sync` reconciles the
|
|
75
|
+
shadow away. When a marketplace's source repo is **absent** (a project we're
|
|
76
|
+
not in, a removed extra repo) the original name-only test is kept, so an
|
|
77
|
+
unrelated sync from another cwd never trashes a plugin whose source simply is
|
|
78
|
+
not reachable right now. Source: `cli/src/lib/plugins/plugins.ts`
|
|
79
|
+
(`cleanOrphanedPluginSkills`, `diffVersionPlugins`, `isOrphanMarketplacePlugin`),
|
|
80
|
+
`cli/src/lib/staleness/writers/plugins.ts`,
|
|
81
|
+
`cli/src/lib/installations/versions.ts`.
|
|
82
|
+
|
|
83
|
+
- **`agents sessions --device all/fleet --json` now actually searches the whole fleet (PHNX-2673).** The documented "search the whole fleet" sentinel was filtered to an empty host set — correct for `--active` and the interactive listing, which fan out by default — but the HISTORICAL `--json` listing does not fan out by default (it stays a deterministic local slice for scripts), so the sentinel was silently dropped and a fleet-wide historical query returned local rows only. Naming devices explicitly (`--device box-a --device box-b`) reached the fleet; `--device all` did not. The `--json` listing path now remembers the sentinel and runs the SAME peer SSH sweep the interactive listing uses (`gatherRemoteList` over the registered online devices, whole-index per peer), merging peer rows in machine-first. A bare `--json` (no `--device`) stays local-only, `--local` still pins to this machine, and dead peers are skipped, never fatal. Source: `cli/src/commands/sessions.ts`.
|
|
84
|
+
|
|
85
|
+
- **`agents pty` now works on macOS/arm64 and Node 25/26 (PHNX-2740).** The pinned
|
|
86
|
+
`@homebridge/node-pty-prebuilt-multiarch@0.13.1` shipped no darwin-arm64 prebuild
|
|
87
|
+
above Node 24 (ABI 137), so on a Mac running Node 25 (ABI 141) or 26 (ABI 147) the
|
|
88
|
+
native binding never resolved and every pty command died with a bare
|
|
89
|
+
`Cannot find module '.../pty.node'` MODULE_NOT_FOUND. Bumped to `0.14.1`, which
|
|
90
|
+
publishes darwin-arm64 (and darwin-x64) prebuilds through ABI 147, covering current
|
|
91
|
+
Node. The PTY sidecar now also **fails loud** when the binding genuinely can't load:
|
|
92
|
+
instead of a raw MODULE_NOT_FOUND it prints the platform, the running Node ABI, and a
|
|
93
|
+
concrete remediation (`npm rebuild @homebridge/node-pty-prebuilt-multiarch` / reinstall
|
|
94
|
+
the CLI), preserving the underlying error. Source: `cli/package.json`,
|
|
95
|
+
`cli/src/lib/pty-server.ts`.
|
|
96
|
+
|
|
97
|
+
- **An upgrade can no longer strand a box with the package installed but no working `agents` command (PHNX-2768).** `agents upgrade` (and every `agents fleet update` box, which runs it) now **owns the global bin links**: after installing and verifying the new version, it checks that `<prefix>/bin/{agents,ag,browser,computer}` resolve to the freshly-installed copy and **restores any the package manager dropped**. This closes the failure that left zion upgraded to 1.22.40 with `/opt/homebrew/bin/{agents,ag,browser,computer}` gone — every `agents` invocation "command not found" until the links were relinked by hand. It covers the sibling entrypoints (`ag`/`browser`/`computer`), not just `agents`. A link it cannot make resolve fails the upgrade **loud** (non-zero exit), so a genuinely-broken box is reported `failed` by the rollout instead of a stranded `ok`/`unverified`. Scope: the npm-prefix POSIX layout — bun and Windows use their own bin shims. Source: `cli/src/lib/self-update.ts` (`ensureGlobalBinLinks`), `cli/src/bootstrap.ts`, `cli/src/commands/ssh.ts`.
|
|
98
|
+
|
|
99
|
+
- **The auto-pulled system repo is verified against its expected origin before a
|
|
100
|
+
fast-forward — a repointed origin is refused, not executed (PHNX-2957).** The
|
|
101
|
+
system repo (`~/.agents/.system/`) ships **hooks** that register as shell
|
|
102
|
+
`command` strings run on every tool event, and its checkout auto-fast-forwards
|
|
103
|
+
from `origin` (on `agents use`, and on the opt-in `AGENTS_AUTO_PULL=1`
|
|
104
|
+
background worker). Neither path verified that `origin` was the repo the
|
|
105
|
+
operator actually chose — so an `origin` repointed to an attacker's fork (or a
|
|
106
|
+
clone seeded from one) would silently fast-forward arbitrary hook code that then
|
|
107
|
+
ran as shell on the next command. Both pull sites now route through
|
|
108
|
+
`tryAutoPullSystemRepo`, which pulls only when `origin` is the canonical system
|
|
109
|
+
repo (`isSystemRepoRemote`) or the exact `AGENTS_SYSTEM_REPO` the operator
|
|
110
|
+
pointed at; an unexpected origin is **refused loud** (`agents use` prints the
|
|
111
|
+
offending remote and the re-point/`AGENTS_SYSTEM_REPO` fix), never pulled. The
|
|
112
|
+
canonical system repo and a legitimate `AGENTS_SYSTEM_REPO` override still sync
|
|
113
|
+
exactly as before — no regression to trusted pulls. Source:
|
|
114
|
+
`cli/src/lib/git.ts` (`isExpectedSystemRepoRemote`, `tryAutoPullSystemRepo`),
|
|
115
|
+
`cli/src/commands/versions.ts`, `cli/src/lib/auto-pull-worker.ts`.
|
|
116
|
+
|
|
117
|
+
- **`agents monitors add --watch-pid <pid>` (PHNX-3023).** A reliable, daemon-polled
|
|
118
|
+
watcher for a backgrounded process — the fix for "will re-invoke me" watchers that
|
|
119
|
+
never fire because a harness's exit hook only wakes the agent when the watched
|
|
120
|
+
process dies, and a watch loop (`gh pr checks --watch`, a long sleep, a tick poll)
|
|
121
|
+
never itself exits. `--watch-pid` has the monitor engine's own poll loop check the
|
|
122
|
+
pid's liveness independent of the arming session, defaults its condition to fire on
|
|
123
|
+
exit, and fails loud at creation time when the pid is already dead instead of
|
|
124
|
+
silently arming a watcher that can never fire. Source: `cli/src/lib/monitors/pid-watch.ts`,
|
|
125
|
+
`cli/src/commands/monitors.ts`.
|
|
126
|
+
|
|
127
|
+
- **`agents doctor --fix` now reconciles hooks / permissions / subagents / rule
|
|
128
|
+
aliases on Windows instead of an unactionable "hold" (PHNX-3187).** On win-mini
|
|
129
|
+
`--fix` healed commands/skills/rules/plugins but reported hooks, permissions,
|
|
130
|
+
subagents, and the `CLAUDE`/`GEMINI` rule aliases as "couldn't reconcile" and
|
|
131
|
+
left the box permanently red — with real guards (`git-guard`, `rm-guard`,
|
|
132
|
+
`secrets-guard`, `ask-user-question-guard`, `public-artifact-guard`)
|
|
133
|
+
uninstalled. Four Windows-specific defects, each fixed at its source: (1)
|
|
134
|
+
**subagents** — `parseSubagentFrontmatter`/`getSubagentBody` split on `'\n'`
|
|
135
|
+
and compared the fence with `=== '---'`, so a git-CRLF-checked-out `AGENT.md`
|
|
136
|
+
(`'---\r'`) was rejected and the subagent silently dropped from discovery, so
|
|
137
|
+
it could never be installed; now CRLF-robust (`/\r?\n/`). (2) **permissions** —
|
|
138
|
+
`buildPermissionsFromGroups` extracted rules with a line regex anchored on the
|
|
139
|
+
closing quote (`"$`), which a trailing `\r` broke, extracting zero rules and
|
|
140
|
+
writing an empty permission set; now CRLF-robust. (3) **rule aliases** — git
|
|
141
|
+
checks the `rules/CLAUDE.md` and `rules/GEMINI.md` symlinks out as plain text
|
|
142
|
+
files on Windows, so `lstat().isSymbolicLink()` was false and the diff treated
|
|
143
|
+
them as independent rule sources no sync could ever produce; a new
|
|
144
|
+
`isCheckedOutSymlink` detector skips them on every platform. (4) **hooks** —
|
|
145
|
+
the heal pass fed the resource diff's extensionless hook names
|
|
146
|
+
(`git-guard`) back to the sync writer, whose `available.hooks` set carries the
|
|
147
|
+
source filename **with** its extension (`git-guard.sh`), so the exact-set match
|
|
148
|
+
found nothing and no flagged hook could be written; a new basename-tolerant
|
|
149
|
+
`resolveHookSelection` maps them, and the orphan sweep still prunes any stale
|
|
150
|
+
extensionless hook copy left in a version home (one would be invisible on
|
|
151
|
+
Windows, lacking both an extension and an exec bit). The subagents writer also now
|
|
152
|
+
surfaces a per-item write failure with its reason instead of swallowing it in a
|
|
153
|
+
bare `catch`, so a genuine failure fails loud rather than reading as a silent
|
|
154
|
+
"hold". Source: `cli/src/lib/subagents.ts`, `cli/src/lib/permissions.ts`,
|
|
155
|
+
`cli/src/lib/doctor-diff.ts`, `cli/src/lib/installations/versions.ts`,
|
|
156
|
+
`cli/src/lib/staleness/writers/subagents.ts`.
|
|
157
|
+
|
|
158
|
+
- Fix the `agents-cli` discovery skill/plugin install commands to use the real GitHub
|
|
159
|
+
repo path `phnx-labs/agi-cli` (they pointed at `phnx-labs/agents-cli`, which only
|
|
160
|
+
resolved via GitHub's rename redirect). The npm package stays `@phnx-labs/agents-cli`.
|
|
161
|
+
Also maps the plugin manifests + skill to `agents-cli-plugin.test.ts` in CI impact
|
|
162
|
+
analysis so a manifest/skill edit runs its test on the PR. (PHNX-3337 review follow-up)
|
|
163
|
+
|
|
164
|
+
- **Ship a cross-harness `agents-cli` discovery skill + Claude plugin marketplace (PHNX-3337).**
|
|
165
|
+
New `skills/agents-cli/SKILL.md` is an authoritative skill whose `description`
|
|
166
|
+
carries the exact intents a developer types — *run multiple coding agents in
|
|
167
|
+
parallel*, *manage multiple Claude Code accounts*, *I hit my usage limit*,
|
|
168
|
+
*resume a session on another machine*, *pin the agent CLI version* — each with a
|
|
169
|
+
verified `agents` command recipe (`teams`, `accounts`, `run --fallback`/`-b`/`auto`,
|
|
170
|
+
`sessions resume`, `add`/`use`). A repo-root `.claude-plugin/marketplace.json` +
|
|
171
|
+
`.claude-plugin/plugin.json` make the repo installable via
|
|
172
|
+
`claude plugin marketplace add phnx-labs/agents-cli` and `npx skills add
|
|
173
|
+
phnx-labs/agents-cli`; the plugin's `source: "./"` bundles the discovery skill
|
|
174
|
+
plus the existing per-command skills. Harness parity is registry-driven, not
|
|
175
|
+
per-harness copies: the one SKILL.md is authored once and the existing
|
|
176
|
+
capability-gated skill sync (`supports(agent, 'skills', …)`) fans it into every
|
|
177
|
+
skill-capable harness home. `claude plugin validate .` passes on the committed
|
|
178
|
+
manifest, and a real (no-mock) test reads the repo-root files and runs the CLI's
|
|
179
|
+
own `validateClaudePluginManifest`. Source:
|
|
180
|
+
`skills/agents-cli/SKILL.md`, `.claude-plugin/marketplace.json`,
|
|
181
|
+
`.claude-plugin/plugin.json`, `cli/src/lib/plugins/agents-cli-plugin.test.ts`,
|
|
182
|
+
`README.md`.
|
|
183
|
+
|
|
184
|
+
- **`agents cloud providers` no longer reports Rush as `ready` when the session token is expired (PHNX-3382).** `capabilities().available` checked only that `~/.rush/user.yaml` existed — it returned `true` even for an expired session, so the provider appeared ready but every dispatch failed with a cryptic HTTP 401. `readToken()` also silently returned an expired token, giving the same bad error on every cloud call (`dispatch`, `status`, `list`, `stream`, `cancel`, `message`). Both paths now check `expires_at` (Unix seconds): `capabilities()` returns `available: false` for a missing file, missing token, or expired session; `readToken()` throws `Rush session expired at <ISO>. Run 'rush login' to refresh.` — an actionable message instead of a 401. The exported `isRushSessionValid(yamlPath?)` helper is testable in isolation. Source: `cli/src/lib/cloud/rush.ts`.
|
|
185
|
+
|
|
186
|
+
- **Traces: session duration is populated for every harness, and the console duration median is active-time, not calendar span (PHNX-3457).** Only claude/codex/droid/gemini/opencode ever derived a `duration_ms` at scan; rush/grok/kimi/cursor/muse/antigravity left it NULL — 52% of sessions, 100% of the dominant `rush` usage — so the Evals console median was computed over only the ~48% that carried it and skewed misleadingly short. `resolveDurationMs` now fills the span at the single upsert boundary from the timestamps the row already stores (a v43→v44 migration backfills existing rows in place, no re-parse), so every harness gets a span. The `agents traces sync` index shard's `medianMs`/`p90Ms` now run over **active time** — span minus every idle gap > 120s (before the first tool call, between calls measured from each call's end, and after the last call to the session end), derived from the ordered `tool_calls` already loaded — so a session resumed after hours or left idle mid-turn no longer inflates them (it killed a 345h calendar-span outlier). The stat keys are unchanged, so the fleet-aggregate worker keeps averaging them; the raw span stays available per session as `SessionDetail.meta.spanMs`, which also gains `activeMs`. Source: `cli/src/lib/session/db.ts`, `cli/src/lib/traces/sync.ts`.
|
|
187
|
+
- **Traces: treemap tiles are drillable — each topic bucket carries example session refs (PHNX-3408).** The `agents traces sync` index shard now emits up to 30 most-recent `sessions` (`{id,title}`) per topic bucket, so the Evals console can drill from a treemap tile into that category's session list instead of rendering every tile display-only. The tile's `count` stays the true total. Source: `cli/src/lib/traces/sync.ts`.
|
|
188
|
+
|
|
189
|
+
- **`agents run --device auto` no longer lands a launch on a fleet box that is logged out for the target harness (PHNX-3466).**
|
|
190
|
+
The `--device auto` placement gate already excluded a device whose harness reads
|
|
191
|
+
signed-out — but it judged a REMOTE candidate by `agents view --json`'s display
|
|
192
|
+
`signedIn`, which is true whenever the box's active/global HOME carries a login even
|
|
193
|
+
if the per-version home the isolated run actually launches has no credential of its
|
|
194
|
+
own. The LOCAL candidate, by contrast, used the strict per-version launch truth
|
|
195
|
+
(`collectRunCandidates` → `isLaunchableSignedIn`). So a worker whose selected harness
|
|
196
|
+
version home was not launchable passed the remote gate, got picked, and the launch
|
|
197
|
+
died at spawn — from AGI EXT the dispatched tab exited 1 and vanished. `agents view
|
|
198
|
+
--json` now emits a per-version `launchable` field (the same `isLaunchableSignedIn`
|
|
199
|
+
signal the local path uses), and remote placement (`viewAgentAccountEligibility`)
|
|
200
|
+
gates on it, so both paths agree. A device with no launchable account for the harness
|
|
201
|
+
is excluded from the `--device auto` candidate set; when that empties the pool the
|
|
202
|
+
existing fail-loud `no healthy device` error fires instead of a silently-lost launch.
|
|
203
|
+
An older remote CLI that omits `launchable` falls back to `signedIn`, so a rolling
|
|
204
|
+
fleet does not regress. No second placement path was added — the single CLI gate is
|
|
205
|
+
hardened, so both the CLI and AGI EXT (which delegates placement to it) benefit.
|
|
206
|
+
Source: `cli/src/commands/view.ts`, `cli/src/lib/view-types.ts`,
|
|
207
|
+
`cli/src/lib/hosts/ready.ts`.
|
|
208
|
+
|
|
209
|
+
- **`agents traces sync` now emits SEGMENTED active-time medians so the console can
|
|
210
|
+
headline agent runs, not the corpus blend (PHNX-3472).** 63% of sessions are
|
|
211
|
+
one-shot queries (≤2 messages, ~15s active), so the blended `medianMs`/`p90Ms` sat
|
|
212
|
+
at ~15s while substantial agent runs have a ~15-minute median — the blend headlined
|
|
213
|
+
neither. The index-shard `stats` now classify each session as an AGENT run (any tool
|
|
214
|
+
call OR more than 8 messages) or INTERACTIVE, and carry `agentMedianMs`/`agentP90Ms`
|
|
215
|
+
(active-time median/p90 over agent sessions), `interactiveMedianMs`, and
|
|
216
|
+
`measuredFraction` (share of sessions with a non-null duration) alongside the
|
|
217
|
+
unchanged blended `medianMs`/`p90Ms`. Reuses the PHNX-3457 active-time computation
|
|
218
|
+
(span minus idle gaps > 120s); no calendar span reintroduced. Source:
|
|
219
|
+
`cli/src/lib/traces/sync.ts`.
|
|
220
|
+
|
|
221
|
+
- **`agents traces sync` classifies session kind and EXCLUDES internal utility calls
|
|
222
|
+
from the Evals corpus (PHNX-3474).** ~68% of the raw corpus is machine plumbing —
|
|
223
|
+
single-shot calls (no tool call AND ≤2 messages) plus known internal-prompt
|
|
224
|
+
signatures (title generation, watchdog ticks, commit-message writes, factory
|
|
225
|
+
workers), all spawned under the `claude` harness by the Rush app — and it poisoned
|
|
226
|
+
every console statistic (count, median, need-attention, tool-error-rate). Each
|
|
227
|
+
session now carries a `kind` (`utility` vs `agent`, `classifySessionKind` in
|
|
228
|
+
`traces/sync.ts`) and every index statistic is computed over the AGENT set ONLY:
|
|
229
|
+
`sessionsImported` is the real agent count (not the raw row count), and the
|
|
230
|
+
medians / `needAttention` / `toolErrorRate` / topic-bucket counts all exclude
|
|
231
|
+
utility. A new top-level `utilityCount` reports how many were dropped. Each session
|
|
232
|
+
ref (topic-tile `sessions[]` and `needsAttention`) also emits `kind` and `harness`
|
|
233
|
+
so the console can filter by both. Pure reclassification — utility rows are tagged
|
|
234
|
+
and excluded at shard-build time, never deleted from `sessions.db`. Source:
|
|
235
|
+
`cli/src/lib/traces/sync.ts`.
|
|
236
|
+
</content>
|
|
237
|
+
</invoke>
|
|
238
|
+
|
|
239
|
+
## 1.22.58
|
|
240
|
+
|
|
241
|
+
- **`agents notify` is deprecated in favor of `agents feed post` (PHNX-3323).** The command still works for existing callers, but it now prints a stderr deprecation notice naming `agents feed post` as the replacement, and `agents notify --help` carries a `[DEPRECATED]` label and examples that use `agents feed post`. Source: `cli/src/commands/send.ts`.
|
|
242
|
+
|
|
243
|
+
- **A remote browser task survives a browser-daemon restart on the driving box (PHNX-2663).** When the local daemon restarted, its in-memory task map was rebuilt from disk for local-CDP profiles but **skipped `ssh://` tunnelled tasks** — so an agent driving a browser host over `--device` got "Unknown browser task" / "Tab not found" for a tab that was still alive on the far side, and gave up. `attachRunningProfile` now re-establishes the SSH tunnel and reconnects CDP for a tunnelled task during rehydrate (reusing `connectSSH`, which attaches to the already-running remote browser via `isOwnTunnel` — it never launches one), carrying the tunnel teardown so it can't leak across the next restart. A failed reconnect falls through to disk reconcile exactly like the local-CDP path. Source: `cli/src/lib/browser/service.ts`.
|
|
244
|
+
|
|
245
|
+
- **`agents prune cleanup` no longer over-reports source-present hooks as orphans (PHNX-2693).** Orphan detection diffed the hook files in a version home against the *registered-hook manifest*, but sync copies helper / test / benchmark scripts into every version home alongside registered hooks without registering them — so each one (e.g. `permission-handler`, `verify-work-state`, `*_test`, `benchmark_*`) read as an orphan, and `agents prune cleanup [hooks | --all]` offered to trash ~1000 in-use files across version homes. It now diffs against the resolved **source** set (user + system + enabled extras hook dirs), so a hook present in any configured source is never an orphan, exactly as the command's own help promised — while a genuinely dead file (present in a home, absent from every source) is still flagged. Same detector backs `agents doctor`'s `orphan` warning. Source: `cli/src/lib/hooks/install.ts`.
|
|
246
|
+
|
|
247
|
+
- **`agents prune cleanup` no longer offers to delete plugin-provided skills (PHNX-3185).** The skill orphan detector compared a version home's skills against `skills/` + `.system/skills/` + extras only, and never credited plugin-bundled skills (`plugins/<plugin>/skills`, `.system/plugins/<plugin>/skills`) — so every skill a plugin legitimately installed (`design`, `plan`, `review`, `blog`, …) read as an orphan. On a real machine that was ~95% false positives, and `agents doctor` framed the same set as `⚠ orphans … → agents prune cleanup --all`, so following the tool's own guidance destroyed the working skill library. The detector now credits plugin skill sources when deciding orphans, and the skill-source resolver finds them too so a plugin skill reads as up-to-date instead of drifted. A genuinely dead skill (no central, extra, or plugin source) is still flagged. Source: `cli/src/lib/plugins/skills.ts`, `cli/src/lib/staleness/writers/sources.ts`.
|
|
248
|
+
|
|
249
|
+
- **`agents sync` stops reporting `✓ reconciled` while the drift it was asked to fix stays put (PHNX-3186, PHNX-2955).** `agents sync status` over-reported four classes of phantom drift that the sync writer never creates and can never clear — so re-running sync forever printed success while a phantom "N drifted / N missing" stuck (42–46 drifted per version observed fleet-wide). `diffVersionResources` now mirrors what the writer actually installs: (1) directory docs (`README`/`AGENTS`/`CLAUDE`/`GEMINI`) in `commands/` are documentation, not commands, so they are no longer counted as missing/drifted/orphan commands; (2) on command-as-skill agents (Codex ≥ 0.117, Kimi) a command whose name collides with a real same-named skill is reported present (the skill wins the shared `skills/<name>/` slot and the wrapper is deliberately never written), not the phantom "missing"; (3) version-gated presence-only kinds (e.g. subagents below the harness floor) are capability-gated so a version that structurally cannot hold the resource is not reported missing it; (4) `diffRules` now finds an agent's non-`.md` instructions file (Cursor's `.cursorrules`), which an `.md`-only scan never saw, so the rules row stops false-reporting `missing`. On top of that, every `agents sync` now runs a **post-reconcile verification**: it re-reads each version it wrote into and, if any drift the reconcile was asked to fix survives, names the exact residual resources instead of printing a bare `✓` (the umbrella line reads `⚠ sync: reconcile INCOMPLETE`, and `--json` carries `residualDrift` with `ok:false`; the exit code stays 0, matching the declined-write precedent so the fleet fan-out never discards the residual payload). Source: `cli/src/lib/doctor-diff.ts`, `cli/src/lib/sync-status.ts`, `cli/src/lib/refresh.ts`, `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/sync.ts`.
|
|
250
|
+
|
|
251
|
+
- **A stale per-version plugin marketplace copy is now reported as drift instead of `everywhere`/`ok` (PHNX-2955).** `describePluginDrift` (the content-aware plugin diff behind `agents sync status` / `agents doctor` / the post-reconcile verification) compared only the *presence* of a mirror's skills/commands, so a skill whose bytes went stale — an edit pulled into `~/.agents/.system` but never refreshed into a version's `marketplaces/…` mirror — read as fully synced while agents ran the OLD skill text. It now content-compares each skill/command present in both central and the mirror and reports `stale skill: <name>` / `stale command: <name>`, so the drift is visible and a `--yes` reconcile (which rewrites the mirror) clears it. (Out of scope, left as follow-ups: the interactive `Already in sync` short-circuit does not yet consult this signal, and `agents plugins update` for marketplace-resolved plugins is unchanged.) Source: `cli/src/lib/doctor-diff.ts`.
|
|
252
|
+
|
|
253
|
+
- **`agents sync` on a box where `~/.agents` is not a git repo adopts it in place instead of silently no-op'ing the pull (PHNX-3239).** A partial install — runtime state present but no `.git` — made the umbrella repo-pull stage fail quietly while `agents sync` still reported success, so fleet dotfiles and resources never propagated and nothing said so. The umbrella now runs the same self-heal `agents sync user` does (`adoptUserRepoIfNeeded`): it materializes the tracked resource set from origin and continues, or fails loud with the reason (and the `agents repo pull user <git-url>` hint) when there is no recorded remote to adopt from — never a silent no-op that looks like success. Source: `cli/src/lib/sync-umbrella.ts`, `cli/src/commands/status.ts`.
|
|
254
|
+
|
|
255
|
+
- **`agents browser sessions --profile <name>` surfaces the profile's real task store instead of an empty one (PHNX-3317).** A browser's live tasks and captures live under a composite runtime dir (`<name>@<device>`), but the sessions reader used the `--profile` value as a raw directory name and read the bare legacy `<name>/` dir — which is empty on a machine where the daemon writes the composite key — so a driving agent saw "no captures" / "Unknown browser task" while 13 real tasks sat one dir over. The reader now resolves a profile to every cache dir that belongs to it via `listProfileCacheDirs`/`keyBelongsToProfile` — the same single rule `status()` and `findTask` already use (RUSH-2709) — so the bare name and its `@<device>` stores are unified in the listing. Source: `cli/src/lib/browser/sessions-list.ts`.
|
|
256
|
+
|
|
257
|
+
- **Failure phenotype folded into the traces insight-engine fingerprint via a persisted per-session cache (PHNX-3327).** `computeInsights()` grouped cross-session failure patterns by `(tool, cause, normalized-error)` but did not carry a `phenotype` (false-termination / premature-completion / out-of-order / failure-to-act) — that classification needs the full derived trajectory, which flat `tool_calls` rows don't have. It is now computed per-session once (`classifyPhenotype`) and persisted in a new mtime+size-keyed `session_phenotypes` cache (same shape as `session_topics` / `session_insights`), then read for the **whole corpus** on every sync and folded in as a fourth grouping dimension. Reading the cache for all rows — never just this run's freshly-parsed batch — is what keeps two identically-signatured sessions in ONE failure cluster regardless of which incremental sync first saw each (the fragmentation an in-memory, batch-only map would cause). The `signature` output (`{ tool, cause, key }`) is unchanged; `FailurePattern` gains a `phenotype` field and the pattern id incorporates it, so the same signature under two phenotypes is two deep-linkable clusters. Source: `cli/src/lib/traces/insights.ts`, `cli/src/lib/traces/sync.ts`, `cli/src/lib/session/db.ts`.
|
|
258
|
+
|
|
259
|
+
- **`agents doctor` and `agents setup` no longer crash on macOS when the Keychain helper source is unavailable (PHNX-3385).** Both read the reserved `auth` secrets bundle as an optional status check; on a macOS box where the bundled/local Keychain helper couldn't be found (a source checkout, or a fresh npm install before `agents setup secrets` downloads the signed helper — RUSH-3100 dropped it from the tarball), that check threw and took the whole command down with it instead of reporting the finding as unavailable. Source: `cli/src/lib/secrets/bundles.ts` (`inspectReservedAuthBundle`), `cli/src/lib/auth-mint.ts` (`hasMintedSetupToken`).
|
|
260
|
+
|
|
261
|
+
- **`agents traces sync` reports a truthful run outcome — recover-then-succeed is `completed`, not `errored` (PHNX-3387).** `SessionDetail.meta.outcome` was derived as `errorCount > 0 ? 'errored' : 'completed'`, so a run that hit a tool failure but *recovered* and finished the task was mislabeled `errored` — and the Evals console's "green run with hidden tool failures" callout (the LangSmith trap) could never fire honestly because `surfacedToolFailures` only existed when the run was already `errored`. Outcome is now derived from a causal-recovery predicate (`recoveredAfterErrors`, shared with the false-termination phenotype in `phenotype.ts`): a run with tool errors is `completed` only when a substantive, non-human-facing tool step succeeded strictly after the last error **and resolved the failed work** — its work signature (the effective program for a shell step, the tool identity otherwise) matches an errored step's. A run that ended in error, punted to a human (`AskUserQuestion`), or whose only post-error success is unrelated work (a failed `bun test` followed by an incidental `ls`) stays `errored` — no regression. `surfacedToolFailures` is retained on `completed` runs so the console can honestly show "recovered from these failures." Source: `cli/src/lib/traces/sync.ts`, `cli/src/lib/traces/phenotype.ts`.
|
|
262
|
+
|
|
263
|
+
- **`--strategy balanced` no longer routes to a weekly-exhausted Claude account on worker boxes; a missing usage snapshot is treated as unknown (fails closed) instead of full capacity (PHNX-3392).** A worker's usage cache is empty (the setup-token lacks the `user:profile` scope, so `/api/oauth/usage` 403s — RUSH-2392), and `capacityWeight(null, …)` scored that absence as 100 — max weight — making a blind, actually-exhausted account the top pick. The null arm now draws `UNVERIFIED_WEIGHT` (1): any verified-healthy account outranks an unverifiable one, while an all-unverified pool still draws a pick. Pinned by spec GWT-E5c, a pure unit test (`capacity.test.ts`), and a real-cache-path regression test proving a persisted 7d-100% `week` window makes the account ineligible. Source: `cli/src/lib/accounting/capacity.ts`, `cli/src/lib/accounting/{capacity.test.ts,rotate.test.ts}`.
|
|
264
|
+
|
|
265
|
+
- **A worker pulling Claude usage from a Windows primary no longer silently blanks its cache (PHNX-3392 follow-up).** The `usage-sync` pull path (`pullUsageFromPrimary`) now wraps the primary's JSON in `stripClixml` before parsing, exactly as every other remote-JSON boundary does. Without it, a headed Windows box serving `__usage-export` prepends a PowerShell CLIXML progress banner to stdout, every pull fails as malformed JSON, the worker's usage cache stays null, and the capacity floor becomes the only thing keeping `--strategy balanced` from picking a blind, exhausted account. Source: `cli/src/lib/accounting/usage-sync.ts`.
|
|
266
|
+
|
|
267
|
+
- **`agents upgrade` self-heals the orphaned npm reify staging dir that dead-ended it forever (PHNX-3393).** npm stages an upgrade in a hidden sibling directory next to the running install, deterministically named from the install's path, then renames it into place; a crash mid-upgrade left that directory behind non-empty, and every subsequent upgrade's rename onto the same path failed `ENOTEMPTY` — recurring across boxes with no re-run ever clearing it. `agents upgrade` now sweeps any such orphan before reifying, and `agents doctor --fix` / the daemon's periodic self-heal do the same fleet-wide (with a 10-minute age guard so a concurrent in-flight upgrade is never touched). Source: `cli/src/lib/self-update.ts` (`sweepStaleInstallStaging`), `cli/src/bootstrap.ts`, `cli/src/lib/self-heal/checks/install-staging.ts`.
|
|
268
|
+
|
|
269
|
+
- **The Evals console's "all agents" view now renders real fleet data (PHNX-3397).** The CLI writes traces per-device (`<userId>/<hostname>/…`), but the console asks for the fleet-wide `<userId>/all/…` view, which was never a stored key — so every request 404'd and the console fell back to sample data. The traces Worker now synthesizes `/all` on read: it lists the owner's device prefixes (one delimited list, not every session object) and merges their shards for `all/index.json`, and resolves `all/sessions/<id>.json` to whichever device holds that session. A single device is an exact passthrough; across devices, counts sum and failure patterns fold by id (median/latency are session-weighted approximations, exact when filtered to one device). No CLI change and no extra upload — the per-device shards already in R2 are the source. Owner auth is unchanged: `/all` still requires a Phoenix bearer whose id matches the path prefix. Source: `cli/src/lib/traces/worker-template.ts`.
|
|
270
|
+
|
|
271
|
+
- **`agents traces sync` no longer silently ships a stale Evals dashboard when the sessions DB is busy (PHNX-3401).** `buildIndexShard` writes freshly-classified topics/insights back into the local `sessions.db` while building the console index. If another process (e.g. the Rush app) holds that DB, the write waits out the 30s `busy_timeout` and throws `SQLITE_BUSY` — which used to escape the build and get swallowed by a bare `catch {}`, so per-session shards uploaded but the aggregated index (wasted-time, failure clusters, latency) never refreshed and the console sat stale while `sync` reported a clean run. The cache write-backs are now best-effort (they only warm a cache the shard doesn't read from), and a genuine index build/upload failure is surfaced as `SyncResult.indexError` plus a `⚠ console index not refreshed` warning instead of a silent success. Source: `cli/src/lib/traces/sync.ts`, `cli/src/commands/traces.ts`.
|
|
272
|
+
|
|
273
|
+
- **A release now redeploys the managed share OG-cover Worker so prod can't drift from the shipped template (PHNX-3403).** The Worker that renders share preview cards was deployed only by a manual `agents artifacts share update` run, decoupled from the release train — so a release could publish a CLI whose `worker-template.ts` changed while the deployed Worker stayed stale (the gap that made PHNX-2835 look shipped while every new share still 404'd its cover). `release.sh` now takes `--deploy-worker <auto|on|off>` (default `auto`): after publish, on the home base, it redeploys the Worker from the just-published CLI when the shipped template differs from what the endpoint deployed (`auto`), always (`on`), or never (`off`). The decision is a pure local render+hash via the new `agents artifacts share update --check` (needs no Cloudflare credentials), so an unchanged template is a cheap no-op; the promote preflight verifies the share endpoint and `cloudflare.com` token on the home base before publishing, and a deploy failure fails the release loud with the manual fallback. Source: `cli/scripts/release.sh`, `cli/scripts/promote-home-base-probe.sh`, `cli/src/commands/share.ts`.
|
|
274
|
+
|
|
275
|
+
- **A browser task created over `agents ssh <host> 'agents browser …'` is owned by the real caller, not `UNRESOLVED@<host>` (PHNX-3405).** The `agents ssh` browser-drive passthrough stamped the fleet-remote consent marker (`AGENTS_FLEET_REMOTE=1`, PHNX-3065) but never forwarded the caller's actor, so the peer's daemon re-resolved identity from its own empty env and collapsed every such task to `UNRESOLVED@<host>` — making the browser look contended by phantom owners. `markFleetRemote` (the single seam behind `buildSshInvocation`, the fleet passthrough, and `leasedBoxRemoteCmd`) now forwards `AGENTS_ACTOR*`/`GIT_*` alongside the marker, in both POSIX and PowerShell dialects, marker-first so the consent gate and idempotency guard are unchanged. This closes the same RUSH-2028 actor-provenance gap on the `agents ssh` seam that the `--device` dispatch already closed via `withActorEnv`. Source: `cli/src/lib/devices/connect.ts`.
|
|
276
|
+
|
|
277
|
+
- **Monitor `--run` actions now require an authenticated account and resolve routine hooks (PHNX-3406).** Claude routine selection verifies the balanced account with Claude's own local `auth status` instead of trusting stale identity metadata, then launches with the operator HOME plus that version's `CLAUDE_CONFIG_DIR`, matching ordinary `agents run` so both the native login and portable `~/...` hooks resolve. `agents monitors list` and its JSON output also surface the latest asynchronous action failure, and the docs clarify that `--run` creates a new headless conversation rather than resuming the caller. Source: `cli/src/lib/{sandbox,daemon/runner}.ts`, `cli/src/commands/monitors.ts`, `cli/docs/monitors.md`.
|
|
278
|
+
|
|
279
|
+
- **`agents sessions fork` now works cross-device and cross-harness (PHNX-3409).** Forking a session that lived on another box used to dead-end with `transcript not found` — fork resolved the source with a local-only lookup and copied the transcript off local disk, and it only worked for Claude. It now resolves the source across the fleet (the same path `agents sessions preview` uses) and launches a NEW same-harness session, load-balanced, seeded with a recap of the source (label, cwd, ticket, last state, changed-files, and the source id for `/continue`) — so a session on any device, in any REPL harness, forks and picks up where it left off. Add `--device <host>` to place the sibling on the fleet or `--terminal` to open it in a fresh tab; `--name` still labels it. Source: `cli/src/commands/fork.ts`, `cli/src/lib/session/fork.ts`.
|
|
280
|
+
|
|
281
|
+
- **`agents routines` now says WHY a routine failed, not just that it did (PHNX-3410).** The list, the interactive detail's Recent runs, and `agents routines logs` showed only the status enum (`failed`/`missed`/`skipped`) — so diagnosing a routine that keeps failing on `auth_failed: Please run /login`, a wedged active-run lock, a readiness block, or a missed slot meant drilling into the run dir. A new `runFailureReason(run)` maps the `RunMeta` already on disk (`errorMessage` / `readiness` / `skipReason` / `missed`) to one compact line, rendered inline in the `Last Status` cell, the detail's Recent runs, and a `reason:` line in the `logs` header. Visibility only — no scheduling behavior changes. Source: `cli/src/commands/routines.ts`.
|
|
282
|
+
|
|
283
|
+
- **The daemon warm-tick session indexer no longer re-derives an active non-Claude/Codex session's whole tool history every tick — the root cause behind the wedged event loop and browser-IPC stalls (PHNX-3411).** The symptom fix (above) made a wedged browser daemon fail loud; this removes what wedged it. The warm-tick indexer parsed AND re-redacted every tool call in a changed session on every tick, in `toolIndexMode: 'replace'`. claude/codex resumed from a stored parse offset; the other 11 harnesses (kimi, grok, opencode, antigravity, cursor, droid, …) did not, so an active large session on the interactive hub re-sanitized its entire tool history several times a second — synchronous work that blocked the daemon event loop, so `browser.sock` accepted connections it never answered. The full-file harness path is incremental now too: the tool ledger records how many events were folded plus the collector snapshot, and a later scan of the same append-only stream re-derives only the newly appended tool calls (`mode: 'append'`), falling back to a full re-scan for a first scan, a truncated/rewritten transcript, or an extractor-version bump. Measured on a 400-tool-call Grok session indexed once per tick, re-derive+redact work drops from 80,200 tool calls to 400 over 400 ticks (~201× less), with the index content byte-for-byte identical to a full re-parse. Source: `cli/src/lib/session/{db,tool-calls,tool-store}.ts`.
|
|
284
|
+
|
|
285
|
+
- **`agents browser` detects a wedged daemon instead of hanging on it (PHNX-3411).** A unix-socket `connect` succeeds at the kernel level even when the browser daemon's shared event loop is blocked, so `isDaemonReachable` (a bare accept probe) reported a busy-but-unresponsive daemon as healthy — `browser.sock` accepted every connection while the daemon never replied, and cross-device browser drives to a busy hub (e.g. zion) hung or surfaced a confusing `Timeout waiting for browser daemon socket`. Readiness is now a *reply*: before serving a verb the CLI requires the daemon to actually answer a `version` probe (`isDaemonResponsive`, retried so a transient GC pause never condemns a healthy daemon), and a reachable-but-wedged daemon fails loud — naming the blocked event loop, the daemon log, and the `agents browser stop --daemon` reset — instead of hanging (which also fixes `reconcileDaemonVersion`'s own untimed version probe). Also adds a critical-path regression guard proving the daemon's session-index warm tick resumes an actively-growing Claude session incrementally (per-tick cost tracks appended bytes, never a full O(session) re-parse) — the invariant whose violation would re-starve browser IPC. Source: `cli/src/lib/browser/ipc.ts`, `cli/src/lib/session/__tests__/active-session-warmtick-incremental.test.ts`.
|
|
286
|
+
|
|
287
|
+
- **A routine whose dispatch account is signed out now records a `blocked` run instead of a doomed `failed` one (PHNX-3415).** When the daemon fires a routine, it preflights the rotation-resolved account against the auth-health cache; if that account is provably signed out (`revoked` or `unconfigured` — the "OAuth revoked" / "Please run /login" cases), it records a terminal `blocked` / `agent_auth_failed` run carrying the exact re-login repair (`agents run <agent>@<v> -- login`) and spawns nothing — instead of launching a run that 401s, lands as `failed`, and burns a session (surfaced from the 2026-08-28 reliability investigation, where signed-out accounts were the top cause of routine failures). The check is cache-only and fails **open** on any still-usable or indeterminate verdict (`rate_limited`/`unverified`/`expired`/`error`/absent), so a stale probe never wedges a routine. Broadens and corrects the prior preflight, which only caught `revoked` and mislabeled it `failed`. Source: `cli/src/lib/routine-readiness.ts` (`fireTimeAuthReadiness`), `cli/src/lib/daemon/runner.ts`; spec RT-12.
|
|
288
|
+
|
|
289
|
+
- **`agents fleet apply` on an empty roster now tells you how to fix it (PHNX-3422).** When `fleet.devices` is `{}` (the default on a fresh box), `apply`/`apply --plan` used to print a gray `No target devices — nothing to apply.` that read like the command was broken. It now prints an actionable `fleet.devices is empty — nothing to converge.` plus the two ways to declare a roster (`fleet: { devices: all }` or a named map), and still exits 0. Other zero-target cases (a `devices: all` fleet with no online peers, or a named roster whose every device was unresolved — off-tailnet/ignored/typo) keep the plain note, since their reason was already surfaced above. Source: `cli/src/lib/fleet/manifest.ts` (`emptyTargetsMessage`), `cli/src/commands/apply.ts`.
|
|
290
|
+
|
|
291
|
+
- **Traces now attributes a failed tool call's OWN blocking duration to wasted time, not just the bounded gap to the next call (PHNX-3437).** `tool_calls` gains an `end_timestamp` column (the call's result-record time, already available at ingestion) so the insight engine books `end - start` — the time a call actually blocked — as its wasted time, independent of whether another call follows. Before, a call that hung for minutes and then failed registered as ~0 waste whenever it was the last call in its session or was followed quickly by an unrelated call. This is what makes a fail-fast fix measurable on the Evals console: a channel that stops hanging ~5.5 minutes on stdin and instead fails in under a second (PHNX-3407) drops from ~5.5 minutes of attributed waste to ~0. The existing 30-minute-bounded inter-call gap heuristic is kept as the fallback for rows an older extractor stored (NULL end), so nothing crashes or yields NaN, and — when the end time is known — the post-call retry/stall gap is measured from the call's END so the blocking duration is never double-counted. `wastedMs`/`wastedMsTotal`/`failurePatterns` shard shapes are unchanged. Source: `cli/src/lib/session/{db.ts,tool-calls.ts,tool-store.ts}`, `cli/src/lib/traces/{sync.ts,insights.ts}`.
|
|
292
|
+
|
|
293
|
+
- **`agents view`: an idle Claude session window no longer renders as a full red "unavailable" block (PHNX-3453).** When a Claude account has no usage in the current rolling 5-hour window, the usage API returns `five_hour.utilization = null`, so no `S:` session bar exists — and `agents view` drew that missing window as a solid red `█████ unavailable`, which reads as "100% used / maxed-out", the opposite of what it means. It now renders a neutral dim dashed row `┄┄┄┄┄ unavailable`, visually distinct from both a real `0%` (`░░░░░`) and a full `100%` (`█████`). The reading itself is unchanged and honest — an idle account genuinely has no current 5-hour number; the bar fills in the moment that account runs Claude, or on `agents view claude --refresh` when the box isn't being usage-endpoint rate-limited. Source: `cli/src/lib/accounting/usage.ts` (`formatUsageSummary`, `NO_DATA`).
|
|
294
|
+
|
|
3
295
|
## 1.22.57
|
|
4
296
|
|
|
5
297
|
- **Removed two duplicate commands: `agents list` and `agents trash restore` (PHNX-3391).** `agents list` was a long-deprecated full duplicate of `agents view` (it already printed "agents list is now agents view" before delegating) — `agents view` is now the one version-listing surface. `agents trash restore <target>` was an exact duplicate of the top-level `agents restore <target>` (both call `restoreVersion`); `agents restore` stays, and `agents trash` keeps `trash list`. Both removed names now error as retired top-level commands rather than auto-correcting. Source: `cli/src/commands/versions.ts`, `cli/src/commands/trash.ts`.
|
|
@@ -20,6 +312,8 @@
|
|
|
20
312
|
|
|
21
313
|
- **Managed-share Open Graph cards render in Cloudflare instead of failing on every new publish (PHNX-2835).** The Worker upload now carries Yoga and resvg as compiled WebAssembly modules alongside the bundled JavaScript and fonts. The previous single-file bundle decoded the WASM into byte arrays and compiled it at request time; Node accepted that in tests, but Cloudflare workerd forbids runtime WASM code generation, so a new share's lazy `<slug>.png` render failed. A real-workerd regression test now renders and validates the PNG, and a genuine renderer failure returns a diagnostic `500` instead of falling through as a missing cover. Source: `cli/src/lib/share/{worker-template,provision}.ts`.
|
|
22
314
|
|
|
315
|
+
- **A dangling per-harness default account no longer hard-fails every launch (PHNX-3326).** When `agents.yaml` `accounts.defaults.<agent>` pointed at an account id absent from the local registry, `resolveSpawnAccount` threw `Unknown account '<uuid>'` and the launch exited before doing any work. The default is now a preference, not a requirement: a stale pointer emits a warning naming the harness and the remedy (`agents accounts clear-default <agent>`), then falls back to the existing balanced-selection path. `accounts set-default`/`switch` now store the portable account name instead of the per-device uuid; legacy uuid entries continue to resolve, and `accounts remove`/`switch` correctly treat both name- and id-stored defaults as references. Source: `cli/src/lib/account-registry.ts`, `cli/src/commands/accounts.ts`.
|
|
316
|
+
|
|
23
317
|
## 1.22.55
|
|
24
318
|
|
|
25
319
|
- **Managed shares generate their Open Graph cover server-side (PHNX-2835).** Publishing HTML to `share.agents-cli.sh` no longer launches local Chromium. The Worker lazily renders a deterministic 1200×630 AGI card with Satori and resvg-wasm, bundles Inter and JetBrains Mono, inherits the canonical page's visibility gate, and caches the PNG in R2. BYO endpoints keep their local screenshot fallback. An explicit missing `AGENTS_SHARE_BROWSER` or `PUPPETEER_EXECUTABLE_PATH` now fails loudly instead of silently falling through. Source: `cli/src/lib/share/{capture,publish,worker-template}.ts`.
|
package/README.md
CHANGED
|
@@ -59,6 +59,20 @@ Everything here — and every other command in this README — is free and needs
|
|
|
59
59
|
The command surface teaches setup through `agents setup` and group-level `--help`.
|
|
60
60
|
The durable system model starts at [`cli/docs/README.md`](cli/docs/README.md).
|
|
61
61
|
|
|
62
|
+
**What `agents setup` installs, and what auto-updates.** Setup clones a small public
|
|
63
|
+
**system repo** (`phnx-labs/.agents-system`) into `~/.agents/.system/`. It ships the
|
|
64
|
+
default resources — including **hooks**, which run as shell commands on tool events —
|
|
65
|
+
and its checkout **fast-forwards from its origin** when you run `agents use` (and, only
|
|
66
|
+
if you opt in with `AGENTS_AUTO_PULL=1`, in the background). Two safeguards bound that:
|
|
67
|
+
the pull is `merge --ff-only` (it can never rewrite your local history), and it is
|
|
68
|
+
**verified against the expected origin** — a checkout whose `origin` is not the canonical
|
|
69
|
+
system repo (or the exact repo you named in `AGENTS_SYSTEM_REPO`) is refused, never
|
|
70
|
+
pulled, so a repointed remote cannot slip hook code onto your machine. To pin or opt out:
|
|
71
|
+
set `AGENTS_SYSTEM_REPO=gh:you/your-fork` to track your own audited copy, run
|
|
72
|
+
`agents setup --no-system-repo` to skip the clone entirely, or check out a specific tag
|
|
73
|
+
in `~/.agents/.system/` (a fast-forward only advances a moving branch, so a detached tag
|
|
74
|
+
stays put).
|
|
75
|
+
|
|
62
76
|
**Learn (concepts):** [Loop + graph engineering](https://agi-cli.sh/learn/loop-and-graph-engineering) · [Teams as graph engineering](https://agi-cli.sh/learn/teams-graph-engineering) · [Sessions · index + cross-device](https://agi-cli.sh/learn/sessions-index) · [Distributed fleet execution](https://agi-cli.sh/learn/distributed-fleet). Also: [harness engineering](https://agi-cli.sh/learn/harness-engineering) · [visual longform](https://share.agents-cli.sh/muqsitnawaz/agents-loop-and-graph-engineering).
|
|
63
77
|
|
|
64
78
|
Already installed? `agents upgrade` updates agi-cli itself to the latest version (`agents upgrade 1.2.3` for a specific version or dist-tag, `-y` to skip the confirm prompt). The command is `upgrade` on every platform -- do not reach for `agents update`, which updates an installed **agent harness**, not agi-cli (and on macOS, `agents helper update` is a third thing: it reinstalls the keychain helper).
|
|
@@ -898,6 +912,21 @@ Skills, commands, and subagents are declarative and never trip the gate. The gat
|
|
|
898
912
|
|
|
899
913
|
Plugins live in the user repo (`~/.agents/plugins/`), not inside any single version home. Switching Claude via `agents use claude@<v>` re-syncs the plugin into the new version automatically — no re-install. New Claude versions added later pick it up on their first sync. Project-level `<repo>/.agents/plugins/<name>/` overrides a same-named user plugin (resolution is project > user > system, same as every other resource).
|
|
900
914
|
|
|
915
|
+
### Install the agents-cli skill in any agent
|
|
916
|
+
|
|
917
|
+
This repo is itself a Claude plugin marketplace and a [skills.sh](https://skills.sh) source. The `agents-cli` skill teaches any coding agent (Claude Code, Codex, Cursor, …) how to drive the `agents` CLI — so when you ask *"how do I run multiple coding agents in parallel?"* the agent surfaces `agents teams` instead of guessing.
|
|
918
|
+
|
|
919
|
+
```bash
|
|
920
|
+
# Claude Code — add this repo as a marketplace, then install the plugin
|
|
921
|
+
claude plugin marketplace add phnx-labs/agi-cli
|
|
922
|
+
claude plugin install agents-cli@agents-cli
|
|
923
|
+
|
|
924
|
+
# skills.sh — install the skill directly from the repo
|
|
925
|
+
npx skills add phnx-labs/agi-cli
|
|
926
|
+
```
|
|
927
|
+
|
|
928
|
+
The manifest is `.claude-plugin/marketplace.json` (validate with `claude plugin validate .`); the skill source is [`skills/agents-cli/SKILL.md`](skills/agents-cli/SKILL.md). Its `description` carries the exact intents the runtime matches against — *run multiple coding agents in parallel*, *manage multiple Claude Code accounts*, *I hit my usage limit*, *resume a session on another machine*, *pin the agent CLI version* — each with a verified command recipe.
|
|
929
|
+
|
|
901
930
|
---
|
|
902
931
|
|
|
903
932
|
## Make it yours
|
package/dist/bootstrap.js
CHANGED
|
@@ -29,7 +29,7 @@ const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
|
29
29
|
const packageJsonPath = path.join(__dirname, '..', 'package.json');
|
|
30
30
|
const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf-8'));
|
|
31
31
|
const VERSION = packageJson.version;
|
|
32
|
-
import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, } from './lib/self-update.js';
|
|
32
|
+
import { NPM_PACKAGE_NAME, deriveGlobalPrefix, detectPackageManager, ensureGlobalBinLinks, installPackageIntoPrefix, installPackageWithBun, verifyInstalledVersion, refreshAliasShims, downloadVerifiedTarball, sweepStaleInstallStaging, } from './lib/self-update.js';
|
|
33
33
|
import { registerUpgradeCommand } from './commands/upgrade.js';
|
|
34
34
|
// Detect dev/working-tree builds and default the noisy startup steps off.
|
|
35
35
|
// Three cases trip this:
|
|
@@ -321,6 +321,13 @@ async function installResolvedPackage(metadata) {
|
|
|
321
321
|
// trusted .tgz. A mismatch throws and nothing below runs — fail closed.
|
|
322
322
|
const tarball = await downloadVerifiedTarball(metadata.tarball, metadata.integrity);
|
|
323
323
|
try {
|
|
324
|
+
// Clear any orphaned npm reify staging dir from a prior crashed upgrade
|
|
325
|
+
// BEFORE the package manager stages the new one (PHNX-3393) — otherwise
|
|
326
|
+
// npm's rename onto that exact, deterministic path fails ENOTEMPTY and
|
|
327
|
+
// every subsequent upgrade dead-ends there forever. bun does not use
|
|
328
|
+
// npm's retire-path staging scheme, so this only needs to run once, ahead
|
|
329
|
+
// of both package-manager branches below.
|
|
330
|
+
sweepStaleInstallStaging(packageRoot);
|
|
324
331
|
// Upgrade with the package manager that owns this install. A bun global
|
|
325
332
|
// install lives at <bunGlobalDir>/node_modules/... (no `lib` segment), so an
|
|
326
333
|
// `npm install --prefix` would write to <bunGlobalDir>/lib/node_modules and
|
|
@@ -343,6 +350,32 @@ async function installResolvedPackage(metadata) {
|
|
|
343
350
|
}
|
|
344
351
|
verifyInstalledVersion(packageRoot, metadata.version);
|
|
345
352
|
refreshAliasShims(packageRoot);
|
|
353
|
+
// PHNX-2768: the npm install above can leave the package at the new version
|
|
354
|
+
// but the global bin links GONE — the state that stranded zion (package at
|
|
355
|
+
// 1.22.40, `/opt/homebrew/bin/{agents,ag,browser,computer}` missing, every
|
|
356
|
+
// `agents` invocation "command not found"). The upgrade OWNS those links, so
|
|
357
|
+
// it restores any that npm dropped and fails LOUD when one cannot be made to
|
|
358
|
+
// resolve — never returning a box the package upgraded but cannot run. Only
|
|
359
|
+
// the npm-prefix POSIX layout has these symlinks; bun and Windows use their
|
|
360
|
+
// own bin shims and are out of scope.
|
|
361
|
+
if (detectPackageManager(packageRoot) !== 'bun' && process.platform !== 'win32') {
|
|
362
|
+
const prefix = deriveGlobalPrefix(packageRoot);
|
|
363
|
+
const repairs = ensureGlobalBinLinks(packageRoot, prefix);
|
|
364
|
+
const repaired = repairs.filter((r) => r.action === 'repaired');
|
|
365
|
+
const failed = repairs.filter((r) => r.action === 'failed');
|
|
366
|
+
if (repaired.length > 0) {
|
|
367
|
+
console.error(chalk.yellow(`Relinked ${repaired.map((r) => r.name).join(', ')} in ${path.join(prefix, 'bin')} — the install left them missing.`));
|
|
368
|
+
}
|
|
369
|
+
if (failed.length > 0) {
|
|
370
|
+
const relink = failed
|
|
371
|
+
.map((r) => `ln -sf ${path.relative(path.dirname(r.linkPath), r.target)} ${r.linkPath}`)
|
|
372
|
+
.join(' && ');
|
|
373
|
+
throw new Error(`upgraded to ${metadata.version} but could not restore the ` +
|
|
374
|
+
`${failed.map((r) => r.name).join(', ')} command link${failed.length === 1 ? '' : 's'} in ` +
|
|
375
|
+
`${path.join(prefix, 'bin')} (${failed.map((r) => r.error).join('; ')}). ` +
|
|
376
|
+
`The box has the new package but no working \`agents\` — relink manually: ${relink}`);
|
|
377
|
+
}
|
|
378
|
+
}
|
|
346
379
|
// The npm install above runs with --ignore-scripts, so the postinstall that
|
|
347
380
|
// installs the macOS Keychain helper never fires on upgrade. Force-refresh the
|
|
348
381
|
// helper here so a user upgrading FROM a broken build (e.g. the entitlement-less
|
|
@@ -720,6 +753,11 @@ async function runUpgrade(version, options) {
|
|
|
720
753
|
return;
|
|
721
754
|
spinner.fail(`Upgrade failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
722
755
|
console.log(chalk.gray(`Run manually: agents upgrade ${version ? version + ' ' : ''}--yes`));
|
|
756
|
+
// A failed upgrade MUST exit non-zero (PHNX-2768). The fleet rollout
|
|
757
|
+
// keys a box `ok` on `agents upgrade` exiting 0 alone; exiting 0 on
|
|
758
|
+
// failure is what let a stranded box (package upgraded, bin links gone)
|
|
759
|
+
// be reported merely `unverified` instead of `failed`.
|
|
760
|
+
process.exitCode = 1;
|
|
723
761
|
}
|
|
724
762
|
}
|
|
725
763
|
function registerUpgradeRuntimeCommand(p) {
|
|
@@ -342,13 +342,17 @@ export function setDefaultAccount(agentRaw, name) {
|
|
|
342
342
|
}
|
|
343
343
|
assertNativeAccountNameable(account.agent);
|
|
344
344
|
}
|
|
345
|
-
|
|
345
|
+
// Reference by NAME, not id: defaults sync fleet-wide with `agents repo push/pull`
|
|
346
|
+
// while account ids are minted per-device, so an id ref breaks on every other
|
|
347
|
+
// machine ("Unknown account '<uuid>'"). Names are the portable handle — the
|
|
348
|
+
// registry resolves both, and existing uuid entries still resolve.
|
|
349
|
+
updateMeta(meta => ({ ...meta, accounts: { ...meta.accounts, defaults: { ...meta.accounts?.defaults, [agent]: account.name } } }));
|
|
346
350
|
return { agent, account };
|
|
347
351
|
}
|
|
348
352
|
async function switchAccountRows(agent) {
|
|
349
353
|
const accounts = await listSwitchableAccounts(agent);
|
|
350
354
|
const candidates = await collectRunCandidates(agent);
|
|
351
|
-
const
|
|
355
|
+
const defaultValue = readMeta().accounts?.defaults?.[agent];
|
|
352
356
|
return accounts.map(account => {
|
|
353
357
|
const candidate = account.kind === 'native'
|
|
354
358
|
? candidates.find(row => row.accountKey === account.identityKey) ?? null
|
|
@@ -357,7 +361,7 @@ async function switchAccountRows(agent) {
|
|
|
357
361
|
accountName: account.name,
|
|
358
362
|
kind: account.kind,
|
|
359
363
|
detail: account.kind === 'provider' ? account.provider : (account.identityLabel ?? account.identityKey),
|
|
360
|
-
current: account.id ===
|
|
364
|
+
current: account.id === defaultValue || account.name === defaultValue,
|
|
361
365
|
candidate,
|
|
362
366
|
};
|
|
363
367
|
});
|
package/dist/commands/apply.js
CHANGED
|
@@ -16,7 +16,7 @@ import { machineId } from '../lib/session/sync/config.js';
|
|
|
16
16
|
import { loadDevices } from '../lib/devices/registry.js';
|
|
17
17
|
import { isHostPinned, managedKnownHostsPath } from '../lib/devices/known-hosts.js';
|
|
18
18
|
import { ensureDevicesRegistered } from '../lib/devices/sync.js';
|
|
19
|
-
import { readFleetFile, resolveDesired } from '../lib/fleet/manifest.js';
|
|
19
|
+
import { readFleetFile, resolveDesired, emptyTargetsMessage } from '../lib/fleet/manifest.js';
|
|
20
20
|
import { snapshotAuth } from '../lib/fleet/auth-sync.js';
|
|
21
21
|
import { AUTH_BUNDLE_NAME } from '../lib/secrets/bundles.js';
|
|
22
22
|
import { agentIdOf, diffFleet, probeDevice, runFleetApply, pool, sourceHome, expandAllSpecs, rosterNeedsVersions, fleetSecretsBundles, } from '../lib/fleet/apply.js';
|
|
@@ -188,7 +188,15 @@ async function runApply(opts) {
|
|
|
188
188
|
if (opts.login === false)
|
|
189
189
|
desired = desired.map((d) => ({ ...d, login: 'skip' }));
|
|
190
190
|
if (desired.length === 0) {
|
|
191
|
-
|
|
191
|
+
const msg = emptyTargetsMessage(manifest);
|
|
192
|
+
if (msg.style === 'hint') {
|
|
193
|
+
console.log(chalk.yellow(msg.lines[0]));
|
|
194
|
+
for (const line of msg.lines.slice(1))
|
|
195
|
+
console.log(chalk.gray(` ${line}`));
|
|
196
|
+
}
|
|
197
|
+
else {
|
|
198
|
+
console.log(chalk.gray(msg.lines[0]));
|
|
199
|
+
}
|
|
192
200
|
return;
|
|
193
201
|
}
|
|
194
202
|
// Snapshot source auth once for every agent named anywhere in the profile.
|
package/dist/commands/fork.d.ts
CHANGED
|
@@ -1,19 +1,32 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* `agents sessions fork <session>` — branch an existing conversation into a new,
|
|
3
|
-
* independent session you can continue separately. The original is untouched.
|
|
4
|
-
* Also exposed as the hidden top-level alias `agents fork` (back-compat).
|
|
5
|
-
*
|
|
6
|
-
* Thin command layer; the copy/register logic lives in `lib/session/fork.ts`.
|
|
7
|
-
*/
|
|
8
1
|
import type { Command } from 'commander';
|
|
9
2
|
interface ForkOptions {
|
|
10
3
|
name?: string;
|
|
4
|
+
device?: string;
|
|
5
|
+
/** Open the sibling in a real terminal tab instead of in-place; optional backend. */
|
|
6
|
+
terminal?: string | boolean;
|
|
7
|
+
}
|
|
8
|
+
/**
|
|
9
|
+
* The two process boundaries fork crosses — a preview subprocess (cross-fleet
|
|
10
|
+
* resolve + digest) and the sibling launch. Injectable so the resolve→recap→run
|
|
11
|
+
* argv logic is unit-tested without spawning real CLIs.
|
|
12
|
+
*/
|
|
13
|
+
export interface ForkDeps {
|
|
14
|
+
/** Run `agents sessions preview <sub…>` and capture stdout + exit status. */
|
|
15
|
+
runPreview: (sub: string[]) => {
|
|
16
|
+
status: number | null;
|
|
17
|
+
stdout: string;
|
|
18
|
+
};
|
|
19
|
+
/** Launch `agents <sub…>` inheriting stdio; returns its exit status. */
|
|
20
|
+
launch: (sub: string[]) => {
|
|
21
|
+
status: number | null;
|
|
22
|
+
};
|
|
11
23
|
}
|
|
12
24
|
/**
|
|
13
|
-
* Resolve the source
|
|
14
|
-
*
|
|
25
|
+
* Resolve the source cross-fleet, build a recap from its preview digest, and
|
|
26
|
+
* launch a same-harness sibling seeded with that recap. Shared by
|
|
27
|
+
* `agents sessions fork` and the `agents fork` alias.
|
|
15
28
|
*/
|
|
16
|
-
export declare function runFork(sessionArg: string, options: ForkOptions): Promise<void>;
|
|
29
|
+
export declare function runFork(sessionArg: string, options: ForkOptions, deps?: ForkDeps): Promise<void>;
|
|
17
30
|
/**
|
|
18
31
|
* Register `agents sessions fork <session>` — the canonical surface (fork is a
|
|
19
32
|
* session operation, so it lives under the `sessions` group).
|