@phnx-labs/agents-cli 1.21.3 → 1.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/CHANGELOG.md +136 -0
  2. package/README.md +32 -3
  3. package/dist/bin/agents +0 -0
  4. package/dist/commands/computer-actions.js +1 -0
  5. package/dist/commands/doctor.js +34 -2
  6. package/dist/commands/exec.d.ts +27 -0
  7. package/dist/commands/exec.js +123 -6
  8. package/dist/commands/models.js +36 -1
  9. package/dist/commands/projects.js +120 -62
  10. package/dist/commands/sessions-backfill.d.ts +32 -0
  11. package/dist/commands/sessions-backfill.js +186 -0
  12. package/dist/commands/sessions.d.ts +17 -1
  13. package/dist/commands/sessions.js +317 -18
  14. package/dist/commands/teams.js +1 -1
  15. package/dist/commands/worktree.d.ts +3 -3
  16. package/dist/commands/worktree.js +35 -4
  17. package/dist/lib/app-bundle-install.d.ts +17 -0
  18. package/dist/lib/app-bundle-install.js +94 -0
  19. package/dist/lib/daemon.d.ts +5 -1
  20. package/dist/lib/daemon.js +63 -14
  21. package/dist/lib/devices/doctor-findings.js +12 -4
  22. package/dist/lib/devices/doctor-overview-cache.d.ts +45 -0
  23. package/dist/lib/devices/doctor-overview-cache.js +168 -0
  24. package/dist/lib/devices/fleet.js +7 -2
  25. package/dist/lib/devices/resolve-target.d.ts +6 -0
  26. package/dist/lib/devices/resolve-target.js +9 -3
  27. package/dist/lib/devices/self-host.d.ts +9 -0
  28. package/dist/lib/devices/self-host.js +61 -0
  29. package/dist/lib/exec.js +39 -8
  30. package/dist/lib/hosts/dispatch.d.ts +12 -0
  31. package/dist/lib/hosts/dispatch.js +23 -6
  32. package/dist/lib/hosts/passthrough.js +8 -6
  33. package/dist/lib/hosts/reconnect.d.ts +38 -0
  34. package/dist/lib/hosts/reconnect.js +85 -4
  35. package/dist/lib/hosts/run-target.js +14 -2
  36. package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
  37. package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
  38. package/dist/lib/menubar/install-menubar.js +27 -23
  39. package/dist/lib/model-tiers.d.ts +54 -0
  40. package/dist/lib/model-tiers.js +229 -0
  41. package/dist/lib/models.d.ts +3 -0
  42. package/dist/lib/models.js +44 -7
  43. package/dist/lib/pricing/prices.json +16 -1
  44. package/dist/lib/project-focus.d.ts +42 -0
  45. package/dist/lib/project-focus.js +80 -0
  46. package/dist/lib/project-schedule.d.ts +75 -0
  47. package/dist/lib/project-schedule.js +110 -0
  48. package/dist/lib/project-status.d.ts +7 -0
  49. package/dist/lib/project-status.js +9 -0
  50. package/dist/lib/projects.d.ts +11 -2
  51. package/dist/lib/projects.js +57 -9
  52. package/dist/lib/redact.d.ts +2 -0
  53. package/dist/lib/redact.js +22 -0
  54. package/dist/lib/remote-agents-json.d.ts +2 -0
  55. package/dist/lib/remote-agents-json.js +3 -3
  56. package/dist/lib/rotate.d.ts +84 -1
  57. package/dist/lib/rotate.js +155 -5
  58. package/dist/lib/runner.d.ts +4 -2
  59. package/dist/lib/runner.js +13 -4
  60. package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
  61. package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
  62. package/dist/lib/secrets/install-helper.js +28 -31
  63. package/dist/lib/session/bash-command.js +60 -9
  64. package/dist/lib/session/db.d.ts +7 -1
  65. package/dist/lib/session/db.js +301 -32
  66. package/dist/lib/session/discover.d.ts +40 -7
  67. package/dist/lib/session/discover.js +144 -83
  68. package/dist/lib/session/parse.d.ts +8 -1
  69. package/dist/lib/session/parse.js +83 -32
  70. package/dist/lib/session/remote-list.d.ts +71 -0
  71. package/dist/lib/session/remote-list.js +410 -2
  72. package/dist/lib/session/shell-programs.d.ts +15 -0
  73. package/dist/lib/session/shell-programs.js +359 -0
  74. package/dist/lib/session/tool-calls.d.ts +88 -0
  75. package/dist/lib/session/tool-calls.js +612 -0
  76. package/dist/lib/session/tool-index.d.ts +100 -0
  77. package/dist/lib/session/tool-index.js +773 -0
  78. package/dist/lib/session/tool-store.d.ts +15 -0
  79. package/dist/lib/session/tool-store.js +198 -0
  80. package/dist/lib/session/types.d.ts +7 -0
  81. package/dist/lib/state.d.ts +10 -1
  82. package/dist/lib/state.js +11 -2
  83. package/dist/lib/teams/remoteWorktree.d.ts +3 -4
  84. package/dist/lib/teams/remoteWorktree.js +3 -4
  85. package/dist/lib/teams/worktree.d.ts +11 -1
  86. package/dist/lib/teams/worktree.js +42 -4
  87. package/dist/lib/types.d.ts +17 -0
  88. package/dist/lib/types.js +17 -0
  89. package/dist/lib/versions.js +69 -22
  90. package/package.json +3 -1
package/CHANGELOG.md CHANGED
@@ -1,5 +1,131 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.22.1
4
+
5
+ - **`agents doctor` de-noise: never-synced and cross-version hook drift are warnings, not criticals (RUSH-2162).** The CRITICAL section now holds only "needs you now" problems — a logged-out account, or a hook/plugin missing from a version you keep synced. A version that was never synced (an old/unused install with nothing installed) and a hook that merely *differs* across versions (installed but stale) are surfaced as WARNINGs instead, cutting the critical count on a busy machine from ~11 to the handful that actually need action. Source: `apps/cli/src/lib/devices/doctor-findings.ts`.
6
+
7
+ - **The "… is damaged and can't be opened" dialog stops — both helper `.app` bundles now install atomically and serialized.** The secrets keychain helper (`Agents CLI.app`) and the menu-bar helper (`MenubarHelper.app`) are each (re)installed on the hot path of ordinary `agents` invocations, and both did a non-atomic `rm -rf dest` + `cp -R src dest` straight onto the live bundle. On a busy box dozens of concurrent invocations raced that path, so a reader (Gatekeeper, or an exec of the bundle) could see a half-written `.app` — a truncated Mach-O / mismatched code signature — which macOS reports as damaged. A new shared installer (`lib/app-bundle-install.ts`, replacing the two duplicated copy functions) stages the copy in a sibling dir and swaps it in with renames (the live bundle is only ever a complete, signed `.app`, and a failed copy never touches it), and serializes concurrent installers behind the shared `withFileLock` with a double-checked skip so a burst copies once instead of stampeding. Source: `apps/cli/src/lib/app-bundle-install.ts`, `apps/cli/src/lib/secrets/install-helper.ts`, `apps/cli/src/lib/menubar/install-menubar.ts`.
8
+
9
+ - **`agents doctor --json` no longer stampedes into dozens of concurrent runs — the overview is singleflighted and cached, with a new `--refresh` to force a live recompute.** The bare `doctor --json` overview probes every host CLI, every agent's sign-in, and every agent×version diff — seconds on an idle box, minutes on a loaded one. The menu-bar helper polls it on a timer with only a per-*process* in-flight guard, so a helper relaunch (or any second poller) each launched its own live compute, and a helper killed mid-run orphaned a `doctor --json` that kept spinning — stacking to dozens of concurrent runs pinning the CPU. Now a fresh snapshot (< 90s) serves instantly from a disk cache, and when a live compute IS needed exactly one runs while every other caller serves its result (a lock-directory singleflight that self-heals if the computer dies). `agents doctor --json --refresh` bypasses the cache. Source: `apps/cli/src/lib/devices/doctor-overview-cache.ts`, `apps/cli/src/commands/doctor.ts`.
10
+
11
+ - **`doctor --json` releases its singleflight lock before it returns.** The overview gate
12
+ fired the lock release without awaiting it on the path where a waiter serves the winner's
13
+ fresh snapshot, so the call returned with the lockfile still on disk. The next caller then
14
+ retried against a lock that was already logically free — the pile-up the gate exists to
15
+ prevent, narrowed to the window between return and unlink. The release is now awaited.
16
+ The existing coalescing test failed 5 times in 15 runs before this and 0 in 15 after.
17
+ Source: `apps/cli/src/lib/devices/doctor-overview-cache.ts`.
18
+
19
+ - **`agents add grok@latest` no longer strands the freshly-downloaded binary in the old version's home.** When the post-install version probe (`<cli> --version`) transiently failed right after grok's self-updating installer exited, `installVersion` silently fell back to the literal string `'latest'` as the resolved version — creating a bogus `versions/grok/latest/` directory and defeating `relocateGrokBinaryToVersionHome`'s exact-filename match (its regex could never match `grok-latest-...`, since the real file is named `grok-<semver>-<platform>`). The real multi-hundred-MB binary was left behind in the PREVIOUS default's downloads dir, and `agents view grok` never listed the new version as installed even though `agents add` reported success. The probe now retries briefly instead of silently falling back, and fails loudly if it still can't resolve a version rather than corrupting the version bookkeeping. Relocation also now self-heals: if the current `~/.grok` symlink target has nothing matching, it sweeps every other installed grok version home for a binary stranded by a past occurrence of this bug. Source: `apps/cli/src/lib/versions.ts`.
20
+
21
+ - **A blocked menu-bar row now takes you to the session (RUSH-2110).** A NEEDS-YOU row
22
+ exists because an agent is waiting on you, but its only action was "Reveal working
23
+ dir", which unblocks nothing — you still had to go find the session by hand. Blocked
24
+ rows now lead with **Focus session**, which runs `agents focus <id>`: attach the live
25
+ terminal, or open a new tab and resume, cross-host. Reveal stays underneath. Both
26
+ render paths are covered — the single inline row and each entry inside a collapsed
27
+ multi-waiter group. A row the engine could not identify (a cloud task, a stale
28
+ sentinel) simply omits the item rather than offering an action that would do nothing.
29
+ Source: `apps/cli/menubar/Sources/MenubarHelper/StatusItemController.swift`,
30
+ `AgentsCLI.swift`.
31
+
32
+ - **Two projects sharing one monorepo checkout are no longer indistinguishable.** Session,
33
+ activity, and feed attribution anchored a project on `root ?? defaultPath`, so a subproject
34
+ whose `root` is the monorepo and whose `defaultPath` is a subdir collapsed onto the same
35
+ path as its umbrella — the longest-match tiebreak had nothing to separate them, and work in
36
+ `rush/apps/cli` counted toward whichever definition happened to be listed first. A
37
+ `defaultPath` nested under `root` now takes precedence over that `root` (the root says where
38
+ the checkout is; `defaultPath` says which work is this project's), and each bound repo's
39
+ checkout and subpath anchor too. A narrowed `root` still covers the rest of its checkout as
40
+ a fallback, so a lone project defined with `--path` keeps attributing work across its own
41
+ repo instead of only inside the subdir. Source: `apps/cli/src/lib/projects.ts`.
42
+
43
+ - **`agents projects view <name>` now shows more than `status`, not less.** The command you
44
+ open to learn everything about one project built its own short list — root, repos, a raw
45
+ Linear project id, an issue count, milestones — and never called the card renderer, so it
46
+ omitted the agents roster, merged PRs and release, focus areas, the schedule verdict,
47
+ tickets, and artifacts that `status` had shown all along. `view` and `status` now gather
48
+ through one function and render through one card; `view` adds every milestone (instead of
49
+ just the next) and the stored definition in full underneath — each repo with its subpath and
50
+ checkout, each context with its purpose, each integration with its URL. It also takes
51
+ `--window <days>` to match `status`. Source: `apps/cli/src/commands/projects.ts`.
52
+ - **The `agents` roster on the card lists live sessions only.** It included every matched
53
+ session, so a card headed `23 live` went on to print `claude · crashed ×25` — the corpses the
54
+ `dead` row already reports, counted twice and contradicting the headline. Both now derive
55
+ from one `isDeadStatus` predicate, pinned by a test across every `ActiveStatus`. Source:
56
+ `apps/cli/src/lib/project-status.ts`.
57
+
58
+ - **`--host <self>` and the fleet-health fan-out now short-circuit ALL of the
59
+ local machine's names, not just its short hostname (RUSH-2114).** A `--host`
60
+ target or fleet probe that referenced this box by its **tailscale dnsName**
61
+ (`zion.tail1a85a1.ts.net`) slipped past a `=== machineId()` check and SSH'd to
62
+ the local box over its own name; on a loaded machine that self-SSH'd `doctor
63
+ --json` orphaned on timeout and piled up until the host was crushed. A new
64
+ `isSelfHost()` matches every identity the box answers to (short id, loopback,
65
+ tailscale dnsName + its short form) and gates all four self-checks — the
66
+ generic `--host` passthrough (`maybeRunOnHost`), the `--devices`-all fan-out
67
+ (`runFleetPassthrough`), `remoteFleetTargets`, and `runFleet` — so a
68
+ self-reference runs locally instead of self-SSHing. Source:
69
+ `apps/cli/src/lib/devices/self-host.ts`, `apps/cli/src/lib/hosts/passthrough.ts`,
70
+ `apps/cli/src/lib/devices/fleet.ts`.
71
+
72
+ ## 1.22.0
73
+
74
+ - **`agents run auto` — full-auto dispatch (RUSH-2132).** `run auto` composes all three routing layers: host (14d launch affinity, unless `--host` is given), harness (installed CLIs weighted by best-account headroom), and account (the configured strategy). `balanced`/`available` now exit nonzero when every installed account is unhealthy — naming each excluded account, the earliest window reset, and the `--strategy pinned` escape hatch — instead of warning "falling back to defaults" and launching the exhausted pinned default. The error text is a machine-readable contract (`no healthy` + `resets <iso-time>`) the Factory watchdog tail-detects for rotate cooldowns. Source: `apps/cli/src/lib/rotate.ts`, `apps/cli/src/commands/exec.ts`, `apps/cli/src/lib/runner.ts`.
75
+
76
+ - **Bash-command summaries are faster and recognize more of what actually ran (#1830).**
77
+ `classifyBashCommand` (behind `agents sessions` / `agents activity` summaries) tokenized
78
+ the *entire* command — every pipeline segment, multi-KB heredoc bodies included — just to
79
+ read the leading executable, costing up to ~1ms on a big `cat <<HEREDOC …`. It now
80
+ tokenizes only the head of the first simple command. Coverage gaps that dumped commands
81
+ into a raw `other` pile are closed too: a `cd` prefix separated by `;` or a newline (not
82
+ just `&&`) unwraps to the real command, a path/tilde executable
83
+ (`~/.agents/skills/linear/scripts/linear`) resolves by basename, and the repo's own
84
+ toolchain (`agents`, `linear`, plus `rmdir`) is recognized — `agents` was the single top
85
+ unrecognized token. `ag` stays the silver searcher, not an `agents` alias. Source:
86
+ `apps/cli/src/lib/session/bash-command.ts`.
87
+
88
+ - **`agents computer describe` now counts toward `usedComputer`.** Every other
89
+ verb (`click`, `type`, `key`, `screenshot`, `run`, …) fires the
90
+ `computer.action` event via `emitComputerAction`; `describe` never did, so a
91
+ session that only ran `agents computer describe` read back
92
+ `usedComputer=false` — a false-negative in the sessions preview. A new
93
+ completeness-guard test pins every registered `agents computer` verb command
94
+ to a matching `emitComputerAction` call so a future verb can't ship the same
95
+ gap silently. Source: `apps/cli/src/commands/computer-actions.ts`,
96
+ `apps/cli/src/commands/computer-actions.test.ts`.
97
+
98
+ - **Pick a model by cost tier — `--model cheap|default|best|ultra` — on `agents run` and `agents teams add`.** Instead of a concrete id that churns per release and differs per harness, a tier resolves per `(harness, installed version)` to a model that version actually ships, ranked by the provider's own lineup (`opus/sonnet/haiku/fable`; Codex "frontier/balanced/fast" → Sol/Terra/Luna), then price, then size tokens. Single-model harnesses (Grok) map the tiers to reasoning effort; Droid uses a curated credit-multiplier map capped at 2x. An unsupported tier clamps to the nearest lower one; an unresolvable tier drops the flag and falls back to the harness default. Concrete model ids keep working unchanged. `agents models [agent[@version]]` now prints the per-harness tier map (with `~$/Mtok` where priced) and emits `tiers` in `--json`, and Droid joins the model-capable set. Also fixes the Claude catalog extractor returning 0 models on the newest native-binary format (a fallback id scan), and refreshes `prices.json` with the GPT-5.6 Sol/Terra/Luna series. Source: `apps/cli/src/lib/model-tiers.ts`, `apps/cli/src/lib/models.ts`, `apps/cli/src/lib/exec.ts`, `apps/cli/src/commands/models.ts`, `apps/cli/docs/model-tiers.md`.
99
+
100
+ - **`agents projects status` says what was worked on and what the dates prove.** Two new lines.
101
+ `focus` ranks the directories the window's commits landed in, read from the local checkout
102
+ with `git log --name-only` — no API call, no credential, no rate-limit budget, measured at
103
+ 0.23s over a 897-commit week. Changelog fragments and lockfiles are excluded from the
104
+ ranking: this repo files one fragment per PR, so `.changelog` otherwise ranked second and
105
+ presented PR count as an area of focus. `schedule` states what the milestone dates prove —
106
+ `overdue by N days`, `due in N days`, `N milestones, no issues filed against any`, or
107
+ `none dated`. Source: `apps/cli/src/lib/project-focus.ts`, `project-schedule.ts`.
108
+ - **The schedule line will never say "on track".** That verdict needs either project start and
109
+ target dates to interpolate expected progress, or a scope-history series to extrapolate a
110
+ finish date. Probed against a live workspace, all of them are absent (`health: null`,
111
+ `startDate`/`targetDate` null, `scopeHistory` and `completedScopeHistory` empty), so an
112
+ on-track or at-risk chip would be fabricated — and a confident wrong answer on a status card
113
+ is unfalsifiable from the card. When a human posts a Linear project health update, it is
114
+ relayed and attributed (`per Linear: atRisk`), never synthesized.
115
+
116
+ - **The `--device`/`--host` auto-reconnect loop no longer trusts a remote-origin exit code of 255 as "the SSH link dropped."** `reattachRemoteSession`'s `connected` flag is set as soon as the fast SSH preflight probe succeeds, before the actual reattach runs — so if the remote command it drives (`agents sessions focus <id> --local --attach-only`) ever exited 255 for a reason that had nothing to do with the SSH transport, that would be indistinguishable from the link itself dropping, refill the retry budget every cycle, and loop forever — printing "attempt 1/6" on every cycle and leaving the terminal full of aborted-TTY escape codes. The remote invocation is now wrapped in `bash -lc` so that whatever exit code it decides on, a 255 is remapped to 254 before this process sees it, closing that gap in the exit-code channel regardless of which remote-side path or peer `agents` version might produce it. A genuinely recurring *local* SSH failure can still refill the retry budget on every attempt by design (unchanged, tracked separately: phnx-labs/agents-cli#1884). Source: `apps/cli/src/lib/hosts/reconnect.ts`.
117
+
118
+ - **`agents sessions` can query distinct tool calls and count static Bash program occurrences locally or across the fleet.** Use `--include tools`, repeat `--query` with `tool:`, `program:`, `input:`, `output:`, `status:`, `exit:`, or `error:` fields, and add `--fleet` for live SSH fan-out. `--count` reports exact occurrence, containing-call, and session totals from ordered `wrapper`/`effective` rows without reparsing; synced mirrors are partitioned by origin so fleet evidence and totals do not duplicate sessions. Historical parsing is explicit and resumable through `agents sessions backfill tools`; normal scans index new and changed sessions once. Codex orchestration wrappers are parsed statically so only literal `tools.exec_command` commands reach the Bash AST, never wrapper code. Each device keeps a redacted, bounded relational SQLite/FTS5 cache, queries perform no transcript I/O or index writes, and no embeddings, vector database, or model calls are used. A sampling script explicitly backfills then extracts redacted shell-command origins from 50–100 sessions over the last seven days into a 16 MiB maximum artifact.
119
+
120
+ - **Local team worktrees base on freshly-fetched `origin/<default>`, not `HEAD`.**
121
+ `createWorktree` (and `agents worktree provision` for new branches) now
122
+ `git fetch origin` then `worktree add -b … origin/<default>`, matching
123
+ `createRemoteWorktree`. Previously local teammates forked from the
124
+ orchestrator's current `HEAD`, so a stale checkout made every teammate write
125
+ on old code and only surface the conflict at merge. Source:
126
+ `apps/cli/src/lib/teams/worktree.ts`, `apps/cli/src/commands/worktree.ts`,
127
+ `apps/cli/docs/teams.md`.
128
+
3
129
  ## 1.21.3
4
130
 
5
131
  - **`agents projects import --from-factory` stops printing raw git errors.** Reading each
@@ -33,6 +159,16 @@
33
159
  `agents sessions`'s perf sample for `command.end` now carries the session id
34
160
  and agent instead of being anonymous. Source: `apps/cli/src/lib/session/prompt.ts`,
35
161
  `apps/cli/src/lib/session/parse.ts`, `apps/cli/src/index.ts`.
162
+ - **`agents routines status` no longer reports "stopped" for a live scheduler, and
163
+ `agents routines start` can't spawn a second one.** The daemon writes its pid file
164
+ once (on claim/start) but rewrites the heartbeat every tick. If the pid file was lost
165
+ while the daemon kept ticking — an earlier status check clearing a stale/reused pid, or
166
+ the file removed out from under a live daemon — `status` read only the pid file and
167
+ reported `stopped` for a scheduler that was in fact running and firing jobs, while
168
+ `claimDaemonInstance()` would start a concurrent `JobScheduler` that double-fires every
169
+ routine. `isDaemonRunning()` and the single-instance claim now also trust a fresh
170
+ heartbeat whose pid is alive, re-adopting the pid file to heal the desync.
171
+ Source: `apps/cli/src/lib/daemon.ts`.
36
172
 
37
173
  ## 1.21.2
38
174
 
package/README.md CHANGED
@@ -161,7 +161,18 @@ agents run claude@
161
161
  agents run codex@ "review this branch"
162
162
  ```
163
163
 
164
- `--strategy balanced` spreads work across available versions of the same agent -- useful when you have multiple accounts and want to avoid burning through one.
164
+ `--strategy balanced` spreads work across available versions of the same agent -- useful when you have multiple accounts and want to avoid burning through one. When every account is rate-limited, the run exits nonzero naming each excluded account and the earliest window reset (use `--strategy pinned` to force the default) -- it never launches into an exhausted account.
165
+
166
+ ### Don't care which harness? `agents run auto`
167
+
168
+ ```bash
169
+ # Picks the host (14d usage affinity), the harness (installed CLIs weighted by
170
+ # best-account headroom), and the account (balanced) -- all three layers.
171
+ agents run auto "summarize recent commits"
172
+ agents run auto --host yosemite-s0 "fix the flaky test" # pin the host layer
173
+ ```
174
+
175
+ `run auto` excludes any harness whose accounts are all rate-limited or signed out, and exits nonzero with the earliest reset time when nothing anywhere is healthy.
165
176
 
166
177
  A trailing `@` opens an account picker before either an interactive or prompt-based run. Each installed version shows its account identity, exact version, login state, plan, and every available session, weekly, or monthly limit. Logged-out, rate-limited, and out-of-credit accounts remain visible with the reason they cannot be selected; signed-in accounts whose provider does not expose quota data stay selectable and say `limits unavailable`. The choice pins only that run and does not change your default version.
167
178
 
@@ -244,11 +255,26 @@ agents sessions a1b2c3d4 --markdown
244
255
 
245
256
  # Just the last 3 turns, user messages only
246
257
  agents sessions a1b2c3d4 --last 3 --include user
258
+
259
+ # Calls in recent Codex sessions on one device
260
+ agents sessions --include tools --agent codex --device mac-mini --since 7d
261
+
262
+ # One session where two different calls match; query every online device
263
+ agents sessions --include tools \
264
+ --query 'program:git input:merge' \
265
+ --query 'program:gh output:CONFLICT' \
266
+ --fleet --json
267
+
268
+ # Count pre-indexed static git sites, containing calls, and sessions
269
+ agents sessions --include tools --query 'program:git' --count --fleet --json
270
+
271
+ # Populate historical tool rows once on each device
272
+ agents sessions backfill tools --fleet
247
273
  ```
248
274
 
249
275
  Interactive picker when you're in a terminal. Structured output (`--json`, `--markdown`, filtered by role or turn count) when piped.
250
276
 
251
- Backed by a SQLite + FTS5 index at `~/.agents/.history/sessions/sessions.db` with incremental scanning -- warm reads in ~100ms. External tools can consume `--json` output as a programmatic observability layer; see [docs/05-sessions.md](apps/cli/docs/05-sessions.md) for the schema and [docs/06-observability.md](apps/cli/docs/06-observability.md) for the consumption patterns.
277
+ Backed by a SQLite + FTS5 index at `~/.agents/.history/sessions/sessions.db` with incremental scanning -- warm reads in ~100ms. Tool-call evidence is redacted and bounded before it is cached; repeated `--query` clauses must match distinct calls in one session. Tool queries read SQLite only: `agents sessions backfill tools` performs the one-time historical parse, while normal incremental scans index new and changed sessions. The index stores ordered static Bash program sites, so `--count` reports occurrences, containing tool calls, and distinct sessions without reparsing. `--fleet` executes one origin partition per device, so synced mirrors cannot duplicate compact evidence or counts returned over SSH; transcript bodies stay on their origin machine. This uses relational SQLite rows and literal FTS5 only, with no embeddings, vector database, or model calls. External tools can consume `--json` output as a programmatic observability layer; see [docs/05-sessions.md](apps/cli/docs/05-sessions.md) for the schemas and [docs/06-observability.md](apps/cli/docs/06-observability.md) for the consumption patterns.
252
278
 
253
279
  ### Live state, and catching up fast
254
280
 
@@ -1095,9 +1121,12 @@ Conversations with Claude, Codex, legacy Gemini, and other agents scatter across
1095
1121
  ```bash
1096
1122
  agents sessions "auth middleware" # Full-text search across all agents
1097
1123
  agents sessions --agent claude --since 7d
1124
+ agents sessions --include tools --query 'program:git' --fleet --json
1125
+ agents sessions --include tools --query 'program:git' --count --fleet --json
1126
+ agents sessions backfill tools --fleet
1098
1127
  ```
1099
1128
 
1100
- The index lives at `~/.agents/.history/sessions/sessions.db` (SQLite + FTS5). Nothing leaves your machine. See [Sessions](#sessions-across-agents) for full usage.
1129
+ The index lives at `~/.agents/.history/sessions/sessions.db` (SQLite + FTS5). A local query stays on the machine; an explicit `--fleet` tool query sends only redacted, bounded match evidence or aggregate counts over SSH. Historical tool parsing is explicit via `sessions backfill tools`; queries never parse transcripts. See [Sessions](#sessions-across-agents) for full usage.
1101
1130
 
1102
1131
  ### Secrets
1103
1132
 
package/dist/bin/agents CHANGED
Binary file
@@ -413,6 +413,7 @@ export function registerActionCommands(program) {
413
413
  if (opts.depth != null)
414
414
  params.max_depth = opts.depth;
415
415
  const res = unwrap(await client.call('describe', params));
416
+ emitComputerAction('describe', pid, opts, { depth: opts.depth });
416
417
  // The tree is inherently structured — always JSON, pretty unless --json.
417
418
  console.log(JSON.stringify(opts.json ? res : res.tree ?? res, null, 2));
418
419
  });
@@ -4,6 +4,7 @@ import { addHostOption } from '../lib/hosts/option.js';
4
4
  import { buildRemoteAgentsInvocation } from '../lib/hosts/remote-cmd.js';
5
5
  import { loadDevices, isControlDevice } from '../lib/devices/registry.js';
6
6
  import { fanOutDevices, planFleetTargets, remoteFleetTargets } from '../lib/devices/fleet.js';
7
+ import { enterDoctorOverviewGate, writeDoctorOverviewCache } from '../lib/devices/doctor-overview-cache.js';
7
8
  import { fleetDialTarget } from '../lib/devices/connect.js';
8
9
  import { compareFleetInventories } from '../lib/devices/fleet-divergence.js';
9
10
  import { collectLocalFleetInventory } from '../lib/devices/fleet-inventory.js';
@@ -1155,12 +1156,18 @@ export function registerDoctorCommand(program) {
1155
1156
  .option('--devices', 'Check agent readiness AND cross-device harness divergence (missing resources/versions, repo drift) on every registered device (alias --hosts)')
1156
1157
  .option('--hosts', 'Alias of --devices')
1157
1158
  .option('--check', 'CI drift gate: exit non-zero when any installed version is out of sync (stale or never-synced), zero when clean. Combine with --devices to gate the whole fleet.')
1159
+ .option('--refresh', 'Bypass the cached overview snapshot: recompute the bare `doctor --json` overview live and refresh the shared cache that the menu-bar and other pollers read')
1158
1160
  .option('-q, --quiet', 'With --check, suppress per-version lines; print only the one-line verdict');
1159
1161
  setHelpSections(doctorCmd, {
1160
1162
  examples: `
1161
1163
  # Overview: CLI availability + sync status + orphans across all defaults
1162
1164
  agents doctor
1163
1165
 
1166
+ # Machine-readable overview (served from a ~90s cache for pollers like the
1167
+ # menu-bar helper); --refresh recomputes live and refreshes that cache
1168
+ agents doctor --json
1169
+ agents doctor --json --refresh
1170
+
1164
1171
  # Full per-resource report for the active default
1165
1172
  agents doctor claude@default
1166
1173
 
@@ -1288,6 +1295,26 @@ export function registerDoctorCommand(program) {
1288
1295
  return;
1289
1296
  }
1290
1297
  if (!target) {
1298
+ // Singleflight + short-TTL cache for the bare `doctor --json` overview.
1299
+ // This overview probes every host CLI, every agent's sign-in, and every
1300
+ // agent×version diff — seconds on an idle box, minutes on a loaded one.
1301
+ // The menu-bar helper polls it with only a per-*process* in-flight guard,
1302
+ // so a helper relaunch (or any second poller) each launched its own live
1303
+ // compute, and a helper killed mid-run orphaned a `doctor --json` that
1304
+ // kept spinning — stacking to dozens of concurrent runs pinning the CPU
1305
+ // (RUSH-2153). Now: a fresh snapshot serves instantly, and when a compute
1306
+ // IS needed exactly one runs while every other caller serves its result.
1307
+ // A crashed computer never wedges the gate — the lock is stolen once its
1308
+ // directory mtime goes stale (see enterDoctorOverviewGate).
1309
+ let releaseOverviewGate;
1310
+ if (opts.json) {
1311
+ const gate = await enterDoctorOverviewGate({ forceRefresh: !!opts.refresh });
1312
+ if (gate.cached !== null) {
1313
+ console.log(gate.cached);
1314
+ return;
1315
+ }
1316
+ releaseOverviewGate = gate.release;
1317
+ }
1291
1318
  const clis = checkAllClis();
1292
1319
  const syncRows = checkSyncStatus(cwd);
1293
1320
  const orphanRows = countOrphans();
@@ -1355,7 +1382,7 @@ export function registerDoctorCommand(program) {
1355
1382
  isolatedVersions,
1356
1383
  });
1357
1384
  if (opts.json) {
1358
- console.log(JSON.stringify({
1385
+ const overviewPayload = {
1359
1386
  clis,
1360
1387
  signIn,
1361
1388
  // Cached auth-health rollup for THIS host — lets `agents fleet status`
@@ -1398,7 +1425,12 @@ export function registerDoctorCommand(program) {
1398
1425
  branch: m.branch,
1399
1426
  fetchedAt: m.fetchedAt,
1400
1427
  })),
1401
- }, null, 2));
1428
+ };
1429
+ // Persist for the next poller and release the singleflight lock BEFORE
1430
+ // printing, so a concurrent caller picks up the fresh snapshot at once.
1431
+ writeDoctorOverviewCache(overviewPayload);
1432
+ releaseOverviewGate?.();
1433
+ console.log(JSON.stringify(overviewPayload, null, 2));
1402
1434
  return;
1403
1435
  }
1404
1436
  // Single-machine hybrid: CRITICAL section + one `▸ <machine>` block.
@@ -7,6 +7,7 @@
7
7
  */
8
8
  import { type Command } from 'commander';
9
9
  import type { ExecEffort } from '../lib/exec.js';
10
+ import { RUN_AUTO_KEYWORD } from '../lib/types.js';
10
11
  import { type SshGResult } from '../lib/hosts/ssh-config.js';
11
12
  export interface RunAccountPickerRequest {
12
13
  requested: boolean;
@@ -41,6 +42,32 @@ export declare function runAccountPickerConflicts(options: {
41
42
  on?: string;
42
43
  computer?: string;
43
44
  }): string[];
45
+ export { RUN_AUTO_KEYWORD };
46
+ /**
47
+ * Whether `run auto` should default its host layer to the affinity pick (the
48
+ * same machinery as `--device auto`). False when the caller pinned any host
49
+ * flag, and false when this process was itself dispatched by a host run — the
50
+ * dispatcher exports AGENTS_RUN_AUTO_HOST_RESOLVED=1 into the remote SHELL
51
+ * (hosts/dispatch.ts remoteRunShellPrelude) because it already resolved the
52
+ * host layer, and re-picking here would chain-hop the run across the fleet.
53
+ * Pure so the pinning matrix is unit-testable.
54
+ */
55
+ export declare function runAutoDefaultsToAffinity(options: {
56
+ host?: string;
57
+ device?: string;
58
+ on?: string;
59
+ computer?: string;
60
+ }, env?: NodeJS.ProcessEnv): boolean;
61
+ /**
62
+ * Whether an interactive host dispatch must mint a correlation launch id and
63
+ * resolve the remote session via the launch-id join (RUSH-2034), rather than
64
+ * trusting a pre-known session id. `run auto` ALWAYS joins: the harness is
65
+ * picked on the remote, so an explicit --session-id is only adopted when the
66
+ * pick lands on claude — pre-registering it would strand a stale session-index
67
+ * entry naming an id a non-claude pick never used (RUSH-2132). Pure so the
68
+ * decision matrix is unit-testable.
69
+ */
70
+ export declare function hostInteractiveNeedsCorrelationId(runAgent: string, hostSessionId: string | undefined, resumeId: string | undefined): boolean;
44
71
  /** The host descriptor fields the `--copy-creds` security gate reads. */
45
72
  export interface CopyCredsGateHost {
46
73
  name: string;
@@ -7,6 +7,7 @@
7
7
  */
8
8
  import { Option } from 'commander';
9
9
  import chalk from 'chalk';
10
+ import { RUN_AUTO_KEYWORD } from '../lib/types.js';
10
11
  import { setHelpSections } from '../lib/help.js';
11
12
  import { isInteractiveTerminal, isPromptCancelled, requireInteractiveSelection } from './utils.js';
12
13
  import { getUserAgentsDir } from '../lib/state.js';
@@ -66,6 +67,39 @@ export function runAccountPickerConflicts(options) {
66
67
  function isValidAgent(agent) {
67
68
  return agent in AGENTS;
68
69
  }
70
+ // Reserved `<agent>` keyword for `agents run auto` — canonical definition in
71
+ // lib/types.ts (shared with the host dispatch layer); re-exported here.
72
+ export { RUN_AUTO_KEYWORD };
73
+ /**
74
+ * Whether `run auto` should default its host layer to the affinity pick (the
75
+ * same machinery as `--device auto`). False when the caller pinned any host
76
+ * flag, and false when this process was itself dispatched by a host run — the
77
+ * dispatcher exports AGENTS_RUN_AUTO_HOST_RESOLVED=1 into the remote SHELL
78
+ * (hosts/dispatch.ts remoteRunShellPrelude) because it already resolved the
79
+ * host layer, and re-picking here would chain-hop the run across the fleet.
80
+ * Pure so the pinning matrix is unit-testable.
81
+ */
82
+ export function runAutoDefaultsToAffinity(options, env = process.env) {
83
+ if (hostTargetGiven(options).length > 0)
84
+ return false;
85
+ return env.AGENTS_RUN_AUTO_HOST_RESOLVED !== '1';
86
+ }
87
+ /**
88
+ * Whether an interactive host dispatch must mint a correlation launch id and
89
+ * resolve the remote session via the launch-id join (RUSH-2034), rather than
90
+ * trusting a pre-known session id. `run auto` ALWAYS joins: the harness is
91
+ * picked on the remote, so an explicit --session-id is only adopted when the
92
+ * pick lands on claude — pre-registering it would strand a stale session-index
93
+ * entry naming an id a non-claude pick never used (RUSH-2132). Pure so the
94
+ * decision matrix is unit-testable.
95
+ */
96
+ export function hostInteractiveNeedsCorrelationId(runAgent, hostSessionId, resumeId) {
97
+ if (resumeId)
98
+ return false;
99
+ if (runAgent === RUN_AUTO_KEYWORD)
100
+ return true;
101
+ return !hostSessionId && isSessionTrackedAgent(runAgent);
102
+ }
69
103
  /** Build a one-line banner describing which version the strategy picked. */
70
104
  function formatRotationBanner(result, verb = 'balanced') {
71
105
  const { picked, healthy, excluded } = result;
@@ -483,7 +517,7 @@ export function registerRunCommand(program) {
483
517
  .description('Execute an agent. Pass a prompt for headless runs; omit it to launch the agent interactively.')
484
518
  .option('-m, --mode <mode>', 'How much the agent can do: plan (read-only), edit (can write files), auto (smart classifier auto-approves safe ops, prompts for risky), skip (bypass all permission prompts). \'full\' accepted as alias for skip.', 'plan')
485
519
  .option('-e, --effort <effort>', 'Reasoning effort: low | medium | high | xhigh | max | auto (claude and codex only)', 'auto')
486
- .option('--model <model>', 'Override the model directly (e.g., claude-opus-4-6)')
520
+ .option('--model <model>', 'Cost tier (cheap|default|best|ultra) or a concrete model id; tiers resolve per harness+version to a supported model')
487
521
  .option('--env <key=value>', 'Pass environment variable to the agent (repeatable, e.g., --env DEBUG=1 --env API_KEY=xyz)', (val, prev) => [...prev, val], [])
488
522
  .option('--secrets <bundle>', 'Inject a secrets bundle (repeatable). Values resolve from macOS Keychain at run time. See `agents secrets`.', (val, prev) => [...prev, val], [])
489
523
  .option('--no-auto-secrets', 'Skip auto-injection of secrets declared by a workflow\'s frontmatter `secrets:` field. Has no effect on bare-agent runs.')
@@ -561,6 +595,11 @@ export function registerRunCommand(program) {
561
595
  # Pick a signed-in account/version for only this run
562
596
  agents run claude@
563
597
 
598
+ # Full-auto: affinity-pick the host, then the harness with the most
599
+ # account headroom, then a balanced account on it
600
+ agents run auto "fix the flaky test" --mode edit
601
+ agents run auto --host yosemite-s0 "fix the flaky test" # pin the host
602
+
564
603
  # Open the session in a terminal tab — detected from where your sessions
565
604
  # already run (Ghostty / iTerm / Terminal.app); force one with a value
566
605
  agents run claude --terminal
@@ -599,6 +638,13 @@ export function registerRunCommand(program) {
599
638
  balanced distribute load across healthy accounts by remaining capacity (default)
600
639
  A version/account is skipped when it is rate-limited right now — any usage window (incl. the 5-hour session window) at 100%, matching the 'agents view' badge.
601
640
  --balanced is shorthand for --strategy balanced. Ignored when @version is pinned, when a profile is used, or with --fallback.
641
+ Zero healthy accounts under balanced/available exits nonzero naming each
642
+ excluded account and the earliest window reset — use --strategy pinned to force.
643
+
644
+ 'auto' harness (agents run auto): picks the host (14d usage affinity,
645
+ unless --host is given), the harness (installed CLIs weighted by
646
+ best-account headroom), and the account (the strategy above). Zero
647
+ healthy accounts on any harness exits nonzero with the earliest reset.
602
648
 
603
649
  Account picker: append @ with no version (agents run claude@) to choose one
604
650
  installed account for this run. Rows show identity, login state, plan,
@@ -682,6 +728,32 @@ export function registerRunCommand(program) {
682
728
  process.exit(1);
683
729
  }
684
730
  }
731
+ // `agents run auto`: the reserved harness keyword — full-auto dispatch
732
+ // (host affinity → cross-harness balance → account balance, RUSH-2132).
733
+ if (normalizedAgentSpec.split('@')[0] === RUN_AUTO_KEYWORD && normalizedAgentSpec !== RUN_AUTO_KEYWORD) {
734
+ console.error(chalk.red(`agents run auto picks the harness itself — a @version pin does not apply. ` +
735
+ `Pin a concrete harness instead: agents run <harness>@<version>.`));
736
+ process.exit(1);
737
+ }
738
+ const autoHarnessRequested = normalizedAgentSpec === RUN_AUTO_KEYWORD;
739
+ if (autoHarnessRequested) {
740
+ // `auto` is reserved. If a future harness registers that id, the
741
+ // keyword collides — fail loud rather than silently shadow the harness.
742
+ if (RUN_AUTO_KEYWORD in AGENTS) {
743
+ console.error(chalk.red(`'${RUN_AUTO_KEYWORD}' is now a registered harness and collides with the reserved 'run auto' keyword. ` +
744
+ `Run the harness by name instead.`));
745
+ process.exit(1);
746
+ }
747
+ if (accountPickerRequested) {
748
+ console.error(chalk.red(`agents run auto picks the harness and account itself — the trailing-@ account picker needs a concrete harness (agents run <harness>@).`));
749
+ process.exit(1);
750
+ }
751
+ // Host layer: with no explicit --host/--device, default to the
752
+ // affinity pick. Skipped on a host-dispatched run — its dispatcher
753
+ // already resolved this layer (see runAutoDefaultsToAffinity).
754
+ if (runAutoDefaultsToAffinity(options))
755
+ options.device = 'auto';
756
+ }
685
757
  // --device auto / --host auto (and deprecated --smart): affinity-pick host.
686
758
  // Harness is always the agent the user typed — never auto-picked.
687
759
  // Affinity failure degrades to local (does not kill the run).
@@ -1062,6 +1134,10 @@ export function registerRunCommand(program) {
1062
1134
  process.exit(1);
1063
1135
  }
1064
1136
  const hostName = hostGiven[0];
1137
+ // Note: a `run auto` dispatch needs no marker forwarded from here — the
1138
+ // dispatch layer (hosts/dispatch.ts remoteRunShellPrelude) exports the
1139
+ // chain-hop guard into the remote shell for BOTH interactive and
1140
+ // headless paths, keyed off the agent name being `auto`.
1065
1141
  const { resolveHostRunTarget, resolveHostSessionId, dispatchPromptToHost, HostResolutionError } = await import('../lib/hosts/run-target.js');
1066
1142
  const { runInteractiveOnHost } = await import('../lib/hosts/dispatch.js');
1067
1143
  const { registerInteractiveHostSession } = await import('../lib/hosts/session-index.js');
@@ -1242,11 +1318,16 @@ export function registerRunCommand(program) {
1242
1318
  // the stream we resolve the id by one ssh read of the remote hook
1243
1319
  // record — the same launch-id join used locally (RUSH-2034). Not
1244
1320
  // needed for Claude (id forced) or resume (id already known).
1245
- const correlationLaunchId = !hostSessionId && !resumeId && isSessionTrackedAgent(runAgent) ? randomUUID() : undefined;
1321
+ // `run auto` ALWAYS joins: the remote picks the harness, so an
1322
+ // explicit --session-id is only adopted by a claude pick.
1323
+ const correlationLaunchId = hostInteractiveNeedsCorrelationId(runAgent, hostSessionId, resumeId) ? randomUUID() : undefined;
1246
1324
  const hostEnv = correlationLaunchId
1247
1325
  ? [...options.env, `AGENT_LAUNCH_ID=${correlationLaunchId}`]
1248
1326
  : options.env;
1249
- if (hostSessionId) {
1327
+ // `run auto` never pre-registers: the explicit id is only real when
1328
+ // the remote pick lands on claude. The launch-id join below records
1329
+ // the id the remote ACTUALLY used, whatever the pick.
1330
+ if (hostSessionId && runAgent !== RUN_AUTO_KEYWORD) {
1250
1331
  registerInteractiveHostSession({
1251
1332
  cwd: process.cwd(),
1252
1333
  host: host.name,
@@ -1316,7 +1397,11 @@ export function registerRunCommand(program) {
1316
1397
  // re-attach the live pane automatically instead of exiting — the user
1317
1398
  // never has to notice the drop and `agents sessions focus` by hand.
1318
1399
  // `raw` runs aren't tmux wrapped, so there is nothing to reconnect to.
1319
- const reconnectId = hostSessionId ?? resolvedRemoteId ?? resumeId;
1400
+ // For `run auto` prefer the join-resolved id (the harness the remote
1401
+ // ACTUALLY picked) over the explicit --session-id only claude adopts.
1402
+ const reconnectId = (runAgent === RUN_AUTO_KEYWORD
1403
+ ? resolvedRemoteId ?? hostSessionId
1404
+ : hostSessionId ?? resolvedRemoteId) ?? resumeId;
1320
1405
  if (reconnectId && !isRaw) {
1321
1406
  const { reconnectInteractiveSession, SSH_CONN_FAILURE } = await import('../lib/hosts/reconnect.js');
1322
1407
  if (exitCode === SSH_CONN_FAILURE) {
@@ -1482,7 +1567,7 @@ export function registerRunCommand(program) {
1482
1567
  await warnUnpushedWork(resumeExec.cwd ?? process.cwd());
1483
1568
  process.exit(resumeExit);
1484
1569
  }
1485
- const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported, isHeadlessSecretsContext }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports }, { shareRuntimeEnv },] = await Promise.all([
1570
+ const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported, isHeadlessSecretsContext }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports }, { shareRuntimeEnv },] = await Promise.all([
1486
1571
  import('../lib/exec.js'),
1487
1572
  import('../lib/agents.js'),
1488
1573
  import('../lib/profiles.js'),
@@ -1537,7 +1622,28 @@ export function registerRunCommand(program) {
1537
1622
  process.exit(1);
1538
1623
  }
1539
1624
  }
1540
- if (isValidAgent(rawAgent)) {
1625
+ if (autoHarnessRequested) {
1626
+ // Harness layer (RUSH-2132): weighted pick across installed harnesses
1627
+ // by best-account headroom. Zero healthy accounts anywhere fails loud
1628
+ // — launching a default "because it's there" is how a rotate loop
1629
+ // hammers an exhausted account.
1630
+ const byHarness = await collectHarnessCandidates();
1631
+ const harnessPick = pickHarnessWeighted(byHarness);
1632
+ if (!harnessPick) {
1633
+ console.error(chalk.red(formatNoHealthyHarnessError(classifyHarnessCandidates(byHarness))));
1634
+ process.exit(1);
1635
+ }
1636
+ agent = harnessPick.picked.agent;
1637
+ if (!options.quiet) {
1638
+ process.stderr.write(chalk.gray(formatHarnessPickBanner(harnessPick) + '\n'));
1639
+ }
1640
+ // --session-id keeps its claude-only semantics: honored when auto
1641
+ // picks claude, ignored (loudly) otherwise.
1642
+ if (options.sessionId && agent !== 'claude' && !options.quiet) {
1643
+ process.stderr.write(chalk.yellow(`[agents] --session-id ignored: auto picked ${agent} (only claude accepts a forced session id)\n`));
1644
+ }
1645
+ }
1646
+ else if (isValidAgent(rawAgent)) {
1541
1647
  agent = rawAgent;
1542
1648
  }
1543
1649
  else if (profileExists(rawAgent)) {
@@ -1929,6 +2035,15 @@ export function registerRunCommand(program) {
1929
2035
  else {
1930
2036
  try {
1931
2037
  const resolved = await resolveRunVersion(agent, strategy, cwd);
2038
+ if (resolved.exhausted) {
2039
+ // Fail loud (RUSH-2132): the old behavior warned "found no
2040
+ // usable version; falling back to defaults" and launched the
2041
+ // pinned default anyway — the exact move that loops a rotate
2042
+ // into an exhausted account. The message text is a contract
2043
+ // the Factory watchdog tail-detects; do not reword it.
2044
+ console.error(chalk.red(formatNoHealthyAccountError(agent, strategy, resolved.exhausted)));
2045
+ process.exit(1);
2046
+ }
1932
2047
  if (resolved.version) {
1933
2048
  version = resolved.version;
1934
2049
  rotationResult = resolved.rotation;
@@ -1938,6 +2053,8 @@ export function registerRunCommand(program) {
1938
2053
  }
1939
2054
  }
1940
2055
  else if (!options.quiet) {
2056
+ // No installed version at all (not "accounts exhausted" — that
2057
+ // fails loud above): keep the pre-existing default resolution.
1941
2058
  process.stderr.write(chalk.yellow(`[agents] strategy ${strategy} found no usable ${agent} version; falling back to defaults\n`));
1942
2059
  }
1943
2060
  }