@phnx-labs/agents-cli 1.21.3 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +67 -0
  2. package/README.md +32 -3
  3. package/dist/bin/agents +0 -0
  4. package/dist/commands/computer-actions.js +1 -0
  5. package/dist/commands/exec.d.ts +27 -0
  6. package/dist/commands/exec.js +123 -6
  7. package/dist/commands/models.js +36 -1
  8. package/dist/commands/projects.js +22 -2
  9. package/dist/commands/sessions-backfill.d.ts +32 -0
  10. package/dist/commands/sessions-backfill.js +186 -0
  11. package/dist/commands/sessions.d.ts +17 -1
  12. package/dist/commands/sessions.js +317 -18
  13. package/dist/commands/teams.js +1 -1
  14. package/dist/commands/worktree.d.ts +3 -3
  15. package/dist/commands/worktree.js +35 -4
  16. package/dist/lib/daemon.d.ts +5 -1
  17. package/dist/lib/daemon.js +63 -14
  18. package/dist/lib/devices/resolve-target.d.ts +6 -0
  19. package/dist/lib/devices/resolve-target.js +9 -3
  20. package/dist/lib/exec.js +39 -8
  21. package/dist/lib/hosts/dispatch.d.ts +12 -0
  22. package/dist/lib/hosts/dispatch.js +23 -6
  23. package/dist/lib/hosts/reconnect.d.ts +38 -0
  24. package/dist/lib/hosts/reconnect.js +85 -4
  25. package/dist/lib/hosts/run-target.js +14 -2
  26. package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
  27. package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
  28. package/dist/lib/model-tiers.d.ts +54 -0
  29. package/dist/lib/model-tiers.js +229 -0
  30. package/dist/lib/models.d.ts +3 -0
  31. package/dist/lib/models.js +44 -7
  32. package/dist/lib/pricing/prices.json +16 -1
  33. package/dist/lib/project-focus.d.ts +42 -0
  34. package/dist/lib/project-focus.js +80 -0
  35. package/dist/lib/project-schedule.d.ts +75 -0
  36. package/dist/lib/project-schedule.js +110 -0
  37. package/dist/lib/redact.d.ts +2 -0
  38. package/dist/lib/redact.js +22 -0
  39. package/dist/lib/remote-agents-json.d.ts +2 -0
  40. package/dist/lib/remote-agents-json.js +3 -3
  41. package/dist/lib/rotate.d.ts +84 -1
  42. package/dist/lib/rotate.js +155 -5
  43. package/dist/lib/runner.d.ts +4 -2
  44. package/dist/lib/runner.js +13 -4
  45. package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
  46. package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
  47. package/dist/lib/session/bash-command.js +60 -9
  48. package/dist/lib/session/db.d.ts +7 -1
  49. package/dist/lib/session/db.js +301 -32
  50. package/dist/lib/session/discover.d.ts +40 -7
  51. package/dist/lib/session/discover.js +144 -83
  52. package/dist/lib/session/parse.d.ts +8 -1
  53. package/dist/lib/session/parse.js +83 -32
  54. package/dist/lib/session/remote-list.d.ts +71 -0
  55. package/dist/lib/session/remote-list.js +410 -2
  56. package/dist/lib/session/shell-programs.d.ts +15 -0
  57. package/dist/lib/session/shell-programs.js +359 -0
  58. package/dist/lib/session/tool-calls.d.ts +88 -0
  59. package/dist/lib/session/tool-calls.js +612 -0
  60. package/dist/lib/session/tool-index.d.ts +100 -0
  61. package/dist/lib/session/tool-index.js +773 -0
  62. package/dist/lib/session/tool-store.d.ts +15 -0
  63. package/dist/lib/session/tool-store.js +198 -0
  64. package/dist/lib/session/types.d.ts +7 -0
  65. package/dist/lib/state.d.ts +10 -1
  66. package/dist/lib/state.js +11 -2
  67. package/dist/lib/teams/remoteWorktree.d.ts +3 -4
  68. package/dist/lib/teams/remoteWorktree.js +3 -4
  69. package/dist/lib/teams/worktree.d.ts +11 -1
  70. package/dist/lib/teams/worktree.js +42 -4
  71. package/dist/lib/types.d.ts +17 -0
  72. package/dist/lib/types.js +17 -0
  73. package/package.json +3 -1
package/CHANGELOG.md CHANGED
@@ -1,5 +1,62 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.22.0
4
+
5
+ - **`agents run auto` — full-auto dispatch (RUSH-2132).** `run auto` composes all three routing layers: host (14d launch affinity, unless `--host` is given), harness (installed CLIs weighted by best-account headroom), and account (the configured strategy). `balanced`/`available` now exit nonzero when every installed account is unhealthy — naming each excluded account, the earliest window reset, and the `--strategy pinned` escape hatch — instead of warning "falling back to defaults" and launching the exhausted pinned default. The error text is a machine-readable contract (`no healthy` + `resets <iso-time>`) the Factory watchdog tail-detects for rotate cooldowns. Source: `apps/cli/src/lib/rotate.ts`, `apps/cli/src/commands/exec.ts`, `apps/cli/src/lib/runner.ts`.
6
+
7
+ - **Bash-command summaries are faster and recognize more of what actually ran (#1830).**
8
+ `classifyBashCommand` (behind `agents sessions` / `agents activity` summaries) tokenized
9
+ the *entire* command — every pipeline segment, multi-KB heredoc bodies included — just to
10
+ read the leading executable, costing up to ~1ms on a big `cat <<HEREDOC …`. It now
11
+ tokenizes only the head of the first simple command. Coverage gaps that dumped commands
12
+ into a raw `other` pile are closed too: a `cd` prefix separated by `;` or a newline (not
13
+ just `&&`) unwraps to the real command, a path/tilde executable
14
+ (`~/.agents/skills/linear/scripts/linear`) resolves by basename, and the repo's own
15
+ toolchain (`agents`, `linear`, plus `rmdir`) is recognized — `agents` was the single top
16
+ unrecognized token. `ag` stays the silver searcher, not an `agents` alias. Source:
17
+ `apps/cli/src/lib/session/bash-command.ts`.
18
+
19
+ - **`agents computer describe` now counts toward `usedComputer`.** Every other
20
+ verb (`click`, `type`, `key`, `screenshot`, `run`, …) fires the
21
+ `computer.action` event via `emitComputerAction`; `describe` never did, so a
22
+ session that only ran `agents computer describe` read back
23
+ `usedComputer=false` — a false-negative in the sessions preview. A new
24
+ completeness-guard test pins every registered `agents computer` verb command
25
+ to a matching `emitComputerAction` call so a future verb can't ship the same
26
+ gap silently. Source: `apps/cli/src/commands/computer-actions.ts`,
27
+ `apps/cli/src/commands/computer-actions.test.ts`.
28
+
29
+ - **Pick a model by cost tier — `--model cheap|default|best|ultra` — on `agents run` and `agents teams add`.** Instead of a concrete id that churns per release and differs per harness, a tier resolves per `(harness, installed version)` to a model that version actually ships, ranked by the provider's own lineup (`opus/sonnet/haiku/fable`; Codex "frontier/balanced/fast" → Sol/Terra/Luna), then price, then size tokens. Single-model harnesses (Grok) map the tiers to reasoning effort; Droid uses a curated credit-multiplier map capped at 2x. An unsupported tier clamps to the nearest lower one; an unresolvable tier drops the flag and falls back to the harness default. Concrete model ids keep working unchanged. `agents models [agent[@version]]` now prints the per-harness tier map (with `~$/Mtok` where priced) and emits `tiers` in `--json`, and Droid joins the model-capable set. Also fixes the Claude catalog extractor returning 0 models on the newest native-binary format (a fallback id scan), and refreshes `prices.json` with the GPT-5.6 Sol/Terra/Luna series. Source: `apps/cli/src/lib/model-tiers.ts`, `apps/cli/src/lib/models.ts`, `apps/cli/src/lib/exec.ts`, `apps/cli/src/commands/models.ts`, `apps/cli/docs/model-tiers.md`.
30
+
31
+ - **`agents projects status` says what was worked on and what the dates prove.** Two new lines.
32
+ `focus` ranks the directories the window's commits landed in, read from the local checkout
33
+ with `git log --name-only` — no API call, no credential, no rate-limit budget, measured at
34
+ 0.23s over a 897-commit week. Changelog fragments and lockfiles are excluded from the
35
+ ranking: this repo files one fragment per PR, so `.changelog` otherwise ranked second and
36
+ presented PR count as an area of focus. `schedule` states what the milestone dates prove —
37
+ `overdue by N days`, `due in N days`, `N milestones, no issues filed against any`, or
38
+ `none dated`. Source: `apps/cli/src/lib/project-focus.ts`, `project-schedule.ts`.
39
+ - **The schedule line will never say "on track".** That verdict needs either project start and
40
+ target dates to interpolate expected progress, or a scope-history series to extrapolate a
41
+ finish date. Probed against a live workspace, all of them are absent (`health: null`,
42
+ `startDate`/`targetDate` null, `scopeHistory` and `completedScopeHistory` empty), so an
43
+ on-track or at-risk chip would be fabricated — and a confident wrong answer on a status card
44
+ is unfalsifiable from the card. When a human posts a Linear project health update, it is
45
+ relayed and attributed (`per Linear: atRisk`), never synthesized.
46
+
47
+ - **The `--device`/`--host` auto-reconnect loop no longer trusts a remote-origin exit code of 255 as "the SSH link dropped."** `reattachRemoteSession`'s `connected` flag is set as soon as the fast SSH preflight probe succeeds, before the actual reattach runs — so if the remote command it drives (`agents sessions focus <id> --local --attach-only`) ever exited 255 for a reason that had nothing to do with the SSH transport, that would be indistinguishable from the link itself dropping, refill the retry budget every cycle, and loop forever — printing "attempt 1/6" on every cycle and leaving the terminal full of aborted-TTY escape codes. The remote invocation is now wrapped in `bash -lc` so that whatever exit code it decides on, a 255 is remapped to 254 before this process sees it, closing that gap in the exit-code channel regardless of which remote-side path or peer `agents` version might produce it. A genuinely recurring *local* SSH failure can still refill the retry budget on every attempt by design (unchanged, tracked separately: phnx-labs/agents-cli#1884). Source: `apps/cli/src/lib/hosts/reconnect.ts`.
48
+
49
+ - **`agents sessions` can query distinct tool calls and count static Bash program occurrences locally or across the fleet.** Use `--include tools`, repeat `--query` with `tool:`, `program:`, `input:`, `output:`, `status:`, `exit:`, or `error:` fields, and add `--fleet` for live SSH fan-out. `--count` reports exact occurrence, containing-call, and session totals from ordered `wrapper`/`effective` rows without reparsing; synced mirrors are partitioned by origin so fleet evidence and totals do not duplicate sessions. Historical parsing is explicit and resumable through `agents sessions backfill tools`; normal scans index new and changed sessions once. Codex orchestration wrappers are parsed statically so only literal `tools.exec_command` commands reach the Bash AST, never wrapper code. Each device keeps a redacted, bounded relational SQLite/FTS5 cache, queries perform no transcript I/O or index writes, and no embeddings, vector database, or model calls are used. A sampling script explicitly backfills then extracts redacted shell-command origins from 50–100 sessions over the last seven days into a 16 MiB maximum artifact.
50
+
51
+ - **Local team worktrees base on freshly-fetched `origin/<default>`, not `HEAD`.**
52
+ `createWorktree` (and `agents worktree provision` for new branches) now
53
+ `git fetch origin` then `worktree add -b … origin/<default>`, matching
54
+ `createRemoteWorktree`. Previously local teammates forked from the
55
+ orchestrator's current `HEAD`, so a stale checkout made every teammate write
56
+ on old code and only surface the conflict at merge. Source:
57
+ `apps/cli/src/lib/teams/worktree.ts`, `apps/cli/src/commands/worktree.ts`,
58
+ `apps/cli/docs/teams.md`.
59
+
3
60
  ## 1.21.3
4
61
 
5
62
  - **`agents projects import --from-factory` stops printing raw git errors.** Reading each
@@ -33,6 +90,16 @@
33
90
  `agents sessions`'s perf sample for `command.end` now carries the session id
34
91
  and agent instead of being anonymous. Source: `apps/cli/src/lib/session/prompt.ts`,
35
92
  `apps/cli/src/lib/session/parse.ts`, `apps/cli/src/index.ts`.
93
+ - **`agents routines status` no longer reports "stopped" for a live scheduler, and
94
+ `agents routines start` can't spawn a second one.** The daemon writes its pid file
95
+ once (on claim/start) but rewrites the heartbeat every tick. If the pid file was lost
96
+ while the daemon kept ticking — an earlier status check clearing a stale/reused pid, or
97
+ the file removed out from under a live daemon — `status` read only the pid file and
98
+ reported `stopped` for a scheduler that was in fact running and firing jobs, while
99
+ `claimDaemonInstance()` would start a concurrent `JobScheduler` that double-fires every
100
+ routine. `isDaemonRunning()` and the single-instance claim now also trust a fresh
101
+ heartbeat whose pid is alive, re-adopting the pid file to heal the desync.
102
+ Source: `apps/cli/src/lib/daemon.ts`.
36
103
 
37
104
  ## 1.21.2
38
105
 
package/README.md CHANGED
@@ -161,7 +161,18 @@ agents run claude@
161
161
  agents run codex@ "review this branch"
162
162
  ```
163
163
 
164
- `--strategy balanced` spreads work across available versions of the same agent -- useful when you have multiple accounts and want to avoid burning through one.
164
+ `--strategy balanced` spreads work across available versions of the same agent -- useful when you have multiple accounts and want to avoid burning through one. When every account is rate-limited, the run exits nonzero naming each excluded account and the earliest window reset (use `--strategy pinned` to force the default) -- it never launches into an exhausted account.
165
+
166
+ ### Don't care which harness? `agents run auto`
167
+
168
+ ```bash
169
+ # Picks the host (14d usage affinity), the harness (installed CLIs weighted by
170
+ # best-account headroom), and the account (balanced) -- all three layers.
171
+ agents run auto "summarize recent commits"
172
+ agents run auto --host yosemite-s0 "fix the flaky test" # pin the host layer
173
+ ```
174
+
175
+ `run auto` excludes any harness whose accounts are all rate-limited or signed out, and exits nonzero with the earliest reset time when nothing anywhere is healthy.
165
176
 
166
177
  A trailing `@` opens an account picker before either an interactive or prompt-based run. Each installed version shows its account identity, exact version, login state, plan, and every available session, weekly, or monthly limit. Logged-out, rate-limited, and out-of-credit accounts remain visible with the reason they cannot be selected; signed-in accounts whose provider does not expose quota data stay selectable and say `limits unavailable`. The choice pins only that run and does not change your default version.
167
178
 
@@ -244,11 +255,26 @@ agents sessions a1b2c3d4 --markdown
244
255
 
245
256
  # Just the last 3 turns, user messages only
246
257
  agents sessions a1b2c3d4 --last 3 --include user
258
+
259
+ # Calls in recent Codex sessions on one device
260
+ agents sessions --include tools --agent codex --device mac-mini --since 7d
261
+
262
+ # One session where two different calls match; query every online device
263
+ agents sessions --include tools \
264
+ --query 'program:git input:merge' \
265
+ --query 'program:gh output:CONFLICT' \
266
+ --fleet --json
267
+
268
+ # Count pre-indexed static git sites, containing calls, and sessions
269
+ agents sessions --include tools --query 'program:git' --count --fleet --json
270
+
271
+ # Populate historical tool rows once on each device
272
+ agents sessions backfill tools --fleet
247
273
  ```
248
274
 
249
275
  Interactive picker when you're in a terminal. Structured output (`--json`, `--markdown`, filtered by role or turn count) when piped.
250
276
 
251
- Backed by a SQLite + FTS5 index at `~/.agents/.history/sessions/sessions.db` with incremental scanning -- warm reads in ~100ms. External tools can consume `--json` output as a programmatic observability layer; see [docs/05-sessions.md](apps/cli/docs/05-sessions.md) for the schema and [docs/06-observability.md](apps/cli/docs/06-observability.md) for the consumption patterns.
277
+ Backed by a SQLite + FTS5 index at `~/.agents/.history/sessions/sessions.db` with incremental scanning -- warm reads in ~100ms. Tool-call evidence is redacted and bounded before it is cached; repeated `--query` clauses must match distinct calls in one session. Tool queries read SQLite only: `agents sessions backfill tools` performs the one-time historical parse, while normal incremental scans index new and changed sessions. The index stores ordered static Bash program sites, so `--count` reports occurrences, containing tool calls, and distinct sessions without reparsing. `--fleet` executes one origin partition per device, so synced mirrors cannot duplicate compact evidence or counts returned over SSH; transcript bodies stay on their origin machine. This uses relational SQLite rows and literal FTS5 only, with no embeddings, vector database, or model calls. External tools can consume `--json` output as a programmatic observability layer; see [docs/05-sessions.md](apps/cli/docs/05-sessions.md) for the schemas and [docs/06-observability.md](apps/cli/docs/06-observability.md) for the consumption patterns.
252
278
 
253
279
  ### Live state, and catching up fast
254
280
 
@@ -1095,9 +1121,12 @@ Conversations with Claude, Codex, legacy Gemini, and other agents scatter across
1095
1121
  ```bash
1096
1122
  agents sessions "auth middleware" # Full-text search across all agents
1097
1123
  agents sessions --agent claude --since 7d
1124
+ agents sessions --include tools --query 'program:git' --fleet --json
1125
+ agents sessions --include tools --query 'program:git' --count --fleet --json
1126
+ agents sessions backfill tools --fleet
1098
1127
  ```
1099
1128
 
1100
- The index lives at `~/.agents/.history/sessions/sessions.db` (SQLite + FTS5). Nothing leaves your machine. See [Sessions](#sessions-across-agents) for full usage.
1129
+ The index lives at `~/.agents/.history/sessions/sessions.db` (SQLite + FTS5). A local query stays on the machine; an explicit `--fleet` tool query sends only redacted, bounded match evidence or aggregate counts over SSH. Historical tool parsing is explicit via `sessions backfill tools`; queries never parse transcripts. See [Sessions](#sessions-across-agents) for full usage.
1101
1130
 
1102
1131
  ### Secrets
1103
1132
 
package/dist/bin/agents CHANGED
Binary file
@@ -413,6 +413,7 @@ export function registerActionCommands(program) {
413
413
  if (opts.depth != null)
414
414
  params.max_depth = opts.depth;
415
415
  const res = unwrap(await client.call('describe', params));
416
+ emitComputerAction('describe', pid, opts, { depth: opts.depth });
416
417
  // The tree is inherently structured — always JSON, pretty unless --json.
417
418
  console.log(JSON.stringify(opts.json ? res : res.tree ?? res, null, 2));
418
419
  });
@@ -7,6 +7,7 @@
7
7
  */
8
8
  import { type Command } from 'commander';
9
9
  import type { ExecEffort } from '../lib/exec.js';
10
+ import { RUN_AUTO_KEYWORD } from '../lib/types.js';
10
11
  import { type SshGResult } from '../lib/hosts/ssh-config.js';
11
12
  export interface RunAccountPickerRequest {
12
13
  requested: boolean;
@@ -41,6 +42,32 @@ export declare function runAccountPickerConflicts(options: {
41
42
  on?: string;
42
43
  computer?: string;
43
44
  }): string[];
45
+ export { RUN_AUTO_KEYWORD };
46
+ /**
47
+ * Whether `run auto` should default its host layer to the affinity pick (the
48
+ * same machinery as `--device auto`). False when the caller pinned any host
49
+ * flag, and false when this process was itself dispatched by a host run — the
50
+ * dispatcher exports AGENTS_RUN_AUTO_HOST_RESOLVED=1 into the remote SHELL
51
+ * (hosts/dispatch.ts remoteRunShellPrelude) because it already resolved the
52
+ * host layer, and re-picking here would chain-hop the run across the fleet.
53
+ * Pure so the pinning matrix is unit-testable.
54
+ */
55
+ export declare function runAutoDefaultsToAffinity(options: {
56
+ host?: string;
57
+ device?: string;
58
+ on?: string;
59
+ computer?: string;
60
+ }, env?: NodeJS.ProcessEnv): boolean;
61
+ /**
62
+ * Whether an interactive host dispatch must mint a correlation launch id and
63
+ * resolve the remote session via the launch-id join (RUSH-2034), rather than
64
+ * trusting a pre-known session id. `run auto` ALWAYS joins: the harness is
65
+ * picked on the remote, so an explicit --session-id is only adopted when the
66
+ * pick lands on claude — pre-registering it would strand a stale session-index
67
+ * entry naming an id a non-claude pick never used (RUSH-2132). Pure so the
68
+ * decision matrix is unit-testable.
69
+ */
70
+ export declare function hostInteractiveNeedsCorrelationId(runAgent: string, hostSessionId: string | undefined, resumeId: string | undefined): boolean;
44
71
  /** The host descriptor fields the `--copy-creds` security gate reads. */
45
72
  export interface CopyCredsGateHost {
46
73
  name: string;
@@ -7,6 +7,7 @@
7
7
  */
8
8
  import { Option } from 'commander';
9
9
  import chalk from 'chalk';
10
+ import { RUN_AUTO_KEYWORD } from '../lib/types.js';
10
11
  import { setHelpSections } from '../lib/help.js';
11
12
  import { isInteractiveTerminal, isPromptCancelled, requireInteractiveSelection } from './utils.js';
12
13
  import { getUserAgentsDir } from '../lib/state.js';
@@ -66,6 +67,39 @@ export function runAccountPickerConflicts(options) {
66
67
  function isValidAgent(agent) {
67
68
  return agent in AGENTS;
68
69
  }
70
+ // Reserved `<agent>` keyword for `agents run auto` — canonical definition in
71
+ // lib/types.ts (shared with the host dispatch layer); re-exported here.
72
+ export { RUN_AUTO_KEYWORD };
73
+ /**
74
+ * Whether `run auto` should default its host layer to the affinity pick (the
75
+ * same machinery as `--device auto`). False when the caller pinned any host
76
+ * flag, and false when this process was itself dispatched by a host run — the
77
+ * dispatcher exports AGENTS_RUN_AUTO_HOST_RESOLVED=1 into the remote SHELL
78
+ * (hosts/dispatch.ts remoteRunShellPrelude) because it already resolved the
79
+ * host layer, and re-picking here would chain-hop the run across the fleet.
80
+ * Pure so the pinning matrix is unit-testable.
81
+ */
82
+ export function runAutoDefaultsToAffinity(options, env = process.env) {
83
+ if (hostTargetGiven(options).length > 0)
84
+ return false;
85
+ return env.AGENTS_RUN_AUTO_HOST_RESOLVED !== '1';
86
+ }
87
+ /**
88
+ * Whether an interactive host dispatch must mint a correlation launch id and
89
+ * resolve the remote session via the launch-id join (RUSH-2034), rather than
90
+ * trusting a pre-known session id. `run auto` ALWAYS joins: the harness is
91
+ * picked on the remote, so an explicit --session-id is only adopted when the
92
+ * pick lands on claude — pre-registering it would strand a stale session-index
93
+ * entry naming an id a non-claude pick never used (RUSH-2132). Pure so the
94
+ * decision matrix is unit-testable.
95
+ */
96
+ export function hostInteractiveNeedsCorrelationId(runAgent, hostSessionId, resumeId) {
97
+ if (resumeId)
98
+ return false;
99
+ if (runAgent === RUN_AUTO_KEYWORD)
100
+ return true;
101
+ return !hostSessionId && isSessionTrackedAgent(runAgent);
102
+ }
69
103
  /** Build a one-line banner describing which version the strategy picked. */
70
104
  function formatRotationBanner(result, verb = 'balanced') {
71
105
  const { picked, healthy, excluded } = result;
@@ -483,7 +517,7 @@ export function registerRunCommand(program) {
483
517
  .description('Execute an agent. Pass a prompt for headless runs; omit it to launch the agent interactively.')
484
518
  .option('-m, --mode <mode>', 'How much the agent can do: plan (read-only), edit (can write files), auto (smart classifier auto-approves safe ops, prompts for risky), skip (bypass all permission prompts). \'full\' accepted as alias for skip.', 'plan')
485
519
  .option('-e, --effort <effort>', 'Reasoning effort: low | medium | high | xhigh | max | auto (claude and codex only)', 'auto')
486
- .option('--model <model>', 'Override the model directly (e.g., claude-opus-4-6)')
520
+ .option('--model <model>', 'Cost tier (cheap|default|best|ultra) or a concrete model id; tiers resolve per harness+version to a supported model')
487
521
  .option('--env <key=value>', 'Pass environment variable to the agent (repeatable, e.g., --env DEBUG=1 --env API_KEY=xyz)', (val, prev) => [...prev, val], [])
488
522
  .option('--secrets <bundle>', 'Inject a secrets bundle (repeatable). Values resolve from macOS Keychain at run time. See `agents secrets`.', (val, prev) => [...prev, val], [])
489
523
  .option('--no-auto-secrets', 'Skip auto-injection of secrets declared by a workflow\'s frontmatter `secrets:` field. Has no effect on bare-agent runs.')
@@ -561,6 +595,11 @@ export function registerRunCommand(program) {
561
595
  # Pick a signed-in account/version for only this run
562
596
  agents run claude@
563
597
 
598
+ # Full-auto: affinity-pick the host, then the harness with the most
599
+ # account headroom, then a balanced account on it
600
+ agents run auto "fix the flaky test" --mode edit
601
+ agents run auto --host yosemite-s0 "fix the flaky test" # pin the host
602
+
564
603
  # Open the session in a terminal tab — detected from where your sessions
565
604
  # already run (Ghostty / iTerm / Terminal.app); force one with a value
566
605
  agents run claude --terminal
@@ -599,6 +638,13 @@ export function registerRunCommand(program) {
599
638
  balanced distribute load across healthy accounts by remaining capacity (default)
600
639
  A version/account is skipped when it is rate-limited right now — any usage window (incl. the 5-hour session window) at 100%, matching the 'agents view' badge.
601
640
  --balanced is shorthand for --strategy balanced. Ignored when @version is pinned, when a profile is used, or with --fallback.
641
+ Zero healthy accounts under balanced/available exits nonzero naming each
642
+ excluded account and the earliest window reset — use --strategy pinned to force.
643
+
644
+ 'auto' harness (agents run auto): picks the host (14d usage affinity,
645
+ unless --host is given), the harness (installed CLIs weighted by
646
+ best-account headroom), and the account (the strategy above). Zero
647
+ healthy accounts on any harness exits nonzero with the earliest reset.
602
648
 
603
649
  Account picker: append @ with no version (agents run claude@) to choose one
604
650
  installed account for this run. Rows show identity, login state, plan,
@@ -682,6 +728,32 @@ export function registerRunCommand(program) {
682
728
  process.exit(1);
683
729
  }
684
730
  }
731
+ // `agents run auto`: the reserved harness keyword — full-auto dispatch
732
+ // (host affinity → cross-harness balance → account balance, RUSH-2132).
733
+ if (normalizedAgentSpec.split('@')[0] === RUN_AUTO_KEYWORD && normalizedAgentSpec !== RUN_AUTO_KEYWORD) {
734
+ console.error(chalk.red(`agents run auto picks the harness itself — a @version pin does not apply. ` +
735
+ `Pin a concrete harness instead: agents run <harness>@<version>.`));
736
+ process.exit(1);
737
+ }
738
+ const autoHarnessRequested = normalizedAgentSpec === RUN_AUTO_KEYWORD;
739
+ if (autoHarnessRequested) {
740
+ // `auto` is reserved. If a future harness registers that id, the
741
+ // keyword collides — fail loud rather than silently shadow the harness.
742
+ if (RUN_AUTO_KEYWORD in AGENTS) {
743
+ console.error(chalk.red(`'${RUN_AUTO_KEYWORD}' is now a registered harness and collides with the reserved 'run auto' keyword. ` +
744
+ `Run the harness by name instead.`));
745
+ process.exit(1);
746
+ }
747
+ if (accountPickerRequested) {
748
+ console.error(chalk.red(`agents run auto picks the harness and account itself — the trailing-@ account picker needs a concrete harness (agents run <harness>@).`));
749
+ process.exit(1);
750
+ }
751
+ // Host layer: with no explicit --host/--device, default to the
752
+ // affinity pick. Skipped on a host-dispatched run — its dispatcher
753
+ // already resolved this layer (see runAutoDefaultsToAffinity).
754
+ if (runAutoDefaultsToAffinity(options))
755
+ options.device = 'auto';
756
+ }
685
757
  // --device auto / --host auto (and deprecated --smart): affinity-pick host.
686
758
  // Harness is always the agent the user typed — never auto-picked.
687
759
  // Affinity failure degrades to local (does not kill the run).
@@ -1062,6 +1134,10 @@ export function registerRunCommand(program) {
1062
1134
  process.exit(1);
1063
1135
  }
1064
1136
  const hostName = hostGiven[0];
1137
+ // Note: a `run auto` dispatch needs no marker forwarded from here — the
1138
+ // dispatch layer (hosts/dispatch.ts remoteRunShellPrelude) exports the
1139
+ // chain-hop guard into the remote shell for BOTH interactive and
1140
+ // headless paths, keyed off the agent name being `auto`.
1065
1141
  const { resolveHostRunTarget, resolveHostSessionId, dispatchPromptToHost, HostResolutionError } = await import('../lib/hosts/run-target.js');
1066
1142
  const { runInteractiveOnHost } = await import('../lib/hosts/dispatch.js');
1067
1143
  const { registerInteractiveHostSession } = await import('../lib/hosts/session-index.js');
@@ -1242,11 +1318,16 @@ export function registerRunCommand(program) {
1242
1318
  // the stream we resolve the id by one ssh read of the remote hook
1243
1319
  // record — the same launch-id join used locally (RUSH-2034). Not
1244
1320
  // needed for Claude (id forced) or resume (id already known).
1245
- const correlationLaunchId = !hostSessionId && !resumeId && isSessionTrackedAgent(runAgent) ? randomUUID() : undefined;
1321
+ // `run auto` ALWAYS joins: the remote picks the harness, so an
1322
+ // explicit --session-id is only adopted by a claude pick.
1323
+ const correlationLaunchId = hostInteractiveNeedsCorrelationId(runAgent, hostSessionId, resumeId) ? randomUUID() : undefined;
1246
1324
  const hostEnv = correlationLaunchId
1247
1325
  ? [...options.env, `AGENT_LAUNCH_ID=${correlationLaunchId}`]
1248
1326
  : options.env;
1249
- if (hostSessionId) {
1327
+ // `run auto` never pre-registers: the explicit id is only real when
1328
+ // the remote pick lands on claude. The launch-id join below records
1329
+ // the id the remote ACTUALLY used, whatever the pick.
1330
+ if (hostSessionId && runAgent !== RUN_AUTO_KEYWORD) {
1250
1331
  registerInteractiveHostSession({
1251
1332
  cwd: process.cwd(),
1252
1333
  host: host.name,
@@ -1316,7 +1397,11 @@ export function registerRunCommand(program) {
1316
1397
  // re-attach the live pane automatically instead of exiting — the user
1317
1398
  // never has to notice the drop and `agents sessions focus` by hand.
1318
1399
  // `raw` runs aren't tmux wrapped, so there is nothing to reconnect to.
1319
- const reconnectId = hostSessionId ?? resolvedRemoteId ?? resumeId;
1400
+ // For `run auto` prefer the join-resolved id (the harness the remote
1401
+ // ACTUALLY picked) over the explicit --session-id only claude adopts.
1402
+ const reconnectId = (runAgent === RUN_AUTO_KEYWORD
1403
+ ? resolvedRemoteId ?? hostSessionId
1404
+ : hostSessionId ?? resolvedRemoteId) ?? resumeId;
1320
1405
  if (reconnectId && !isRaw) {
1321
1406
  const { reconnectInteractiveSession, SSH_CONN_FAILURE } = await import('../lib/hosts/reconnect.js');
1322
1407
  if (exitCode === SSH_CONN_FAILURE) {
@@ -1482,7 +1567,7 @@ export function registerRunCommand(program) {
1482
1567
  await warnUnpushedWork(resumeExec.cwd ?? process.cwd());
1483
1568
  process.exit(resumeExit);
1484
1569
  }
1485
- const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported, isHeadlessSecretsContext }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports }, { shareRuntimeEnv },] = await Promise.all([
1570
+ const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported, isHeadlessSecretsContext }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports }, { shareRuntimeEnv },] = await Promise.all([
1486
1571
  import('../lib/exec.js'),
1487
1572
  import('../lib/agents.js'),
1488
1573
  import('../lib/profiles.js'),
@@ -1537,7 +1622,28 @@ export function registerRunCommand(program) {
1537
1622
  process.exit(1);
1538
1623
  }
1539
1624
  }
1540
- if (isValidAgent(rawAgent)) {
1625
+ if (autoHarnessRequested) {
1626
+ // Harness layer (RUSH-2132): weighted pick across installed harnesses
1627
+ // by best-account headroom. Zero healthy accounts anywhere fails loud
1628
+ // — launching a default "because it's there" is how a rotate loop
1629
+ // hammers an exhausted account.
1630
+ const byHarness = await collectHarnessCandidates();
1631
+ const harnessPick = pickHarnessWeighted(byHarness);
1632
+ if (!harnessPick) {
1633
+ console.error(chalk.red(formatNoHealthyHarnessError(classifyHarnessCandidates(byHarness))));
1634
+ process.exit(1);
1635
+ }
1636
+ agent = harnessPick.picked.agent;
1637
+ if (!options.quiet) {
1638
+ process.stderr.write(chalk.gray(formatHarnessPickBanner(harnessPick) + '\n'));
1639
+ }
1640
+ // --session-id keeps its claude-only semantics: honored when auto
1641
+ // picks claude, ignored (loudly) otherwise.
1642
+ if (options.sessionId && agent !== 'claude' && !options.quiet) {
1643
+ process.stderr.write(chalk.yellow(`[agents] --session-id ignored: auto picked ${agent} (only claude accepts a forced session id)\n`));
1644
+ }
1645
+ }
1646
+ else if (isValidAgent(rawAgent)) {
1541
1647
  agent = rawAgent;
1542
1648
  }
1543
1649
  else if (profileExists(rawAgent)) {
@@ -1929,6 +2035,15 @@ export function registerRunCommand(program) {
1929
2035
  else {
1930
2036
  try {
1931
2037
  const resolved = await resolveRunVersion(agent, strategy, cwd);
2038
+ if (resolved.exhausted) {
2039
+ // Fail loud (RUSH-2132): the old behavior warned "found no
2040
+ // usable version; falling back to defaults" and launched the
2041
+ // pinned default anyway — the exact move that loops a rotate
2042
+ // into an exhausted account. The message text is a contract
2043
+ // the Factory watchdog tail-detects; do not reword it.
2044
+ console.error(chalk.red(formatNoHealthyAccountError(agent, strategy, resolved.exhausted)));
2045
+ process.exit(1);
2046
+ }
1932
2047
  if (resolved.version) {
1933
2048
  version = resolved.version;
1934
2049
  rotationResult = resolved.rotation;
@@ -1938,6 +2053,8 @@ export function registerRunCommand(program) {
1938
2053
  }
1939
2054
  }
1940
2055
  else if (!options.quiet) {
2056
+ // No installed version at all (not "accounts exhausted" — that
2057
+ // fails loud above): keep the pre-existing default resolution.
1941
2058
  process.stderr.write(chalk.yellow(`[agents] strategy ${strategy} found no usable ${agent} version; falling back to defaults\n`));
1942
2059
  }
1943
2060
  }
@@ -11,9 +11,11 @@ import { homeDir } from '../lib/platform/index.js';
11
11
  import { resolveAgentName, formatAgentError, agentLabel, } from '../lib/agents.js';
12
12
  import { listInstalledVersions, getGlobalDefault, resolveVersion, resolveVersionAlias } from '../lib/versions.js';
13
13
  import { getModelCatalog, locateModelSource } from '../lib/models.js';
14
+ import { resolveTierMap, MODEL_TIERS } from '../lib/model-tiers.js';
15
+ import { getModelPricing } from '../lib/pricing/index.js';
14
16
  import { terminalWidth, truncateToWidth, stringWidth } from '../lib/session/width.js';
15
17
  import { wrapJoined } from './inspect.js';
16
- const MODEL_CAPABLE_AGENTS = ['claude', 'codex', 'opencode', 'cursor', 'openclaw', 'antigravity', 'kimi', 'grok'];
18
+ const MODEL_CAPABLE_AGENTS = ['claude', 'codex', 'opencode', 'cursor', 'openclaw', 'antigravity', 'kimi', 'grok', 'droid'];
17
19
  /**
18
20
  * Agents that don't necessarily install under ~/.agents/versions (cursor ships
19
21
  * via a curl script). For these, fall back to the PATH binary and synthesize
@@ -52,6 +54,7 @@ export function registerModelsCommand(program) {
52
54
  agent,
53
55
  version,
54
56
  catalog: getModelCatalog(agent, version),
57
+ tiers: resolveTierMap(agent, version),
55
58
  }));
56
59
  console.log(JSON.stringify(out, null, 2));
57
60
  return;
@@ -122,8 +125,16 @@ function printCatalog(agent, version, isDefault, options) {
122
125
  const tag = isDefault ? chalk.gray(' (default)') : '';
123
126
  const header = `${agentLabel(agent)} ${chalk.bold(version)}${tag}`;
124
127
  console.log(header);
128
+ // Cost tiers first -- the thing an orchestrating agent reads to pick a model.
129
+ printTiers(agent, version);
125
130
  const src = locateModelSource(agent, version);
126
131
  if (!src) {
132
+ if (agent === 'droid') {
133
+ // Droid has no extractable catalog (no models CLI/API/config); the curated
134
+ // tier map above is the whole surface.
135
+ console.log(chalk.gray(' (Droid has no model list command; tiers are a curated, credit-multiplier map.)'));
136
+ return;
137
+ }
127
138
  console.log(chalk.yellow(` Could not locate model source for ${agent}@${version}.`));
128
139
  console.log(chalk.gray(` Expected the agent's CLI bundle or native binary under ~/.agents/.history/versions/${agent}/${version}/.`));
129
140
  return;
@@ -171,6 +182,30 @@ function printCatalog(agent, version, isDefault, options) {
171
182
  }
172
183
  }
173
184
  }
185
+ /** Rough blended $/Mtok label for a model id, or '' when unpriced. */
186
+ function priceLabel(id) {
187
+ const p = getModelPricing(id);
188
+ if (!p)
189
+ return chalk.gray(' --');
190
+ const perM = (p.inputPerToken + p.outputPerToken) * 1e6;
191
+ return chalk.gray(` ~$${perM.toFixed(0)}/Mtok`);
192
+ }
193
+ /** Print the cheap/default/best/ultra tier map for an (agent, version). */
194
+ function printTiers(agent, version) {
195
+ const map = resolveTierMap(agent, version);
196
+ if (!MODEL_TIERS.some((t) => map[t].model))
197
+ return;
198
+ console.log(chalk.gray(' tiers:'));
199
+ for (const t of MODEL_TIERS) {
200
+ const r = map[t];
201
+ if (!r.model)
202
+ continue;
203
+ const eff = r.effort ? chalk.gray(` @${r.effort}`) : '';
204
+ const clamp = r.clampedFrom ? chalk.gray(' (clamped)') : '';
205
+ console.log(` ${chalk.cyan(t.padEnd(8))} ${chalk.bold(r.model)}${eff}${priceLabel(r.model)}${clamp}`);
206
+ }
207
+ console.log();
208
+ }
174
209
  /** Abbreviate a path by replacing the home directory with ~. */
175
210
  function shortPath(p) {
176
211
  return p.replace(homeDir(), '~');
@@ -29,6 +29,8 @@ import { rollupSessionsByProject, liveDeadSplit, enrichProjectSignals, formatPro
29
29
  import { fetchLinearProjectCounts } from '../lib/linear-project-counts.js';
30
30
  import { listLinearProjects, pickLinearProject } from '../lib/linear-projects.js';
31
31
  import { checkRepoSlug } from '../lib/project-doctor.js';
32
+ import { readFocusAreas } from '../lib/project-focus.js';
33
+ import { formatVerdict, scheduleVerdict } from '../lib/project-schedule.js';
32
34
  import { buildFactoryImportCandidates, buildLinearImportCandidates, validateImportOpts, } from '../lib/project-import.js';
33
35
  /** Recursion guard: a peer answering a probe fan-out never re-fans-out itself. */
34
36
  export const PROJECTS_NO_FANOUT_ENV = 'AGENTS_PROJECTS_LOCAL';
@@ -268,7 +270,9 @@ function statusBar(r) {
268
270
  }
269
271
  function renderCard(def, r, remote, fleet, linear, nowMs = Date.now(),
270
272
  /** How many milestones to print. `status` shows the next one; `view` shows all. */
271
- milestoneLimit = 1) {
273
+ milestoneLimit = 1,
274
+ /** Directories the window's work landed in, from local git. */
275
+ focus = []) {
272
276
  // The headline counts LIVE agents. It used to be every matched session, which
273
277
  // read `39 agents` on a project where 19 had crashed. `planPct` used to sit
274
278
  // here too and is gone: it summed each session's latest checklist snapshot,
@@ -304,6 +308,14 @@ milestoneLimit = 1) {
304
308
  for (const line of formatMilestoneLines(linear?.milestones ?? [], linear?.nextMilestone, nowMs, milestoneLimit)) {
305
309
  console.log(line);
306
310
  }
311
+ // What the dates prove — never an invented "on track". See project-schedule.ts.
312
+ const verdict = linear?.milestones?.length ? formatVerdict(scheduleVerdict(linear.milestones, nowMs)) : undefined;
313
+ if (verdict) {
314
+ console.log(` ${chalk.dim('schedule')} ${verdict.warn ? chalk.yellow(verdict.text) : verdict.text}`);
315
+ }
316
+ if (focus.length) {
317
+ console.log(` ${chalk.dim('focus')} ${focus.map((f) => `${f.path} ${chalk.dim(String(f.touches))}`).join(chalk.dim(' · '))}`);
318
+ }
307
319
  if (r && r.tickets.length) {
308
320
  console.log(` ${chalk.dim('tickets')} ${r.tickets.slice(0, 8).join(' · ')}${r.tickets.length > 8 ? ' …' : ''}`);
309
321
  }
@@ -583,6 +595,8 @@ export function registerProjectsCommands(program) {
583
595
  // log but skips the gh calls + Linear counts (both are network).
584
596
  const remote = new Map();
585
597
  const linear = new Map();
598
+ // Local git, no API, no rate limit — measured 0.23s over a 897-commit week.
599
+ const focus = new Map();
586
600
  await Promise.all(defs.map(async (d) => {
587
601
  const skipRemote = opts.remote === false;
588
602
  const [sig, counts] = await Promise.all([
@@ -594,6 +608,8 @@ export function registerProjectsCommands(program) {
594
608
  remote.set(d.name, sig);
595
609
  if (counts)
596
610
  linear.set(d.name, counts);
611
+ if (d.root)
612
+ focus.set(d.name, await readFocusAreas(expandLocalHome(d.root), windowDays));
597
613
  }));
598
614
  /** This def's slice of the fleet probe, in its own target order. */
599
615
  const fleetFor = (d) => {
@@ -612,6 +628,10 @@ export function registerProjectsCommands(program) {
612
628
  byStatus: r?.byStatus ?? {},
613
629
  members: r?.members ?? [],
614
630
  plan: r?.plan ?? { done: 0, total: 0 },
631
+ schedule: linear.get(d.name)?.milestones?.length
632
+ ? scheduleVerdict(linear.get(d.name).milestones, nowMs)
633
+ : null,
634
+ focus: focus.get(d.name) ?? [],
615
635
  live: r ? liveDeadSplit(r.byStatus).live : 0,
616
636
  dead: r ? liveDeadSplit(r.byStatus).dead : 0,
617
637
  openPrs: r?.openPrs ?? [],
@@ -630,7 +650,7 @@ export function registerProjectsCommands(program) {
630
650
  return;
631
651
  }
632
652
  for (const d of defs) {
633
- renderCard(d, roll.get(d.name), remote.get(d.name), opts.fleet ? fleetFor(d) : undefined, linear.get(d.name), nowMs);
653
+ renderCard(d, roll.get(d.name), remote.get(d.name), opts.fleet ? fleetFor(d) : undefined, linear.get(d.name), nowMs, 1, focus.get(d.name) ?? []);
634
654
  }
635
655
  if (fleetSkipped.length > 0)
636
656
  process.stdout.write(formatFleetSkippedNote(fleetSkipped));
@@ -0,0 +1,32 @@
1
+ import type { Command } from 'commander';
2
+ import { type ToolIndexCoverage } from '../lib/session/tool-index.js';
3
+ export interface ToolBackfillMachineResult {
4
+ machine: string;
5
+ indexedFiles: number;
6
+ indexedCalls: number;
7
+ coverage: ToolIndexCoverage;
8
+ }
9
+ export interface ToolBackfillEnvelope {
10
+ schemaVersion: 1;
11
+ kind: 'tools-backfill';
12
+ generatedAt: string;
13
+ complete: boolean;
14
+ machines: ToolBackfillMachineResult[];
15
+ }
16
+ interface ToolBackfillOptions {
17
+ agent?: string;
18
+ project?: string;
19
+ since?: string;
20
+ until?: string;
21
+ unmanaged?: boolean;
22
+ teams?: boolean;
23
+ json?: boolean;
24
+ local?: boolean;
25
+ fleet?: boolean;
26
+ host?: string[] | string;
27
+ device?: string[] | string;
28
+ }
29
+ export declare function backfillToolsLocal(options: ToolBackfillOptions, oneBatch?: boolean): Promise<ToolBackfillMachineResult>;
30
+ export declare function runToolsBackfill(options: ToolBackfillOptions): Promise<ToolBackfillEnvelope>;
31
+ export declare function registerSessionsBackfillCommand(sessionsCmd: Command): void;
32
+ export {};