@phnx-labs/agents-cli 1.22.60 → 1.22.61

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/CHANGELOG.md +49 -0
  2. package/dist/cli/command-registry.d.ts +1 -0
  3. package/dist/cli/command-registry.js +2 -0
  4. package/dist/commands/browser.js +9 -4
  5. package/dist/commands/doctor.js +1 -1
  6. package/dist/commands/exec.js +35 -1
  7. package/dist/commands/harness-hooks.d.ts +55 -0
  8. package/dist/commands/harness-hooks.js +104 -0
  9. package/dist/commands/harness-wizard.d.ts +33 -14
  10. package/dist/commands/harness-wizard.js +53 -23
  11. package/dist/commands/harness.d.ts +14 -0
  12. package/dist/commands/harness.js +86 -5
  13. package/dist/commands/reminders.d.ts +9 -0
  14. package/dist/commands/reminders.js +49 -0
  15. package/dist/commands/run-account-picker.d.ts +14 -0
  16. package/dist/commands/run-account-picker.js +13 -0
  17. package/dist/commands/teams.d.ts +1 -1
  18. package/dist/commands/teams.js +9 -3
  19. package/dist/index.js +9 -0
  20. package/dist/lib/accounting/rotate.d.ts +63 -0
  21. package/dist/lib/accounting/rotate.js +229 -13
  22. package/dist/lib/browser/drivers/local.d.ts +11 -0
  23. package/dist/lib/browser/drivers/local.js +26 -0
  24. package/dist/lib/browser/profiles.js +8 -6
  25. package/dist/lib/browser/service.d.ts +12 -8
  26. package/dist/lib/browser/service.js +38 -10
  27. package/dist/lib/claude-statusline.d.ts +14 -1
  28. package/dist/lib/claude-statusline.js +27 -2
  29. package/dist/lib/daemon/runner.js +17 -2
  30. package/dist/lib/devices/doctor-findings.d.ts +1 -1
  31. package/dist/lib/devices/doctor-findings.js +22 -4
  32. package/dist/lib/doctor-diff.d.ts +21 -5
  33. package/dist/lib/doctor-diff.js +242 -76
  34. package/dist/lib/feed/events.d.ts +1 -1
  35. package/dist/lib/feed/events.js +25 -16
  36. package/dist/lib/github/gh-overload.d.ts +58 -0
  37. package/dist/lib/github/gh-overload.js +246 -0
  38. package/dist/lib/github/rest.d.ts +64 -0
  39. package/dist/lib/github/rest.js +111 -0
  40. package/dist/lib/harness-connection-test.d.ts +57 -0
  41. package/dist/lib/harness-connection-test.js +80 -0
  42. package/dist/lib/heal.js +8 -3
  43. package/dist/lib/installations/shims.d.ts +22 -0
  44. package/dist/lib/installations/shims.js +104 -0
  45. package/dist/lib/linear-project-counts.js +8 -0
  46. package/dist/lib/linear-rate-limit.d.ts +26 -0
  47. package/dist/lib/linear-rate-limit.js +163 -0
  48. package/dist/lib/mcp.d.ts +9 -0
  49. package/dist/lib/mcp.js +37 -1
  50. package/dist/lib/open-url.js +5 -3
  51. package/dist/lib/permissions.d.ts +28 -0
  52. package/dist/lib/permissions.js +156 -1
  53. package/dist/lib/refresh.js +9 -1
  54. package/dist/lib/reminders.d.ts +29 -0
  55. package/dist/lib/reminders.js +88 -0
  56. package/dist/lib/resource-content-diff.d.ts +33 -0
  57. package/dist/lib/resource-content-diff.js +103 -0
  58. package/dist/lib/rules/compile.d.ts +7 -0
  59. package/dist/lib/rules/compile.js +7 -1
  60. package/dist/lib/session/active.d.ts +41 -4
  61. package/dist/lib/session/active.js +58 -7
  62. package/dist/lib/session/host-link.d.ts +22 -0
  63. package/dist/lib/session/host-link.js +40 -4
  64. package/dist/lib/session/trajectory.d.ts +42 -0
  65. package/dist/lib/session/trajectory.js +46 -27
  66. package/dist/lib/ssh-exec.d.ts +30 -0
  67. package/dist/lib/ssh-exec.js +37 -5
  68. package/dist/lib/startup/command-registry.js +1 -1
  69. package/dist/lib/subagents-registry.d.ts +18 -0
  70. package/dist/lib/subagents-registry.js +79 -0
  71. package/dist/lib/teams/agents.d.ts +12 -0
  72. package/dist/lib/teams/agents.js +51 -0
  73. package/dist/lib/traces/schema2-build.d.ts +85 -0
  74. package/dist/lib/traces/schema2-build.js +637 -0
  75. package/dist/lib/traces/schema2-danger.d.ts +36 -0
  76. package/dist/lib/traces/schema2-danger.js +185 -0
  77. package/dist/lib/traces/schema2.d.ts +149 -0
  78. package/dist/lib/traces/schema2.js +20 -0
  79. package/dist/lib/traces/sync.d.ts +93 -0
  80. package/dist/lib/traces/sync.js +75 -22
  81. package/dist/lib/traces/worker-template.js +5 -0
  82. package/dist/lib/uninstall.js +10 -1
  83. package/dist/lib/workflows.d.ts +11 -0
  84. package/dist/lib/workflows.js +67 -8
  85. package/package.json +1 -1
package/CHANGELOG.md CHANGED
@@ -1,7 +1,56 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.22.61
4
+
5
+ - **`agents harness` wizard: model catalog, connection test, and edit matrix (PHNX-2218/2220/2221/2222).** The create/edit wizard now picks the model from the host's own catalog (`getModelCatalog`) with a free-text escape hatch, gates the endpoint step to hosts that actually carry one, and runs a real pre-save connection test — `agents run <name> "say alive in one word" --headless --timeout 60s` classified into pass / auth / endpoint / model — behind a confirm with `--test`/`--no-test`, offering keep / edit / delete on failure rather than saving a broken harness silently. A resolver-sourced `harnessEditable` matrix disables (with a reason) any param the host's API format can't carry. Source: `apps/cli/src/commands/harness.ts`, `apps/cli/src/commands/harness-wizard.ts`, `apps/cli/src/commands/harness-hooks.ts`, `apps/cli/src/lib/harness-connection-test.ts`.
6
+
7
+ - **Cap per-machine Linear API spend with a shared per-key request budget (PHNX-2310).** With ~13 concurrent drain agents on one Linear API key on a single box, nothing stopped them collectively exhausting Linear's 2,500 requests/hour limit and throttling ticket-status reads for every agent on that box. The `agents projects` card fetch now reserves against a proactive request budget keyed by the API key before each request; when the hourly pool is spent it serves the last cached snapshot (marked stale) instead of forcing a 429. The pool is lock-free on-disk state under the machine-local cache (one stamp file per request, a sliding one-hour window), so the many separate agent processes on a box coordinate without a shared mutable document to race on. Scope is per machine: the state is not fleet-synced, so two boxes sharing one key each budget independently — cross-device aggregation is a known limitation, not yet covered. This complements the existing read cache and reactive 429 backoff. Source: `cli/src/lib/linear-rate-limit.ts`, `cli/src/lib/linear-project-counts.ts`.
8
+
9
+ - **`agents browser` drives your RUNNING Arc instead of spawning a second one (PHNX-2399).** Arc is single-instance: relaunching the Arc binary with a fresh `--user-data-dir` produced a stray window and no CDP endpoint rather than a debuggable instance. An Arc profile now attaches to the Arc you already have open (point its endpoint at the port you launched Arc's remote debugging on), and when that Arc exposes no CDP endpoint on the profile's port it **fails loud** with the one relaunch that fixes it (`open -a Arc --args --remote-debugging-port=<port>`) — it never silently launches a duplicate. Bind the profile to the tab/Space you want to drive with `--target-filter` (accepted for `--browser arc`, not just `--electron`): it is consulted when Arc picks which existing tab to drive, so `navigate` reuses your bound Space instead of an unrelated tab, and refuses rather than borrowing one when nothing matches. The documented first-use flow — `agents browser navigate --profile arc --url …` on a fresh profile — now attaches to an open tab instead of erroring. Opening a brand-new tab remains the one CDP-only op Arc can't survive (`Target.createTarget` crashes it), so tab-creating verbs fail clearly and steer you to reuse an existing tab or a Chromium-family browser. Source: `cli/src/lib/browser/drivers/local.ts`, `cli/src/lib/browser/service.ts`, `cli/src/commands/browser.ts`, `cli/src/lib/open-url.ts`.
10
+
11
+ - **`agents sessions watch --json` rows now carry a canonical `phase` (PHNX-2484).** Each live/recoverable session row projects a coarse lifecycle bucket — `running | waiting | failed | done | idle` — derived once at the source from the finalized status (`derivePhase` in `cli/src/lib/session/active.ts`), so thin-client consumers (the AGI EXT Fleet panel) read `phase` off the row instead of re-deriving it from the status word. It also fixes a drift a status-only re-derivation had: `orphaned` and `crashed` now bucket to `failed` (dangling/dead — needs attention) rather than falling through to `idle` and hiding a dead agent.
12
+
13
+ - **`balanced`/`available` routing never auto-picks an account from entirely stale usage (PHNX-2526).** When every signed-in account's usage snapshot is older than the 5-minute decision horizon and none can be verified, the initial route no longer guesses on a stale number (the yosemite-s1 incident, where a 26h–2.7d-old snapshot read 48% while the account was at its weekly cap). An interactive run now shows the account picker so a human chooses with the stale numbers in view; an unattended run (`--headless`/`--json`/no TTY, or a routine) fails loud with `NO_VERIFIED_USAGE`. The stale candidates are preserved only for bounded post-rejection failover, never the first pick. A pool that is merely *blind* (no snapshot at all — a worker box whose usage endpoint 403s) is unaffected and still draws a pick. Source: `apps/cli/src/lib/accounting/rotate.ts`, `apps/cli/src/commands/exec.ts`, `apps/cli/src/lib/daemon/runner.ts`.
14
+
15
+ - **Raise SSH `ControlPersist` to 10 minutes so repeated fleet touches reuse warm connections (PHNX-2582).** The OpenSSH multiplex master now lives for 10 minutes after its last use (`SSH_CONTROL_PERSIST_SECONDS`) instead of 60 seconds. The master only helps a repeated touch of the same host that arrives while it is still alive, and the ad-hoc `--device` / fan-out calls that touch a box repeatedly (`sessions --active`, `fleet ping`, `doctor`, `teams`, a `--device` command run a few times) arrive in bursts spread over minutes — so at the old 60 s window any two more than a minute apart both paid a fresh TCP+auth handshake (~100-300 ms direct, up to ~500 ms relayed). 10 minutes keeps a multi-minute burst warm (~3-5 ms per reuse) while staying bounded so an idle master to a slept host is reaped rather than lingering. Source: `cli/src/lib/ssh-exec.ts`.
16
+
17
+ - **A running agent whose owning IDE window died is now surfaced as `orphaned` (PHNX-3183).** `foldHostLink` promoted only `idle`/`input_required` sessions to `orphaned`; a still-`running` agent was never flagged, so a genuinely-stranded remote agent — alive in tmux after its owning window crashed, rebooted, or dropped its SSH link — stayed hidden as a healthy "running" row. It now promotes a `running` session to `orphaned` when, and only when, its owning window was LOST: the window's registry slice went stale past `HOST_HEARTBEAT_STALE_MS` after having republished it (`hostWindowLost`). Client ABSENCE alone still never flags a running agent — a tmux attached-client count of zero is the normal steady state for a detached `agents run --device` pane (RUSH-3125), so promoting on it would relabel every unattended remote agent as orphaned (the over-report reverted in `6d973b823`). This narrows spec `SES-18a` to the window-loss case rather than the blanket precondition-drop that was reverted. Source: `cli/src/lib/session/host-link.ts`, `cli/src/lib/session/active.ts`, `cli/docs/specifications.md`, `cli/docs/sessions.md`.
18
+
19
+ - **Teammates can no longer silently self-merge their own PRs (PHNX-3236).** A
20
+ write-capable `agents teams` teammate authenticates as the repo owner, so it
21
+ could merge its own PR straight past the required non-author-review gate (the
22
+ RUSH-2988 wave-1 dispatch self-merged PRs #1817/#1820 with zero reviews). Team
23
+ dispatch now appends a self-merge boundary (`TEAMMATE_PR_POLICY`, via one
24
+ `withTeammatePrPolicy` helper) to every write-capable (non-plan) teammate's
25
+ prompt, fresh or resumed, at **all three** dispatch surfaces — local, `--device`
26
+ remote (both via `buildRunArgv`), and `--cloud` (`cloudDispatchOptions`) — so
27
+ none can drift: open the PR and hand it off; do not merge your own PR without a
28
+ posted non-author verdict. The hard enforcement is `merge-guard.sh`, the PreToolUse
29
+ hook a teammate inherits, whose verdict check now excludes the PR author's own
30
+ reviews/comments (`phnx-labs/.agents` #395) — but it only reaches hook-capable
31
+ local/remote teammates; cloud teammates (provider sandbox, no inherited hook) and
32
+ hook-incapable harnesses (Warp) get the prompt as their only layer, a residual
33
+ documented in `cli/AGENTS.md` §6 that server-side branch protection closes. Source:
34
+ `cli/src/lib/teams/agents.ts`, `cli/src/commands/teams.ts`.
35
+
36
+ - **Blanket `Bash` now auto-approves shell on Grok, so fleet agents stop punting on `ssh`/`scp`/`agents ssh` (PHNX-3294).** A blanket `Bash` grant was translated to a Grok rule with `pattern:"*"`, but Grok's `*` is a SINGLE-level wildcard, so it never auto-approved a multi-token command like `ssh host cmd` or `scp a b` — a Grok agent on a box without `permission_mode=always-approve` prompted and handed the command back to the user. Blanket `Bash` (and its `Bash(*)`/`Bash(**)` forms, allow and deny) now emit Grok's documented "bare prefix matches all invocations" rule — a bash rule with no `pattern` key — the true allow-all-shell form, matching how Kimi, Droid, and Claude already express it. This reads back as `Bash(*)`, so the round-trip is unchanged. Source: `cli/src/lib/permissions.ts`.
37
+
38
+ - **`agents traces sync`: per-session roster in `index.json` (PHNX-3483).** The index shard now carries `sessions` — one flat scalar row per AGENT session (`id`, `title`, `harness`, `model`, `repo`, `mode`, `projectType`, `startedAt`, `durationMs`, `toolCount`, `errorCount`, `needsAttention`, best-effort `costUsd`) — so the Rush console can filter the session set and re-aggregate its headline metrics client-side instead of being stuck with the pre-rolled scalars. `durationMs` is the ACTIVE duration (`sessionActiveMs`, the value behind `stats.medianMs`; 0 when unmeasured) and `mode` encodes the AGENT-vs-INTERACTIVE segmentation (`headless`/`interactive`) so a mode-split median over the measured rows reproduces `stats.agentMedianMs`/`stats.interactiveMedianMs` (the segmented stats skip null-duration rows, which the roster still carries at `durationMs: 0`). Utility rows are excluded, so the roster length equals `stats.sessionsImported`. Source: `apps/cli/src/lib/traces/sync.ts`.
39
+
40
+ - **`agents insights perf run` now actually records `agent.run` timings (PHNX-3497).** The `startup` sub-phase surface added in PHNX-3468 had no data on a foreground run: `recordPerfTiming` wrote the perf-spool row behind a fire-and-forget `import().then(...)`, and the `agents run` process exited before that deferred write landed — so every foreground run's `agent.run` sample was silently lost. The spool append is now synchronous (`recordSample` is an `appendFileSync` that never opens SQLite), so `agent.run` rows and their `startup` p50/p90 phase line persist and show up. Source: `cli/src/lib/feed/events.ts`.
41
+
42
+ - **`gh pr checks` now escapes the shared GraphQL rate limit automatically — no new command to learn (PHNX-3501).** The whole fleet shares one GitHub token, and the merge loop's `gh pr checks/view/list` are all GraphQL-backed, so the fleet collectively drains GitHub's 5000-point/hr GraphQL budget while REST core sits ~95% idle — after which every agent's `gh pr checks --watch` dies with `GraphQL: API rate limit already exceeded`. A generated PATH shim (like the existing `browser` shim) now transparently intercepts the agent's trained `gh pr checks` and routes it to a hidden `agents __gh` verb backed by REST (`commits/{sha}/check-runs` + `/status`, anchored to the PR's live head SHA); every other `gh` verb execs the real binary byte-for-byte. `--watch` owns the REST poll loop (head-SHA-anchored, so a superseded run's red can never be reported — closes PHNX-3042); a one-shot runs real gh first and falls back to REST only on the exact rate-limit stderr. The shim self-heals — a leftover shim execs real gh whenever agents-cli is gone or uninstalled, so it can never break `gh`. Installed on `agents sync` (POSIX in v1). Source: `cli/src/lib/github/rest.ts`, `cli/src/lib/github/gh-overload.ts`, `cli/src/lib/installations/shims.ts`.
43
+
44
+ - **`agents doctor` now detects content drift for mcp, permissions, subagents, workflows, and memory (PHNX-3504).** The resource-diff engine compared these kinds by NAME only (or, for workflows/memory, not at all), so an edited MCP command/args/env, a swapped permission rule, an edited subagent prompt, a changed workflow, or an edited knowledge-memory fact all read as a false `ok` under an unchanged name. Every kind `syncResourcesToVersion` writes is now content-aware: mcp structurally compares the home server def against the resolved source; subagents re-render the source through the registry transform and byte-compare; workflows compare layout-aware per harness (copied tree or transformed file); memory (`~/.agents/memory/*.md` facts, tracked via the `.agents-cli-memory.json` manifest) byte-compares each fact; permissions compare per-rule in the harness's native vocabulary for the representable harnesses (claude/opencode/cursor/droid/openclaw) and stay presence-only with an honest `detail: 'format cannot verify content'` for the lossy TOML/flag harnesses rather than faking `ok`. `workflows` and `memory` are now real `--kind` filters; `promptcuts` (not version-scoped) is dropped. A completeness test binds `DOCTOR_ALL_KINDS` to the writer set so a future synced kind cannot silently become a blind spot, and `agents doctor --fix` reaches the newly-covered kinds. Source: `cli/src/lib/doctor-diff.ts`, `cli/src/lib/resource-content-diff.ts`, `cli/src/lib/subagents-registry.ts`, `cli/src/lib/mcp.ts`, `cli/src/lib/permissions.ts`, `cli/src/lib/workflows.ts`, `cli/src/lib/heal.ts`, `cli/src/lib/devices/doctor-findings.ts`.
45
+
46
+ - **A balanced/available route now records WHY it picked an account, so a bad pick is debuggable from `agents events` alone (PHNX-3564).** `rotation.resolved` / `rotation.unresolved` used to carry only `{ version, healthy: <n>, excluded: <n> }` — a device-local version number (meaningless across the fleet) and two counts. When a run landed on a weekly-maxed, logged-out, or absent account (PHNX-3505 / PHNX-3479), the event could not say which org, how fresh its usage was, or why it won the draw, so every occurrence was debugged blind. `buildRotationDecisionEvent` (`cli/src/lib/accounting/rotate.ts`) now emits a full decision record: `picked { usageKey (per-org quota key), email, tier }`; a `pickReason` read straight from the router's own result (`verified-weighted` for the healthy path, `unverified-<tier>-draw` for a blind/stale fallback, `refused-no-verified` for the fail-closed exit); a per-candidate map `{ usageKey, accountKey, providerAccount, tier (verified/stale/blind), source, capturedAt, ageMs, windows, eligible, excludedReason, credentialVerdict }` keyed by pool index (not version or usageKey, which a RUSH-3182 provider-account pool shares across rows); and a `freshness` tally `{ verified, stale, blind }` over the healthy pool. `usageKey` is the org uuid — the only identity that joins the same account across devices — and `ageMs` is measured by the routing host's clock, so a `last_seen` snapshot showing `tier: verified` beside a large `ageMs` is the cross-host clock-skew failure made visible. Two details are dictated by the event sink's generic sanitizer (`feed/events.ts`): candidates are a keyed object (not an array) because arrays are truncated to 10 entries, and the verdict field is named `credentialVerdict` (not `authVerdict`) because any key matching `/auth/i` is redacted — an end-to-end test drives a real `emit()` into a redirected sink and asserts the persisted line to lock both in. This event now also carries account **emails** to the local-only event log (`~/.agents/.history/events`, `0600`) — a new local PII surface for this event type. Pure observability: no routing-logic change, and every field is read from data the router already computed; the build+emit is wrapped so an observability bug can never abort a launch. Live-verified on zion (8 claude accounts: 1 verified, 7 stale — every synced row already past the 5-min decision window on arrival). Source: `cli/src/lib/accounting/rotate.ts`, `cli/src/lib/accounting/rotate.test.ts`.
47
+
48
+ - **`agents reminders` + Claude statusline reminders.** Personal operating reminders kept in `~/.agents/reminders/reminders.yaml` (each a `short`/`full` pair) now surface succinctly in the Claude statusline — one per session, chosen deterministically from the session id so concurrent agents each show a different one and it stays stable within a session. Presence of the file with at least one entry is the opt-in; a malformed file is swallowed by the statusline (a broken prompt is worse than a missing line) but surfaced by `agents reminders`. The file syncs across the fleet via `agents repo push/pull`. Source: `cli/src/lib/reminders.ts`, `cli/src/commands/reminders.ts`, `cli/src/lib/claude-statusline.ts`.
49
+
3
50
  ## 1.22.60
4
51
 
52
+ - **Richer schema-2 trace shards (PHNX-3442).** The per-session trace shard the evals console reads is now a typed `ToolExecution` discriminated union instead of a thin `SessionStep` with one `output` blob. Bash steps carry per-action argv, effective program, build/test/git/network categories, and a conservative `danger`/`DESTRUCTIVE` classification (recursive-force-delete, `git reset --hard`, `git clean -f`, force push incl. the `+refspec` form, `git checkout -- .`, `DROP TABLE`/`TRUNCATE`/`DELETE`-without-`WHERE`, `kill -9`, `dd of=`, `mkfs`, redirects over block devices); edit/write steps carry file mutations, diff hunks, and a cross-step revert ledger (only an exact hash round-trip stamps `revertedByStep`, never a partial edit); read/grep steps carry counts; permission/hook events are first-class steps. It emits by default with no flag: the prix/web console already decodes both schema 1 and schema 2 and is the only shard consumer, so the change is backward-compatible by construction — a fleet-wide producer must not depend on an operator setting an env var. The bash unwrap/tokenize/classify, trajectory step-pairing (`pairSteps`), and shard meta helpers are reused, not reimplemented. `category`/`risk`/`categoryMetrics` are deliberately omitted — the consumer backfills neutral defaults. Source: `cli/src/lib/traces/schema2.ts`, `schema2-build.ts`, `schema2-danger.ts`, `cli/src/lib/traces/sync.ts`, `cli/src/lib/session/trajectory.ts`.
53
+
5
54
  - **Detect stranded uncommitted work in `agents teams status` (PHNX-2951).** A teammate that exits `COMPLETED` with no PR and uncommitted changes in its local worktree now reports `STRANDED` instead of `done`, and the status line names the worktree path so the work can be rescued before cleanup. The stranded count is computed over the full team even when `--filter` narrows the displayed agents, so `agents teams status <team> --filter running` still surfaces stranded completed teammates. Team rollups and `teams list --status` classify these teams as `stranded`, not `done`. Source: `cli/src/lib/teams/delivery.ts`, `cli/src/lib/teams/api.ts`, `cli/src/commands/teams.ts`.
6
55
 
7
56
  - **Track agent-launch boot cost: persist + surface the `startup` phase (PHNX-3468).** `agent.run` already timed a `startup` phase (entering `spawnAgent` → child spawn) but only the total run duration reached the perf warehouse, so boot cost was untrackable across the fleet. The timer now persists its sub-phase marks into each `perf.timing` sample's `meta_json`, `aggregateSamples` folds them into a per-label `phases` break-out (p50/p90 over the samples that carried each phase), and `agents insights perf run` renders a `└ startup: p50 … p90 … (n=…)` sub-line under `agent.run`. This is the measurement the boot-perf work is graded against. Source: `cli/src/lib/feed/events.ts`, `cli/src/lib/perf/db.ts`, `cli/src/lib/perf/types.ts`, `cli/src/commands/perf.ts`.
@@ -91,6 +91,7 @@ export declare const loadAccounts: ModuleLoader;
91
91
  export declare const loadDaemon: ModuleLoader;
92
92
  export declare const loadAuth: ModuleLoader;
93
93
  export declare const loadTraces: ModuleLoader;
94
+ export declare const loadReminders: ModuleLoader;
94
95
  /**
95
96
  * Commands whose modules pull in the SQLite-backed session/cloud stack. They are
96
97
  * registered AFTER `applyGlobalHelpConventions` (mirroring main's order: help
@@ -94,6 +94,7 @@ export const loadAccounts = async () => (await import('../commands/accounts.js')
94
94
  export const loadDaemon = async () => (await import('../commands/daemon.js')).registerDaemonCommand;
95
95
  export const loadAuth = async () => (await import('../commands/auth.js')).registerAuthCommand;
96
96
  export const loadTraces = async () => (await import('../commands/traces.js')).registerTracesCommands;
97
+ export const loadReminders = async () => (await import('../commands/reminders.js')).registerRemindersCommand;
97
98
  /**
98
99
  * Commands whose modules pull in the SQLite-backed session/cloud stack. They are
99
100
  * registered AFTER `applyGlobalHelpConventions` (mirroring main's order: help
@@ -128,6 +129,7 @@ export const COMMAND_LOADERS = {
128
129
  view: [loadView],
129
130
  inspect: [loadInspect],
130
131
  feedback: [loadFeedback],
132
+ reminders: [loadReminders],
131
133
  commands: [loadCommands],
132
134
  hooks: [loadHooks],
133
135
  skills: [loadSkills],
@@ -602,7 +602,7 @@ function registerProfilesCommands(browser) {
602
602
  .option('--position <X,Y>', 'Window position on screen, e.g. 80,80')
603
603
  .option('--binary <path>', 'Absolute path to the browser/app binary (required with --browser custom)')
604
604
  .option('--electron', 'Treat this profile as an Electron desktop app: never call Target.createTarget; bind to the visible window using --target-filter or the skip-invisible heuristic')
605
- .option('--target-filter <expr>', 'Pick the visible CDP page target when the app exposes more than one. Format: url:<substring> or title:<substring>')
605
+ .option('--target-filter <expr>', 'Pick the existing CDP page target to drive (Electron apps, and Arc — which reuses an open tab / Space rather than creating one). Format: url:<substring> or title:<substring>')
606
606
  .action(async (name, opts) => {
607
607
  try {
608
608
  assertRegistrableProfileName(name);
@@ -629,8 +629,13 @@ function registerProfilesCommands(browser) {
629
629
  console.error('--target-filter must be url:<substring> or title:<substring> (non-empty value, no leading whitespace)');
630
630
  process.exit(1);
631
631
  }
632
- if (!opts.electron) {
633
- console.error('--target-filter requires --electron (the filter is only consulted on Electron profiles)');
632
+ // The filter picks WHICH existing page target to drive without ever
633
+ // calling Target.createTarget — the two profiles that do that are
634
+ // Electron apps and Arc (which crashes on tab creation, so it drives an
635
+ // existing tab / Space, PHNX-2399). Any other browser opens its own tab
636
+ // and never consults the filter, so requiring it there would be a lie.
637
+ if (!opts.electron && opts.browser !== 'arc') {
638
+ console.error('--target-filter requires --electron or --browser arc (the filter is only consulted on profiles that reuse an existing tab)');
634
639
  process.exit(1);
635
640
  }
636
641
  }
@@ -704,7 +709,7 @@ function registerProfilesCommands(browser) {
704
709
  .option('--binary <path>', 'Absolute path to the browser/app binary')
705
710
  .option('--electron', 'Treat this profile as an Electron desktop app')
706
711
  .option('--no-electron', 'Stop treating it as an Electron app')
707
- .option('--target-filter <expr>', "url:<substring> or title:<substring>; requires --electron (pass '' to clear)")
712
+ .option('--target-filter <expr>', "url:<substring> or title:<substring>; consulted on Electron and Arc profiles (pass '' to clear)")
708
713
  .option('--json', 'Output machine-readable JSON')
709
714
  .action(async (name, opts) => {
710
715
  // The browser type and the name are identity, not settings: both key the
@@ -1458,7 +1458,7 @@ export function registerDoctorCommand(program) {
1458
1458
  .option('--json', 'Output machine-readable JSON')
1459
1459
  .option('--diff', 'In target mode, include unified diffs for divergent files')
1460
1460
  .option('--fix', 'Heal gaps: install missing resources, repair invalid plugin manifests, refresh stale plugins, reconcile drift, and purge stale/legacy agents-cli installs (npx-cache, pre-1.22.30, unsafe helper installer) when a fixed peer exists')
1461
- .option('--kind <kinds>', 'Restrict to comma-separated resource kinds (commands,skills,hooks,rules,mcp,permissions,subagents,plugins,promptcuts)')
1461
+ .option('--kind <kinds>', 'Restrict to comma-separated resource kinds (commands,skills,hooks,rules,mcp,permissions,subagents,plugins,workflows,memory)')
1462
1462
  .option('--cwd <path>', 'Resolution cwd for project layer detection (default: process.cwd())')
1463
1463
  .option('--adopt <agent>', "Take over the agent's native launcher that shadows the shim (symlink it to the version-managed shim; reversible with --release)")
1464
1464
  .option('--release <agent>', 'Undo --adopt: restore the native launcher agents-cli previously adopted')
@@ -1773,7 +1773,7 @@ agents run auto --device yosemite-s0 "fix the flaky test" # pin the device
1773
1773
  }
1774
1774
  process.exit(resumeExit);
1775
1775
  }
1776
- const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, implicitModeFor, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, readProfile, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError, signInRecoverableCandidates }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports, capableAgents }, { shareRuntimeEnv },] = await Promise.all([
1776
+ const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, implicitModeFor, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, readProfile, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError, formatNoVerifiedUsageError, signInRecoverableCandidates }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports, capableAgents }, { shareRuntimeEnv },] = await Promise.all([
1777
1777
  import('../lib/exec.js'),
1778
1778
  import('../lib/agents.js'),
1779
1779
  import('../lib/profiles.js'),
@@ -2439,6 +2439,40 @@ agents run auto --device yosemite-s0 "fix the flaky test" # pin the device
2439
2439
  process.exit(1);
2440
2440
  }
2441
2441
  }
2442
+ else if (resolved.noVerifiedUsage) {
2443
+ // Entirely stale usage (PHNX-2526): every eligible account carries
2444
+ // a stale-but-present usage number and none is verified, so the
2445
+ // route would be a guess. NEVER auto-pick it. Interactive: show the
2446
+ // account picker so a human chooses with the (stale) numbers in
2447
+ // view. Unattended: fail loud with NO_VERIFIED_USAGE. The stale
2448
+ // pool survives ONLY as `resolved.rotation.healthy` for bounded
2449
+ // post-rejection failover, never as the initial pick.
2450
+ const { noVerifiedUsageDecision, pickRunAccountCandidate } = await import('./run-account-picker.js');
2451
+ const decision = noVerifiedUsageDecision({
2452
+ tty: isInteractiveTerminal(),
2453
+ json: options.json === true,
2454
+ headless: options.headless === true,
2455
+ });
2456
+ if (decision === 'picker') {
2457
+ const selected = await pickRunAccountCandidate(agent);
2458
+ // A cancelled picker launches nothing — same contract as the
2459
+ // trailing-@ account picker and the sign-in launch above.
2460
+ if (!selected)
2461
+ return;
2462
+ version = selected.version;
2463
+ // Keep the rotation so mid-run failover can still cascade across
2464
+ // the other (stale) healthy accounts after a real rejection.
2465
+ rotationResult = resolved.rotation;
2466
+ if (!options.quiet) {
2467
+ const identity = selected.accountLabel || 'signed-in account';
2468
+ process.stderr.write(chalk.gray(`[agents] no fresh usage for any ${agent} account — you picked ${identity} · ${agent}@${selected.version}\n`));
2469
+ }
2470
+ }
2471
+ else {
2472
+ console.error(chalk.red(formatNoVerifiedUsageError(agent, strategy, resolved.rotation?.healthy ?? [])));
2473
+ process.exit(1);
2474
+ }
2475
+ }
2442
2476
  else if (resolved.version) {
2443
2477
  version = resolved.version;
2444
2478
  rotationResult = resolved.rotation;
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Production {@link WizardHooks} for the harness create/edit wizard.
3
+ *
4
+ * The engine ({@link ./harness-wizard.js}) is pure and hook-driven; this module
5
+ * supplies the three real extension points the sibling subtasks fill:
6
+ * - `pickModel` — a catalog pick from the host's own model list
7
+ * (`getModelCatalog`), with a free-text escape hatch and a
8
+ * free-text FALLBACK when the host exposes no catalog
9
+ * (RUSH-2220).
10
+ * - `connectionTest` — a real pre-save smoke test through `agents run`
11
+ * (RUSH-2221), delegated to {@link runHarnessConnectionTest}.
12
+ * - `editable` — the resolver-sourced per-host editability matrix
13
+ * (RUSH-2222), {@link defaultEditable}.
14
+ *
15
+ * Kept out of the engine so the engine stays testable with a scripted IO and no
16
+ * catalog probe, keychain read, or subprocess.
17
+ */
18
+ import type { AgentId } from '../lib/types.js';
19
+ import { type ModelInfo } from '../lib/models.js';
20
+ import { type WizardHooks, type WizardIO, type WizardChoice } from './harness-wizard.js';
21
+ /**
22
+ * The installed version whose model catalog to read for a host. The wizard has
23
+ * no version in hand for a fresh create, so it reads the host's default (or its
24
+ * sole installed) version — the same version a bare `agents run <host>` uses.
25
+ * Null when the host has no installed version to probe (→ free-text model).
26
+ */
27
+ export declare function catalogVersionFor(host: AgentId): string | null;
28
+ /**
29
+ * Build the model `select` choices from a catalog. Pure, so the labelling +
30
+ * escape-hatch rows are unit-tested with no catalog probe. Every list ends with
31
+ * a "type a custom id" row so a model the catalog doesn't list is always
32
+ * reachable; in edit mode a "keep current" row leads.
33
+ */
34
+ export declare function buildModelChoices(models: ModelInfo[], current?: string): WizardChoice<string>[];
35
+ /**
36
+ * Prompt over an already-resolved catalog: a `select` of the models plus the
37
+ * keep-current / custom-id rows, mapping the sentinel choices back to a concrete
38
+ * model id. Split from the catalog probe so the KEEP / CUSTOM / pick branches are
39
+ * unit-tested with a scripted IO and no installed agent.
40
+ */
41
+ export declare function chooseModelFromCatalog(io: WizardIO, models: ModelInfo[], current: string | undefined): Promise<string>;
42
+ /**
43
+ * The catalog-backed model pick (RUSH-2220). Returns the chosen model id, or
44
+ * `null` to fall through to the engine's free-text prompt when the host exposes
45
+ * no probeable catalog — so a host we can't enumerate degrades to today's
46
+ * free-text behaviour rather than blocking.
47
+ */
48
+ export declare function pickModel(io: WizardIO, host: AgentId | undefined, version: string | undefined, current: string | undefined): Promise<string | null>;
49
+ /**
50
+ * Assemble the production hook set the harness commands drive the wizard with.
51
+ * The connection-test hook reads the assembled draft's `name` — the caller writes
52
+ * the profile to disk first, then invokes the test through the real `agents run`
53
+ * path against that name.
54
+ */
55
+ export declare function harnessHooks(): WizardHooks;
@@ -0,0 +1,104 @@
1
+ /**
2
+ * Production {@link WizardHooks} for the harness create/edit wizard.
3
+ *
4
+ * The engine ({@link ./harness-wizard.js}) is pure and hook-driven; this module
5
+ * supplies the three real extension points the sibling subtasks fill:
6
+ * - `pickModel` — a catalog pick from the host's own model list
7
+ * (`getModelCatalog`), with a free-text escape hatch and a
8
+ * free-text FALLBACK when the host exposes no catalog
9
+ * (RUSH-2220).
10
+ * - `connectionTest` — a real pre-save smoke test through `agents run`
11
+ * (RUSH-2221), delegated to {@link runHarnessConnectionTest}.
12
+ * - `editable` — the resolver-sourced per-host editability matrix
13
+ * (RUSH-2222), {@link defaultEditable}.
14
+ *
15
+ * Kept out of the engine so the engine stays testable with a scripted IO and no
16
+ * catalog probe, keychain read, or subprocess.
17
+ */
18
+ import chalk from 'chalk';
19
+ import { getModelCatalog } from '../lib/models.js';
20
+ import { getGlobalDefault, listInstalledVersions } from '../lib/installations/versions.js';
21
+ import { runHarnessConnectionTest } from '../lib/harness-connection-test.js';
22
+ import { defaultEditable, } from './harness-wizard.js';
23
+ /** Sentinel select values for the two non-catalog rows in the model pick. */
24
+ const CUSTOM_MODEL = '__custom_model__';
25
+ const KEEP_MODEL = '__keep_model__';
26
+ /**
27
+ * The installed version whose model catalog to read for a host. The wizard has
28
+ * no version in hand for a fresh create, so it reads the host's default (or its
29
+ * sole installed) version — the same version a bare `agents run <host>` uses.
30
+ * Null when the host has no installed version to probe (→ free-text model).
31
+ */
32
+ export function catalogVersionFor(host) {
33
+ return getGlobalDefault(host) || listInstalledVersions(host)[0] || null;
34
+ }
35
+ /**
36
+ * Build the model `select` choices from a catalog. Pure, so the labelling +
37
+ * escape-hatch rows are unit-tested with no catalog probe. Every list ends with
38
+ * a "type a custom id" row so a model the catalog doesn't list is always
39
+ * reachable; in edit mode a "keep current" row leads.
40
+ */
41
+ export function buildModelChoices(models, current) {
42
+ const choices = [];
43
+ if (current)
44
+ choices.push({ name: `Keep current (${current})`, value: KEEP_MODEL });
45
+ for (const m of models) {
46
+ const tags = [];
47
+ if (m.alias)
48
+ tags.push(m.alias);
49
+ if (m.isDefault)
50
+ tags.push('default');
51
+ const tail = tags.length ? chalk.gray(` (${tags.join(', ')})`) : '';
52
+ const label = m.displayName && m.displayName !== m.id ? `${m.id}${chalk.gray(' ' + m.displayName)}` : m.id;
53
+ choices.push({ name: `${label}${tail}`, value: m.id });
54
+ }
55
+ choices.push({ name: 'Type a custom model id…', value: CUSTOM_MODEL });
56
+ return choices;
57
+ }
58
+ /**
59
+ * Prompt over an already-resolved catalog: a `select` of the models plus the
60
+ * keep-current / custom-id rows, mapping the sentinel choices back to a concrete
61
+ * model id. Split from the catalog probe so the KEEP / CUSTOM / pick branches are
62
+ * unit-tested with a scripted IO and no installed agent.
63
+ */
64
+ export async function chooseModelFromCatalog(io, models, current) {
65
+ const choice = await io.select({
66
+ message: 'Model',
67
+ choices: buildModelChoices(models, current),
68
+ });
69
+ if (choice === KEEP_MODEL)
70
+ return current ?? '';
71
+ if (choice === CUSTOM_MODEL)
72
+ return io.input({ message: 'Model id', default: current });
73
+ return choice;
74
+ }
75
+ /**
76
+ * The catalog-backed model pick (RUSH-2220). Returns the chosen model id, or
77
+ * `null` to fall through to the engine's free-text prompt when the host exposes
78
+ * no probeable catalog — so a host we can't enumerate degrades to today's
79
+ * free-text behaviour rather than blocking.
80
+ */
81
+ export async function pickModel(io, host, version, current) {
82
+ if (!host)
83
+ return null;
84
+ const resolvedVersion = version || catalogVersionFor(host);
85
+ if (!resolvedVersion)
86
+ return null;
87
+ const catalog = getModelCatalog(host, resolvedVersion);
88
+ if (!catalog || catalog.models.length === 0)
89
+ return null;
90
+ return chooseModelFromCatalog(io, catalog.models, current);
91
+ }
92
+ /**
93
+ * Assemble the production hook set the harness commands drive the wizard with.
94
+ * The connection-test hook reads the assembled draft's `name` — the caller writes
95
+ * the profile to disk first, then invokes the test through the real `agents run`
96
+ * path against that name.
97
+ */
98
+ export function harnessHooks() {
99
+ return {
100
+ pickModel,
101
+ connectionTest: (draft) => runHarnessConnectionTest(draft.name),
102
+ editable: defaultEditable,
103
+ };
104
+ }
@@ -28,6 +28,8 @@
28
28
  */
29
29
  import type { AgentId } from '../lib/types.js';
30
30
  import { type Profile } from '../lib/profiles.js';
31
+ import type { ConnectionTestResult } from '../lib/harness-connection-test.js';
32
+ export type { ConnectionTestResult } from '../lib/harness-connection-test.js';
31
33
  /** Whether the wizard is creating a new harness or editing an existing one. */
32
34
  export type WizardMode = 'create' | 'edit';
33
35
  /** A single `select` choice. `disabled` greys the row (used by the edit matrix). */
@@ -129,23 +131,40 @@ export interface HarnessEditable {
129
131
  version: boolean;
130
132
  fallback: boolean;
131
133
  }
134
+ /** One field's editability plus, when disabled, the one-line reason to surface. */
135
+ export interface EditableField {
136
+ enabled: boolean;
137
+ /** Set only when `enabled` is false — the greyed field's stated reason. */
138
+ reason?: string;
139
+ }
140
+ /** Per-host editability with a reason attached to every disabled field. */
141
+ export interface HarnessEditability {
142
+ model: EditableField;
143
+ baseUrl: EditableField;
144
+ auth: EditableField;
145
+ version: EditableField;
146
+ fallback: EditableField;
147
+ }
148
+ /**
149
+ * The per-harness editability matrix (RUSH-2222). Which of a harness's params
150
+ * this host's API format actually lets you change, each disabled field carrying
151
+ * the reason the wizard greys it with.
152
+ *
153
+ * Sourced ENTIRELY from the same maps the run-time resolver reads —
154
+ * `baseUrlEnvKeyForHost` (endpoint slot), `authEnvKeyForHost` (auth env), and
155
+ * `isSelfUpdatingAgent` (pinnable version) — never a table hardcoded alongside
156
+ * them, so the wizard's enable/disable can never drift from what a run actually
157
+ * honors (repo rule: the capability table stays truthful, in lockstep with the
158
+ * code). A disabled param is never a silent no-op — the wizard shows its reason
159
+ * and the flag path fails loud (`forkProfile`'s base-URL throw is the precedent).
160
+ */
161
+ export declare function harnessEditable(host: AgentId): HarnessEditability;
132
162
  /**
133
- * Resolver-sourced editability, the scaffold default behind the RUSH-2222 seam.
134
- * Every field is read from the same maps the run-time resolver uses
135
- * (`baseUrlEnvKeyForHost` / `authEnvKeyForHost` / `isSelfUpdatingAgent`), never a
136
- * table hardcoded alongside them — so the wizard's enable/disable can never drift
137
- * from what a run actually honors (repo rule: the capability table stays truthful,
138
- * in lockstep with the code). RUSH-2222 may replace this via {@link WizardHooks.editable}
139
- * to add per-param reasons; it must stay sourced from the resolver.
163
+ * The boolean projection of {@link harnessEditable} — the scaffold default behind
164
+ * the RUSH-2222 {@link WizardHooks.editable} seam. Derived from the reason-carrying
165
+ * matrix so the two can never disagree.
140
166
  */
141
167
  export declare function defaultEditable(host: AgentId): HarnessEditable;
142
- /** Outcome of a connection test (RUSH-2221 fills the real classifier). */
143
- export interface ConnectionTestResult {
144
- ok: boolean;
145
- /** Machine-readable class when it failed (auth / endpoint / model / unknown). */
146
- reason?: 'auth' | 'endpoint' | 'model' | 'unknown';
147
- message?: string;
148
- }
149
168
  /**
150
169
  * Extension points the sibling subtasks fill without touching the engine. Each is
151
170
  * a real no-op-by-default seam: absent, the scaffold uses today's behavior; none
@@ -33,21 +33,49 @@ import { listBundles } from '../lib/secrets/bundles.js';
33
33
  import { AGENTS, ALL_AGENT_IDS, isSelfUpdatingAgent, resolveAgentName } from '../lib/agents.js';
34
34
  import { readAccountRegistry } from '../lib/account-registry.js';
35
35
  /**
36
- * Resolver-sourced editability, the scaffold default behind the RUSH-2222 seam.
37
- * Every field is read from the same maps the run-time resolver uses
38
- * (`baseUrlEnvKeyForHost` / `authEnvKeyForHost` / `isSelfUpdatingAgent`), never a
39
- * table hardcoded alongside them — so the wizard's enable/disable can never drift
40
- * from what a run actually honors (repo rule: the capability table stays truthful,
41
- * in lockstep with the code). RUSH-2222 may replace this via {@link WizardHooks.editable}
42
- * to add per-param reasons; it must stay sourced from the resolver.
36
+ * The per-harness editability matrix (RUSH-2222). Which of a harness's params
37
+ * this host's API format actually lets you change, each disabled field carrying
38
+ * the reason the wizard greys it with.
39
+ *
40
+ * Sourced ENTIRELY from the same maps the run-time resolver reads —
41
+ * `baseUrlEnvKeyForHost` (endpoint slot), `authEnvKeyForHost` (auth env), and
42
+ * `isSelfUpdatingAgent` (pinnable version) — never a table hardcoded alongside
43
+ * them, so the wizard's enable/disable can never drift from what a run actually
44
+ * honors (repo rule: the capability table stays truthful, in lockstep with the
45
+ * code). A disabled param is never a silent no-op — the wizard shows its reason
46
+ * and the flag path fails loud (`forkProfile`'s base-URL throw is the precedent).
47
+ */
48
+ export function harnessEditable(host) {
49
+ const hasEndpoint = baseUrlEnvKeyForHost(host) !== null;
50
+ const hasAuth = authEnvKeyForHost(host) !== null;
51
+ const selfUpdating = isSelfUpdatingAgent(host);
52
+ return {
53
+ model: { enabled: true },
54
+ baseUrl: hasEndpoint
55
+ ? { enabled: true }
56
+ : { enabled: false, reason: `host '${host}' has no custom-endpoint slot — base URL not applicable` },
57
+ auth: hasAuth
58
+ ? { enabled: true }
59
+ : { enabled: false, reason: `host '${host}' manages its own login — no auth to edit` },
60
+ version: selfUpdating
61
+ ? { enabled: false, reason: `host '${host}' self-updates — its version can't be pinned` }
62
+ : { enabled: true },
63
+ fallback: { enabled: true },
64
+ };
65
+ }
66
+ /**
67
+ * The boolean projection of {@link harnessEditable} — the scaffold default behind
68
+ * the RUSH-2222 {@link WizardHooks.editable} seam. Derived from the reason-carrying
69
+ * matrix so the two can never disagree.
43
70
  */
44
71
  export function defaultEditable(host) {
72
+ const e = harnessEditable(host);
45
73
  return {
46
- model: true,
47
- baseUrl: baseUrlEnvKeyForHost(host) !== null,
48
- auth: authEnvKeyForHost(host) !== null,
49
- version: !isSelfUpdatingAgent(host),
50
- fallback: true,
74
+ model: e.model.enabled,
75
+ baseUrl: e.baseUrl.enabled,
76
+ auth: e.auth.enabled,
77
+ version: e.version.enabled,
78
+ fallback: e.fallback.enabled,
51
79
  };
52
80
  }
53
81
  /** Resolve the host CLI a fork `source` runs under (the host `buildFork` will use). */
@@ -219,8 +247,10 @@ export function createSteps() {
219
247
  // rest replaces the old silent-drop (`profileFromHostModel` discards a
220
248
  // base URL the host can't honor) with an explicit reason.
221
249
  const host = d.host ?? hostForSource(d.source);
222
- if (host && baseUrlEnvKeyForHost(host) === null) {
223
- return { disabled: `host '${host}' has no custom-endpoint slot — base URL not applicable` };
250
+ if (host) {
251
+ const cap = harnessEditable(host).baseUrl;
252
+ if (!cap.enabled)
253
+ return { disabled: cap.reason };
224
254
  }
225
255
  return 'run';
226
256
  },
@@ -285,14 +315,14 @@ function currentBaseUrl(p) {
285
315
  export function editSteps(original) {
286
316
  const host = original.host.agent;
287
317
  const editableFor = (hooks) => (hooks.editable ?? defaultEditable)(host);
288
- // decide() has no access to hooks, so gate on the resolver default; a hook that
289
- // narrows editability further is applied inside run(). The scaffold's default is
290
- // the resolver truth, which is what the matrix subtask builds on.
291
- const cap = defaultEditable(host);
318
+ // decide() has no access to hooks, so gate on the resolver-sourced matrix; a
319
+ // hook that narrows editability further is applied inside run(). The matrix is
320
+ // the resolver truth (RUSH-2222), reasons and all.
321
+ const cap = harnessEditable(host);
292
322
  return [
293
323
  {
294
324
  id: 'model',
295
- decide: () => (cap.model ? 'run' : { disabled: `host '${host}' does not support a pinned model` }),
325
+ decide: () => (cap.model.enabled ? 'run' : { disabled: cap.model.reason }),
296
326
  async run(io, d, hooks) {
297
327
  if (!editableFor(hooks).model)
298
328
  return;
@@ -301,7 +331,7 @@ export function editSteps(original) {
301
331
  },
302
332
  {
303
333
  id: 'baseUrl',
304
- decide: () => (cap.baseUrl ? 'run' : { disabled: `host '${host}' has no custom-endpoint slot` }),
334
+ decide: () => (cap.baseUrl.enabled ? 'run' : { disabled: cap.baseUrl.reason }),
305
335
  async run(io, d, hooks) {
306
336
  if (!editableFor(hooks).baseUrl)
307
337
  return;
@@ -311,7 +341,7 @@ export function editSteps(original) {
311
341
  },
312
342
  {
313
343
  id: 'account',
314
- decide: () => (cap.auth ? 'run' : { disabled: `host '${host}' manages its own login — no auth to edit` }),
344
+ decide: () => (cap.auth.enabled ? 'run' : { disabled: cap.auth.reason }),
315
345
  async run(io, d, hooks) {
316
346
  if (!editableFor(hooks).auth)
317
347
  return;
@@ -327,7 +357,7 @@ export function editSteps(original) {
327
357
  },
328
358
  {
329
359
  id: 'version',
330
- decide: () => cap.version ? 'run' : { disabled: `host '${host}' self-updates — its version can't be pinned` },
360
+ decide: () => (cap.version.enabled ? 'run' : { disabled: cap.version.reason }),
331
361
  async run(io, d, hooks) {
332
362
  if (!editableFor(hooks).version)
333
363
  return;
@@ -339,7 +369,7 @@ export function editSteps(original) {
339
369
  },
340
370
  {
341
371
  id: 'fallback',
342
- decide: () => (cap.fallback ? 'run' : 'skip'),
372
+ decide: () => (cap.fallback.enabled ? 'run' : 'skip'),
343
373
  async run(io, d) {
344
374
  d.fallbackModel = await io.input({
345
375
  message: 'Fallback model (same-host rate-limit retry; blank for none)',
@@ -32,6 +32,8 @@ export interface ForkOptions {
32
32
  fromSecrets?: string;
33
33
  keyStdin?: boolean;
34
34
  force?: boolean;
35
+ /** Tri-state pre-save connection test: true = force, false = skip, undefined = ask on a TTY. */
36
+ test?: boolean;
35
37
  }
36
38
  /** Options accepted by `agents harness edit`. */
37
39
  export interface EditOptions {
@@ -47,6 +49,8 @@ export interface EditOptions {
47
49
  /** `<bundle>` or `<bundle>:<key>` — see {@link applyFromSecrets} in ./profiles.js. */
48
50
  fromSecrets?: string;
49
51
  keyStdin?: boolean;
52
+ /** Tri-state pre-save connection test: true = force, false = skip, undefined = ask on a TTY. */
53
+ test?: boolean;
50
54
  }
51
55
  /**
52
56
  * Build the new harness for `agents harness fork <source> <name>`.
@@ -79,6 +83,16 @@ export declare function forkNeedsWizard(source: string | undefined, name: string
79
83
  * `agents harness add <preset-name>` (no flags) still resolves via the preset
80
84
  * fallback instead of being routed into the wizard. */
81
85
  export declare function addNeedsWizard(name: string | undefined, opts: AddProfileOptions): boolean;
86
+ /** Whether the pre-save connection test runs, or must be asked for on a TTY. */
87
+ export type ConnectionTestGate = 'on' | 'off' | 'ask';
88
+ /**
89
+ * Resolve the tri-state connection-test gate (RUSH-2221), pure so the branching
90
+ * is unit-tested with no prompt or spawn. `--test` forces it on, `--no-test`
91
+ * forces it off, and with neither flag a TTY is asked (default yes) while a
92
+ * non-interactive caller (`--key-stdin`, piped, CI) skips it — so scripting stays
93
+ * non-interactive unless it opts in with `--test`.
94
+ */
95
+ export declare function connectionTestGate(testFlag: boolean | undefined, interactive: boolean): ConnectionTestGate;
82
96
  /**
83
97
  * Map a finished edit-wizard draft onto {@link EditOptions}, keeping only the
84
98
  * fields the user actually changed from the profile's current values. Unchanged