@phnx-labs/agents-cli 1.22.59 → 1.22.61
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +71 -0
- package/dist/cli/command-registry.d.ts +1 -0
- package/dist/cli/command-registry.js +2 -0
- package/dist/commands/browser.js +9 -4
- package/dist/commands/doctor.js +1 -1
- package/dist/commands/exec.js +35 -1
- package/dist/commands/harness-hooks.d.ts +55 -0
- package/dist/commands/harness-hooks.js +104 -0
- package/dist/commands/harness-wizard.d.ts +33 -14
- package/dist/commands/harness-wizard.js +53 -23
- package/dist/commands/harness.d.ts +14 -0
- package/dist/commands/harness.js +86 -5
- package/dist/commands/perf.js +10 -0
- package/dist/commands/reminders.d.ts +9 -0
- package/dist/commands/reminders.js +49 -0
- package/dist/commands/run-account-picker.d.ts +14 -0
- package/dist/commands/run-account-picker.js +13 -0
- package/dist/commands/sessions-picker.d.ts +13 -0
- package/dist/commands/sessions-picker.js +17 -8
- package/dist/commands/sessions.js +13 -11
- package/dist/commands/teams-picker.js +20 -6
- package/dist/commands/teams.d.ts +3 -3
- package/dist/commands/teams.js +86 -24
- package/dist/index.js +9 -0
- package/dist/lib/accounting/rotate.d.ts +63 -0
- package/dist/lib/accounting/rotate.js +240 -16
- package/dist/lib/accounting/usage-sync.d.ts +12 -2
- package/dist/lib/accounting/usage-sync.js +34 -6
- package/dist/lib/browser/drivers/local.d.ts +11 -0
- package/dist/lib/browser/drivers/local.js +26 -0
- package/dist/lib/browser/profiles.js +8 -6
- package/dist/lib/browser/service.d.ts +12 -8
- package/dist/lib/browser/service.js +38 -10
- package/dist/lib/claude-statusline.d.ts +14 -1
- package/dist/lib/claude-statusline.js +27 -2
- package/dist/lib/daemon/runner.js +17 -2
- package/dist/lib/devices/doctor-findings.d.ts +1 -1
- package/dist/lib/devices/doctor-findings.js +22 -4
- package/dist/lib/doctor-diff.d.ts +21 -5
- package/dist/lib/doctor-diff.js +242 -76
- package/dist/lib/feed/events.d.ts +1 -1
- package/dist/lib/feed/events.js +28 -15
- package/dist/lib/github/gh-overload.d.ts +58 -0
- package/dist/lib/github/gh-overload.js +246 -0
- package/dist/lib/github/rest.d.ts +64 -0
- package/dist/lib/github/rest.js +111 -0
- package/dist/lib/harness-connection-test.d.ts +57 -0
- package/dist/lib/harness-connection-test.js +80 -0
- package/dist/lib/heal.js +8 -3
- package/dist/lib/installations/shims.d.ts +22 -0
- package/dist/lib/installations/shims.js +104 -0
- package/dist/lib/linear-project-counts.js +8 -0
- package/dist/lib/linear-rate-limit.d.ts +26 -0
- package/dist/lib/linear-rate-limit.js +163 -0
- package/dist/lib/mcp.d.ts +9 -0
- package/dist/lib/mcp.js +37 -1
- package/dist/lib/open-url.js +5 -3
- package/dist/lib/perf/db.d.ts +1 -1
- package/dist/lib/perf/db.js +53 -2
- package/dist/lib/perf/types.d.ts +14 -0
- package/dist/lib/permissions.d.ts +28 -0
- package/dist/lib/permissions.js +156 -1
- package/dist/lib/refresh.js +9 -1
- package/dist/lib/reminders.d.ts +29 -0
- package/dist/lib/reminders.js +88 -0
- package/dist/lib/resource-content-diff.d.ts +33 -0
- package/dist/lib/resource-content-diff.js +103 -0
- package/dist/lib/rules/compile.d.ts +7 -0
- package/dist/lib/rules/compile.js +7 -1
- package/dist/lib/session/active.d.ts +41 -4
- package/dist/lib/session/active.js +58 -7
- package/dist/lib/session/host-link.d.ts +22 -0
- package/dist/lib/session/host-link.js +40 -4
- package/dist/lib/session/live-metadata.js +3 -3
- package/dist/lib/session/trajectory.d.ts +42 -0
- package/dist/lib/session/trajectory.js +46 -27
- package/dist/lib/ssh-exec.d.ts +30 -0
- package/dist/lib/ssh-exec.js +37 -5
- package/dist/lib/startup/command-registry.js +1 -1
- package/dist/lib/subagents-registry.d.ts +18 -0
- package/dist/lib/subagents-registry.js +79 -0
- package/dist/lib/teams/agents.d.ts +12 -0
- package/dist/lib/teams/agents.js +51 -0
- package/dist/lib/teams/api.d.ts +8 -0
- package/dist/lib/teams/api.js +50 -6
- package/dist/lib/teams/delivery.d.ts +14 -4
- package/dist/lib/teams/delivery.js +15 -5
- package/dist/lib/traces/schema2-build.d.ts +85 -0
- package/dist/lib/traces/schema2-build.js +637 -0
- package/dist/lib/traces/schema2-danger.d.ts +36 -0
- package/dist/lib/traces/schema2-danger.js +185 -0
- package/dist/lib/traces/schema2.d.ts +149 -0
- package/dist/lib/traces/schema2.js +20 -0
- package/dist/lib/traces/sync.d.ts +93 -0
- package/dist/lib/traces/sync.js +75 -22
- package/dist/lib/traces/worker-template.js +5 -0
- package/dist/lib/uninstall.js +10 -1
- package/dist/lib/workflows.d.ts +11 -0
- package/dist/lib/workflows.js +67 -8
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,76 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.22.61
|
|
4
|
+
|
|
5
|
+
- **`agents harness` wizard: model catalog, connection test, and edit matrix (PHNX-2218/2220/2221/2222).** The create/edit wizard now picks the model from the host's own catalog (`getModelCatalog`) with a free-text escape hatch, gates the endpoint step to hosts that actually carry one, and runs a real pre-save connection test — `agents run <name> "say alive in one word" --headless --timeout 60s` classified into pass / auth / endpoint / model — behind a confirm with `--test`/`--no-test`, offering keep / edit / delete on failure rather than saving a broken harness silently. A resolver-sourced `harnessEditable` matrix disables (with a reason) any param the host's API format can't carry. Source: `apps/cli/src/commands/harness.ts`, `apps/cli/src/commands/harness-wizard.ts`, `apps/cli/src/commands/harness-hooks.ts`, `apps/cli/src/lib/harness-connection-test.ts`.
|
|
6
|
+
|
|
7
|
+
- **Cap per-machine Linear API spend with a shared per-key request budget (PHNX-2310).** With ~13 concurrent drain agents on one Linear API key on a single box, nothing stopped them collectively exhausting Linear's 2,500 requests/hour limit and throttling ticket-status reads for every agent on that box. The `agents projects` card fetch now reserves against a proactive request budget keyed by the API key before each request; when the hourly pool is spent it serves the last cached snapshot (marked stale) instead of forcing a 429. The pool is lock-free on-disk state under the machine-local cache (one stamp file per request, a sliding one-hour window), so the many separate agent processes on a box coordinate without a shared mutable document to race on. Scope is per machine: the state is not fleet-synced, so two boxes sharing one key each budget independently — cross-device aggregation is a known limitation, not yet covered. This complements the existing read cache and reactive 429 backoff. Source: `cli/src/lib/linear-rate-limit.ts`, `cli/src/lib/linear-project-counts.ts`.
|
|
8
|
+
|
|
9
|
+
- **`agents browser` drives your RUNNING Arc instead of spawning a second one (PHNX-2399).** Arc is single-instance: relaunching the Arc binary with a fresh `--user-data-dir` produced a stray window and no CDP endpoint rather than a debuggable instance. An Arc profile now attaches to the Arc you already have open (point its endpoint at the port you launched Arc's remote debugging on), and when that Arc exposes no CDP endpoint on the profile's port it **fails loud** with the one relaunch that fixes it (`open -a Arc --args --remote-debugging-port=<port>`) — it never silently launches a duplicate. Bind the profile to the tab/Space you want to drive with `--target-filter` (accepted for `--browser arc`, not just `--electron`): it is consulted when Arc picks which existing tab to drive, so `navigate` reuses your bound Space instead of an unrelated tab, and refuses rather than borrowing one when nothing matches. The documented first-use flow — `agents browser navigate --profile arc --url …` on a fresh profile — now attaches to an open tab instead of erroring. Opening a brand-new tab remains the one CDP-only op Arc can't survive (`Target.createTarget` crashes it), so tab-creating verbs fail clearly and steer you to reuse an existing tab or a Chromium-family browser. Source: `cli/src/lib/browser/drivers/local.ts`, `cli/src/lib/browser/service.ts`, `cli/src/commands/browser.ts`, `cli/src/lib/open-url.ts`.
|
|
10
|
+
|
|
11
|
+
- **`agents sessions watch --json` rows now carry a canonical `phase` (PHNX-2484).** Each live/recoverable session row projects a coarse lifecycle bucket — `running | waiting | failed | done | idle` — derived once at the source from the finalized status (`derivePhase` in `cli/src/lib/session/active.ts`), so thin-client consumers (the AGI EXT Fleet panel) read `phase` off the row instead of re-deriving it from the status word. It also fixes a drift a status-only re-derivation had: `orphaned` and `crashed` now bucket to `failed` (dangling/dead — needs attention) rather than falling through to `idle` and hiding a dead agent.
|
|
12
|
+
|
|
13
|
+
- **`balanced`/`available` routing never auto-picks an account from entirely stale usage (PHNX-2526).** When every signed-in account's usage snapshot is older than the 5-minute decision horizon and none can be verified, the initial route no longer guesses on a stale number (the yosemite-s1 incident, where a 26h–2.7d-old snapshot read 48% while the account was at its weekly cap). An interactive run now shows the account picker so a human chooses with the stale numbers in view; an unattended run (`--headless`/`--json`/no TTY, or a routine) fails loud with `NO_VERIFIED_USAGE`. The stale candidates are preserved only for bounded post-rejection failover, never the first pick. A pool that is merely *blind* (no snapshot at all — a worker box whose usage endpoint 403s) is unaffected and still draws a pick. Source: `apps/cli/src/lib/accounting/rotate.ts`, `apps/cli/src/commands/exec.ts`, `apps/cli/src/lib/daemon/runner.ts`.
|
|
14
|
+
|
|
15
|
+
- **Raise SSH `ControlPersist` to 10 minutes so repeated fleet touches reuse warm connections (PHNX-2582).** The OpenSSH multiplex master now lives for 10 minutes after its last use (`SSH_CONTROL_PERSIST_SECONDS`) instead of 60 seconds. The master only helps a repeated touch of the same host that arrives while it is still alive, and the ad-hoc `--device` / fan-out calls that touch a box repeatedly (`sessions --active`, `fleet ping`, `doctor`, `teams`, a `--device` command run a few times) arrive in bursts spread over minutes — so at the old 60 s window any two more than a minute apart both paid a fresh TCP+auth handshake (~100-300 ms direct, up to ~500 ms relayed). 10 minutes keeps a multi-minute burst warm (~3-5 ms per reuse) while staying bounded so an idle master to a slept host is reaped rather than lingering. Source: `cli/src/lib/ssh-exec.ts`.
|
|
16
|
+
|
|
17
|
+
- **A running agent whose owning IDE window died is now surfaced as `orphaned` (PHNX-3183).** `foldHostLink` promoted only `idle`/`input_required` sessions to `orphaned`; a still-`running` agent was never flagged, so a genuinely-stranded remote agent — alive in tmux after its owning window crashed, rebooted, or dropped its SSH link — stayed hidden as a healthy "running" row. It now promotes a `running` session to `orphaned` when, and only when, its owning window was LOST: the window's registry slice went stale past `HOST_HEARTBEAT_STALE_MS` after having republished it (`hostWindowLost`). Client ABSENCE alone still never flags a running agent — a tmux attached-client count of zero is the normal steady state for a detached `agents run --device` pane (RUSH-3125), so promoting on it would relabel every unattended remote agent as orphaned (the over-report reverted in `6d973b823`). This narrows spec `SES-18a` to the window-loss case rather than the blanket precondition-drop that was reverted. Source: `cli/src/lib/session/host-link.ts`, `cli/src/lib/session/active.ts`, `cli/docs/specifications.md`, `cli/docs/sessions.md`.
|
|
18
|
+
|
|
19
|
+
- **Teammates can no longer silently self-merge their own PRs (PHNX-3236).** A
|
|
20
|
+
write-capable `agents teams` teammate authenticates as the repo owner, so it
|
|
21
|
+
could merge its own PR straight past the required non-author-review gate (the
|
|
22
|
+
RUSH-2988 wave-1 dispatch self-merged PRs #1817/#1820 with zero reviews). Team
|
|
23
|
+
dispatch now appends a self-merge boundary (`TEAMMATE_PR_POLICY`, via one
|
|
24
|
+
`withTeammatePrPolicy` helper) to every write-capable (non-plan) teammate's
|
|
25
|
+
prompt, fresh or resumed, at **all three** dispatch surfaces — local, `--device`
|
|
26
|
+
remote (both via `buildRunArgv`), and `--cloud` (`cloudDispatchOptions`) — so
|
|
27
|
+
none can drift: open the PR and hand it off; do not merge your own PR without a
|
|
28
|
+
posted non-author verdict. The hard enforcement is `merge-guard.sh`, the PreToolUse
|
|
29
|
+
hook a teammate inherits, whose verdict check now excludes the PR author's own
|
|
30
|
+
reviews/comments (`phnx-labs/.agents` #395) — but it only reaches hook-capable
|
|
31
|
+
local/remote teammates; cloud teammates (provider sandbox, no inherited hook) and
|
|
32
|
+
hook-incapable harnesses (Warp) get the prompt as their only layer, a residual
|
|
33
|
+
documented in `cli/AGENTS.md` §6 that server-side branch protection closes. Source:
|
|
34
|
+
`cli/src/lib/teams/agents.ts`, `cli/src/commands/teams.ts`.
|
|
35
|
+
|
|
36
|
+
- **Blanket `Bash` now auto-approves shell on Grok, so fleet agents stop punting on `ssh`/`scp`/`agents ssh` (PHNX-3294).** A blanket `Bash` grant was translated to a Grok rule with `pattern:"*"`, but Grok's `*` is a SINGLE-level wildcard, so it never auto-approved a multi-token command like `ssh host cmd` or `scp a b` — a Grok agent on a box without `permission_mode=always-approve` prompted and handed the command back to the user. Blanket `Bash` (and its `Bash(*)`/`Bash(**)` forms, allow and deny) now emit Grok's documented "bare prefix matches all invocations" rule — a bash rule with no `pattern` key — the true allow-all-shell form, matching how Kimi, Droid, and Claude already express it. This reads back as `Bash(*)`, so the round-trip is unchanged. Source: `cli/src/lib/permissions.ts`.
|
|
37
|
+
|
|
38
|
+
- **`agents traces sync`: per-session roster in `index.json` (PHNX-3483).** The index shard now carries `sessions` — one flat scalar row per AGENT session (`id`, `title`, `harness`, `model`, `repo`, `mode`, `projectType`, `startedAt`, `durationMs`, `toolCount`, `errorCount`, `needsAttention`, best-effort `costUsd`) — so the Rush console can filter the session set and re-aggregate its headline metrics client-side instead of being stuck with the pre-rolled scalars. `durationMs` is the ACTIVE duration (`sessionActiveMs`, the value behind `stats.medianMs`; 0 when unmeasured) and `mode` encodes the AGENT-vs-INTERACTIVE segmentation (`headless`/`interactive`) so a mode-split median over the measured rows reproduces `stats.agentMedianMs`/`stats.interactiveMedianMs` (the segmented stats skip null-duration rows, which the roster still carries at `durationMs: 0`). Utility rows are excluded, so the roster length equals `stats.sessionsImported`. Source: `apps/cli/src/lib/traces/sync.ts`.
|
|
39
|
+
|
|
40
|
+
- **`agents insights perf run` now actually records `agent.run` timings (PHNX-3497).** The `startup` sub-phase surface added in PHNX-3468 had no data on a foreground run: `recordPerfTiming` wrote the perf-spool row behind a fire-and-forget `import().then(...)`, and the `agents run` process exited before that deferred write landed — so every foreground run's `agent.run` sample was silently lost. The spool append is now synchronous (`recordSample` is an `appendFileSync` that never opens SQLite), so `agent.run` rows and their `startup` p50/p90 phase line persist and show up. Source: `cli/src/lib/feed/events.ts`.
|
|
41
|
+
|
|
42
|
+
- **`gh pr checks` now escapes the shared GraphQL rate limit automatically — no new command to learn (PHNX-3501).** The whole fleet shares one GitHub token, and the merge loop's `gh pr checks/view/list` are all GraphQL-backed, so the fleet collectively drains GitHub's 5000-point/hr GraphQL budget while REST core sits ~95% idle — after which every agent's `gh pr checks --watch` dies with `GraphQL: API rate limit already exceeded`. A generated PATH shim (like the existing `browser` shim) now transparently intercepts the agent's trained `gh pr checks` and routes it to a hidden `agents __gh` verb backed by REST (`commits/{sha}/check-runs` + `/status`, anchored to the PR's live head SHA); every other `gh` verb execs the real binary byte-for-byte. `--watch` owns the REST poll loop (head-SHA-anchored, so a superseded run's red can never be reported — closes PHNX-3042); a one-shot runs real gh first and falls back to REST only on the exact rate-limit stderr. The shim self-heals — a leftover shim execs real gh whenever agents-cli is gone or uninstalled, so it can never break `gh`. Installed on `agents sync` (POSIX in v1). Source: `cli/src/lib/github/rest.ts`, `cli/src/lib/github/gh-overload.ts`, `cli/src/lib/installations/shims.ts`.
|
|
43
|
+
|
|
44
|
+
- **`agents doctor` now detects content drift for mcp, permissions, subagents, workflows, and memory (PHNX-3504).** The resource-diff engine compared these kinds by NAME only (or, for workflows/memory, not at all), so an edited MCP command/args/env, a swapped permission rule, an edited subagent prompt, a changed workflow, or an edited knowledge-memory fact all read as a false `ok` under an unchanged name. Every kind `syncResourcesToVersion` writes is now content-aware: mcp structurally compares the home server def against the resolved source; subagents re-render the source through the registry transform and byte-compare; workflows compare layout-aware per harness (copied tree or transformed file); memory (`~/.agents/memory/*.md` facts, tracked via the `.agents-cli-memory.json` manifest) byte-compares each fact; permissions compare per-rule in the harness's native vocabulary for the representable harnesses (claude/opencode/cursor/droid/openclaw) and stay presence-only with an honest `detail: 'format cannot verify content'` for the lossy TOML/flag harnesses rather than faking `ok`. `workflows` and `memory` are now real `--kind` filters; `promptcuts` (not version-scoped) is dropped. A completeness test binds `DOCTOR_ALL_KINDS` to the writer set so a future synced kind cannot silently become a blind spot, and `agents doctor --fix` reaches the newly-covered kinds. Source: `cli/src/lib/doctor-diff.ts`, `cli/src/lib/resource-content-diff.ts`, `cli/src/lib/subagents-registry.ts`, `cli/src/lib/mcp.ts`, `cli/src/lib/permissions.ts`, `cli/src/lib/workflows.ts`, `cli/src/lib/heal.ts`, `cli/src/lib/devices/doctor-findings.ts`.
|
|
45
|
+
|
|
46
|
+
- **A balanced/available route now records WHY it picked an account, so a bad pick is debuggable from `agents events` alone (PHNX-3564).** `rotation.resolved` / `rotation.unresolved` used to carry only `{ version, healthy: <n>, excluded: <n> }` — a device-local version number (meaningless across the fleet) and two counts. When a run landed on a weekly-maxed, logged-out, or absent account (PHNX-3505 / PHNX-3479), the event could not say which org, how fresh its usage was, or why it won the draw, so every occurrence was debugged blind. `buildRotationDecisionEvent` (`cli/src/lib/accounting/rotate.ts`) now emits a full decision record: `picked { usageKey (per-org quota key), email, tier }`; a `pickReason` read straight from the router's own result (`verified-weighted` for the healthy path, `unverified-<tier>-draw` for a blind/stale fallback, `refused-no-verified` for the fail-closed exit); a per-candidate map `{ usageKey, accountKey, providerAccount, tier (verified/stale/blind), source, capturedAt, ageMs, windows, eligible, excludedReason, credentialVerdict }` keyed by pool index (not version or usageKey, which a RUSH-3182 provider-account pool shares across rows); and a `freshness` tally `{ verified, stale, blind }` over the healthy pool. `usageKey` is the org uuid — the only identity that joins the same account across devices — and `ageMs` is measured by the routing host's clock, so a `last_seen` snapshot showing `tier: verified` beside a large `ageMs` is the cross-host clock-skew failure made visible. Two details are dictated by the event sink's generic sanitizer (`feed/events.ts`): candidates are a keyed object (not an array) because arrays are truncated to 10 entries, and the verdict field is named `credentialVerdict` (not `authVerdict`) because any key matching `/auth/i` is redacted — an end-to-end test drives a real `emit()` into a redirected sink and asserts the persisted line to lock both in. This event now also carries account **emails** to the local-only event log (`~/.agents/.history/events`, `0600`) — a new local PII surface for this event type. Pure observability: no routing-logic change, and every field is read from data the router already computed; the build+emit is wrapped so an observability bug can never abort a launch. Live-verified on zion (8 claude accounts: 1 verified, 7 stale — every synced row already past the 5-min decision window on arrival). Source: `cli/src/lib/accounting/rotate.ts`, `cli/src/lib/accounting/rotate.test.ts`.
|
|
47
|
+
|
|
48
|
+
- **`agents reminders` + Claude statusline reminders.** Personal operating reminders kept in `~/.agents/reminders/reminders.yaml` (each a `short`/`full` pair) now surface succinctly in the Claude statusline — one per session, chosen deterministically from the session id so concurrent agents each show a different one and it stays stable within a session. Presence of the file with at least one entry is the opt-in; a malformed file is swallowed by the statusline (a broken prompt is worse than a missing line) but surfaced by `agents reminders`. The file syncs across the fleet via `agents repo push/pull`. Source: `cli/src/lib/reminders.ts`, `cli/src/commands/reminders.ts`, `cli/src/lib/claude-statusline.ts`.
|
|
49
|
+
|
|
50
|
+
## 1.22.60
|
|
51
|
+
|
|
52
|
+
- **Richer schema-2 trace shards (PHNX-3442).** The per-session trace shard the evals console reads is now a typed `ToolExecution` discriminated union instead of a thin `SessionStep` with one `output` blob. Bash steps carry per-action argv, effective program, build/test/git/network categories, and a conservative `danger`/`DESTRUCTIVE` classification (recursive-force-delete, `git reset --hard`, `git clean -f`, force push incl. the `+refspec` form, `git checkout -- .`, `DROP TABLE`/`TRUNCATE`/`DELETE`-without-`WHERE`, `kill -9`, `dd of=`, `mkfs`, redirects over block devices); edit/write steps carry file mutations, diff hunks, and a cross-step revert ledger (only an exact hash round-trip stamps `revertedByStep`, never a partial edit); read/grep steps carry counts; permission/hook events are first-class steps. It emits by default with no flag: the prix/web console already decodes both schema 1 and schema 2 and is the only shard consumer, so the change is backward-compatible by construction — a fleet-wide producer must not depend on an operator setting an env var. The bash unwrap/tokenize/classify, trajectory step-pairing (`pairSteps`), and shard meta helpers are reused, not reimplemented. `category`/`risk`/`categoryMetrics` are deliberately omitted — the consumer backfills neutral defaults. Source: `cli/src/lib/traces/schema2.ts`, `schema2-build.ts`, `schema2-danger.ts`, `cli/src/lib/traces/sync.ts`, `cli/src/lib/session/trajectory.ts`.
|
|
53
|
+
|
|
54
|
+
- **Detect stranded uncommitted work in `agents teams status` (PHNX-2951).** A teammate that exits `COMPLETED` with no PR and uncommitted changes in its local worktree now reports `STRANDED` instead of `done`, and the status line names the worktree path so the work can be rescued before cleanup. The stranded count is computed over the full team even when `--filter` narrows the displayed agents, so `agents teams status <team> --filter running` still surfaces stranded completed teammates. Team rollups and `teams list --status` classify these teams as `stranded`, not `done`. Source: `cli/src/lib/teams/delivery.ts`, `cli/src/lib/teams/api.ts`, `cli/src/commands/teams.ts`.
|
|
55
|
+
|
|
56
|
+
- **Track agent-launch boot cost: persist + surface the `startup` phase (PHNX-3468).** `agent.run` already timed a `startup` phase (entering `spawnAgent` → child spawn) but only the total run duration reached the perf warehouse, so boot cost was untrackable across the fleet. The timer now persists its sub-phase marks into each `perf.timing` sample's `meta_json`, `aggregateSamples` folds them into a per-label `phases` break-out (p50/p90 over the samples that carried each phase), and `agents insights perf run` renders a `└ startup: p50 … p90 … (n=…)` sub-line under `agent.run`. This is the measurement the boot-perf work is graded against. Source: `cli/src/lib/feed/events.ts`, `cli/src/lib/perf/db.ts`, `cli/src/lib/perf/types.ts`, `cli/src/commands/perf.ts`.
|
|
57
|
+
|
|
58
|
+
- **`agents sessions preview <id>` now pulls the transcript digest from the
|
|
59
|
+
owning device for host-dispatched sessions (PHNX-3481).** A host dispatch
|
|
60
|
+
leaves a synthetic local index row with `machine = <peer>` and an empty
|
|
61
|
+
`filePath`, but that row does not carry the live fan-out's `_remote` marker.
|
|
62
|
+
Full-UUID preview resolved the local row and then treated it as a local live
|
|
63
|
+
session, rendering "full transcript not indexed here" without making any SSH
|
|
64
|
+
hop. Transcript reads now follow one file-aware predicate: an explicit remote
|
|
65
|
+
row still reads on its peer; a row naming another machine also reads there
|
|
66
|
+
when no transcript exists on this disk; and a synced mirror with a real local
|
|
67
|
+
file still renders locally. The direct preview, `sessions <id>`, picker view,
|
|
68
|
+
and picker/browser digest pane all share that rule. Live-registry metadata now
|
|
69
|
+
also stamps `_remote` when its execution machine differs from this box, and an
|
|
70
|
+
unreachable owner remains a loud error instead of a local placeholder.
|
|
71
|
+
Source: `cli/src/commands/sessions-picker.ts`, `cli/src/commands/sessions.ts`,
|
|
72
|
+
`cli/src/lib/session/live-metadata.ts`.
|
|
73
|
+
|
|
3
74
|
## 1.22.59
|
|
4
75
|
|
|
5
76
|
- **`agents devices disable/prefer` now actually change `--device auto` placement (PHNX-2092).**
|
|
@@ -91,6 +91,7 @@ export declare const loadAccounts: ModuleLoader;
|
|
|
91
91
|
export declare const loadDaemon: ModuleLoader;
|
|
92
92
|
export declare const loadAuth: ModuleLoader;
|
|
93
93
|
export declare const loadTraces: ModuleLoader;
|
|
94
|
+
export declare const loadReminders: ModuleLoader;
|
|
94
95
|
/**
|
|
95
96
|
* Commands whose modules pull in the SQLite-backed session/cloud stack. They are
|
|
96
97
|
* registered AFTER `applyGlobalHelpConventions` (mirroring main's order: help
|
|
@@ -94,6 +94,7 @@ export const loadAccounts = async () => (await import('../commands/accounts.js')
|
|
|
94
94
|
export const loadDaemon = async () => (await import('../commands/daemon.js')).registerDaemonCommand;
|
|
95
95
|
export const loadAuth = async () => (await import('../commands/auth.js')).registerAuthCommand;
|
|
96
96
|
export const loadTraces = async () => (await import('../commands/traces.js')).registerTracesCommands;
|
|
97
|
+
export const loadReminders = async () => (await import('../commands/reminders.js')).registerRemindersCommand;
|
|
97
98
|
/**
|
|
98
99
|
* Commands whose modules pull in the SQLite-backed session/cloud stack. They are
|
|
99
100
|
* registered AFTER `applyGlobalHelpConventions` (mirroring main's order: help
|
|
@@ -128,6 +129,7 @@ export const COMMAND_LOADERS = {
|
|
|
128
129
|
view: [loadView],
|
|
129
130
|
inspect: [loadInspect],
|
|
130
131
|
feedback: [loadFeedback],
|
|
132
|
+
reminders: [loadReminders],
|
|
131
133
|
commands: [loadCommands],
|
|
132
134
|
hooks: [loadHooks],
|
|
133
135
|
skills: [loadSkills],
|
package/dist/commands/browser.js
CHANGED
|
@@ -602,7 +602,7 @@ function registerProfilesCommands(browser) {
|
|
|
602
602
|
.option('--position <X,Y>', 'Window position on screen, e.g. 80,80')
|
|
603
603
|
.option('--binary <path>', 'Absolute path to the browser/app binary (required with --browser custom)')
|
|
604
604
|
.option('--electron', 'Treat this profile as an Electron desktop app: never call Target.createTarget; bind to the visible window using --target-filter or the skip-invisible heuristic')
|
|
605
|
-
.option('--target-filter <expr>', 'Pick the
|
|
605
|
+
.option('--target-filter <expr>', 'Pick the existing CDP page target to drive (Electron apps, and Arc — which reuses an open tab / Space rather than creating one). Format: url:<substring> or title:<substring>')
|
|
606
606
|
.action(async (name, opts) => {
|
|
607
607
|
try {
|
|
608
608
|
assertRegistrableProfileName(name);
|
|
@@ -629,8 +629,13 @@ function registerProfilesCommands(browser) {
|
|
|
629
629
|
console.error('--target-filter must be url:<substring> or title:<substring> (non-empty value, no leading whitespace)');
|
|
630
630
|
process.exit(1);
|
|
631
631
|
}
|
|
632
|
-
|
|
633
|
-
|
|
632
|
+
// The filter picks WHICH existing page target to drive without ever
|
|
633
|
+
// calling Target.createTarget — the two profiles that do that are
|
|
634
|
+
// Electron apps and Arc (which crashes on tab creation, so it drives an
|
|
635
|
+
// existing tab / Space, PHNX-2399). Any other browser opens its own tab
|
|
636
|
+
// and never consults the filter, so requiring it there would be a lie.
|
|
637
|
+
if (!opts.electron && opts.browser !== 'arc') {
|
|
638
|
+
console.error('--target-filter requires --electron or --browser arc (the filter is only consulted on profiles that reuse an existing tab)');
|
|
634
639
|
process.exit(1);
|
|
635
640
|
}
|
|
636
641
|
}
|
|
@@ -704,7 +709,7 @@ function registerProfilesCommands(browser) {
|
|
|
704
709
|
.option('--binary <path>', 'Absolute path to the browser/app binary')
|
|
705
710
|
.option('--electron', 'Treat this profile as an Electron desktop app')
|
|
706
711
|
.option('--no-electron', 'Stop treating it as an Electron app')
|
|
707
|
-
.option('--target-filter <expr>', "url:<substring> or title:<substring>;
|
|
712
|
+
.option('--target-filter <expr>', "url:<substring> or title:<substring>; consulted on Electron and Arc profiles (pass '' to clear)")
|
|
708
713
|
.option('--json', 'Output machine-readable JSON')
|
|
709
714
|
.action(async (name, opts) => {
|
|
710
715
|
// The browser type and the name are identity, not settings: both key the
|
package/dist/commands/doctor.js
CHANGED
|
@@ -1458,7 +1458,7 @@ export function registerDoctorCommand(program) {
|
|
|
1458
1458
|
.option('--json', 'Output machine-readable JSON')
|
|
1459
1459
|
.option('--diff', 'In target mode, include unified diffs for divergent files')
|
|
1460
1460
|
.option('--fix', 'Heal gaps: install missing resources, repair invalid plugin manifests, refresh stale plugins, reconcile drift, and purge stale/legacy agents-cli installs (npx-cache, pre-1.22.30, unsafe helper installer) when a fixed peer exists')
|
|
1461
|
-
.option('--kind <kinds>', 'Restrict to comma-separated resource kinds (commands,skills,hooks,rules,mcp,permissions,subagents,plugins,
|
|
1461
|
+
.option('--kind <kinds>', 'Restrict to comma-separated resource kinds (commands,skills,hooks,rules,mcp,permissions,subagents,plugins,workflows,memory)')
|
|
1462
1462
|
.option('--cwd <path>', 'Resolution cwd for project layer detection (default: process.cwd())')
|
|
1463
1463
|
.option('--adopt <agent>', "Take over the agent's native launcher that shadows the shim (symlink it to the version-managed shim; reversible with --release)")
|
|
1464
1464
|
.option('--release <agent>', 'Undo --adopt: restore the native launcher agents-cli previously adopted')
|
package/dist/commands/exec.js
CHANGED
|
@@ -1773,7 +1773,7 @@ agents run auto --device yosemite-s0 "fix the flaky test" # pin the device
|
|
|
1773
1773
|
}
|
|
1774
1774
|
process.exit(resumeExit);
|
|
1775
1775
|
}
|
|
1776
|
-
const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, implicitModeFor, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, readProfile, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError, signInRecoverableCandidates }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports, capableAgents }, { shareRuntimeEnv },] = await Promise.all([
|
|
1776
|
+
const [{ buildExecCommand, parseExecEnv, execAgent, runWithFallback, normalizeMode, resolveMode, implicitModeFor, headlessPlanStallCommand, nativeResume, resolveInteractive, inferredInteractiveWithoutTty }, { ALL_AGENT_IDS, ACCOUNT_INSPECTION_AGENT_IDS, agentLabel, supportsAccountInspection }, { profileExists, readProfile, resolveProfileForRun }, { readAndResolveBundleEnv, describeBundle, assertRemoteBundleFlagsUnsupported }, { splitBundleRef, resolveHostSshTarget, remoteResolveEnv }, { getConfiguredRunStrategy, normalizeRunStrategy, resolveRunVersion, rotationFailoverChain, shouldArmRotationFailover, RUN_STRATEGIES, collectHarnessCandidates, pickHarnessWeighted, classifyHarnessCandidates, formatHarnessPickBanner, formatNoHealthyHarnessError, formatNoHealthyAccountError, formatNoVerifiedUsageError, signInRecoverableCandidates }, { getGlobalDefault, getVersionHomePath, resolveVersion, resolveVersionAlias, ensureAgentRunnable }, { buildDiscoveredPlugin, loadPluginManifest, syncPluginToVersion }, { parseWorkflowFrontmatter, resolveWorkflowRef, resolveAllowedSubagents, pruneStaleWorkflowSubagents, ensureSubagentDispatchTool }, { resolveRunDefaults }, { getMcpServersByName, buildWorkflowMcpConfig }, { supports, capableAgents }, { shareRuntimeEnv },] = await Promise.all([
|
|
1777
1777
|
import('../lib/exec.js'),
|
|
1778
1778
|
import('../lib/agents.js'),
|
|
1779
1779
|
import('../lib/profiles.js'),
|
|
@@ -2439,6 +2439,40 @@ agents run auto --device yosemite-s0 "fix the flaky test" # pin the device
|
|
|
2439
2439
|
process.exit(1);
|
|
2440
2440
|
}
|
|
2441
2441
|
}
|
|
2442
|
+
else if (resolved.noVerifiedUsage) {
|
|
2443
|
+
// Entirely stale usage (PHNX-2526): every eligible account carries
|
|
2444
|
+
// a stale-but-present usage number and none is verified, so the
|
|
2445
|
+
// route would be a guess. NEVER auto-pick it. Interactive: show the
|
|
2446
|
+
// account picker so a human chooses with the (stale) numbers in
|
|
2447
|
+
// view. Unattended: fail loud with NO_VERIFIED_USAGE. The stale
|
|
2448
|
+
// pool survives ONLY as `resolved.rotation.healthy` for bounded
|
|
2449
|
+
// post-rejection failover, never as the initial pick.
|
|
2450
|
+
const { noVerifiedUsageDecision, pickRunAccountCandidate } = await import('./run-account-picker.js');
|
|
2451
|
+
const decision = noVerifiedUsageDecision({
|
|
2452
|
+
tty: isInteractiveTerminal(),
|
|
2453
|
+
json: options.json === true,
|
|
2454
|
+
headless: options.headless === true,
|
|
2455
|
+
});
|
|
2456
|
+
if (decision === 'picker') {
|
|
2457
|
+
const selected = await pickRunAccountCandidate(agent);
|
|
2458
|
+
// A cancelled picker launches nothing — same contract as the
|
|
2459
|
+
// trailing-@ account picker and the sign-in launch above.
|
|
2460
|
+
if (!selected)
|
|
2461
|
+
return;
|
|
2462
|
+
version = selected.version;
|
|
2463
|
+
// Keep the rotation so mid-run failover can still cascade across
|
|
2464
|
+
// the other (stale) healthy accounts after a real rejection.
|
|
2465
|
+
rotationResult = resolved.rotation;
|
|
2466
|
+
if (!options.quiet) {
|
|
2467
|
+
const identity = selected.accountLabel || 'signed-in account';
|
|
2468
|
+
process.stderr.write(chalk.gray(`[agents] no fresh usage for any ${agent} account — you picked ${identity} · ${agent}@${selected.version}\n`));
|
|
2469
|
+
}
|
|
2470
|
+
}
|
|
2471
|
+
else {
|
|
2472
|
+
console.error(chalk.red(formatNoVerifiedUsageError(agent, strategy, resolved.rotation?.healthy ?? [])));
|
|
2473
|
+
process.exit(1);
|
|
2474
|
+
}
|
|
2475
|
+
}
|
|
2442
2476
|
else if (resolved.version) {
|
|
2443
2477
|
version = resolved.version;
|
|
2444
2478
|
rotationResult = resolved.rotation;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Production {@link WizardHooks} for the harness create/edit wizard.
|
|
3
|
+
*
|
|
4
|
+
* The engine ({@link ./harness-wizard.js}) is pure and hook-driven; this module
|
|
5
|
+
* supplies the three real extension points the sibling subtasks fill:
|
|
6
|
+
* - `pickModel` — a catalog pick from the host's own model list
|
|
7
|
+
* (`getModelCatalog`), with a free-text escape hatch and a
|
|
8
|
+
* free-text FALLBACK when the host exposes no catalog
|
|
9
|
+
* (RUSH-2220).
|
|
10
|
+
* - `connectionTest` — a real pre-save smoke test through `agents run`
|
|
11
|
+
* (RUSH-2221), delegated to {@link runHarnessConnectionTest}.
|
|
12
|
+
* - `editable` — the resolver-sourced per-host editability matrix
|
|
13
|
+
* (RUSH-2222), {@link defaultEditable}.
|
|
14
|
+
*
|
|
15
|
+
* Kept out of the engine so the engine stays testable with a scripted IO and no
|
|
16
|
+
* catalog probe, keychain read, or subprocess.
|
|
17
|
+
*/
|
|
18
|
+
import type { AgentId } from '../lib/types.js';
|
|
19
|
+
import { type ModelInfo } from '../lib/models.js';
|
|
20
|
+
import { type WizardHooks, type WizardIO, type WizardChoice } from './harness-wizard.js';
|
|
21
|
+
/**
|
|
22
|
+
* The installed version whose model catalog to read for a host. The wizard has
|
|
23
|
+
* no version in hand for a fresh create, so it reads the host's default (or its
|
|
24
|
+
* sole installed) version — the same version a bare `agents run <host>` uses.
|
|
25
|
+
* Null when the host has no installed version to probe (→ free-text model).
|
|
26
|
+
*/
|
|
27
|
+
export declare function catalogVersionFor(host: AgentId): string | null;
|
|
28
|
+
/**
|
|
29
|
+
* Build the model `select` choices from a catalog. Pure, so the labelling +
|
|
30
|
+
* escape-hatch rows are unit-tested with no catalog probe. Every list ends with
|
|
31
|
+
* a "type a custom id" row so a model the catalog doesn't list is always
|
|
32
|
+
* reachable; in edit mode a "keep current" row leads.
|
|
33
|
+
*/
|
|
34
|
+
export declare function buildModelChoices(models: ModelInfo[], current?: string): WizardChoice<string>[];
|
|
35
|
+
/**
|
|
36
|
+
* Prompt over an already-resolved catalog: a `select` of the models plus the
|
|
37
|
+
* keep-current / custom-id rows, mapping the sentinel choices back to a concrete
|
|
38
|
+
* model id. Split from the catalog probe so the KEEP / CUSTOM / pick branches are
|
|
39
|
+
* unit-tested with a scripted IO and no installed agent.
|
|
40
|
+
*/
|
|
41
|
+
export declare function chooseModelFromCatalog(io: WizardIO, models: ModelInfo[], current: string | undefined): Promise<string>;
|
|
42
|
+
/**
|
|
43
|
+
* The catalog-backed model pick (RUSH-2220). Returns the chosen model id, or
|
|
44
|
+
* `null` to fall through to the engine's free-text prompt when the host exposes
|
|
45
|
+
* no probeable catalog — so a host we can't enumerate degrades to today's
|
|
46
|
+
* free-text behaviour rather than blocking.
|
|
47
|
+
*/
|
|
48
|
+
export declare function pickModel(io: WizardIO, host: AgentId | undefined, version: string | undefined, current: string | undefined): Promise<string | null>;
|
|
49
|
+
/**
|
|
50
|
+
* Assemble the production hook set the harness commands drive the wizard with.
|
|
51
|
+
* The connection-test hook reads the assembled draft's `name` — the caller writes
|
|
52
|
+
* the profile to disk first, then invokes the test through the real `agents run`
|
|
53
|
+
* path against that name.
|
|
54
|
+
*/
|
|
55
|
+
export declare function harnessHooks(): WizardHooks;
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Production {@link WizardHooks} for the harness create/edit wizard.
|
|
3
|
+
*
|
|
4
|
+
* The engine ({@link ./harness-wizard.js}) is pure and hook-driven; this module
|
|
5
|
+
* supplies the three real extension points the sibling subtasks fill:
|
|
6
|
+
* - `pickModel` — a catalog pick from the host's own model list
|
|
7
|
+
* (`getModelCatalog`), with a free-text escape hatch and a
|
|
8
|
+
* free-text FALLBACK when the host exposes no catalog
|
|
9
|
+
* (RUSH-2220).
|
|
10
|
+
* - `connectionTest` — a real pre-save smoke test through `agents run`
|
|
11
|
+
* (RUSH-2221), delegated to {@link runHarnessConnectionTest}.
|
|
12
|
+
* - `editable` — the resolver-sourced per-host editability matrix
|
|
13
|
+
* (RUSH-2222), {@link defaultEditable}.
|
|
14
|
+
*
|
|
15
|
+
* Kept out of the engine so the engine stays testable with a scripted IO and no
|
|
16
|
+
* catalog probe, keychain read, or subprocess.
|
|
17
|
+
*/
|
|
18
|
+
import chalk from 'chalk';
|
|
19
|
+
import { getModelCatalog } from '../lib/models.js';
|
|
20
|
+
import { getGlobalDefault, listInstalledVersions } from '../lib/installations/versions.js';
|
|
21
|
+
import { runHarnessConnectionTest } from '../lib/harness-connection-test.js';
|
|
22
|
+
import { defaultEditable, } from './harness-wizard.js';
|
|
23
|
+
/** Sentinel select values for the two non-catalog rows in the model pick. */
|
|
24
|
+
const CUSTOM_MODEL = '__custom_model__';
|
|
25
|
+
const KEEP_MODEL = '__keep_model__';
|
|
26
|
+
/**
|
|
27
|
+
* The installed version whose model catalog to read for a host. The wizard has
|
|
28
|
+
* no version in hand for a fresh create, so it reads the host's default (or its
|
|
29
|
+
* sole installed) version — the same version a bare `agents run <host>` uses.
|
|
30
|
+
* Null when the host has no installed version to probe (→ free-text model).
|
|
31
|
+
*/
|
|
32
|
+
export function catalogVersionFor(host) {
|
|
33
|
+
return getGlobalDefault(host) || listInstalledVersions(host)[0] || null;
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* Build the model `select` choices from a catalog. Pure, so the labelling +
|
|
37
|
+
* escape-hatch rows are unit-tested with no catalog probe. Every list ends with
|
|
38
|
+
* a "type a custom id" row so a model the catalog doesn't list is always
|
|
39
|
+
* reachable; in edit mode a "keep current" row leads.
|
|
40
|
+
*/
|
|
41
|
+
export function buildModelChoices(models, current) {
|
|
42
|
+
const choices = [];
|
|
43
|
+
if (current)
|
|
44
|
+
choices.push({ name: `Keep current (${current})`, value: KEEP_MODEL });
|
|
45
|
+
for (const m of models) {
|
|
46
|
+
const tags = [];
|
|
47
|
+
if (m.alias)
|
|
48
|
+
tags.push(m.alias);
|
|
49
|
+
if (m.isDefault)
|
|
50
|
+
tags.push('default');
|
|
51
|
+
const tail = tags.length ? chalk.gray(` (${tags.join(', ')})`) : '';
|
|
52
|
+
const label = m.displayName && m.displayName !== m.id ? `${m.id}${chalk.gray(' ' + m.displayName)}` : m.id;
|
|
53
|
+
choices.push({ name: `${label}${tail}`, value: m.id });
|
|
54
|
+
}
|
|
55
|
+
choices.push({ name: 'Type a custom model id…', value: CUSTOM_MODEL });
|
|
56
|
+
return choices;
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Prompt over an already-resolved catalog: a `select` of the models plus the
|
|
60
|
+
* keep-current / custom-id rows, mapping the sentinel choices back to a concrete
|
|
61
|
+
* model id. Split from the catalog probe so the KEEP / CUSTOM / pick branches are
|
|
62
|
+
* unit-tested with a scripted IO and no installed agent.
|
|
63
|
+
*/
|
|
64
|
+
export async function chooseModelFromCatalog(io, models, current) {
|
|
65
|
+
const choice = await io.select({
|
|
66
|
+
message: 'Model',
|
|
67
|
+
choices: buildModelChoices(models, current),
|
|
68
|
+
});
|
|
69
|
+
if (choice === KEEP_MODEL)
|
|
70
|
+
return current ?? '';
|
|
71
|
+
if (choice === CUSTOM_MODEL)
|
|
72
|
+
return io.input({ message: 'Model id', default: current });
|
|
73
|
+
return choice;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* The catalog-backed model pick (RUSH-2220). Returns the chosen model id, or
|
|
77
|
+
* `null` to fall through to the engine's free-text prompt when the host exposes
|
|
78
|
+
* no probeable catalog — so a host we can't enumerate degrades to today's
|
|
79
|
+
* free-text behaviour rather than blocking.
|
|
80
|
+
*/
|
|
81
|
+
export async function pickModel(io, host, version, current) {
|
|
82
|
+
if (!host)
|
|
83
|
+
return null;
|
|
84
|
+
const resolvedVersion = version || catalogVersionFor(host);
|
|
85
|
+
if (!resolvedVersion)
|
|
86
|
+
return null;
|
|
87
|
+
const catalog = getModelCatalog(host, resolvedVersion);
|
|
88
|
+
if (!catalog || catalog.models.length === 0)
|
|
89
|
+
return null;
|
|
90
|
+
return chooseModelFromCatalog(io, catalog.models, current);
|
|
91
|
+
}
|
|
92
|
+
/**
|
|
93
|
+
* Assemble the production hook set the harness commands drive the wizard with.
|
|
94
|
+
* The connection-test hook reads the assembled draft's `name` — the caller writes
|
|
95
|
+
* the profile to disk first, then invokes the test through the real `agents run`
|
|
96
|
+
* path against that name.
|
|
97
|
+
*/
|
|
98
|
+
export function harnessHooks() {
|
|
99
|
+
return {
|
|
100
|
+
pickModel,
|
|
101
|
+
connectionTest: (draft) => runHarnessConnectionTest(draft.name),
|
|
102
|
+
editable: defaultEditable,
|
|
103
|
+
};
|
|
104
|
+
}
|
|
@@ -28,6 +28,8 @@
|
|
|
28
28
|
*/
|
|
29
29
|
import type { AgentId } from '../lib/types.js';
|
|
30
30
|
import { type Profile } from '../lib/profiles.js';
|
|
31
|
+
import type { ConnectionTestResult } from '../lib/harness-connection-test.js';
|
|
32
|
+
export type { ConnectionTestResult } from '../lib/harness-connection-test.js';
|
|
31
33
|
/** Whether the wizard is creating a new harness or editing an existing one. */
|
|
32
34
|
export type WizardMode = 'create' | 'edit';
|
|
33
35
|
/** A single `select` choice. `disabled` greys the row (used by the edit matrix). */
|
|
@@ -129,23 +131,40 @@ export interface HarnessEditable {
|
|
|
129
131
|
version: boolean;
|
|
130
132
|
fallback: boolean;
|
|
131
133
|
}
|
|
134
|
+
/** One field's editability plus, when disabled, the one-line reason to surface. */
|
|
135
|
+
export interface EditableField {
|
|
136
|
+
enabled: boolean;
|
|
137
|
+
/** Set only when `enabled` is false — the greyed field's stated reason. */
|
|
138
|
+
reason?: string;
|
|
139
|
+
}
|
|
140
|
+
/** Per-host editability with a reason attached to every disabled field. */
|
|
141
|
+
export interface HarnessEditability {
|
|
142
|
+
model: EditableField;
|
|
143
|
+
baseUrl: EditableField;
|
|
144
|
+
auth: EditableField;
|
|
145
|
+
version: EditableField;
|
|
146
|
+
fallback: EditableField;
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* The per-harness editability matrix (RUSH-2222). Which of a harness's params
|
|
150
|
+
* this host's API format actually lets you change, each disabled field carrying
|
|
151
|
+
* the reason the wizard greys it with.
|
|
152
|
+
*
|
|
153
|
+
* Sourced ENTIRELY from the same maps the run-time resolver reads —
|
|
154
|
+
* `baseUrlEnvKeyForHost` (endpoint slot), `authEnvKeyForHost` (auth env), and
|
|
155
|
+
* `isSelfUpdatingAgent` (pinnable version) — never a table hardcoded alongside
|
|
156
|
+
* them, so the wizard's enable/disable can never drift from what a run actually
|
|
157
|
+
* honors (repo rule: the capability table stays truthful, in lockstep with the
|
|
158
|
+
* code). A disabled param is never a silent no-op — the wizard shows its reason
|
|
159
|
+
* and the flag path fails loud (`forkProfile`'s base-URL throw is the precedent).
|
|
160
|
+
*/
|
|
161
|
+
export declare function harnessEditable(host: AgentId): HarnessEditability;
|
|
132
162
|
/**
|
|
133
|
-
*
|
|
134
|
-
*
|
|
135
|
-
*
|
|
136
|
-
* table hardcoded alongside them — so the wizard's enable/disable can never drift
|
|
137
|
-
* from what a run actually honors (repo rule: the capability table stays truthful,
|
|
138
|
-
* in lockstep with the code). RUSH-2222 may replace this via {@link WizardHooks.editable}
|
|
139
|
-
* to add per-param reasons; it must stay sourced from the resolver.
|
|
163
|
+
* The boolean projection of {@link harnessEditable} — the scaffold default behind
|
|
164
|
+
* the RUSH-2222 {@link WizardHooks.editable} seam. Derived from the reason-carrying
|
|
165
|
+
* matrix so the two can never disagree.
|
|
140
166
|
*/
|
|
141
167
|
export declare function defaultEditable(host: AgentId): HarnessEditable;
|
|
142
|
-
/** Outcome of a connection test (RUSH-2221 fills the real classifier). */
|
|
143
|
-
export interface ConnectionTestResult {
|
|
144
|
-
ok: boolean;
|
|
145
|
-
/** Machine-readable class when it failed (auth / endpoint / model / unknown). */
|
|
146
|
-
reason?: 'auth' | 'endpoint' | 'model' | 'unknown';
|
|
147
|
-
message?: string;
|
|
148
|
-
}
|
|
149
168
|
/**
|
|
150
169
|
* Extension points the sibling subtasks fill without touching the engine. Each is
|
|
151
170
|
* a real no-op-by-default seam: absent, the scaffold uses today's behavior; none
|
|
@@ -33,21 +33,49 @@ import { listBundles } from '../lib/secrets/bundles.js';
|
|
|
33
33
|
import { AGENTS, ALL_AGENT_IDS, isSelfUpdatingAgent, resolveAgentName } from '../lib/agents.js';
|
|
34
34
|
import { readAccountRegistry } from '../lib/account-registry.js';
|
|
35
35
|
/**
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
36
|
+
* The per-harness editability matrix (RUSH-2222). Which of a harness's params
|
|
37
|
+
* this host's API format actually lets you change, each disabled field carrying
|
|
38
|
+
* the reason the wizard greys it with.
|
|
39
|
+
*
|
|
40
|
+
* Sourced ENTIRELY from the same maps the run-time resolver reads —
|
|
41
|
+
* `baseUrlEnvKeyForHost` (endpoint slot), `authEnvKeyForHost` (auth env), and
|
|
42
|
+
* `isSelfUpdatingAgent` (pinnable version) — never a table hardcoded alongside
|
|
43
|
+
* them, so the wizard's enable/disable can never drift from what a run actually
|
|
44
|
+
* honors (repo rule: the capability table stays truthful, in lockstep with the
|
|
45
|
+
* code). A disabled param is never a silent no-op — the wizard shows its reason
|
|
46
|
+
* and the flag path fails loud (`forkProfile`'s base-URL throw is the precedent).
|
|
47
|
+
*/
|
|
48
|
+
export function harnessEditable(host) {
|
|
49
|
+
const hasEndpoint = baseUrlEnvKeyForHost(host) !== null;
|
|
50
|
+
const hasAuth = authEnvKeyForHost(host) !== null;
|
|
51
|
+
const selfUpdating = isSelfUpdatingAgent(host);
|
|
52
|
+
return {
|
|
53
|
+
model: { enabled: true },
|
|
54
|
+
baseUrl: hasEndpoint
|
|
55
|
+
? { enabled: true }
|
|
56
|
+
: { enabled: false, reason: `host '${host}' has no custom-endpoint slot — base URL not applicable` },
|
|
57
|
+
auth: hasAuth
|
|
58
|
+
? { enabled: true }
|
|
59
|
+
: { enabled: false, reason: `host '${host}' manages its own login — no auth to edit` },
|
|
60
|
+
version: selfUpdating
|
|
61
|
+
? { enabled: false, reason: `host '${host}' self-updates — its version can't be pinned` }
|
|
62
|
+
: { enabled: true },
|
|
63
|
+
fallback: { enabled: true },
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* The boolean projection of {@link harnessEditable} — the scaffold default behind
|
|
68
|
+
* the RUSH-2222 {@link WizardHooks.editable} seam. Derived from the reason-carrying
|
|
69
|
+
* matrix so the two can never disagree.
|
|
43
70
|
*/
|
|
44
71
|
export function defaultEditable(host) {
|
|
72
|
+
const e = harnessEditable(host);
|
|
45
73
|
return {
|
|
46
|
-
model:
|
|
47
|
-
baseUrl:
|
|
48
|
-
auth:
|
|
49
|
-
version:
|
|
50
|
-
fallback:
|
|
74
|
+
model: e.model.enabled,
|
|
75
|
+
baseUrl: e.baseUrl.enabled,
|
|
76
|
+
auth: e.auth.enabled,
|
|
77
|
+
version: e.version.enabled,
|
|
78
|
+
fallback: e.fallback.enabled,
|
|
51
79
|
};
|
|
52
80
|
}
|
|
53
81
|
/** Resolve the host CLI a fork `source` runs under (the host `buildFork` will use). */
|
|
@@ -219,8 +247,10 @@ export function createSteps() {
|
|
|
219
247
|
// rest replaces the old silent-drop (`profileFromHostModel` discards a
|
|
220
248
|
// base URL the host can't honor) with an explicit reason.
|
|
221
249
|
const host = d.host ?? hostForSource(d.source);
|
|
222
|
-
if (host
|
|
223
|
-
|
|
250
|
+
if (host) {
|
|
251
|
+
const cap = harnessEditable(host).baseUrl;
|
|
252
|
+
if (!cap.enabled)
|
|
253
|
+
return { disabled: cap.reason };
|
|
224
254
|
}
|
|
225
255
|
return 'run';
|
|
226
256
|
},
|
|
@@ -285,14 +315,14 @@ function currentBaseUrl(p) {
|
|
|
285
315
|
export function editSteps(original) {
|
|
286
316
|
const host = original.host.agent;
|
|
287
317
|
const editableFor = (hooks) => (hooks.editable ?? defaultEditable)(host);
|
|
288
|
-
// decide() has no access to hooks, so gate on the resolver
|
|
289
|
-
// narrows editability further is applied inside run(). The
|
|
290
|
-
// the resolver truth,
|
|
291
|
-
const cap =
|
|
318
|
+
// decide() has no access to hooks, so gate on the resolver-sourced matrix; a
|
|
319
|
+
// hook that narrows editability further is applied inside run(). The matrix is
|
|
320
|
+
// the resolver truth (RUSH-2222), reasons and all.
|
|
321
|
+
const cap = harnessEditable(host);
|
|
292
322
|
return [
|
|
293
323
|
{
|
|
294
324
|
id: 'model',
|
|
295
|
-
decide: () => (cap.model ? 'run' : { disabled:
|
|
325
|
+
decide: () => (cap.model.enabled ? 'run' : { disabled: cap.model.reason }),
|
|
296
326
|
async run(io, d, hooks) {
|
|
297
327
|
if (!editableFor(hooks).model)
|
|
298
328
|
return;
|
|
@@ -301,7 +331,7 @@ export function editSteps(original) {
|
|
|
301
331
|
},
|
|
302
332
|
{
|
|
303
333
|
id: 'baseUrl',
|
|
304
|
-
decide: () => (cap.baseUrl ? 'run' : { disabled:
|
|
334
|
+
decide: () => (cap.baseUrl.enabled ? 'run' : { disabled: cap.baseUrl.reason }),
|
|
305
335
|
async run(io, d, hooks) {
|
|
306
336
|
if (!editableFor(hooks).baseUrl)
|
|
307
337
|
return;
|
|
@@ -311,7 +341,7 @@ export function editSteps(original) {
|
|
|
311
341
|
},
|
|
312
342
|
{
|
|
313
343
|
id: 'account',
|
|
314
|
-
decide: () => (cap.auth ? 'run' : { disabled:
|
|
344
|
+
decide: () => (cap.auth.enabled ? 'run' : { disabled: cap.auth.reason }),
|
|
315
345
|
async run(io, d, hooks) {
|
|
316
346
|
if (!editableFor(hooks).auth)
|
|
317
347
|
return;
|
|
@@ -327,7 +357,7 @@ export function editSteps(original) {
|
|
|
327
357
|
},
|
|
328
358
|
{
|
|
329
359
|
id: 'version',
|
|
330
|
-
decide: () => cap.version ? 'run' : { disabled:
|
|
360
|
+
decide: () => (cap.version.enabled ? 'run' : { disabled: cap.version.reason }),
|
|
331
361
|
async run(io, d, hooks) {
|
|
332
362
|
if (!editableFor(hooks).version)
|
|
333
363
|
return;
|
|
@@ -339,7 +369,7 @@ export function editSteps(original) {
|
|
|
339
369
|
},
|
|
340
370
|
{
|
|
341
371
|
id: 'fallback',
|
|
342
|
-
decide: () => (cap.fallback ? 'run' : 'skip'),
|
|
372
|
+
decide: () => (cap.fallback.enabled ? 'run' : 'skip'),
|
|
343
373
|
async run(io, d) {
|
|
344
374
|
d.fallbackModel = await io.input({
|
|
345
375
|
message: 'Fallback model (same-host rate-limit retry; blank for none)',
|
|
@@ -32,6 +32,8 @@ export interface ForkOptions {
|
|
|
32
32
|
fromSecrets?: string;
|
|
33
33
|
keyStdin?: boolean;
|
|
34
34
|
force?: boolean;
|
|
35
|
+
/** Tri-state pre-save connection test: true = force, false = skip, undefined = ask on a TTY. */
|
|
36
|
+
test?: boolean;
|
|
35
37
|
}
|
|
36
38
|
/** Options accepted by `agents harness edit`. */
|
|
37
39
|
export interface EditOptions {
|
|
@@ -47,6 +49,8 @@ export interface EditOptions {
|
|
|
47
49
|
/** `<bundle>` or `<bundle>:<key>` — see {@link applyFromSecrets} in ./profiles.js. */
|
|
48
50
|
fromSecrets?: string;
|
|
49
51
|
keyStdin?: boolean;
|
|
52
|
+
/** Tri-state pre-save connection test: true = force, false = skip, undefined = ask on a TTY. */
|
|
53
|
+
test?: boolean;
|
|
50
54
|
}
|
|
51
55
|
/**
|
|
52
56
|
* Build the new harness for `agents harness fork <source> <name>`.
|
|
@@ -79,6 +83,16 @@ export declare function forkNeedsWizard(source: string | undefined, name: string
|
|
|
79
83
|
* `agents harness add <preset-name>` (no flags) still resolves via the preset
|
|
80
84
|
* fallback instead of being routed into the wizard. */
|
|
81
85
|
export declare function addNeedsWizard(name: string | undefined, opts: AddProfileOptions): boolean;
|
|
86
|
+
/** Whether the pre-save connection test runs, or must be asked for on a TTY. */
|
|
87
|
+
export type ConnectionTestGate = 'on' | 'off' | 'ask';
|
|
88
|
+
/**
|
|
89
|
+
* Resolve the tri-state connection-test gate (RUSH-2221), pure so the branching
|
|
90
|
+
* is unit-tested with no prompt or spawn. `--test` forces it on, `--no-test`
|
|
91
|
+
* forces it off, and with neither flag a TTY is asked (default yes) while a
|
|
92
|
+
* non-interactive caller (`--key-stdin`, piped, CI) skips it — so scripting stays
|
|
93
|
+
* non-interactive unless it opts in with `--test`.
|
|
94
|
+
*/
|
|
95
|
+
export declare function connectionTestGate(testFlag: boolean | undefined, interactive: boolean): ConnectionTestGate;
|
|
82
96
|
/**
|
|
83
97
|
* Map a finished edit-wizard draft onto {@link EditOptions}, keeping only the
|
|
84
98
|
* fields the user actually changed from the profile's current values. Unchanged
|