@phnx-labs/agents-cli 1.22.57 → 1.22.59
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +294 -0
- package/README.md +29 -0
- package/dist/bootstrap.js +39 -1
- package/dist/commands/accounts.js +7 -3
- package/dist/commands/apply.js +10 -2
- package/dist/commands/fork.d.ts +23 -10
- package/dist/commands/fork.js +115 -58
- package/dist/commands/monitors.js +198 -23
- package/dist/commands/prune.js +5 -3
- package/dist/commands/routines.d.ts +8 -0
- package/dist/commands/routines.js +57 -3
- package/dist/commands/routines.test-fixture.js +5 -0
- package/dist/commands/send.d.ts +2 -1
- package/dist/commands/send.js +7 -5
- package/dist/commands/sessions-picker.d.ts +11 -0
- package/dist/commands/sessions-picker.js +16 -0
- package/dist/commands/sessions-stats.js +37 -5
- package/dist/commands/sessions.js +40 -5
- package/dist/commands/share.d.ts +14 -0
- package/dist/commands/share.js +43 -2
- package/dist/commands/ssh.js +12 -1
- package/dist/commands/status.js +1 -1
- package/dist/commands/sync.js +83 -7
- package/dist/commands/traces.js +7 -0
- package/dist/commands/versions.js +12 -4
- package/dist/commands/view.js +7 -2
- package/dist/index.d.ts +1 -1
- package/dist/index.js +6 -1
- package/dist/lib/account-registry.d.ts +5 -1
- package/dist/lib/account-registry.js +47 -14
- package/dist/lib/accounting/capacity.d.ts +18 -7
- package/dist/lib/accounting/capacity.js +19 -8
- package/dist/lib/accounting/usage-sync.d.ts +29 -1
- package/dist/lib/accounting/usage-sync.js +76 -2
- package/dist/lib/accounting/usage.js +7 -1
- package/dist/lib/auth-mint.d.ts +11 -1
- package/dist/lib/auth-mint.js +21 -6
- package/dist/lib/auto-pull-worker.js +7 -2
- package/dist/lib/browser/ipc.d.ts +8 -0
- package/dist/lib/browser/ipc.js +87 -0
- package/dist/lib/browser/service.d.ts +19 -0
- package/dist/lib/browser/service.js +96 -11
- package/dist/lib/browser/sessions-list.js +10 -1
- package/dist/lib/cloud/rush.d.ts +7 -0
- package/dist/lib/cloud/rush.js +29 -1
- package/dist/lib/daemon/daemon.d.ts +22 -0
- package/dist/lib/daemon/daemon.js +39 -0
- package/dist/lib/daemon/runner.d.ts +3 -0
- package/dist/lib/daemon/runner.js +86 -45
- package/dist/lib/daemon/session-index-service.js +9 -1
- package/dist/lib/daemon/usage-sync-service.d.ts +3 -3
- package/dist/lib/daemon/usage-sync-service.js +14 -8
- package/dist/lib/daemon-services.js +1 -1
- package/dist/lib/daemon-ticks.d.ts +15 -0
- package/dist/lib/daemon-ticks.js +26 -0
- package/dist/lib/device-config.d.ts +5 -1
- package/dist/lib/device-config.js +2 -2
- package/dist/lib/devices/connect.d.ts +17 -8
- package/dist/lib/devices/connect.js +31 -14
- package/dist/lib/devices/health.js +5 -1
- package/dist/lib/devices/pool.d.ts +25 -2
- package/dist/lib/devices/pool.js +32 -2
- package/dist/lib/devices/stats-cache.d.ts +0 -6
- package/dist/lib/devices/stats-cache.js +2 -9
- package/dist/lib/doctor-diff.d.ts +14 -0
- package/dist/lib/doctor-diff.js +120 -9
- package/dist/lib/fleet/manifest.d.ts +17 -0
- package/dist/lib/fleet/manifest.js +26 -0
- package/dist/lib/git.d.ts +38 -0
- package/dist/lib/git.js +58 -0
- package/dist/lib/hooks/install.d.ts +27 -11
- package/dist/lib/hooks/install.js +42 -17
- package/dist/lib/hosts/ready.d.ts +8 -0
- package/dist/lib/hosts/ready.js +13 -2
- package/dist/lib/hosts/reconnect.d.ts +52 -203
- package/dist/lib/hosts/reconnect.js +64 -284
- package/dist/lib/installations/migrate.d.ts +6 -120
- package/dist/lib/installations/migrate.js +27 -259
- package/dist/lib/installations/shims.d.ts +13 -95
- package/dist/lib/installations/shims.js +22 -139
- package/dist/lib/installations/store.js +1 -1
- package/dist/lib/installations/versions.d.ts +43 -133
- package/dist/lib/installations/versions.js +94 -206
- package/dist/lib/monitors/config.d.ts +71 -3
- package/dist/lib/monitors/config.js +100 -12
- package/dist/lib/monitors/pid-watch.d.ts +35 -0
- package/dist/lib/monitors/pid-watch.js +45 -0
- package/dist/lib/monitors/remote.d.ts +18 -0
- package/dist/lib/monitors/remote.js +11 -0
- package/dist/lib/permissions.js +7 -2
- package/dist/lib/plugins/plugins.d.ts +17 -3
- package/dist/lib/plugins/plugins.js +84 -9
- package/dist/lib/plugins/skills.d.ts +8 -1
- package/dist/lib/plugins/skills.js +18 -2
- package/dist/lib/pty-server.d.ts +14 -0
- package/dist/lib/pty-server.js +49 -5
- package/dist/lib/refresh.d.ts +9 -0
- package/dist/lib/refresh.js +3 -1
- package/dist/lib/routine-readiness.d.ts +15 -1
- package/dist/lib/routine-readiness.js +41 -0
- package/dist/lib/sandbox.d.ts +4 -1
- package/dist/lib/sandbox.js +30 -1
- package/dist/lib/secrets/agent.d.ts +80 -225
- package/dist/lib/secrets/agent.js +139 -401
- package/dist/lib/secrets/bundles.d.ts +73 -222
- package/dist/lib/secrets/bundles.js +168 -467
- package/dist/lib/secrets/drivers/rush.js +5 -0
- package/dist/lib/secrets/reaper.d.ts +28 -70
- package/dist/lib/secrets/reaper.js +30 -85
- package/dist/lib/secrets/remote.d.ts +42 -129
- package/dist/lib/secrets/remote.js +55 -173
- package/dist/lib/self-heal/checks/install-staging.d.ts +4 -0
- package/dist/lib/self-heal/checks/install-staging.js +96 -0
- package/dist/lib/self-heal/registry.js +2 -0
- package/dist/lib/self-heal/types.d.ts +1 -1
- package/dist/lib/self-update.d.ts +65 -0
- package/dist/lib/self-update.js +138 -0
- package/dist/lib/session/active.d.ts +13 -1
- package/dist/lib/session/active.js +2 -0
- package/dist/lib/session/cloud.js +5 -0
- package/dist/lib/session/db.d.ts +51 -6
- package/dist/lib/session/db.js +266 -20
- package/dist/lib/session/fork.d.ts +45 -26
- package/dist/lib/session/fork.js +32 -95
- package/dist/lib/session/tool-calls.d.ts +43 -1
- package/dist/lib/session/tool-calls.js +74 -44
- package/dist/lib/session/tool-store.d.ts +33 -2
- package/dist/lib/session/tool-store.js +56 -3
- package/dist/lib/smart-launch.d.ts +6 -0
- package/dist/lib/smart-launch.js +5 -2
- package/dist/lib/staleness/writers/plugins.js +5 -2
- package/dist/lib/staleness/writers/sources.d.ts +5 -0
- package/dist/lib/staleness/writers/sources.js +2 -1
- package/dist/lib/staleness/writers/subagents.js +13 -3
- package/dist/lib/state.d.ts +7 -4
- package/dist/lib/state.js +7 -4
- package/dist/lib/subagents.js +8 -2
- package/dist/lib/sync-status.d.ts +22 -0
- package/dist/lib/sync-status.js +27 -0
- package/dist/lib/sync-umbrella.d.ts +9 -0
- package/dist/lib/sync-umbrella.js +21 -2
- package/dist/lib/teams/scheduler.d.ts +10 -0
- package/dist/lib/teams/scheduler.js +8 -0
- package/dist/lib/traces/insights.d.ts +47 -14
- package/dist/lib/traces/insights.js +92 -21
- package/dist/lib/traces/phenotype.d.ts +23 -3
- package/dist/lib/traces/phenotype.js +72 -24
- package/dist/lib/traces/sync.d.ts +128 -6
- package/dist/lib/traces/sync.js +294 -35
- package/dist/lib/traces/worker-template.js +154 -1
- package/dist/lib/view-types.d.ts +12 -0
- package/package.json +2 -2
|
@@ -1,131 +1,42 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Auto-reconnect for an interactive `agents run --device` session whose
|
|
3
|
-
*
|
|
2
|
+
* Auto-reconnect for an interactive `agents run --device` session whose SSH link
|
|
3
|
+
* dropped. ssh exit 255 triggers bounded-backoff re-attaches via the peer's own
|
|
4
|
+
* `agents sessions focus <id> --local`, which joins a surviving tmux pane or
|
|
5
|
+
* resumes the session in place. The retry window bounds unproductive streaks,
|
|
6
|
+
* resetting when an attach reaches the host and holds the pane.
|
|
4
7
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* sshd session and a blink SIGHUPs it: the in-flight turn is lost, and the
|
|
11
|
-
* reattach RESUMES the harness session from disk instead. Both outcomes go
|
|
12
|
-
* through the same verb below; what changed in PHNX-3316 is that "resumed"
|
|
13
|
-
* is once again an honest, expected result rather than a corpse this file
|
|
14
|
-
* pretended was alive (the pre-RUSH-3125 bug), and RUSH-3125's forced wrap —
|
|
15
|
-
* which made the pane a guarantee by overriding the operator's tmux.enabled —
|
|
16
|
-
* is gone with it.
|
|
17
|
-
*
|
|
18
|
-
* `sshStream` reports that drop as exit code 255 (ssh's
|
|
19
|
-
* own connection-layer failure; see ssh-exec.ts). Without this, exec.ts would
|
|
20
|
-
* `process.exit(255)` and the user would have to notice, find the session id, and
|
|
21
|
-
* `agents sessions focus` by hand. Instead we re-attach over SSH automatically,
|
|
22
|
-
* with bounded backoff, until the user detaches cleanly (the remote returns 0),
|
|
23
|
-
* the agent exits (a non-255 code), or the user interrupts the wait with
|
|
24
|
-
* Ctrl-C ({@link waitOrInterrupt} → 130).
|
|
25
|
-
*
|
|
26
|
-
* The re-attach reuses the peer's OWN recovery verb — `agents sessions focus <id>
|
|
27
|
-
* --local` — which JOINS the live local tmux pane there (a second client, no fork)
|
|
28
|
-
* when it exists, and RESUMES the session in place when there is no pane — which
|
|
29
|
-
* is every drop on a default (wrap-off) box. Dropping `--attach-only` is
|
|
30
|
-
* deliberate: a reattach that finds no pane must not dead-end at a bare shell
|
|
31
|
-
* (the RUSH-2085 bug), it must fall through to resume so the user is put back
|
|
32
|
-
* into the agent. There is one re-attach implementation (the peer's focus) to
|
|
33
|
-
* keep in sync.
|
|
34
|
-
*
|
|
35
|
-
* **What it takes to refill the budget: reached the host AND held the pane.** ssh
|
|
36
|
-
* returns 255 for BOTH "couldn't connect at all" and "connected, then the link
|
|
37
|
-
* dropped." A failed connect still takes up to `ConnectTimeout` (10s) to return, so
|
|
38
|
-
* the duration of the whole call can't tell the two apart — a threshold below the
|
|
39
|
-
* connect timeout would classify every hung connect as a live session and retry
|
|
40
|
-
* forever under a sustained outage (the exact failure this feature exists to
|
|
41
|
-
* survive). So each attempt runs a fast preflight probe FIRST, and only once that
|
|
42
|
-
* probe proves the host reachable does the interactive attach run — which means the
|
|
43
|
-
* ATTACH's own duration is a clean signal, measured with the connect phase already
|
|
44
|
-
* behind it. A reattach that reached the host and then held for at least
|
|
45
|
-
* {@link MIN_HOLD_MS} refills the retry budget, so a long session that blinks all
|
|
46
|
-
* day keeps reconnecting. Everything else — never reached the host, or reached it
|
|
47
|
-
* and died right back — counts against the budget, so a sustained outage and a
|
|
48
|
-
* fast-flapping link both give up once {@link RECONNECT_WINDOW_MS} of unproductive
|
|
49
|
-
* retrying has elapsed.
|
|
50
|
-
*
|
|
51
|
-
* The retry policy is a pure state machine (`reconnectStep`) so it is unit-tested
|
|
52
|
-
* without touching SSH; the loop (`reconnectInteractiveSession`) only adds the real
|
|
53
|
-
* preflight + `sshStream` re-attach and the wait.
|
|
54
|
-
*
|
|
55
|
-
* **255 from the REMOTE side should never be trusted as "the link dropped."**
|
|
56
|
-
* `reattachRemoteSession`'s `connected` flag is set as soon as the fast preflight
|
|
57
|
-
* probe succeeds — it says nothing about whether the interactive attach/resume
|
|
58
|
-
* that follows actually put the user back into the agent. If the REMOTE command
|
|
59
|
-
* (`agents sessions focus <id> --local`) itself ever happened to exit 255 for a
|
|
60
|
-
* reason that has nothing to do with the ssh transport, `sshStream` would return
|
|
61
|
-
* that same 255, `reconnectStep` couldn't tell it apart from a genuine drop, and
|
|
62
|
-
* `connected: true` would refill the retry budget forever — "attempt 1/N" printed
|
|
63
|
-
* on every single cycle, the terminal filling with aborted-TTY escape-code
|
|
64
|
-
* garbage, the retry window never actually bounding anything.
|
|
65
|
-
*
|
|
66
|
-
* The resume fall-through only widens that surface — the peer's focus now runs a
|
|
67
|
-
* full recovery path (`resumeSessionInPlace` / `runOnPeer`) on a dead pane, any
|
|
68
|
-
* step of which could in principle exit 255 for its own reasons. So the channel-
|
|
69
|
-
* level defense is what matters, not an audit of which branch can fire:
|
|
70
|
-
* {@link wrapRemoteExitCode} wraps the entire remote command so that whatever exit
|
|
71
|
-
* code it decides on, a 255 is remapped to {@link REMOTE_EXIT_255_REMAPPED} before
|
|
72
|
-
* `sshStream` ever sees it, regardless of which internal branch produced it and
|
|
73
|
-
* regardless of the peer's `agents` version (the remap happens in the shell
|
|
74
|
-
* wrapper THIS process sends).
|
|
75
|
-
*
|
|
76
|
-
* A recurring LOCAL ssh failure used to defeat the budget the same way, and the
|
|
77
|
-
* remap alone did not close it (agents-cli#1884): `connected` was set by the
|
|
78
|
-
* preflight probe and said nothing about whether the attach that followed held, so
|
|
79
|
-
* a fast-flapping link — or an attach that died at the TTY-negotiation stage every
|
|
80
|
-
* single time — refilled the budget on every cycle, printed "attempt 1" forever,
|
|
81
|
-
* and left the retry window bounding nothing. The {@link MIN_HOLD_MS} floor above is
|
|
82
|
-
* what closes it: an attach that dies immediately is not a reconnection, so the
|
|
83
|
-
* budget drains and the loop gives up with {@link unstableNotice}. A flat total-
|
|
84
|
-
* attempt or wall-clock ceiling that ignored `connected` was the alternative and is
|
|
85
|
-
* deliberately NOT taken — any fixed total eventually strands the all-day-blinking
|
|
86
|
-
* session this feature exists for, while the hold floor only ever stops a loop that
|
|
87
|
-
* is failing to put the user back into the agent.
|
|
8
|
+
* A remote interactive session MUST be tmux-wrapped: with the wrap on, the agent
|
|
9
|
+
* runs in a DETACHED tmux pane on the peer, so a blink kills only the ssh client
|
|
10
|
+
* and the reattach rejoins the live pane; with it off, the agent is a child of
|
|
11
|
+
* the sshd session and a blink SIGHUPs it, losing the in-flight turn (the
|
|
12
|
+
* reattach then resumes the harness from disk). See lib/exec.ts `runInTmux`.
|
|
88
13
|
*/
|
|
89
14
|
import { sshExec, sshStream, shellQuote, SSH_CONN_FAILURE_CODE } from '../ssh-exec.js';
|
|
90
15
|
import { hostIdentityArgs, sshTargetFor } from './types.js';
|
|
91
16
|
import { RUN_AUTO_KEYWORD } from '../types.js';
|
|
92
|
-
/** ssh's connection-layer failure code —
|
|
93
|
-
* than the remote command exiting on its own. Re-exported from ssh-exec.ts, which
|
|
94
|
-
* owns the ssh invocation, so the two cannot drift apart. */
|
|
17
|
+
/** ssh's connection-layer failure code — re-exported from ssh-exec.ts. */
|
|
95
18
|
export const SSH_CONN_FAILURE = SSH_CONN_FAILURE_CODE;
|
|
96
|
-
/**
|
|
97
|
-
*
|
|
98
|
-
*
|
|
19
|
+
/**
|
|
20
|
+
* ssh returns 255 for BOTH "couldn't connect" and "connected then dropped", so a
|
|
21
|
+
* remote-origin 255 (the focus command's own exit) is remapped to 254 before the
|
|
22
|
+
* reconnect loop sees it — inside the loop, 255 therefore always means a network
|
|
23
|
+
* drop, never a code the remote command chose.
|
|
24
|
+
*/
|
|
99
25
|
export const REMOTE_EXIT_255_REMAPPED = 254;
|
|
100
26
|
/**
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* A wall-clock window, not an attempt count, because what it has to outlast is
|
|
104
|
-
* measured in minutes: a laptop lid close, a Wi-Fi handoff, a VPN or Tailscale
|
|
105
|
-
* re-auth, a router reboot. The previous `MAX_ATTEMPTS = 6` over a 2/4/8/16/30/30
|
|
106
|
-
* backoff gave up after about **90 seconds** — shorter than any of them
|
|
107
|
-
* (RUSH-3125). Worse, timers are suspended across sleep, so on wake the whole
|
|
108
|
-
* backoff fired back-to-back before the network was up and the budget was gone in
|
|
109
|
-
* seconds.
|
|
110
|
-
*
|
|
111
|
-
* It bounds the STREAK, not the session: a reattach that reconnects and HOLDS
|
|
112
|
-
* resets it to zero ({@link refillsBudget}), so a session that blinks all day
|
|
113
|
-
* still reconnects every time. That is the property the file header insists on,
|
|
114
|
-
* and a flat total would break it — this changes only how a *streak* is bounded.
|
|
27
|
+
* Wall-clock window bounding unproductive reconnect streaks, not the total
|
|
28
|
+
* session. A reattach that reaches the host and holds resets the budget.
|
|
115
29
|
*/
|
|
116
30
|
export const RECONNECT_WINDOW_MS = 15 * 60_000;
|
|
117
|
-
/** Backoff curve
|
|
118
|
-
* closes. Capped so a long outage keeps probing at a useful cadence instead of
|
|
119
|
-
* drifting out to multi-minute gaps and missing the moment the link returns. */
|
|
31
|
+
/** Backoff curve: 2s, 4s, 8s, 16s, 30s, then 30s until the window closes. */
|
|
120
32
|
const BASE_BACKOFF_MS = 2_000;
|
|
121
33
|
const MAX_BACKOFF_MS = 30_000;
|
|
122
|
-
/**
|
|
123
|
-
*
|
|
124
|
-
*
|
|
125
|
-
*
|
|
126
|
-
*
|
|
127
|
-
|
|
128
|
-
* should stop retrying, not one it should keep re-entering. */
|
|
34
|
+
/**
|
|
35
|
+
* A genuine reconnection must hold the remote pane this long before refilling
|
|
36
|
+
* the retry budget. It is the minimum time to clear TTY negotiation and confirm
|
|
37
|
+
* a working session, distinguishing that from an attach that reconnects and
|
|
38
|
+
* immediately re-drops; otherwise a flapping link would retry forever.
|
|
39
|
+
*/
|
|
129
40
|
export const MIN_HOLD_MS = 10_000;
|
|
130
41
|
export function initialReconnectState() {
|
|
131
42
|
return { attempt: 0, unproductiveMs: 0 };
|
|
@@ -134,36 +45,20 @@ export function initialReconnectState() {
|
|
|
134
45
|
export function backoffMs(attempt) {
|
|
135
46
|
return Math.min(BASE_BACKOFF_MS * 2 ** attempt, MAX_BACKOFF_MS);
|
|
136
47
|
}
|
|
137
|
-
/**
|
|
138
|
-
* Did this attempt genuinely put the user back into the agent? Only such an attempt
|
|
139
|
-
* refills the retry budget — it must have reached the host AND held the pane for at
|
|
140
|
-
* least {@link MIN_HOLD_MS}. An attach that reached the host and died right back is
|
|
141
|
-
* a flapping link, not a reconnection, and counts against the budget like an
|
|
142
|
-
* unreachable host (agents-cli#1884; see the file header).
|
|
143
|
-
*/
|
|
48
|
+
/** True when the attempt both connected and held long enough to refill the budget. */
|
|
144
49
|
export function refillsBudget(outcome) {
|
|
145
50
|
return outcome.connected && outcome.heldMs >= MIN_HOLD_MS;
|
|
146
51
|
}
|
|
147
52
|
/**
|
|
148
|
-
* Decide
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
* - a non-255 code means the remote command spoke for itself (clean detach = 0,
|
|
152
|
-
* agent exit / no live session = non-zero) → stop and surface that code.
|
|
153
|
-
* - a 255 means the link dropped → retry, unless the budget is spent.
|
|
154
|
-
* - a 255 from an attempt that reconnected AND held ({@link refillsBudget})
|
|
155
|
-
* refills the budget first; every other 255 counts against it, so a host that
|
|
156
|
-
* stays unreachable — and a link that keeps dropping the attach immediately —
|
|
157
|
-
* both give up once the retry window closes.
|
|
53
|
+
* Decide the next action from a run/re-attach outcome. Non-255 exits stop and
|
|
54
|
+
* surface that code; 255 retries until the unproductive window is spent.
|
|
158
55
|
*/
|
|
159
56
|
export function reconnectStep(state, outcome) {
|
|
160
57
|
if (outcome.code !== SSH_CONN_FAILURE)
|
|
161
58
|
return { action: 'stop', code: outcome.code };
|
|
162
59
|
const productive = refillsBudget(outcome);
|
|
163
60
|
const attempts = productive ? 0 : state.attempt;
|
|
164
|
-
// The attach's own duration counts against the window
|
|
165
|
-
// burned on every attempt is real elapsed time the user is waiting, and
|
|
166
|
-
// ignoring it would stretch a "15 minute" window well past fifteen minutes.
|
|
61
|
+
// The attach's own duration counts against the window.
|
|
167
62
|
const burned = productive ? 0 : state.unproductiveMs + outcome.heldMs;
|
|
168
63
|
if (burned >= RECONNECT_WINDOW_MS)
|
|
169
64
|
return { action: 'stop', code: SSH_CONN_FAILURE };
|
|
@@ -182,68 +77,33 @@ export function formatDuration(ms) {
|
|
|
182
77
|
const sec = total % 60;
|
|
183
78
|
return m > 0 ? `${m}m${String(sec).padStart(2, '0')}s` : `${sec}s`;
|
|
184
79
|
}
|
|
185
|
-
/**
|
|
186
|
-
* Notice shown before each reconnect wait.
|
|
187
|
-
*
|
|
188
|
-
* Says how long is left in the window rather than "attempt 2/6": with a
|
|
189
|
-
* wall-clock budget the attempt number no longer tells the user when this stops,
|
|
190
|
-
* and "how much longer will it keep trying" is the actual question during an
|
|
191
|
-
* outage. Ctrl-C is advertised because a user who wants their shell back should
|
|
192
|
-
* not have to guess whether interrupting is safe — it is (the agent keeps
|
|
193
|
-
* running on the peer, which is the whole point).
|
|
194
|
-
*/
|
|
80
|
+
/** Notice shown before each reconnect wait. */
|
|
195
81
|
export function reconnectNotice(target, host, attempt, waitMs, remainingMs) {
|
|
196
82
|
const secs = Math.round(waitMs / 1000);
|
|
197
83
|
const when = secs <= 1 ? 'now' : `in ${secs}s`;
|
|
198
84
|
return `\nConnection to ${host} dropped — ${targetLabel(target)} is still running there.`
|
|
199
85
|
+ `\n Reconnecting ${when} · ${formatDuration(remainingMs)} left · attempt ${attempt} · Ctrl-C to stop\n`;
|
|
200
86
|
}
|
|
201
|
-
/** Notice shown once the retry budget is spent on
|
|
202
|
-
* Hands back the one verb that re-enters the terminal — attach the live pane if it
|
|
203
|
-
* survived, else resume. */
|
|
87
|
+
/** Notice shown once the retry budget is spent on an unreachable host. */
|
|
204
88
|
export function exhaustedNotice(target, host) {
|
|
205
89
|
return `\nCouldn't reconnect to ${host} after ${formatDuration(RECONNECT_WINDOW_MS)}. The agent may still be running — get back in when the network is back:\n${recoveryHint(target, host)}`;
|
|
206
90
|
}
|
|
207
|
-
/** Notice shown when the
|
|
208
|
-
* the host and the connection dropped again within {@link MIN_HOLD_MS}. Saying
|
|
209
|
-
* "couldn't reconnect" there would be false — it did reconnect and could not stay
|
|
210
|
-
* — and the user needs to know the link, not the host, is the problem. It claims
|
|
211
|
-
* no count of successful reconnections: the budget can also be spent by a run of
|
|
212
|
-
* unreachable attempts followed by one that reconnected and dropped straight out. */
|
|
91
|
+
/** Notice shown when the host reconnects but drops again within {@link MIN_HOLD_MS}. */
|
|
213
92
|
export function unstableNotice(target, host) {
|
|
214
93
|
const secs = Math.round(MIN_HOLD_MS / 1000);
|
|
215
94
|
return `\nGave up reconnecting to ${host} after ${formatDuration(RECONNECT_WINDOW_MS)} — it kept dropping again within ${secs} seconds of getting back in. The agent may still be running there; reconnect once the link is stable:\n${recoveryHint(target, host)}`;
|
|
216
95
|
}
|
|
217
|
-
/** Notice shown when a reattach
|
|
218
|
-
* ({@link REMOTE_EXIT_255_REMAPPED} — a would-be-255 the remote command decided
|
|
219
|
-
* on for its own reasons, not the ssh transport dropping; see
|
|
220
|
-
* {@link wrapRemoteExitCode}). Distinct from {@link exhaustedNotice}, which is
|
|
221
|
-
* only for a genuinely spent retry budget. */
|
|
96
|
+
/** Notice shown when a reattach ends with a remapped remote-side exit. */
|
|
222
97
|
export function remoteExitNotice(target, host) {
|
|
223
98
|
return `\nReattach to ${targetLabel(target)} on ${host} ended (not a network drop) — get back in, or check whether it's still live:\n${recoveryHint(target, host)}`;
|
|
224
99
|
}
|
|
225
|
-
/**
|
|
226
|
-
* detached on the peer — so this says how to get back rather than implying the
|
|
227
|
-
* work was lost. */
|
|
100
|
+
/** Notice shown when the user stops the wait with Ctrl-C. */
|
|
228
101
|
export function interruptedNotice(target, host) {
|
|
229
102
|
return `\nStopped reconnecting. ${targetLabel(target)} is still running on ${host}:\n${recoveryHint(target, host)}`;
|
|
230
103
|
}
|
|
231
104
|
/**
|
|
232
|
-
* Wrap `cmd` in `bash -lc`
|
|
233
|
-
*
|
|
234
|
-
* a login shell at all; the sibling interactive dispatch in dispatch.ts sends a
|
|
235
|
-
* bare `agents …` with no shell wrapper, so this is a NEW login-shell hop on the
|
|
236
|
-
* reattach path specifically, not something already universal here) with a
|
|
237
|
-
* trailing exit-code remap: whatever `cmd` itself exits with, a 255 becomes
|
|
238
|
-
* {@link REMOTE_EXIT_255_REMAPPED} before the wrapper exits — see the file
|
|
239
|
-
* header for why. Every other code (0, 1, …) passes through unchanged. This
|
|
240
|
-
* carries no PATH bootstrap of its own — `ensureHostReady`/`readyProbe` already
|
|
241
|
-
* gates every `--device` dispatch on `bash -lc 'agents --version'` succeeding
|
|
242
|
-
* before a run is attempted at all, so the peer's login shell resolving `agents`
|
|
243
|
-
* is an established precondition here too. Pure string-building, so it is
|
|
244
|
-
* unit-tested without SSH (and, since the constructed script is ordinary POSIX,
|
|
245
|
-
* also exercised by actually running it through a real shell in the test — no
|
|
246
|
-
* mock needed).
|
|
105
|
+
* Wrap `cmd` in `bash -lc` with an exit-code remap: 255 becomes
|
|
106
|
+
* {@link REMOTE_EXIT_255_REMAPPED} so a remote-origin 255 cannot pass as ssh drop.
|
|
247
107
|
*/
|
|
248
108
|
export function wrapRemoteExitCode(cmd) {
|
|
249
109
|
const guarded = `${cmd}; rc=$?; [ "$rc" = "${SSH_CONN_FAILURE}" ] && rc=${REMOTE_EXIT_255_REMAPPED}; exit "$rc"`;
|
|
@@ -254,23 +114,10 @@ export function targetLabel(target) {
|
|
|
254
114
|
return target.id.slice(0, 8);
|
|
255
115
|
}
|
|
256
116
|
/**
|
|
257
|
-
* The command to
|
|
258
|
-
*
|
|
259
|
-
* A `session` target has a real id, so `agents sessions resume <id>` does the
|
|
260
|
-
* same attach-else-recover the loop was attempting. The full id is labeled on
|
|
261
|
-
* its own `Session` line, not only inside the command (RUSH-3227). (It used to
|
|
262
|
-
* say `agents reconnect`, which is deprecated and hidden — `commands/reconnect.ts`
|
|
263
|
-
* — so the advice printed at the worst possible moment was itself stale; RUSH-3125.)
|
|
264
|
-
*
|
|
265
|
-
* A `launch` target has no id a user-facing verb accepts: the mapping lives in
|
|
266
|
-
* the peer's hook records, which is exactly why reconnect uses it. So point at
|
|
267
|
-
* the peer's own resolver rather than inventing a launch-id selector on every
|
|
268
|
-
* local command for a string no human ever types.
|
|
117
|
+
* The command a user can run to re-enter a session after reconnect gives up.
|
|
118
|
+
* The full id is printed on its own line so it remains copyable after a drop.
|
|
269
119
|
*/
|
|
270
120
|
export function recoveryHint(target, host) {
|
|
271
|
-
// The full id is labeled on its own line: burying it only inside a command is
|
|
272
|
-
// how a dropped SSH tab used to land on a bare shell with nothing copyable
|
|
273
|
-
// (RUSH-3227). Resume still sits underneath so the user can paste one verb.
|
|
274
121
|
if (target.kind === 'session') {
|
|
275
122
|
return ` Session ${target.id}\n Resume: agents sessions resume ${target.id}\n`;
|
|
276
123
|
}
|
|
@@ -278,22 +125,16 @@ export function recoveryHint(target, host) {
|
|
|
278
125
|
+ ` or pick it: agents sessions --active\n`;
|
|
279
126
|
}
|
|
280
127
|
/**
|
|
281
|
-
* Notice shown when an interactive remote connection
|
|
282
|
-
*
|
|
283
|
-
* about to auto-reconnect. The OpenSSH close line (`Shared connection to …
|
|
284
|
-
* closed.`) names the host and nothing else; this is the handle they need to
|
|
285
|
-
* get back in (RUSH-3227).
|
|
128
|
+
* Notice shown when an interactive remote connection ends and no auto-reconnect
|
|
129
|
+
* follows. Prints the session id/handle so the user can get back in.
|
|
286
130
|
*/
|
|
287
131
|
export function connectionEndedNotice(target, host, opts = {}) {
|
|
288
132
|
const verb = opts.dropped ? 'dropped' : 'closed';
|
|
289
133
|
return `\nConnection to ${host} ${verb}.\n${recoveryHint(target, host)}`;
|
|
290
134
|
}
|
|
291
135
|
/**
|
|
292
|
-
* Banner printed
|
|
293
|
-
*
|
|
294
|
-
* will cover this; it survives in scrollback so the id exists *while* the
|
|
295
|
-
* connection exists, not only after it dies (RUSH-3227 plan B). Launch-id
|
|
296
|
-
* targets are not a resume handle — do not print them here.
|
|
136
|
+
* Banner printed as an interactive `--device` run takes the TTY, so the id is
|
|
137
|
+
* visible in scrollback while the connection exists.
|
|
297
138
|
*/
|
|
298
139
|
export function connectionStartedNotice(target, host) {
|
|
299
140
|
if (target.kind !== 'session')
|
|
@@ -301,10 +142,8 @@ export function connectionStartedNotice(target, host) {
|
|
|
301
142
|
return `Session ${target.id} on ${host}\n Resume later: agents sessions resume ${target.id}\n`;
|
|
302
143
|
}
|
|
303
144
|
/**
|
|
304
|
-
* The id to print as an interactive `--device` stream starts.
|
|
305
|
-
*
|
|
306
|
-
* Claude's forced id is the other. `run auto` is excluded: its forwarded
|
|
307
|
-
* `--session-id` is only real if the remote pick is Claude.
|
|
145
|
+
* The id to print as an interactive `--device` stream starts. `run auto` is
|
|
146
|
+
* excluded because its forwarded id is only real when the remote picks Claude.
|
|
308
147
|
*/
|
|
309
148
|
export function startConnectionTarget(opts) {
|
|
310
149
|
if (opts.agent === RUN_AUTO_KEYWORD)
|
|
@@ -313,15 +152,8 @@ export function startConnectionTarget(opts) {
|
|
|
313
152
|
return id ? { kind: 'session', id } : undefined;
|
|
314
153
|
}
|
|
315
154
|
/**
|
|
316
|
-
* Decide what happens after an interactive `--device` stream returns.
|
|
317
|
-
*
|
|
318
|
-
* Auto-reconnect fires whenever the link dropped (255) and the run did not opt
|
|
319
|
-
* out with `--raw` (`willReconnect`) — wrap or no wrap. A wrapped run's pane
|
|
320
|
-
* survived, so the reattach rejoins it; a bare run's agent died with the link,
|
|
321
|
-
* so the same verb resumes the session in place from disk (PHNX-3316). `--raw`
|
|
322
|
-
* never reconnects — but it still prints the session id: the user is at a
|
|
323
|
-
* shell, and EXEC-55 does not exempt raw (RUSH-3227). Bundling the notice
|
|
324
|
-
* behind `!isRaw` was the miss.
|
|
155
|
+
* Decide what happens after an interactive `--device` stream returns. Auto-
|
|
156
|
+
* reconnect fires on 255 unless `--raw` opted out; otherwise print recovery info.
|
|
325
157
|
*/
|
|
326
158
|
export function afterInteractiveRemoteExit(opts) {
|
|
327
159
|
if (!opts.target)
|
|
@@ -336,24 +168,8 @@ export function afterInteractiveRemoteExit(opts) {
|
|
|
336
168
|
};
|
|
337
169
|
}
|
|
338
170
|
/**
|
|
339
|
-
* Choose how to name the dropped run
|
|
340
|
-
*
|
|
341
|
-
* Order matters, and the last clause is the fix (RUSH-3125):
|
|
342
|
-
*
|
|
343
|
-
* 1. `run auto` prefers `resolvedId` — the harness the remote ACTUALLY picked,
|
|
344
|
-
* which the launcher's own `--session-id` (adopted only by Claude) may not
|
|
345
|
-
* name. Every other agent prefers the id the launcher forced.
|
|
346
|
-
* 2. `resumeId` covers a run that was continuing a known session.
|
|
347
|
-
* 3. **`launchId` last, and it is what makes this work at all off-Claude.**
|
|
348
|
-
* Every id above either came from the launcher or was read back over SSH —
|
|
349
|
-
* and that read happens after the stream returned, i.e. over the link that
|
|
350
|
-
* just died, so on a real drop it yields nothing. Falling through to the
|
|
351
|
-
* launch id means the reconnect no longer depends on reaching the host to
|
|
352
|
-
* learn what to reconnect to; the peer resolves it locally instead.
|
|
353
|
-
*
|
|
354
|
-
* Returns undefined only when the launcher has no handle at all (a hookless
|
|
355
|
-
* harness with no forced id), which is the one case reconnect genuinely cannot
|
|
356
|
-
* serve. Pure, so the precedence is unit-tested without SSH.
|
|
171
|
+
* Choose how to name the dropped run. `launchId` is last because it is minted
|
|
172
|
+
* locally before the connection exists, so it survives the drop.
|
|
357
173
|
*/
|
|
358
174
|
export function pickReconnectTarget(inputs) {
|
|
359
175
|
const { agent, sessionId, resolvedId, resumeId, launchId } = inputs;
|
|
@@ -366,18 +182,8 @@ export function pickReconnectTarget(inputs) {
|
|
|
366
182
|
return launchId ? { kind: 'launch', id: launchId } : undefined;
|
|
367
183
|
}
|
|
368
184
|
/**
|
|
369
|
-
* The
|
|
370
|
-
*
|
|
371
|
-
* so a stray remote-origin 255 (from this command, whatever produces it — see the
|
|
372
|
-
* file header) can never masquerade as a network drop. No `--attach-only`: focus
|
|
373
|
-
* joins the live pane when it survived, else RESUMES the session in place, so a
|
|
374
|
-
* reattach landing after the pane died recovers the agent instead of dead-ending
|
|
375
|
-
* (RUSH-2085). Split out from {@link reattachRemoteSession} so it is unit-tested
|
|
376
|
-
* without SSH — mirrors `remoteAgentsJsonCommand` in lib/remote-agents-json.ts.
|
|
377
|
-
*
|
|
378
|
-
* A `launch` target passes `--launch-id`, which focus resolves against the hook
|
|
379
|
-
* records on the peer itself — no network read from this side, which is the
|
|
380
|
-
* whole point (see {@link ReconnectTarget}).
|
|
185
|
+
* The peer's recovery verb wrapped so a remote-origin 255 cannot masquerade as
|
|
186
|
+
* a network drop. No `--attach-only`: focus resumes in place when no pane survived.
|
|
381
187
|
*/
|
|
382
188
|
export function reattachRemoteCommand(target) {
|
|
383
189
|
const selector = target.kind === 'launch'
|
|
@@ -389,27 +195,19 @@ export function reattachRemoteCommand(target) {
|
|
|
389
195
|
return wrapRemoteExitCode(inner);
|
|
390
196
|
}
|
|
391
197
|
/**
|
|
392
|
-
* Re-attach the live remote
|
|
393
|
-
*
|
|
394
|
-
*
|
|
395
|
-
* `connected` bit, not the call duration, is what the retry policy keys on. Only on
|
|
396
|
-
* a reachable host do we run the interactive attach-or-resume (which carries no
|
|
397
|
-
* credentials — the agent already runs on the peer — so it rides the normal
|
|
398
|
-
* transport). Returns the ssh exit code (255 = dropped again / unreachable; 0 =
|
|
399
|
-
* clean detach; other = session ended), whether this attempt connected, and how
|
|
400
|
-
* long the attach held — the two inputs {@link refillsBudget} decides on.
|
|
198
|
+
* Re-attach the live remote pane by driving the peer's `agents sessions focus`.
|
|
199
|
+
* A fast preflight probe decides reachability; only then do we run the
|
|
200
|
+
* interactive attach/resume. Returns the exit code, connected bit, and hold time.
|
|
401
201
|
*/
|
|
402
202
|
export function reattachRemoteSession(host, target) {
|
|
403
203
|
const sshTarget = sshTargetFor(host);
|
|
404
204
|
const extraSshArgs = hostIdentityArgs(host);
|
|
405
|
-
// Fresh (non-multiplexed) reachability probe:
|
|
406
|
-
//
|
|
407
|
-
// RUSH-2265: pass host identity on every hop (probe + stream), not only the first.
|
|
205
|
+
// Fresh (non-multiplexed) reachability probe: only a completed handshake is
|
|
206
|
+
// counted as connected.
|
|
408
207
|
const probe = sshExec(sshTarget, 'true', { multiplex: false, extraSshArgs });
|
|
409
208
|
if (probe.code !== 0)
|
|
410
209
|
return { code: SSH_CONN_FAILURE, connected: false, heldMs: 0 };
|
|
411
|
-
// Timed from
|
|
412
|
-
// carries none of the connect phase the file header rules out as a signal.
|
|
210
|
+
// Timed from after the probe returns, so this measures only the attach itself.
|
|
413
211
|
const startedAt = Date.now();
|
|
414
212
|
const code = sshStream(sshTarget, reattachRemoteCommand(target), { tty: true, extraSshArgs });
|
|
415
213
|
return { code, connected: true, heldMs: Date.now() - startedAt };
|
|
@@ -417,14 +215,8 @@ export function reattachRemoteSession(host, target) {
|
|
|
417
215
|
/**
|
|
418
216
|
* Wait `ms`, but return early if the user interrupts.
|
|
419
217
|
*
|
|
420
|
-
*
|
|
421
|
-
*
|
|
422
|
-
* happened, the terminal was left mid-notice, and the user was dropped at a bare
|
|
423
|
-
* shell with no hint that the agent was still alive on the peer — the same
|
|
424
|
-
* dead-end the reconnect exists to prevent. The handler is installed only for
|
|
425
|
-
* the duration of the wait and always removed, so it can never swallow a Ctrl-C
|
|
426
|
-
* meant for the attached agent (during an attach, ssh owns the tty and this
|
|
427
|
-
* process is not in the foreground group anyway).
|
|
218
|
+
* Installs a SIGINT handler only for the wait so Ctrl-C gives a graceful
|
|
219
|
+
* "agent still running" message instead of killing the local process silently.
|
|
428
220
|
*/
|
|
429
221
|
async function waitOrInterrupt(ms) {
|
|
430
222
|
return new Promise((resolve) => {
|
|
@@ -442,27 +234,18 @@ async function waitOrInterrupt(ms) {
|
|
|
442
234
|
process.on('SIGINT', onSigint);
|
|
443
235
|
});
|
|
444
236
|
}
|
|
445
|
-
/**
|
|
446
|
-
* Drive the reconnect loop from the initial run's outcome to a terminal code.
|
|
447
|
-
* Only the real SSH re-attach and the wait are side effects; the decision is
|
|
448
|
-
* {@link reconnectStep}. Returns the exit code the process should ultimately use.
|
|
449
|
-
*/
|
|
237
|
+
/** Drive the reconnect loop from the initial run's outcome to a terminal code. */
|
|
450
238
|
export async function reconnectInteractiveSession(opts) {
|
|
451
239
|
const write = opts.write ?? ((s) => process.stderr.write(s));
|
|
452
240
|
const wait = opts.wait ?? waitOrInterrupt;
|
|
453
241
|
const reattach = opts.reattach ?? reattachRemoteSession;
|
|
454
242
|
let state = initialReconnectState();
|
|
455
|
-
// The initial run
|
|
456
|
-
// hold duration is never measured — exec.ts owns that call — and never needs to
|
|
457
|
-
// be: refilling is a no-op at attempt 0, which is where the loop starts, so this
|
|
458
|
-
// outcome can only ever produce the first retry either way.
|
|
243
|
+
// The initial run connected; its hold duration doesn't matter at attempt 0.
|
|
459
244
|
let outcome = { code: opts.initialExit, connected: true, heldMs: 0 };
|
|
460
245
|
for (;;) {
|
|
461
246
|
const decision = reconnectStep(state, outcome);
|
|
462
247
|
if (decision.action === 'stop') {
|
|
463
|
-
//
|
|
464
|
-
// reattach that spent it, so an unreachable host reads "couldn't reconnect"
|
|
465
|
-
// and a link that reconnected but kept dropping reads as exactly that.
|
|
248
|
+
// Spent budget: unreachable host vs. unstable link need different messages.
|
|
466
249
|
if (decision.code === SSH_CONN_FAILURE) {
|
|
467
250
|
write(outcome.connected
|
|
468
251
|
? unstableNotice(opts.target, opts.host.name)
|
|
@@ -472,17 +255,14 @@ export async function reconnectInteractiveSession(opts) {
|
|
|
472
255
|
write(remoteExitNotice(opts.target, opts.host.name));
|
|
473
256
|
}
|
|
474
257
|
else {
|
|
475
|
-
// Clean detach / agent exit
|
|
476
|
-
// shell and still needs the session id (RUSH-3227). Spent-budget and
|
|
477
|
-
// remapped-255 notices already carry recoveryHint.
|
|
258
|
+
// Clean detach / agent exit: still print the session handle.
|
|
478
259
|
write(connectionEndedNotice(opts.target, opts.host.name));
|
|
479
260
|
}
|
|
480
261
|
return decision.code;
|
|
481
262
|
}
|
|
482
263
|
write(reconnectNotice(opts.target, opts.host.name, decision.state.attempt, decision.waitMs, decision.remainingMs));
|
|
483
264
|
if (await wait(decision.waitMs) === 'interrupted') {
|
|
484
|
-
//
|
|
485
|
-
// agent is and how to return, then exit as an interrupt (128 + SIGINT).
|
|
265
|
+
// User interrupted; the agent is still running on the peer.
|
|
486
266
|
write(interruptedNotice(opts.target, opts.host.name));
|
|
487
267
|
return 130;
|
|
488
268
|
}
|