@cohortapp/agent-sdk 2.18.14 → 2.18.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,9 +22,10 @@
22
22
  *
23
23
  * The daemon is the honest home for this. It is a separate process, it is
24
24
  * already alive on every seat that beats, it already reads the front-door
25
- * state every poll to decide who owns the inbox, and it runs every thirty
26
- * seconds. A seat whose door shuts is then measured in a minute rather than a
27
- * day, by something the shut door cannot take down with it.
25
+ * state every poll to decide who owns the inbox, and it re-checks the door on
26
+ * its own sixty-second cadence (REVIVE_CHECK_INTERVAL_MS). A seat whose door
27
+ * shuts is then measured in a minute or two rather than a day, by something
28
+ * the shut door cannot take down with it.
28
29
  *
29
30
  * Pure: the decision only. The daemon owns the kickstart.
30
31
  */
@@ -38,19 +39,33 @@ export const DEFAULT_REVIVE_BACKOFF_MS = 15 * 60 * 1000;
38
39
  /** How many restarts before the daemon stops and leaves it to a person. */
39
40
  export const DEFAULT_REVIVE_MAX = 3;
40
41
 
42
+ /**
43
+ * Consecutive not-live reads required before the FIRST kickstart.
44
+ *
45
+ * A kickstart drops whatever the session was doing. One bad read is not
46
+ * enough evidence to pay that cost — a heartbeat file can be caught
47
+ * mid-write, or a poll can land in the one second between a session ending
48
+ * a tool call and starting the next. Two reads in a row, SIXTY seconds
49
+ * apart at the daemon's revive-check cadence (REVIVE_CHECK_INTERVAL_MS), is
50
+ * the line Ravi drew after the fleet had already seen what a single-read
51
+ * trigger costs.
52
+ */
53
+ export const DEFAULT_REVIVE_CONFIRM_READS = 2;
54
+
41
55
  /**
42
56
  * Should the daemon restart the front door right now?
43
57
  *
44
58
  * The caller supplies the front-door state it already reads for dispatch, the
45
59
  * age of the session heartbeat, and what this daemon has already tried. Every
46
60
  * bound is explicit because the failure mode of getting this wrong is a seat
47
- * that restarts its own session every thirty seconds forever — which is worse
61
+ * that restarts its own session every sixty seconds forever — which is worse
48
62
  * than the shut door, and is the reason the ladder ends in "stop and say so"
49
63
  * rather than in another attempt.
50
64
  *
51
65
  * @param {{frontDoor?:string, sessionLive?:boolean, silentMs?:number|null,
52
66
  * attempts?:number, lastAttemptAt?:number|null, now:number,
53
- * reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number}} a
67
+ * reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number,
68
+ * notLiveReads?:number, confirmReads?:number}} a
54
69
  * @returns {{revive:boolean, reason:string}}
55
70
  */
56
71
  export function shouldReviveFrontDoor(a) {
@@ -59,6 +74,7 @@ export function shouldReviveFrontDoor(a) {
59
74
  const after = Number.isFinite(x.reviveAfterMs) && x.reviveAfterMs > 0 ? x.reviveAfterMs : DEFAULT_REVIVE_AFTER_MS;
60
75
  const backoff = Number.isFinite(x.backoffMs) && x.backoffMs > 0 ? x.backoffMs : DEFAULT_REVIVE_BACKOFF_MS;
61
76
  const max = Number.isFinite(x.maxAttempts) && x.maxAttempts >= 0 ? x.maxAttempts : DEFAULT_REVIVE_MAX;
77
+ const confirm = Number.isFinite(x.confirmReads) && x.confirmReads > 0 ? x.confirmReads : DEFAULT_REVIVE_CONFIRM_READS;
62
78
 
63
79
  // A seat whose lane is the daemon has no front door to revive: the daemon
64
80
  // itself is answering, and restarting a session job it does not depend on
@@ -82,6 +98,16 @@ export function shouldReviveFrontDoor(a) {
82
98
  if (!Number.isFinite(silentMs)) return { revive: false, reason: "silence-unknown" };
83
99
  if (silentMs < after) return { revive: false, reason: "within-grace" };
84
100
 
101
+ // Two-read hysteresis, gating only the FIRST kickstart: a kickstart drops
102
+ // whatever the session was mid-way through, so the daemon must not act on
103
+ // a single not-live read before it has confirmed the verdict on the next
104
+ // poll. `notLiveReads` undefined means a caller that predates this streak
105
+ // (every existing test above) — treat it as already confirmed so those
106
+ // callers see no change in behaviour; the daemon is the one caller that
107
+ // threads the real streak through.
108
+ const reads = x.notLiveReads === undefined ? confirm : Number(x.notLiveReads);
109
+ if (!(reads >= confirm)) return { revive: false, reason: "confirming" };
110
+
85
111
  const attempts = Number(x.attempts) || 0;
86
112
  if (attempts >= max) return { revive: false, reason: "budget-spent" };
87
113
 
@@ -93,6 +119,277 @@ export function shouldReviveFrontDoor(a) {
93
119
  return { revive: true, reason: `front door silent ${Math.round(silentMs / 1000)}s (attempt ${attempts + 1}/${max})` };
94
120
  }
95
121
 
122
+ /**
123
+ * Should this seat's daemon job be restarted? Local check only — the
124
+ * reciprocal of `shouldReviveFrontDoor`'s job, but it CANNOT be run by the
125
+ * daemon on itself: a process that is alive to ask the question is a process
126
+ * launchctl reports a live pid for, so a live daemon can never observe its own
127
+ * down state. The actor is therefore the hourly `autoupdate.sh` backstop (it
128
+ * already parses `launchctl list` for `daemon_last_exit`), which is exactly
129
+ * the seat-local watchdog that catches a daemon that has stopped while the
130
+ * front door kept beating (Jacob's 2026-09-25 case: heartbeat.json ticking, no
131
+ * overdue handoff, daemon dead). `autoupdate.sh#revive_daemon` is the caller.
132
+ *
133
+ * `launchctl` is read by the caller from `launchctl list`, the same source
134
+ * `sibling_job_audit` and `autoupdate.sh` already parse: "pid" when the
135
+ * column holds a number, "dash" when it holds `-` (loaded, no pid), "absent"
136
+ * when the label is not in the list at all.
137
+ *
138
+ * PHANTOM PID: a silent crash loop can leave `launchctl list` holding a pid
139
+ * whose process has genuinely exited — the label still carries a number, but a
140
+ * `kill -0` on that pid fails because no such process is running, a kickstart
141
+ * yields another dead pid, and nothing lands on stderr. A LISTED pid is not a
142
+ * RUNNING daemon. So the caller decides liveness by `kill -0` on the pid and
143
+ * passes `pidAlive`; a phantom pid (`launchctl:"pid"` but `pidAlive:false`) is
144
+ * treated exactly like "dash" — loaded with no live process — and the beat age
145
+ * decides, instead of the listed pid rubber-stamping the daemon as alive.
146
+ *
147
+ * @param {{launchctl?:string, beatAgeMs?:number, staleMs?:number, pidAlive?:boolean}} a
148
+ * @returns {{revive:boolean, reason:string}}
149
+ */
150
+ export function shouldReviveDaemon(a) {
151
+ const x = a && typeof a === "object" ? a : {};
152
+ const stale = Number.isFinite(x.staleMs) && x.staleMs > 0 ? x.staleMs : DEFAULT_REVIVE_AFTER_MS;
153
+
154
+ // A pid that launchctl lists but the OS does not actually run is a phantom —
155
+ // the same down state as a dash, dressed up as alive. Only an EXPLICIT
156
+ // `pidAlive === false` demotes it; a caller that does not probe liveness
157
+ // (`pidAlive` undefined) keeps the old behaviour, so nothing that never
158
+ // measured it changes.
159
+ const phantom = x.launchctl === "pid" && x.pidAlive === false;
160
+ const state = phantom ? "dash" : x.launchctl;
161
+
162
+ if (state === "pid") return { revive: false, reason: "daemon-alive" };
163
+
164
+ // No job at all is not a wedge to clear — it is a seat that was never
165
+ // provisioned with one, or had it removed, and a kickstart has nothing to
166
+ // aim at.
167
+ if (state === "absent") return { revive: false, reason: "daemon-job-absent" };
168
+
169
+ if (state === "dash") {
170
+ const beatAgeMs = Number(x.beatAgeMs);
171
+ // A loaded job with no live pid and a beat that has gone stale is the down
172
+ // state this exists to catch. A fresh beat under the same launchctl
173
+ // reading means the prior instance is still exiting, not that this one
174
+ // needs help.
175
+ if (!(beatAgeMs > stale)) return { revive: false, reason: "daemon-settling" };
176
+ return { revive: true, reason: phantom ? "daemon-phantom-pid" : "daemon-down" };
177
+ }
178
+
179
+ // Anything else — missing, empty, an error string from a failed parse — is
180
+ // a reading this cannot name a state from, and an unnamed state is never
181
+ // grounds to act.
182
+ return { revive: false, reason: "daemon-state-unknown" };
183
+ }
184
+
185
+ /** How many consecutive observations across the settle window prove continuity. */
186
+ export const DEFAULT_CONFIRM_SAMPLES = 2;
187
+
188
+ /**
189
+ * Did a revive attempt actually work? Checked from OUTSIDE the restart helper,
190
+ * on purpose: a helper that exits 0 has told you it ran, not that it succeeded,
191
+ * and trusting that exit code is the gap that let a shut front door sit
192
+ * unrestarted for days while the job meant to fix it kept reporting clean.
193
+ *
194
+ * ── WHY A ONE-SHOT PID CHECK IS NOT ENOUGH (Jacob, incident-validated) ──────
195
+ *
196
+ * A single "is there a pid, is the beat fresh" read passes two things that are
197
+ * NOT a revive: a PHANTOM pid (launchctl lists it, the process is dead),
198
+ * and a COME-UP-THEN-CRASH loop that beats once, dies, and is relaunched — its
199
+ * local beat file always looks fresh because the crash-looping restart keeps
200
+ * re-stamping it. The bar is "alive-AND-still-beating-server-side", proven two
201
+ * ways, either of which alone is sufficient:
202
+ *
203
+ * 1. UPTIME CONTINUITY — the one thing a crash loop cannot fake. One pid,
204
+ * alive on every read (`pidAlive`), the SAME process identity (start-time,
205
+ * `ps -o lstart`) across the whole settle window: no restart boundary. A
206
+ * crash loop moves its start-time on every restart, so identity drift is
207
+ * the tell. This alone still admits an alive-but-WEDGED process, so it is
208
+ * paired with a fresh presence beat read back FROM THE SERVER — not the
209
+ * local beat file, which the wedged/crash-looping process keeps stamping.
210
+ * `serverBeatFresh:false` (server reachable, beat stale = alive-but-not-
211
+ * publishing) blocks; `undefined` (could not reach the server) does NOT —
212
+ * recovery is never held hostage to org reachability, continuity carries.
213
+ *
214
+ * 2. A COMPLETED UNIT OF WORK — the STRONGER signal when it exists, and
215
+ * SUFFICIENT on its own (Jacob's positive control: A001 recovered dark →
216
+ * 12 authored posts). A crash loop cannot author a real unit of work.
217
+ *
218
+ * Work is sufficient, not necessary: a seat that recovered into an EMPTY queue
219
+ * does real nothing and MUST still pass on continuity alone — an assert that
220
+ * demanded work would fire a revive on every idle seat overnight and get
221
+ * itself disabled at 3am. Continuity-OR-work separates busy-recovered,
222
+ * crash-loop, phantom, AND quiet-recovered.
223
+ *
224
+ * ── THREE COMPLEMENTARY WITHIN-WINDOW CONDITIONS ────────────────────────────
225
+ *
226
+ * Continuity is three independent checks, each catching a failure the other two
227
+ * cannot: pid-recycling breaks start-time, a crash loop breaks pid, and a WRONG
228
+ * process (right pid, wrong binary) breaks comm. The third exists because `ps`
229
+ * is aliased to `pnpm start` on this fleet — a stray match on the wrong process
230
+ * would otherwise read as a revive. When the caller supplies `expectedComm` and
231
+ * a `comm` per sample (from `/bin/ps -p <pid> -o comm=`), a mismatch fails and
232
+ * NAMES `wrong-comm`; a null comm is unknown, not wrong, and continuity carries
233
+ * it. A caller that passes no `expectedComm` sees the pre-comm-check behaviour.
234
+ *
235
+ * @param {{
236
+ * samples?:Array<{pid?:number, alive?:boolean, startTime?:string|null, comm?:string|null}>,
237
+ * serverBeatFresh?:boolean, workObserved?:boolean, minSamples?:number,
238
+ * expectedComm?:string
239
+ * }} a `samples` are the per-read observations across the settle window;
240
+ * `serverBeatFresh` is a fresh presence beat read back from the SERVER;
241
+ * `workObserved` is a completed unit of work since the kickstart;
242
+ * `expectedComm` is the binary name the revived pid must be running.
243
+ * @returns {{ok:boolean, reason:string}}
244
+ */
245
+ export function confirmRevived(a) {
246
+ const x = a && typeof a === "object" ? a : {};
247
+
248
+ // Work is sufficient on its own and is the strongest evidence there is — a
249
+ // crash loop cannot produce a completed unit of work.
250
+ if (x.workObserved === true) return { ok: true, reason: "revived-work" };
251
+
252
+ const min = Number.isFinite(x.minSamples) && x.minSamples > 0 ? x.minSamples : DEFAULT_CONFIRM_SAMPLES;
253
+ const samples = Array.isArray(x.samples) ? x.samples : null;
254
+
255
+ // Fewer reads than the settle window needs cannot establish continuity: the
256
+ // process may be about to crash on its next breath. Not a failure to
257
+ // escalate — just not yet confirmed, so the caller re-checks within budget.
258
+ if (!samples || samples.length < min) return { ok: false, reason: "settling" };
259
+
260
+ // A live pid must appear SOMEWHERE in the window. If none ever did — every
261
+ // read dead — the helper claimed a process that is not there: a phantom pid
262
+ // (launchctl lists it, `kill -0` says dead) or an exit-0-with-no-pid. That is
263
+ // the failure-to-escalate that hands the seat to a person.
264
+ const isAlive = (s) => s && Number.isInteger(s.pid) && s.pid > 0 && s.alive === true;
265
+ const aliveSamples = samples.filter(isAlive);
266
+ if (aliveSamples.length === 0) return { ok: false, reason: "no-pid" };
267
+
268
+ // A dead read FOLLOWED BY a live one is a daemon mid-boot, not a failed
269
+ // revive. The first settle sample is taken the instant the kickstart returns,
270
+ // before a normal slow boot has drawn its first breath — so a dead first read
271
+ // then an alive one is boot-in-progress. Latching that as terminal aborts a
272
+ // revive that was working. A window that is not yet all-alive is still
273
+ // SETTLING: re-check within budget rather than escalate. Only two genuinely
274
+ // ALIVE reads can prove — or refute — continuity.
275
+ if (aliveSamples.length < samples.length) return { ok: false, reason: "settling" };
276
+
277
+ // Every read is alive. Same pid AND same start-time across the window = one
278
+ // instance that never restarted. A restart boundary between two ALIVE reads —
279
+ // a different pid, or the same pid with a different start-time (PID reuse by
280
+ // the relaunched process) — is the fingerprint of a crash loop, never a
281
+ // revive. `startTime` may be null on both when `ps` was unavailable;
282
+ // identical-null still agrees, and the pid continuity carries the degraded
283
+ // case.
284
+ const first = aliveSamples[0];
285
+ const continuous = aliveSamples.every((s) => s.pid === first.pid && s.startTime === first.startTime);
286
+ if (!continuous) return { ok: false, reason: "restart-boundary" };
287
+
288
+ // THIRD within-window condition: the binary under the pid must be the one we
289
+ // kickstarted. Right pid, right start-time, WRONG process is still not a
290
+ // revive — and `ps` aliased to `pnpm start` on this fleet makes a stray match
291
+ // real. Opt-in: only when the caller supplies `expectedComm`. A null comm is
292
+ // unknown (ps unavailable / pid vanished), not wrong — continuity carries it,
293
+ // as a null start-time does; only an OBSERVED, mismatching comm fails. Matched
294
+ // by suffix because `ps -o comm=` prints the absolute path and the expected
295
+ // value is a bare binary name.
296
+ if (x.expectedComm) {
297
+ const want = String(x.expectedComm);
298
+ const observed = aliveSamples.filter((s) => s.comm != null);
299
+ const commOk = observed.every((s) => {
300
+ const got = String(s.comm);
301
+ return got === want || got.endsWith(`/${want}`) || want.endsWith(`/${got}`);
302
+ });
303
+ if (!commOk) return { ok: false, reason: "wrong-comm" };
304
+ }
305
+
306
+ // Alive and continuous, but is it actually WORKING? The local beat file lies
307
+ // in a wedge, so the confirming signal is the presence beat the server
308
+ // recorded. An explicit stale reading (reachable, not publishing) blocks; an
309
+ // unknown reading (unreachable) does not — continuity carries it.
310
+ if (x.serverBeatFresh === false) return { ok: false, reason: "beat-not-advancing" };
311
+
312
+ return { ok: true, reason: "revived" };
313
+ }
314
+
315
+ /**
316
+ * A confirm outcome that must STOP the revive and hand the seat to a person,
317
+ * rather than fire another blind kickstart. A second attempt after a crash-loop
318
+ * assert buries the FIRST failure's log under the loop, so these outcomes are
319
+ * terminal: `no-pid` (phantom / exit-0-with-no-pid), `restart-boundary`
320
+ * (came-up-then-crashed within the settle window), `wrong-comm` (a live pid
321
+ * running the wrong binary — not the process we kickstarted), and
322
+ * `crash-loop-cross-poll` (an unrequested restart caught a poll later — the
323
+ * crash loop whose period outran the settle window). `settling` and
324
+ * `beat-not-advancing` are "not yet" — the caller may re-check within budget.
325
+ *
326
+ * @param {{ok?:boolean, reason?:string}} confirmed
327
+ * @returns {boolean}
328
+ */
329
+ export function isTerminalReviveFailure(confirmed) {
330
+ const r = confirmed && confirmed.reason;
331
+ return r === "no-pid" || r === "restart-boundary" || r === "wrong-comm" || r === "crash-loop-cross-poll";
332
+ }
333
+
334
+ /**
335
+ * Did the session job restart WITHOUT this daemon asking it to, BETWEEN two
336
+ * daemon polls? The unbounded companion to `confirmRevived`'s within-window
337
+ * continuity check.
338
+ *
339
+ * ── WHY WITHIN-WINDOW CONTINUITY IS NOT ENOUGH ──────────────────────────────
340
+ *
341
+ * `confirmRevived` proves continuity across a SETTLE WINDOW (a few seconds). A
342
+ * crash loop whose period EXCEEDS that window shows the SAME identity twice
343
+ * inside it — two alive reads, same pid, same start-time — and false-confirms
344
+ * "revived". A 25s crash-loop period against a 5s settle window is invisible to
345
+ * a within-window check, and WIDENING the window only relocates the threshold:
346
+ * a 65s period beats a 60s window just as cleanly.
347
+ *
348
+ * The fix has no threshold to tune. The daemon persists the last-confirmed
349
+ * {pid, startTime} between its 60s poll iterations. On the NEXT poll, if the job
350
+ * holds a DIFFERENT identity (a different pid, OR the same pid with a moved
351
+ * start-time from PID reuse by the relaunched process) AND this daemon issued NO
352
+ * kickstart in the interval, that is an UNREQUESTED restart — the crash loop, at
353
+ * ANY period. A changed identity WITH a kickstart in between is exactly what a
354
+ * legitimate revive looks like and is never flagged.
355
+ *
356
+ * Pure: the verdict only. The daemon owns the persistence, the kickstart flag,
357
+ * and the halt.
358
+ *
359
+ * @param {{prev?:{pid?:number, startTime?:string|null}|null,
360
+ * curr?:{pid?:number, startTime?:string|null}|null,
361
+ * kickstartIssuedSince?:boolean}} a
362
+ * `prev` is the last poll's confirmed identity (null on the first poll);
363
+ * `curr` is this poll's observed identity; `kickstartIssuedSince` is whether
364
+ * THIS daemon kickstarted the job between the two polls.
365
+ * @returns {{restarted:boolean, reason:string}}
366
+ */
367
+ export function detectUnrequestedRestart(a) {
368
+ const x = a && typeof a === "object" ? a : {};
369
+ const prev = x.prev && typeof x.prev === "object" ? x.prev : null;
370
+ const curr = x.curr && typeof x.curr === "object" ? x.curr : null;
371
+
372
+ // Nothing to compare against yet — the first poll after boot establishes the
373
+ // baseline, it cannot judge drift from one.
374
+ if (!prev || !(Number.isInteger(prev.pid) && prev.pid > 0)) return { restarted: false, reason: "no-prior" };
375
+
376
+ // No live identity this poll is a DOWN door, not an identity change — that is
377
+ // the revive ladder's job. This function speaks only to drift between two
378
+ // observed identities.
379
+ if (!curr || !(Number.isInteger(curr.pid) && curr.pid > 0)) return { restarted: false, reason: "no-current" };
380
+
381
+ const sameIdentity = curr.pid === prev.pid && curr.startTime === prev.startTime;
382
+ if (sameIdentity) return { restarted: false, reason: "continuous" };
383
+
384
+ // Identity changed. If this daemon asked for it, that is a legitimate revive,
385
+ // not the loop.
386
+ if (x.kickstartIssuedSince === true) return { restarted: false, reason: "kickstart-issued" };
387
+
388
+ // Changed identity, no kickstart in between: an unrequested restart — the
389
+ // crash loop, at whatever period it runs.
390
+ return { restarted: true, reason: "unrequested-restart" };
391
+ }
392
+
96
393
  /**
97
394
  * The command that restarts the front-door job. Pure — the caller runs it.
98
395
  *
@@ -19,6 +19,10 @@
19
19
  * - subAgentsRunning over the cap → INFO (more children than expected).
20
20
  * - machine.sessionNote present → WARNING/CRITICAL (frontdoor) — the
21
21
  * seat's front door is not answering.
22
+ * - machine.upgrade.failStreak >= N → WARNING/CRITICAL (rollout) — N
23
+ * consecutive attempts to reach
24
+ * @latest have failed; the seat is
25
+ * stranded on old code.
22
26
  *
23
27
  * Output: a deterministic array `[{ id, severity, kind, detail }, ...]` (the
24
28
  * snapshot's alert shape) sorted critical→warning→info then by id, so identical
@@ -90,6 +94,30 @@ export const DEFAULT_THRESHOLDS = Object.freeze({
90
94
  warnMs: 15 * 60 * 1000,
91
95
  critMs: 2 * 60 * 60 * 1000,
92
96
  }),
97
+ /**
98
+ * How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK.
99
+ *
100
+ * `warnStreak` 3, not 1: one failure is a bad release or a slow mirror and
101
+ * the hourly job is entitled to try again; the failed-target hold already
102
+ * spaces those out. Three separate, completed attempts have failed, which
103
+ * means every fix the seat can apply to itself has been applied three times
104
+ * and the seat is still where it was.
105
+ *
106
+ * `critStreak` 6: past this the seat has been refusing @latest for the best
107
+ * part of a week under the default 24 h hold — which is exactly how long
108
+ * three seats sat on 2.17.0 while every hourly log line was individually
109
+ * correct. A release that cannot reach a colleague is not a machine problem
110
+ * that resolves itself; it waits for a person, like a scope fault.
111
+ */
112
+ upgradeStuck: Object.freeze({
113
+ // THE ONE DEFINITION of the stuck threshold. scripts/fleet/rollout.mjs
114
+ // imports `warnStreak` as its STUCK_STREAK rather than restating it, and
115
+ // the shell copy (autoupdate.sh) is pinned against this value by a
116
+ // source-reading test — because three literals agreeing by comment is the
117
+ // prose-promise-instead-of-a-check pattern this repo bans.
118
+ warnStreak: 3,
119
+ critStreak: 6,
120
+ }),
93
121
  replyDebt: Object.freeze({
94
122
  // ONE withheld reply is already worth saying. This is not a resource gauge
95
123
  // where a low reading is normal noise — it counts people who asked this
@@ -175,6 +203,7 @@ export function deriveAlerts(status, thresholds, now = Date.now()) {
175
203
  push(alerts, frontDoorRule(machine, t.frontDoor, now));
176
204
  push(alerts, scopeFaultRule(machine));
177
205
  push(alerts, repliesWithheldRule(machine, t.replyDebt));
206
+ push(alerts, upgradeStuckRule(machine, t.upgradeStuck, now));
178
207
 
179
208
  alerts.sort((a, b) => {
180
209
  const r = (SEVERITY_RANK[a.severity] ?? 9) - (SEVERITY_RANK[b.severity] ?? 9);
@@ -282,6 +311,71 @@ function repliesWithheldRule(machine, cfg = {}) {
282
311
  };
283
312
  }
284
313
 
314
+ /**
315
+ * THIS SEAT CANNOT GET TO @latest, AND HAS PROVED IT N TIMES.
316
+ *
317
+ * Every other rule in this file reads a gauge. This one reads a COUNT of
318
+ * completed failures, because the failure it names has no gauge: on 2026-09-25
319
+ * three seats were found sitting on 2.17.0 while the fleet ran 2.18.13, and
320
+ * nothing anywhere was in an error state. The launchd job fired hourly, npm
321
+ * answered, the install ran, the health gate did its job, the rollback worked,
322
+ * `last.json` recorded the outcome honestly and the beat carried it. Each hour
323
+ * was a correct, self-contained failure — and the org has no way to see the
324
+ * difference between the first one and the fortieth, which is the only thing
325
+ * that distinguishes "a release is settling" from "a colleague is stranded on
326
+ * code from three weeks ago and will never leave it unaided".
327
+ *
328
+ * WHY THIS RIDES THE EXISTING BEAT AND NOTHING ELSE. The seats this rule is
329
+ * about are, by construction, running the OLDEST code in the fleet — so any
330
+ * new channel it might use is the one channel they do not have. `machine.upgrade`
331
+ * is a field their beat already carries; the count is added to it, and this
332
+ * rule turns it into an alert on the seats new enough to run this file.
333
+ *
334
+ * FOR THE ONES THAT ARE NOT, THIS RULE IS STRUCTURALLY BLIND AND SAYING SO IS
335
+ * PART OF THE DESIGN. A seat too stale to install the SDK runs a collector that
336
+ * never writes `failStreak` and an alerts file that has never heard of this
337
+ * rule — so the alert cannot reach exactly the seats it was written for. The
338
+ * fleet-side derivation covers them from outside: `seatStuck()` in
339
+ * scripts/fleet/rollout.mjs reads the same fact out of a 2.17-era beat, and
340
+ * `propagationOutcome()` there returns reason `"stuck"` so that stage 3 of
341
+ * every publish exits non-zero and names the machine. That path is automatic
342
+ * because publishing is; nobody has to decide to go looking.
343
+ *
344
+ * Both answer (d) the same way: a streak counts ATTEMPTS, and a seat that has
345
+ * not been asked has made none.
346
+ *
347
+ * DISTINCT FROM DRIFT. A `stranded` seat (framework paths pinned in
348
+ * `.maestroignore`, `machine.stranded`) cannot RECEIVE parts of a release it
349
+ * did install; a STUCK seat never installs the release at all. Different
350
+ * cause, different fix, different alert id — they can be true at once.
351
+ */
352
+ function upgradeStuckRule(machine, cfg = {}, now = Date.now()) {
353
+ const u = machine && machine.upgrade;
354
+ if (!u || typeof u !== "object") return null;
355
+ // Strict: the producer (autoupdate.sh via collect.upgradeSummary) writes an
356
+ // integer, and both validate it. A string here means something else wrote
357
+ // this record, and a rule that coerces would be reporting on a shape it does
358
+ // not understand.
359
+ const n = Number.isInteger(u.failStreak) ? u.failStreak : 0;
360
+ if (!(n > 0)) return null;
361
+ const sev = severityFor(n, num(cfg.warnStreak, Infinity), num(cfg.critStreak, Infinity));
362
+ if (!sev) return null;
363
+ const target = typeof u.streakTarget === "string" && u.streakTarget ? u.streakTarget : (typeof u.to === "string" ? u.to : "@latest");
364
+ // A DURATION, not only a tally: four attempts since Monday and four attempts
365
+ // in the last hour are different seats with different urgency, and the count
366
+ // alone cannot tell them apart.
367
+ const sinceMs = typeof u.stuckSince === "string" ? Date.parse(u.stuckSince) : NaN;
368
+ const days = Number.isFinite(sinceMs) ? Math.floor((now - sinceMs) / 86400000) : null;
369
+ const forHow = days === null ? "" : days >= 1 ? ` over ${days} ${days === 1 ? "day" : "days"}` : " today";
370
+ const why = typeof u.reason === "string" && u.reason ? ` Last failure: ${u.reason.split(":")[0]}.` : "";
371
+ return {
372
+ id: "upgrade_stuck",
373
+ severity: sev,
374
+ kind: "rollout",
375
+ detail: `${n} consecutive upgrade attempts to ${target} have failed${forHow} — this seat is running ${typeof u.from === "string" && u.from ? u.from : "older code"} and will not arrive on its own.${why} It needs a person at the machine.`,
376
+ };
377
+ }
378
+
285
379
  function subAgentsRule(status, cfg = {}) {
286
380
  const v = num(status.subAgentsRunning);
287
381
  const cap = num(cfg.infoCap, Infinity);