@cohortapp/agent-sdk 2.18.13 → 2.18.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/bin/maestro.mjs +38 -1
  2. package/docs/runbooks/fleet-rollout.md +58 -7
  3. package/docs/runbooks/recovery-and-failover.md +18 -0
  4. package/lib/assurance/batch.mjs +353 -0
  5. package/lib/assurance/first-reply.mjs +423 -0
  6. package/lib/assurance/notice-voice.mjs +357 -0
  7. package/lib/assurance/plan-note.mjs +43 -0
  8. package/lib/assurance/room-budget.mjs +55 -6
  9. package/lib/cadence-failure-class.mjs +245 -0
  10. package/lib/claude-bin.mjs +26 -7
  11. package/lib/cli/doctor-checks.mjs +149 -1
  12. package/lib/comms/send-gate.mjs +59 -0
  13. package/lib/diagnostics/alerts.mjs +33 -0
  14. package/lib/engine/agents/usage.mjs +45 -0
  15. package/lib/engine/budget.mjs +293 -29
  16. package/lib/engine/cli.mjs +54 -5
  17. package/lib/engine/loop.mjs +30 -0
  18. package/lib/engine/output/json.mjs +26 -0
  19. package/lib/engine/wire/errors.mjs +179 -0
  20. package/lib/engine/wire/search.mjs +44 -8
  21. package/lib/identity/persona.mjs +31 -2
  22. package/lib/org/quota.mjs +27 -0
  23. package/lib/session/config.mjs +4 -0
  24. package/lib/session/identity.mjs +71 -7
  25. package/lib/session/launch-failure.mjs +251 -0
  26. package/lib/session/resume-target.mjs +86 -0
  27. package/lib/telemetry/alerts.mjs +94 -0
  28. package/lib/telemetry/collect.mjs +155 -2
  29. package/lib/upgrade/pinned-drift.mjs +467 -0
  30. package/package.json +1 -1
  31. package/scaffold/config/alerts.yaml +7 -0
  32. package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
  33. package/scripts/ci/check.mjs +3 -0
  34. package/scripts/daemon/agent-daemon.mjs +75 -5
  35. package/scripts/daemon/assurance.mjs +709 -44
  36. package/scripts/daemon/cadence-consumer.mjs +281 -34
  37. package/scripts/daemon/deliver.mjs +109 -0
  38. package/scripts/daemon/dispatcher.mjs +21 -3
  39. package/scripts/daemon/inbox-deferral.mjs +102 -9
  40. package/scripts/daemon/session-lock.mjs +41 -1
  41. package/scripts/emergency-stop.sh +114 -13
  42. package/scripts/fleet/rollout.mjs +256 -10
  43. package/scripts/healthcheck.sh +131 -33
  44. package/scripts/local-triggers/autoupdate.sh +144 -11
  45. package/scripts/resume-operations.sh +101 -6
  46. package/scripts/session/supervisor.mjs +198 -5
@@ -0,0 +1,251 @@
1
+ /**
2
+ * lib/session/launch-failure.mjs — WHY a main-session launch failed, and when
3
+ * the seat should stop trying quietly and say so. Pure; no clock, no fs.
4
+ *
5
+ * ── THE FAULT THIS MODULE EXISTS TO NOT HAVE ────────────────────────────────
6
+ * Measured on James Kirkland's seat, 2026-09-25. The supervisor relaunched
7
+ * every ten minutes for DAYS running
8
+ *
9
+ * claude --resume e953a33e-97a0-4d13-9fd4-6a77cd7a68c9
10
+ * No conversation found with session ID: e953a33e-…
11
+ * exit 1, after ~2 s
12
+ *
13
+ * for a conversation that no longer existed on that machine. Every layer
14
+ * behaved "correctly": launchd relaunched a job that exited non-zero, the
15
+ * rotation budget throttled to one attempt per ten minutes, the log recorded
16
+ * each failure. The seat looked UP the whole time — a live launchd job, a
17
+ * supervisor process, a mux session flickering into existence — and never
18
+ * beat once. Nobody reads that log; nobody could ssh in to read it.
19
+ *
20
+ * Two things were missing and both are here:
21
+ *
22
+ * 1. THE FAILURE HAS A NAME AND THE NAME WAS ON THE SCREEN. "No conversation
23
+ * found with session ID" is not a symptom to infer from timing — it is the
24
+ * runtime telling us the resume target is gone. {@link classifyLaunchFailure}
25
+ * reads it, so the repair (mint a fresh id) is taken on evidence rather
26
+ * than on the 20-second heuristic in `identity#rotationDecision`, which
27
+ * misses a failure that takes 21 seconds and cannot distinguish a dead
28
+ * transcript from a missing binary.
29
+ *
30
+ * 2. N IDENTICAL FAILURES ARE NOT N CHANCES — they are one fault, repeated.
31
+ * A launch that fails the same way three times in a row will fail the
32
+ * fourth time too, and the only useful next step is to tell somebody who
33
+ * is not on this machine. {@link launchFailureStreak} counts consecutive
34
+ * failures WITH THE SAME SIGNATURE across supervisor lifetimes (each
35
+ * launch is a fresh process, so an in-memory counter is always zero —
36
+ * the same trap `first-run#carriedRestarts` exists for), and
37
+ * {@link escalationDecision} says when it stops being a retry and becomes
38
+ * an escalation.
39
+ *
40
+ * The signature deliberately does NOT include the session id: rotating to a
41
+ * fresh id and failing the same way again is the SAME fault (that is exactly
42
+ * what a missing binary does), and a signature that changed every rotation
43
+ * would reset the streak forever — which is how this failure stayed invisible.
44
+ *
45
+ * @module lib/session/launch-failure
46
+ */
47
+
48
+ "use strict";
49
+
50
+ /** Consecutive identical launch failures before the seat escalates. */
51
+ export const IDENTICAL_FAILURE_LIMIT = 3;
52
+
53
+ /**
54
+ * The runtime's own words for "the conversation you asked me to resume is not
55
+ * on this machine". Matched case-insensitively against the tail of the pane /
56
+ * the launcher's stderr. Kept as a list because the wording is the CLI's, not
57
+ * ours, and a second phrasing must be addable without touching the classifier.
58
+ */
59
+ export const RESUME_MISSING_PATTERNS = Object.freeze([
60
+ /no conversation found with session id/i,
61
+ /no conversation found matching/i,
62
+ /session .{0,80}? not found/i,
63
+ ]);
64
+
65
+ /** "the binary is not there" — a shell's 127, and the two spellings of ENOENT. */
66
+ export const BINARY_MISSING_PATTERNS = Object.freeze([
67
+ /command not found/i,
68
+ /no such file or directory/i,
69
+ /\bENOENT\b/,
70
+ ]);
71
+
72
+ const anyMatch = (patterns, text) => patterns.some((re) => re.test(text));
73
+
74
+ /**
75
+ * Pure: name the failure of one launch.
76
+ *
77
+ * `output` is whatever text the launch left behind — a pane capture, the
78
+ * launcher's stderr, or "" when nothing was captured. It is EVIDENCE, and its
79
+ * absence is never treated as proof of anything: with no output the verdict
80
+ * falls back to what the exit code and the mode can carry on their own.
81
+ *
82
+ * @param {object} a
83
+ * @param {"session-id"|"resume"} a.mode how this launch addressed the session
84
+ * @param {number|null} a.exitCode the recorded exit (null = the exit file was never written)
85
+ * @param {string} [a.output] pane tail / stderr, may be ""
86
+ * @param {boolean} [a.beatSeen] did this launch produce a heartbeat?
87
+ * @returns {{kind:"none"|"resume-target-missing"|"binary-missing"|"never-beaten"|"nonzero-exit", proven:boolean, detail:string}}
88
+ */
89
+ export function classifyLaunchFailure(a = {}) {
90
+ const exitCode = a.exitCode === null || a.exitCode === undefined ? null : Number(a.exitCode);
91
+ const output = typeof a.output === "string" ? a.output : "";
92
+ const beatSeen = a.beatSeen === true;
93
+ const failed = exitCode === null || exitCode !== 0;
94
+
95
+ // A launch that beat and then exited 0 is a session that RAN. Nothing to name.
96
+ if (!failed && beatSeen) return { kind: "none", proven: true, detail: "the session ran and exited cleanly" };
97
+ if (!failed) return { kind: "none", proven: false, detail: "the session exited cleanly without beating" };
98
+
99
+ // PROVEN verdicts first: the runtime said what was wrong, in words.
100
+ if (a.mode === "resume" && anyMatch(RESUME_MISSING_PATTERNS, output)) {
101
+ return {
102
+ kind: "resume-target-missing",
103
+ proven: true,
104
+ detail: "the runtime reported that the conversation this launch tried to resume does not exist on this machine",
105
+ };
106
+ }
107
+ if (exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output)) {
108
+ return {
109
+ kind: "binary-missing",
110
+ proven: exitCode === 127 || anyMatch(BINARY_MISSING_PATTERNS, output),
111
+ detail: "the launcher could not find the binary it was told to run",
112
+ };
113
+ }
114
+
115
+ // INFERRED verdicts. `never-beaten` is the honest name for James's seat when
116
+ // the pane was not captured: the launch ended non-zero having never reported
117
+ // in, so whatever it was, it was never a working session.
118
+ if (!beatSeen) {
119
+ return { kind: "never-beaten", proven: false, detail: `the launch ended (exit ${exitCode === null ? "unknown" : exitCode}) without ever writing a heartbeat` };
120
+ }
121
+ return { kind: "nonzero-exit", proven: false, detail: `the session beat, then ended with exit ${exitCode === null ? "unknown" : exitCode}` };
122
+ }
123
+
124
+ /**
125
+ * Pure: the stable identity of a failure, for counting repeats.
126
+ *
127
+ * Mode and kind only. NOT the session id (see the module docblock) and not the
128
+ * exit code when the kind already names the fault — otherwise a flapping exit
129
+ * status would read as a different fault each time and never accumulate.
130
+ *
131
+ * @param {{mode:string, kind:string, exitCode?:number|null}} a
132
+ * @returns {string} e.g. `resume:resume-target-missing`
133
+ */
134
+ export function launchFailureSignature(a = {}) {
135
+ const kind = String(a.kind || "unknown");
136
+ const mode = a.mode === "resume" ? "resume" : "session-id";
137
+ if (kind === "none") return "";
138
+ if (kind === "never-beaten" || kind === "nonzero-exit") {
139
+ const exit = a.exitCode === null || a.exitCode === undefined ? "unknown" : String(Number(a.exitCode));
140
+ return `${mode}:${kind}:${exit}`;
141
+ }
142
+ return `${mode}:${kind}`;
143
+ }
144
+
145
+ /**
146
+ * Pure: parse `state/session/launch-failures.json`. Anything unreadable is
147
+ * "no streak" — a corrupt counter must never be able to escalate on its own.
148
+ * @param {string|null|undefined} text
149
+ * @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null}
150
+ */
151
+ export function parseLaunchFailures(text) {
152
+ if (typeof text !== "string" || !text.trim()) return null;
153
+ let raw;
154
+ try { raw = JSON.parse(text); } catch { return null; }
155
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
156
+ const signature = typeof raw.signature === "string" ? raw.signature : "";
157
+ const streak = Number.isInteger(raw.streak) && raw.streak > 0 ? raw.streak : 0;
158
+ if (!signature || !streak) return null;
159
+ return {
160
+ signature,
161
+ streak,
162
+ firstAt: typeof raw.firstAt === "string" ? raw.firstAt : "",
163
+ lastAt: typeof raw.lastAt === "string" ? raw.lastAt : "",
164
+ escalatedAt: typeof raw.escalatedAt === "string" ? raw.escalatedAt : "",
165
+ };
166
+ }
167
+
168
+ /**
169
+ * Pure: the streak after one more launch outcome.
170
+ *
171
+ * A DIFFERENT signature restarts the count at 1 rather than adding to it: two
172
+ * different faults in a row are two problems, and escalating on their sum
173
+ * would cry wolf. A successful launch (`signature` empty) clears the record
174
+ * entirely — evidence of work, the same reset rule `carriedRestarts` uses.
175
+ *
176
+ * @param {object|null} prior {@link parseLaunchFailures} output
177
+ * @param {{signature:string, at:string}} outcome
178
+ * @returns {{signature:string, streak:number, firstAt:string, lastAt:string, escalatedAt:string}|null} null = clear the file
179
+ */
180
+ export function launchFailureStreak(prior, outcome = {}) {
181
+ const signature = typeof outcome.signature === "string" ? outcome.signature : "";
182
+ const at = typeof outcome.at === "string" ? outcome.at : "";
183
+ if (!signature) return null; // the launch worked: nothing to carry
184
+ if (prior && prior.signature === signature) {
185
+ return { signature, streak: prior.streak + 1, firstAt: prior.firstAt || at, lastAt: at, escalatedAt: prior.escalatedAt || "" };
186
+ }
187
+ return { signature, streak: 1, firstAt: at, lastAt: at, escalatedAt: "" };
188
+ }
189
+
190
+ /**
191
+ * Pure: has this stopped being a retry?
192
+ *
193
+ * `escalate` is true on the launch that REACHES the limit and on every launch
194
+ * after it, because the escalation channel is a beat field that the next
195
+ * supervisor run clears (`attention.json` is unlinked at launch): a one-shot
196
+ * escalation would flicker off and the org would see a seat that healed. The
197
+ * `fresh` flag distinguishes the first crossing, which is the one worth a log
198
+ * line and a distinct wording.
199
+ *
200
+ * @param {{streak:number, limit?:number, escalatedAt?:string}} a
201
+ * @returns {{escalate:boolean, fresh:boolean, streak:number, limit:number, reason:string}}
202
+ */
203
+ export function escalationDecision(a = {}) {
204
+ const limit = Number.isInteger(a.limit) && a.limit > 0 ? a.limit : IDENTICAL_FAILURE_LIMIT;
205
+ const streak = Number.isInteger(a.streak) && a.streak > 0 ? a.streak : 0;
206
+ if (streak < limit) {
207
+ return { escalate: false, fresh: false, streak, limit, reason: `launch failure ${streak}/${limit} — retrying` };
208
+ }
209
+ const fresh = !a.escalatedAt;
210
+ return {
211
+ escalate: true,
212
+ fresh,
213
+ streak,
214
+ limit,
215
+ reason: `${streak} consecutive launches have failed the same way (limit ${limit}) — this is a fault, not a retry`,
216
+ };
217
+ }
218
+
219
+ /**
220
+ * Pure: the one sentence an operator in a browser gets, via the beat's
221
+ * `machine.sessionNote.detail`.
222
+ *
223
+ * THE VERDICT LEADS, THE EXPLANATION TRAILS — because the channel TRUNCATES.
224
+ * `telemetry/collect#sanitizeNoteDetail` caps the detail at 200 characters and
225
+ * appends an ellipsis, so anything after the first sentence may never leave the
226
+ * machine. The first draft of this line ended "…this needs a person" and that
227
+ * clause — the only part that asks for an action — was the part that got cut.
228
+ * So: how many, that retries are not working, then the diagnosis.
229
+ *
230
+ * @param {{signature:string, streak:number, limit?:number, detail:string, sessionId?:string}} a
231
+ * @returns {string}
232
+ */
233
+ export function escalationHint(a = {}) {
234
+ const sig = String(a.signature || "unknown");
235
+ const streak = Number(a.streak) || 0;
236
+ const detail = String(a.detail || "").trim();
237
+ const id = typeof a.sessionId === "string" && a.sessionId ? ` Session ${a.sessionId}.` : "";
238
+ return `${streak}× in a row, identically — retries are not fixing it and this needs a person. ${detail || sig}.${id}`;
239
+ }
240
+
241
+ export default {
242
+ IDENTICAL_FAILURE_LIMIT,
243
+ RESUME_MISSING_PATTERNS,
244
+ BINARY_MISSING_PATTERNS,
245
+ classifyLaunchFailure,
246
+ launchFailureSignature,
247
+ parseLaunchFailures,
248
+ launchFailureStreak,
249
+ escalationDecision,
250
+ escalationHint,
251
+ };
@@ -0,0 +1,86 @@
1
+ /**
2
+ * lib/session/resume-target.mjs — is the conversation we are about to `--resume`
3
+ * actually on this machine?
4
+ *
5
+ * ── WHY THIS IS A PREFLIGHT AND NOT A POST-MORTEM ───────────────────────────
6
+ * The supervisor already had a post-mortem: a resume that died non-zero inside
7
+ * twenty seconds rotates to a fresh id (`identity#rotationDecision`). It is a
8
+ * good heuristic and it is not enough — it cannot tell a dead transcript from a
9
+ * missing binary, it misses a failure that takes twenty-one seconds, and every
10
+ * one of its verdicts costs a full launch, a mux session and a launchd cycle to
11
+ * reach. On James Kirkland's seat that cycle ran every ten minutes for days
12
+ * against a session id that had not existed for just as long.
13
+ *
14
+ * The answer is on disk before we launch. Claude Code keeps each project's
15
+ * transcripts as `~/.claude/projects/<project-slug>/<session-id>.jsonl`, where
16
+ * the slug is the project directory with every non-alphanumeric character
17
+ * replaced by `-`. `--resume` is project-scoped, so the seat's own slug is the
18
+ * only place the id could be.
19
+ *
20
+ * ── FAIL-OPEN, DELIBERATELY ─────────────────────────────────────────────────
21
+ * A wrong "missing" throws away a live transcript; a wrong "unknown" costs one
22
+ * launch and lands in the post-mortem that was already there. So `missing` is
23
+ * returned ONLY from positive evidence: the project directory exists, it holds
24
+ * transcripts, and this id is not among them. An absent directory, an empty
25
+ * one, an unreadable one, an engine that does not use this layout at all — all
26
+ * `unknown`, and the launch proceeds exactly as before.
27
+ *
28
+ * One I/O shell (`readdirSync`/`existsSync`), both injected.
29
+ *
30
+ * @module lib/session/resume-target
31
+ */
32
+
33
+ "use strict";
34
+
35
+ import { readdirSync as fsReaddirSync } from "node:fs";
36
+ import { join } from "node:path";
37
+
38
+ /** Where Claude Code keeps per-project transcripts, relative to the home dir. */
39
+ export const TRANSCRIPT_ROOT = join(".claude", "projects");
40
+
41
+ /**
42
+ * Pure: Claude Code's directory name for a project path — every character that
43
+ * is not a letter or a digit becomes `-` (`/Users/x/hq` → `-Users-x-hq`).
44
+ * @param {string} projectDir
45
+ * @returns {string}
46
+ */
47
+ export function projectSlug(projectDir) {
48
+ return String(projectDir || "").replace(/[^A-Za-z0-9]/g, "-");
49
+ }
50
+
51
+ /**
52
+ * Is `sessionId`'s transcript present for `projectDir`?
53
+ *
54
+ * @param {object} a
55
+ * @param {string} a.homeDir the user's home directory
56
+ * @param {string} a.projectDir the directory the session runs in (the agent root)
57
+ * @param {string} a.sessionId the id this launch would resume
58
+ * @param {"claude"|"cohort"} [a.engine] a non-claude engine does not use this layout → always "unknown"
59
+ * @param {{readdirSync?:Function}} [deps]
60
+ * @returns {{state:"present"|"missing"|"unknown", dir:string, reason:string}}
61
+ */
62
+ export function resumeTargetState(a = {}, deps = {}) {
63
+ const readdirSync = deps.readdirSync || fsReaddirSync;
64
+ const homeDir = typeof a.homeDir === "string" ? a.homeDir : "";
65
+ const sessionId = String(a.sessionId || "").toLowerCase();
66
+ if (a.engine && a.engine !== "claude") return { state: "unknown", dir: "", reason: `engine ${a.engine} does not keep transcripts here` };
67
+ if (!homeDir || !sessionId) return { state: "unknown", dir: "", reason: "no home directory or session id to check" };
68
+ const dir = join(homeDir, TRANSCRIPT_ROOT, projectSlug(a.projectDir));
69
+ let entries;
70
+ try { entries = readdirSync(dir); } catch { return { state: "unknown", dir, reason: "the project's transcript directory is absent or unreadable" }; }
71
+ const names = (Array.isArray(entries) ? entries : []).map((e) => String(e && e.name ? e.name : e));
72
+ const transcripts = names.filter((n) => n.toLowerCase().endsWith(".jsonl"));
73
+ if (transcripts.some((n) => n.toLowerCase() === `${sessionId}.jsonl`)) {
74
+ return { state: "present", dir, reason: "the transcript is on disk" };
75
+ }
76
+ // An EMPTY directory proves nothing: it is equally what a seat looks like the
77
+ // moment before its first session writes, and rotating on it would be a guess.
78
+ if (transcripts.length === 0) return { state: "unknown", dir, reason: "the project's transcript directory holds no transcripts at all" };
79
+ return {
80
+ state: "missing",
81
+ dir,
82
+ reason: `${transcripts.length} transcript(s) on disk for this project and none of them is ${sessionId}`,
83
+ };
84
+ }
85
+
86
+ export default { TRANSCRIPT_ROOT, projectSlug, resumeTargetState };
@@ -19,6 +19,10 @@
19
19
  * - subAgentsRunning over the cap → INFO (more children than expected).
20
20
  * - machine.sessionNote present → WARNING/CRITICAL (frontdoor) — the
21
21
  * seat's front door is not answering.
22
+ * - machine.upgrade.failStreak >= N → WARNING/CRITICAL (rollout) — N
23
+ * consecutive attempts to reach
24
+ * @latest have failed; the seat is
25
+ * stranded on old code.
22
26
  *
23
27
  * Output: a deterministic array `[{ id, severity, kind, detail }, ...]` (the
24
28
  * snapshot's alert shape) sorted critical→warning→info then by id, so identical
@@ -90,6 +94,30 @@ export const DEFAULT_THRESHOLDS = Object.freeze({
90
94
  warnMs: 15 * 60 * 1000,
91
95
  critMs: 2 * 60 * 60 * 1000,
92
96
  }),
97
+ /**
98
+ * How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK.
99
+ *
100
+ * `warnStreak` 3, not 1: one failure is a bad release or a slow mirror and
101
+ * the hourly job is entitled to try again; the failed-target hold already
102
+ * spaces those out. Three separate, completed attempts have failed, which
103
+ * means every fix the seat can apply to itself has been applied three times
104
+ * and the seat is still where it was.
105
+ *
106
+ * `critStreak` 6: past this the seat has been refusing @latest for the best
107
+ * part of a week under the default 24 h hold — which is exactly how long
108
+ * three seats sat on 2.17.0 while every hourly log line was individually
109
+ * correct. A release that cannot reach a colleague is not a machine problem
110
+ * that resolves itself; it waits for a person, like a scope fault.
111
+ */
112
+ upgradeStuck: Object.freeze({
113
+ // THE ONE DEFINITION of the stuck threshold. scripts/fleet/rollout.mjs
114
+ // imports `warnStreak` as its STUCK_STREAK rather than restating it, and
115
+ // the shell copy (autoupdate.sh) is pinned against this value by a
116
+ // source-reading test — because three literals agreeing by comment is the
117
+ // prose-promise-instead-of-a-check pattern this repo bans.
118
+ warnStreak: 3,
119
+ critStreak: 6,
120
+ }),
93
121
  replyDebt: Object.freeze({
94
122
  // ONE withheld reply is already worth saying. This is not a resource gauge
95
123
  // where a low reading is normal noise — it counts people who asked this
@@ -175,6 +203,7 @@ export function deriveAlerts(status, thresholds, now = Date.now()) {
175
203
  push(alerts, frontDoorRule(machine, t.frontDoor, now));
176
204
  push(alerts, scopeFaultRule(machine));
177
205
  push(alerts, repliesWithheldRule(machine, t.replyDebt));
206
+ push(alerts, upgradeStuckRule(machine, t.upgradeStuck, now));
178
207
 
179
208
  alerts.sort((a, b) => {
180
209
  const r = (SEVERITY_RANK[a.severity] ?? 9) - (SEVERITY_RANK[b.severity] ?? 9);
@@ -282,6 +311,71 @@ function repliesWithheldRule(machine, cfg = {}) {
282
311
  };
283
312
  }
284
313
 
314
+ /**
315
+ * THIS SEAT CANNOT GET TO @latest, AND HAS PROVED IT N TIMES.
316
+ *
317
+ * Every other rule in this file reads a gauge. This one reads a COUNT of
318
+ * completed failures, because the failure it names has no gauge: on 2026-09-25
319
+ * three seats were found sitting on 2.17.0 while the fleet ran 2.18.13, and
320
+ * nothing anywhere was in an error state. The launchd job fired hourly, npm
321
+ * answered, the install ran, the health gate did its job, the rollback worked,
322
+ * `last.json` recorded the outcome honestly and the beat carried it. Each hour
323
+ * was a correct, self-contained failure — and the org has no way to see the
324
+ * difference between the first one and the fortieth, which is the only thing
325
+ * that distinguishes "a release is settling" from "a colleague is stranded on
326
+ * code from three weeks ago and will never leave it unaided".
327
+ *
328
+ * WHY THIS RIDES THE EXISTING BEAT AND NOTHING ELSE. The seats this rule is
329
+ * about are, by construction, running the OLDEST code in the fleet — so any
330
+ * new channel it might use is the one channel they do not have. `machine.upgrade`
331
+ * is a field their beat already carries; the count is added to it, and this
332
+ * rule turns it into an alert on the seats new enough to run this file.
333
+ *
334
+ * FOR THE ONES THAT ARE NOT, THIS RULE IS STRUCTURALLY BLIND AND SAYING SO IS
335
+ * PART OF THE DESIGN. A seat too stale to install the SDK runs a collector that
336
+ * never writes `failStreak` and an alerts file that has never heard of this
337
+ * rule — so the alert cannot reach exactly the seats it was written for. The
338
+ * fleet-side derivation covers them from outside: `seatStuck()` in
339
+ * scripts/fleet/rollout.mjs reads the same fact out of a 2.17-era beat, and
340
+ * `propagationOutcome()` there returns reason `"stuck"` so that stage 3 of
341
+ * every publish exits non-zero and names the machine. That path is automatic
342
+ * because publishing is; nobody has to decide to go looking.
343
+ *
344
+ * Both answer (d) the same way: a streak counts ATTEMPTS, and a seat that has
345
+ * not been asked has made none.
346
+ *
347
+ * DISTINCT FROM DRIFT. A `stranded` seat (framework paths pinned in
348
+ * `.maestroignore`, `machine.stranded`) cannot RECEIVE parts of a release it
349
+ * did install; a STUCK seat never installs the release at all. Different
350
+ * cause, different fix, different alert id — they can be true at once.
351
+ */
352
+ function upgradeStuckRule(machine, cfg = {}, now = Date.now()) {
353
+ const u = machine && machine.upgrade;
354
+ if (!u || typeof u !== "object") return null;
355
+ // Strict: the producer (autoupdate.sh via collect.upgradeSummary) writes an
356
+ // integer, and both validate it. A string here means something else wrote
357
+ // this record, and a rule that coerces would be reporting on a shape it does
358
+ // not understand.
359
+ const n = Number.isInteger(u.failStreak) ? u.failStreak : 0;
360
+ if (!(n > 0)) return null;
361
+ const sev = severityFor(n, num(cfg.warnStreak, Infinity), num(cfg.critStreak, Infinity));
362
+ if (!sev) return null;
363
+ const target = typeof u.streakTarget === "string" && u.streakTarget ? u.streakTarget : (typeof u.to === "string" ? u.to : "@latest");
364
+ // A DURATION, not only a tally: four attempts since Monday and four attempts
365
+ // in the last hour are different seats with different urgency, and the count
366
+ // alone cannot tell them apart.
367
+ const sinceMs = typeof u.stuckSince === "string" ? Date.parse(u.stuckSince) : NaN;
368
+ const days = Number.isFinite(sinceMs) ? Math.floor((now - sinceMs) / 86400000) : null;
369
+ const forHow = days === null ? "" : days >= 1 ? ` over ${days} ${days === 1 ? "day" : "days"}` : " today";
370
+ const why = typeof u.reason === "string" && u.reason ? ` Last failure: ${u.reason.split(":")[0]}.` : "";
371
+ return {
372
+ id: "upgrade_stuck",
373
+ severity: sev,
374
+ kind: "rollout",
375
+ detail: `${n} consecutive upgrade attempts to ${target} have failed${forHow} — this seat is running ${typeof u.from === "string" && u.from ? u.from : "older code"} and will not arrive on its own.${why} It needs a person at the machine.`,
376
+ };
377
+ }
378
+
285
379
  function subAgentsRule(status, cfg = {}) {
286
380
  const v = num(status.subAgentsRunning);
287
381
  const cap = num(cfg.infoCap, Infinity);