@cohortapp/agent-sdk 2.18.14 → 2.18.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/runbooks/fleet-rollout.md +45 -1
- package/lib/assurance/batch.mjs +353 -0
- package/lib/assurance/first-reply.mjs +423 -0
- package/lib/assurance/notice-voice.mjs +357 -0
- package/lib/assurance/plan-note.mjs +43 -0
- package/lib/assurance/room-budget.mjs +55 -6
- package/lib/comms/send-gate.mjs +59 -0
- package/lib/identity/persona.mjs +31 -2
- package/lib/org/inbound/hydrate.mjs +35 -1
- package/lib/org/inbound/project.mjs +3 -0
- package/lib/session/frontdoor.mjs +87 -8
- package/lib/session/handoffs.mjs +57 -0
- package/lib/session/inbox-claims.mjs +47 -2
- package/lib/session/revive.mjs +302 -5
- package/lib/telemetry/alerts.mjs +94 -0
- package/lib/telemetry/collect.mjs +250 -2
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +317 -14
- package/scripts/daemon/assurance.mjs +765 -45
- package/scripts/daemon/deliver.mjs +109 -0
- package/scripts/daemon/dispatcher.mjs +21 -3
- package/scripts/daemon/inbox-deferral.mjs +102 -9
- package/scripts/daemon/session-lock.mjs +41 -1
- package/scripts/fleet/rollout.mjs +292 -9
- package/scripts/hooks/pre-write-yaml-validate.mjs +63 -2
- package/scripts/local-triggers/autoupdate.sh +338 -20
package/lib/session/revive.mjs
CHANGED
|
@@ -22,9 +22,10 @@
|
|
|
22
22
|
*
|
|
23
23
|
* The daemon is the honest home for this. It is a separate process, it is
|
|
24
24
|
* already alive on every seat that beats, it already reads the front-door
|
|
25
|
-
* state every poll to decide who owns the inbox, and it
|
|
26
|
-
*
|
|
27
|
-
*
|
|
25
|
+
* state every poll to decide who owns the inbox, and it re-checks the door on
|
|
26
|
+
* its own sixty-second cadence (REVIVE_CHECK_INTERVAL_MS). A seat whose door
|
|
27
|
+
* shuts is then measured in a minute or two rather than a day, by something
|
|
28
|
+
* the shut door cannot take down with it.
|
|
28
29
|
*
|
|
29
30
|
* Pure: the decision only. The daemon owns the kickstart.
|
|
30
31
|
*/
|
|
@@ -38,19 +39,33 @@ export const DEFAULT_REVIVE_BACKOFF_MS = 15 * 60 * 1000;
|
|
|
38
39
|
/** How many restarts before the daemon stops and leaves it to a person. */
|
|
39
40
|
export const DEFAULT_REVIVE_MAX = 3;
|
|
40
41
|
|
|
42
|
+
/**
|
|
43
|
+
* Consecutive not-live reads required before the FIRST kickstart.
|
|
44
|
+
*
|
|
45
|
+
* A kickstart drops whatever the session was doing. One bad read is not
|
|
46
|
+
* enough evidence to pay that cost — a heartbeat file can be caught
|
|
47
|
+
* mid-write, or a poll can land in the one second between a session ending
|
|
48
|
+
* a tool call and starting the next. Two reads in a row, SIXTY seconds
|
|
49
|
+
* apart at the daemon's revive-check cadence (REVIVE_CHECK_INTERVAL_MS), is
|
|
50
|
+
* the line Ravi drew after the fleet had already seen what a single-read
|
|
51
|
+
* trigger costs.
|
|
52
|
+
*/
|
|
53
|
+
export const DEFAULT_REVIVE_CONFIRM_READS = 2;
|
|
54
|
+
|
|
41
55
|
/**
|
|
42
56
|
* Should the daemon restart the front door right now?
|
|
43
57
|
*
|
|
44
58
|
* The caller supplies the front-door state it already reads for dispatch, the
|
|
45
59
|
* age of the session heartbeat, and what this daemon has already tried. Every
|
|
46
60
|
* bound is explicit because the failure mode of getting this wrong is a seat
|
|
47
|
-
* that restarts its own session every
|
|
61
|
+
* that restarts its own session every sixty seconds forever — which is worse
|
|
48
62
|
* than the shut door, and is the reason the ladder ends in "stop and say so"
|
|
49
63
|
* rather than in another attempt.
|
|
50
64
|
*
|
|
51
65
|
* @param {{frontDoor?:string, sessionLive?:boolean, silentMs?:number|null,
|
|
52
66
|
* attempts?:number, lastAttemptAt?:number|null, now:number,
|
|
53
|
-
* reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number
|
|
67
|
+
* reviveAfterMs?:number, backoffMs?:number, maxAttempts?:number,
|
|
68
|
+
* notLiveReads?:number, confirmReads?:number}} a
|
|
54
69
|
* @returns {{revive:boolean, reason:string}}
|
|
55
70
|
*/
|
|
56
71
|
export function shouldReviveFrontDoor(a) {
|
|
@@ -59,6 +74,7 @@ export function shouldReviveFrontDoor(a) {
|
|
|
59
74
|
const after = Number.isFinite(x.reviveAfterMs) && x.reviveAfterMs > 0 ? x.reviveAfterMs : DEFAULT_REVIVE_AFTER_MS;
|
|
60
75
|
const backoff = Number.isFinite(x.backoffMs) && x.backoffMs > 0 ? x.backoffMs : DEFAULT_REVIVE_BACKOFF_MS;
|
|
61
76
|
const max = Number.isFinite(x.maxAttempts) && x.maxAttempts >= 0 ? x.maxAttempts : DEFAULT_REVIVE_MAX;
|
|
77
|
+
const confirm = Number.isFinite(x.confirmReads) && x.confirmReads > 0 ? x.confirmReads : DEFAULT_REVIVE_CONFIRM_READS;
|
|
62
78
|
|
|
63
79
|
// A seat whose lane is the daemon has no front door to revive: the daemon
|
|
64
80
|
// itself is answering, and restarting a session job it does not depend on
|
|
@@ -82,6 +98,16 @@ export function shouldReviveFrontDoor(a) {
|
|
|
82
98
|
if (!Number.isFinite(silentMs)) return { revive: false, reason: "silence-unknown" };
|
|
83
99
|
if (silentMs < after) return { revive: false, reason: "within-grace" };
|
|
84
100
|
|
|
101
|
+
// Two-read hysteresis, gating only the FIRST kickstart: a kickstart drops
|
|
102
|
+
// whatever the session was mid-way through, so the daemon must not act on
|
|
103
|
+
// a single not-live read before it has confirmed the verdict on the next
|
|
104
|
+
// poll. `notLiveReads` undefined means a caller that predates this streak
|
|
105
|
+
// (every existing test above) — treat it as already confirmed so those
|
|
106
|
+
// callers see no change in behaviour; the daemon is the one caller that
|
|
107
|
+
// threads the real streak through.
|
|
108
|
+
const reads = x.notLiveReads === undefined ? confirm : Number(x.notLiveReads);
|
|
109
|
+
if (!(reads >= confirm)) return { revive: false, reason: "confirming" };
|
|
110
|
+
|
|
85
111
|
const attempts = Number(x.attempts) || 0;
|
|
86
112
|
if (attempts >= max) return { revive: false, reason: "budget-spent" };
|
|
87
113
|
|
|
@@ -93,6 +119,277 @@ export function shouldReviveFrontDoor(a) {
|
|
|
93
119
|
return { revive: true, reason: `front door silent ${Math.round(silentMs / 1000)}s (attempt ${attempts + 1}/${max})` };
|
|
94
120
|
}
|
|
95
121
|
|
|
122
|
+
/**
|
|
123
|
+
* Should this seat's daemon job be restarted? Local check only — the
|
|
124
|
+
* reciprocal of `shouldReviveFrontDoor`'s job, but it CANNOT be run by the
|
|
125
|
+
* daemon on itself: a process that is alive to ask the question is a process
|
|
126
|
+
* launchctl reports a live pid for, so a live daemon can never observe its own
|
|
127
|
+
* down state. The actor is therefore the hourly `autoupdate.sh` backstop (it
|
|
128
|
+
* already parses `launchctl list` for `daemon_last_exit`), which is exactly
|
|
129
|
+
* the seat-local watchdog that catches a daemon that has stopped while the
|
|
130
|
+
* front door kept beating (Jacob's 2026-09-25 case: heartbeat.json ticking, no
|
|
131
|
+
* overdue handoff, daemon dead). `autoupdate.sh#revive_daemon` is the caller.
|
|
132
|
+
*
|
|
133
|
+
* `launchctl` is read by the caller from `launchctl list`, the same source
|
|
134
|
+
* `sibling_job_audit` and `autoupdate.sh` already parse: "pid" when the
|
|
135
|
+
* column holds a number, "dash" when it holds `-` (loaded, no pid), "absent"
|
|
136
|
+
* when the label is not in the list at all.
|
|
137
|
+
*
|
|
138
|
+
* PHANTOM PID: a silent crash loop can leave `launchctl list` holding a pid
|
|
139
|
+
* whose process has genuinely exited — the label still carries a number, but a
|
|
140
|
+
* `kill -0` on that pid fails because no such process is running, a kickstart
|
|
141
|
+
* yields another dead pid, and nothing lands on stderr. A LISTED pid is not a
|
|
142
|
+
* RUNNING daemon. So the caller decides liveness by `kill -0` on the pid and
|
|
143
|
+
* passes `pidAlive`; a phantom pid (`launchctl:"pid"` but `pidAlive:false`) is
|
|
144
|
+
* treated exactly like "dash" — loaded with no live process — and the beat age
|
|
145
|
+
* decides, instead of the listed pid rubber-stamping the daemon as alive.
|
|
146
|
+
*
|
|
147
|
+
* @param {{launchctl?:string, beatAgeMs?:number, staleMs?:number, pidAlive?:boolean}} a
|
|
148
|
+
* @returns {{revive:boolean, reason:string}}
|
|
149
|
+
*/
|
|
150
|
+
export function shouldReviveDaemon(a) {
|
|
151
|
+
const x = a && typeof a === "object" ? a : {};
|
|
152
|
+
const stale = Number.isFinite(x.staleMs) && x.staleMs > 0 ? x.staleMs : DEFAULT_REVIVE_AFTER_MS;
|
|
153
|
+
|
|
154
|
+
// A pid that launchctl lists but the OS does not actually run is a phantom —
|
|
155
|
+
// the same down state as a dash, dressed up as alive. Only an EXPLICIT
|
|
156
|
+
// `pidAlive === false` demotes it; a caller that does not probe liveness
|
|
157
|
+
// (`pidAlive` undefined) keeps the old behaviour, so nothing that never
|
|
158
|
+
// measured it changes.
|
|
159
|
+
const phantom = x.launchctl === "pid" && x.pidAlive === false;
|
|
160
|
+
const state = phantom ? "dash" : x.launchctl;
|
|
161
|
+
|
|
162
|
+
if (state === "pid") return { revive: false, reason: "daemon-alive" };
|
|
163
|
+
|
|
164
|
+
// No job at all is not a wedge to clear — it is a seat that was never
|
|
165
|
+
// provisioned with one, or had it removed, and a kickstart has nothing to
|
|
166
|
+
// aim at.
|
|
167
|
+
if (state === "absent") return { revive: false, reason: "daemon-job-absent" };
|
|
168
|
+
|
|
169
|
+
if (state === "dash") {
|
|
170
|
+
const beatAgeMs = Number(x.beatAgeMs);
|
|
171
|
+
// A loaded job with no live pid and a beat that has gone stale is the down
|
|
172
|
+
// state this exists to catch. A fresh beat under the same launchctl
|
|
173
|
+
// reading means the prior instance is still exiting, not that this one
|
|
174
|
+
// needs help.
|
|
175
|
+
if (!(beatAgeMs > stale)) return { revive: false, reason: "daemon-settling" };
|
|
176
|
+
return { revive: true, reason: phantom ? "daemon-phantom-pid" : "daemon-down" };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Anything else — missing, empty, an error string from a failed parse — is
|
|
180
|
+
// a reading this cannot name a state from, and an unnamed state is never
|
|
181
|
+
// grounds to act.
|
|
182
|
+
return { revive: false, reason: "daemon-state-unknown" };
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** How many consecutive observations across the settle window prove continuity. */
|
|
186
|
+
export const DEFAULT_CONFIRM_SAMPLES = 2;
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Did a revive attempt actually work? Checked from OUTSIDE the restart helper,
|
|
190
|
+
* on purpose: a helper that exits 0 has told you it ran, not that it succeeded,
|
|
191
|
+
* and trusting that exit code is the gap that let a shut front door sit
|
|
192
|
+
* unrestarted for days while the job meant to fix it kept reporting clean.
|
|
193
|
+
*
|
|
194
|
+
* ── WHY A ONE-SHOT PID CHECK IS NOT ENOUGH (Jacob, incident-validated) ──────
|
|
195
|
+
*
|
|
196
|
+
* A single "is there a pid, is the beat fresh" read passes two things that are
|
|
197
|
+
* NOT a revive: a PHANTOM pid (launchctl lists it, the process is dead),
|
|
198
|
+
* and a COME-UP-THEN-CRASH loop that beats once, dies, and is relaunched — its
|
|
199
|
+
* local beat file always looks fresh because the crash-looping restart keeps
|
|
200
|
+
* re-stamping it. The bar is "alive-AND-still-beating-server-side", proven two
|
|
201
|
+
* ways, either of which alone is sufficient:
|
|
202
|
+
*
|
|
203
|
+
* 1. UPTIME CONTINUITY — the one thing a crash loop cannot fake. One pid,
|
|
204
|
+
* alive on every read (`pidAlive`), the SAME process identity (start-time,
|
|
205
|
+
* `ps -o lstart`) across the whole settle window: no restart boundary. A
|
|
206
|
+
* crash loop moves its start-time on every restart, so identity drift is
|
|
207
|
+
* the tell. This alone still admits an alive-but-WEDGED process, so it is
|
|
208
|
+
* paired with a fresh presence beat read back FROM THE SERVER — not the
|
|
209
|
+
* local beat file, which the wedged/crash-looping process keeps stamping.
|
|
210
|
+
* `serverBeatFresh:false` (server reachable, beat stale = alive-but-not-
|
|
211
|
+
* publishing) blocks; `undefined` (could not reach the server) does NOT —
|
|
212
|
+
* recovery is never held hostage to org reachability, continuity carries.
|
|
213
|
+
*
|
|
214
|
+
* 2. A COMPLETED UNIT OF WORK — the STRONGER signal when it exists, and
|
|
215
|
+
* SUFFICIENT on its own (Jacob's positive control: A001 recovered dark →
|
|
216
|
+
* 12 authored posts). A crash loop cannot author a real unit of work.
|
|
217
|
+
*
|
|
218
|
+
* Work is sufficient, not necessary: a seat that recovered into an EMPTY queue
|
|
219
|
+
* does real nothing and MUST still pass on continuity alone — an assert that
|
|
220
|
+
* demanded work would fire a revive on every idle seat overnight and get
|
|
221
|
+
* itself disabled at 3am. Continuity-OR-work separates busy-recovered,
|
|
222
|
+
* crash-loop, phantom, AND quiet-recovered.
|
|
223
|
+
*
|
|
224
|
+
* ── THREE COMPLEMENTARY WITHIN-WINDOW CONDITIONS ────────────────────────────
|
|
225
|
+
*
|
|
226
|
+
* Continuity is three independent checks, each catching a failure the other two
|
|
227
|
+
* cannot: pid-recycling breaks start-time, a crash loop breaks pid, and a WRONG
|
|
228
|
+
* process (right pid, wrong binary) breaks comm. The third exists because `ps`
|
|
229
|
+
* is aliased to `pnpm start` on this fleet — a stray match on the wrong process
|
|
230
|
+
* would otherwise read as a revive. When the caller supplies `expectedComm` and
|
|
231
|
+
* a `comm` per sample (from `/bin/ps -p <pid> -o comm=`), a mismatch fails and
|
|
232
|
+
* NAMES `wrong-comm`; a null comm is unknown, not wrong, and continuity carries
|
|
233
|
+
* it. A caller that passes no `expectedComm` sees the pre-comm-check behaviour.
|
|
234
|
+
*
|
|
235
|
+
* @param {{
|
|
236
|
+
* samples?:Array<{pid?:number, alive?:boolean, startTime?:string|null, comm?:string|null}>,
|
|
237
|
+
* serverBeatFresh?:boolean, workObserved?:boolean, minSamples?:number,
|
|
238
|
+
* expectedComm?:string
|
|
239
|
+
* }} a `samples` are the per-read observations across the settle window;
|
|
240
|
+
* `serverBeatFresh` is a fresh presence beat read back from the SERVER;
|
|
241
|
+
* `workObserved` is a completed unit of work since the kickstart;
|
|
242
|
+
* `expectedComm` is the binary name the revived pid must be running.
|
|
243
|
+
* @returns {{ok:boolean, reason:string}}
|
|
244
|
+
*/
|
|
245
|
+
export function confirmRevived(a) {
|
|
246
|
+
const x = a && typeof a === "object" ? a : {};
|
|
247
|
+
|
|
248
|
+
// Work is sufficient on its own and is the strongest evidence there is — a
|
|
249
|
+
// crash loop cannot produce a completed unit of work.
|
|
250
|
+
if (x.workObserved === true) return { ok: true, reason: "revived-work" };
|
|
251
|
+
|
|
252
|
+
const min = Number.isFinite(x.minSamples) && x.minSamples > 0 ? x.minSamples : DEFAULT_CONFIRM_SAMPLES;
|
|
253
|
+
const samples = Array.isArray(x.samples) ? x.samples : null;
|
|
254
|
+
|
|
255
|
+
// Fewer reads than the settle window needs cannot establish continuity: the
|
|
256
|
+
// process may be about to crash on its next breath. Not a failure to
|
|
257
|
+
// escalate — just not yet confirmed, so the caller re-checks within budget.
|
|
258
|
+
if (!samples || samples.length < min) return { ok: false, reason: "settling" };
|
|
259
|
+
|
|
260
|
+
// A live pid must appear SOMEWHERE in the window. If none ever did — every
|
|
261
|
+
// read dead — the helper claimed a process that is not there: a phantom pid
|
|
262
|
+
// (launchctl lists it, `kill -0` says dead) or an exit-0-with-no-pid. That is
|
|
263
|
+
// the failure-to-escalate that hands the seat to a person.
|
|
264
|
+
const isAlive = (s) => s && Number.isInteger(s.pid) && s.pid > 0 && s.alive === true;
|
|
265
|
+
const aliveSamples = samples.filter(isAlive);
|
|
266
|
+
if (aliveSamples.length === 0) return { ok: false, reason: "no-pid" };
|
|
267
|
+
|
|
268
|
+
// A dead read FOLLOWED BY a live one is a daemon mid-boot, not a failed
|
|
269
|
+
// revive. The first settle sample is taken the instant the kickstart returns,
|
|
270
|
+
// before a normal slow boot has drawn its first breath — so a dead first read
|
|
271
|
+
// then an alive one is boot-in-progress. Latching that as terminal aborts a
|
|
272
|
+
// revive that was working. A window that is not yet all-alive is still
|
|
273
|
+
// SETTLING: re-check within budget rather than escalate. Only two genuinely
|
|
274
|
+
// ALIVE reads can prove — or refute — continuity.
|
|
275
|
+
if (aliveSamples.length < samples.length) return { ok: false, reason: "settling" };
|
|
276
|
+
|
|
277
|
+
// Every read is alive. Same pid AND same start-time across the window = one
|
|
278
|
+
// instance that never restarted. A restart boundary between two ALIVE reads —
|
|
279
|
+
// a different pid, or the same pid with a different start-time (PID reuse by
|
|
280
|
+
// the relaunched process) — is the fingerprint of a crash loop, never a
|
|
281
|
+
// revive. `startTime` may be null on both when `ps` was unavailable;
|
|
282
|
+
// identical-null still agrees, and the pid continuity carries the degraded
|
|
283
|
+
// case.
|
|
284
|
+
const first = aliveSamples[0];
|
|
285
|
+
const continuous = aliveSamples.every((s) => s.pid === first.pid && s.startTime === first.startTime);
|
|
286
|
+
if (!continuous) return { ok: false, reason: "restart-boundary" };
|
|
287
|
+
|
|
288
|
+
// THIRD within-window condition: the binary under the pid must be the one we
|
|
289
|
+
// kickstarted. Right pid, right start-time, WRONG process is still not a
|
|
290
|
+
// revive — and `ps` aliased to `pnpm start` on this fleet makes a stray match
|
|
291
|
+
// real. Opt-in: only when the caller supplies `expectedComm`. A null comm is
|
|
292
|
+
// unknown (ps unavailable / pid vanished), not wrong — continuity carries it,
|
|
293
|
+
// as a null start-time does; only an OBSERVED, mismatching comm fails. Matched
|
|
294
|
+
// by suffix because `ps -o comm=` prints the absolute path and the expected
|
|
295
|
+
// value is a bare binary name.
|
|
296
|
+
if (x.expectedComm) {
|
|
297
|
+
const want = String(x.expectedComm);
|
|
298
|
+
const observed = aliveSamples.filter((s) => s.comm != null);
|
|
299
|
+
const commOk = observed.every((s) => {
|
|
300
|
+
const got = String(s.comm);
|
|
301
|
+
return got === want || got.endsWith(`/${want}`) || want.endsWith(`/${got}`);
|
|
302
|
+
});
|
|
303
|
+
if (!commOk) return { ok: false, reason: "wrong-comm" };
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// Alive and continuous, but is it actually WORKING? The local beat file lies
|
|
307
|
+
// in a wedge, so the confirming signal is the presence beat the server
|
|
308
|
+
// recorded. An explicit stale reading (reachable, not publishing) blocks; an
|
|
309
|
+
// unknown reading (unreachable) does not — continuity carries it.
|
|
310
|
+
if (x.serverBeatFresh === false) return { ok: false, reason: "beat-not-advancing" };
|
|
311
|
+
|
|
312
|
+
return { ok: true, reason: "revived" };
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* A confirm outcome that must STOP the revive and hand the seat to a person,
|
|
317
|
+
* rather than fire another blind kickstart. A second attempt after a crash-loop
|
|
318
|
+
* assert buries the FIRST failure's log under the loop, so these outcomes are
|
|
319
|
+
* terminal: `no-pid` (phantom / exit-0-with-no-pid), `restart-boundary`
|
|
320
|
+
* (came-up-then-crashed within the settle window), `wrong-comm` (a live pid
|
|
321
|
+
* running the wrong binary — not the process we kickstarted), and
|
|
322
|
+
* `crash-loop-cross-poll` (an unrequested restart caught a poll later — the
|
|
323
|
+
* crash loop whose period outran the settle window). `settling` and
|
|
324
|
+
* `beat-not-advancing` are "not yet" — the caller may re-check within budget.
|
|
325
|
+
*
|
|
326
|
+
* @param {{ok?:boolean, reason?:string}} confirmed
|
|
327
|
+
* @returns {boolean}
|
|
328
|
+
*/
|
|
329
|
+
export function isTerminalReviveFailure(confirmed) {
|
|
330
|
+
const r = confirmed && confirmed.reason;
|
|
331
|
+
return r === "no-pid" || r === "restart-boundary" || r === "wrong-comm" || r === "crash-loop-cross-poll";
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
/**
|
|
335
|
+
* Did the session job restart WITHOUT this daemon asking it to, BETWEEN two
|
|
336
|
+
* daemon polls? The unbounded companion to `confirmRevived`'s within-window
|
|
337
|
+
* continuity check.
|
|
338
|
+
*
|
|
339
|
+
* ── WHY WITHIN-WINDOW CONTINUITY IS NOT ENOUGH ──────────────────────────────
|
|
340
|
+
*
|
|
341
|
+
* `confirmRevived` proves continuity across a SETTLE WINDOW (a few seconds). A
|
|
342
|
+
* crash loop whose period EXCEEDS that window shows the SAME identity twice
|
|
343
|
+
* inside it — two alive reads, same pid, same start-time — and false-confirms
|
|
344
|
+
* "revived". A 25s crash-loop period against a 5s settle window is invisible to
|
|
345
|
+
* a within-window check, and WIDENING the window only relocates the threshold:
|
|
346
|
+
* a 65s period beats a 60s window just as cleanly.
|
|
347
|
+
*
|
|
348
|
+
* The fix has no threshold to tune. The daemon persists the last-confirmed
|
|
349
|
+
* {pid, startTime} between its 60s poll iterations. On the NEXT poll, if the job
|
|
350
|
+
* holds a DIFFERENT identity (a different pid, OR the same pid with a moved
|
|
351
|
+
* start-time from PID reuse by the relaunched process) AND this daemon issued NO
|
|
352
|
+
* kickstart in the interval, that is an UNREQUESTED restart — the crash loop, at
|
|
353
|
+
* ANY period. A changed identity WITH a kickstart in between is exactly what a
|
|
354
|
+
* legitimate revive looks like and is never flagged.
|
|
355
|
+
*
|
|
356
|
+
* Pure: the verdict only. The daemon owns the persistence, the kickstart flag,
|
|
357
|
+
* and the halt.
|
|
358
|
+
*
|
|
359
|
+
* @param {{prev?:{pid?:number, startTime?:string|null}|null,
|
|
360
|
+
* curr?:{pid?:number, startTime?:string|null}|null,
|
|
361
|
+
* kickstartIssuedSince?:boolean}} a
|
|
362
|
+
* `prev` is the last poll's confirmed identity (null on the first poll);
|
|
363
|
+
* `curr` is this poll's observed identity; `kickstartIssuedSince` is whether
|
|
364
|
+
* THIS daemon kickstarted the job between the two polls.
|
|
365
|
+
* @returns {{restarted:boolean, reason:string}}
|
|
366
|
+
*/
|
|
367
|
+
export function detectUnrequestedRestart(a) {
|
|
368
|
+
const x = a && typeof a === "object" ? a : {};
|
|
369
|
+
const prev = x.prev && typeof x.prev === "object" ? x.prev : null;
|
|
370
|
+
const curr = x.curr && typeof x.curr === "object" ? x.curr : null;
|
|
371
|
+
|
|
372
|
+
// Nothing to compare against yet — the first poll after boot establishes the
|
|
373
|
+
// baseline, it cannot judge drift from one.
|
|
374
|
+
if (!prev || !(Number.isInteger(prev.pid) && prev.pid > 0)) return { restarted: false, reason: "no-prior" };
|
|
375
|
+
|
|
376
|
+
// No live identity this poll is a DOWN door, not an identity change — that is
|
|
377
|
+
// the revive ladder's job. This function speaks only to drift between two
|
|
378
|
+
// observed identities.
|
|
379
|
+
if (!curr || !(Number.isInteger(curr.pid) && curr.pid > 0)) return { restarted: false, reason: "no-current" };
|
|
380
|
+
|
|
381
|
+
const sameIdentity = curr.pid === prev.pid && curr.startTime === prev.startTime;
|
|
382
|
+
if (sameIdentity) return { restarted: false, reason: "continuous" };
|
|
383
|
+
|
|
384
|
+
// Identity changed. If this daemon asked for it, that is a legitimate revive,
|
|
385
|
+
// not the loop.
|
|
386
|
+
if (x.kickstartIssuedSince === true) return { restarted: false, reason: "kickstart-issued" };
|
|
387
|
+
|
|
388
|
+
// Changed identity, no kickstart in between: an unrequested restart — the
|
|
389
|
+
// crash loop, at whatever period it runs.
|
|
390
|
+
return { restarted: true, reason: "unrequested-restart" };
|
|
391
|
+
}
|
|
392
|
+
|
|
96
393
|
/**
|
|
97
394
|
* The command that restarts the front-door job. Pure — the caller runs it.
|
|
98
395
|
*
|
package/lib/telemetry/alerts.mjs
CHANGED
|
@@ -19,6 +19,10 @@
|
|
|
19
19
|
* - subAgentsRunning over the cap → INFO (more children than expected).
|
|
20
20
|
* - machine.sessionNote present → WARNING/CRITICAL (frontdoor) — the
|
|
21
21
|
* seat's front door is not answering.
|
|
22
|
+
* - machine.upgrade.failStreak >= N → WARNING/CRITICAL (rollout) — N
|
|
23
|
+
* consecutive attempts to reach
|
|
24
|
+
* @latest have failed; the seat is
|
|
25
|
+
* stranded on old code.
|
|
22
26
|
*
|
|
23
27
|
* Output: a deterministic array `[{ id, severity, kind, detail }, ...]` (the
|
|
24
28
|
* snapshot's alert shape) sorted critical→warning→info then by id, so identical
|
|
@@ -90,6 +94,30 @@ export const DEFAULT_THRESHOLDS = Object.freeze({
|
|
|
90
94
|
warnMs: 15 * 60 * 1000,
|
|
91
95
|
critMs: 2 * 60 * 60 * 1000,
|
|
92
96
|
}),
|
|
97
|
+
/**
|
|
98
|
+
* How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK.
|
|
99
|
+
*
|
|
100
|
+
* `warnStreak` 3, not 1: one failure is a bad release or a slow mirror and
|
|
101
|
+
* the hourly job is entitled to try again; the failed-target hold already
|
|
102
|
+
* spaces those out. Three separate, completed attempts have failed, which
|
|
103
|
+
* means every fix the seat can apply to itself has been applied three times
|
|
104
|
+
* and the seat is still where it was.
|
|
105
|
+
*
|
|
106
|
+
* `critStreak` 6: past this the seat has been refusing @latest for the best
|
|
107
|
+
* part of a week under the default 24 h hold — which is exactly how long
|
|
108
|
+
* three seats sat on 2.17.0 while every hourly log line was individually
|
|
109
|
+
* correct. A release that cannot reach a colleague is not a machine problem
|
|
110
|
+
* that resolves itself; it waits for a person, like a scope fault.
|
|
111
|
+
*/
|
|
112
|
+
upgradeStuck: Object.freeze({
|
|
113
|
+
// THE ONE DEFINITION of the stuck threshold. scripts/fleet/rollout.mjs
|
|
114
|
+
// imports `warnStreak` as its STUCK_STREAK rather than restating it, and
|
|
115
|
+
// the shell copy (autoupdate.sh) is pinned against this value by a
|
|
116
|
+
// source-reading test — because three literals agreeing by comment is the
|
|
117
|
+
// prose-promise-instead-of-a-check pattern this repo bans.
|
|
118
|
+
warnStreak: 3,
|
|
119
|
+
critStreak: 6,
|
|
120
|
+
}),
|
|
93
121
|
replyDebt: Object.freeze({
|
|
94
122
|
// ONE withheld reply is already worth saying. This is not a resource gauge
|
|
95
123
|
// where a low reading is normal noise — it counts people who asked this
|
|
@@ -175,6 +203,7 @@ export function deriveAlerts(status, thresholds, now = Date.now()) {
|
|
|
175
203
|
push(alerts, frontDoorRule(machine, t.frontDoor, now));
|
|
176
204
|
push(alerts, scopeFaultRule(machine));
|
|
177
205
|
push(alerts, repliesWithheldRule(machine, t.replyDebt));
|
|
206
|
+
push(alerts, upgradeStuckRule(machine, t.upgradeStuck, now));
|
|
178
207
|
|
|
179
208
|
alerts.sort((a, b) => {
|
|
180
209
|
const r = (SEVERITY_RANK[a.severity] ?? 9) - (SEVERITY_RANK[b.severity] ?? 9);
|
|
@@ -282,6 +311,71 @@ function repliesWithheldRule(machine, cfg = {}) {
|
|
|
282
311
|
};
|
|
283
312
|
}
|
|
284
313
|
|
|
314
|
+
/**
|
|
315
|
+
* THIS SEAT CANNOT GET TO @latest, AND HAS PROVED IT N TIMES.
|
|
316
|
+
*
|
|
317
|
+
* Every other rule in this file reads a gauge. This one reads a COUNT of
|
|
318
|
+
* completed failures, because the failure it names has no gauge: on 2026-09-25
|
|
319
|
+
* three seats were found sitting on 2.17.0 while the fleet ran 2.18.13, and
|
|
320
|
+
* nothing anywhere was in an error state. The launchd job fired hourly, npm
|
|
321
|
+
* answered, the install ran, the health gate did its job, the rollback worked,
|
|
322
|
+
* `last.json` recorded the outcome honestly and the beat carried it. Each hour
|
|
323
|
+
* was a correct, self-contained failure — and the org has no way to see the
|
|
324
|
+
* difference between the first one and the fortieth, which is the only thing
|
|
325
|
+
* that distinguishes "a release is settling" from "a colleague is stranded on
|
|
326
|
+
* code from three weeks ago and will never leave it unaided".
|
|
327
|
+
*
|
|
328
|
+
* WHY THIS RIDES THE EXISTING BEAT AND NOTHING ELSE. The seats this rule is
|
|
329
|
+
* about are, by construction, running the OLDEST code in the fleet — so any
|
|
330
|
+
* new channel it might use is the one channel they do not have. `machine.upgrade`
|
|
331
|
+
* is a field their beat already carries; the count is added to it, and this
|
|
332
|
+
* rule turns it into an alert on the seats new enough to run this file.
|
|
333
|
+
*
|
|
334
|
+
* FOR THE ONES THAT ARE NOT, THIS RULE IS STRUCTURALLY BLIND AND SAYING SO IS
|
|
335
|
+
* PART OF THE DESIGN. A seat too stale to install the SDK runs a collector that
|
|
336
|
+
* never writes `failStreak` and an alerts file that has never heard of this
|
|
337
|
+
* rule — so the alert cannot reach exactly the seats it was written for. The
|
|
338
|
+
* fleet-side derivation covers them from outside: `seatStuck()` in
|
|
339
|
+
* scripts/fleet/rollout.mjs reads the same fact out of a 2.17-era beat, and
|
|
340
|
+
* `propagationOutcome()` there returns reason `"stuck"` so that stage 3 of
|
|
341
|
+
* every publish exits non-zero and names the machine. That path is automatic
|
|
342
|
+
* because publishing is; nobody has to decide to go looking.
|
|
343
|
+
*
|
|
344
|
+
* Both answer (d) the same way: a streak counts ATTEMPTS, and a seat that has
|
|
345
|
+
* not been asked has made none.
|
|
346
|
+
*
|
|
347
|
+
* DISTINCT FROM DRIFT. A `stranded` seat (framework paths pinned in
|
|
348
|
+
* `.maestroignore`, `machine.stranded`) cannot RECEIVE parts of a release it
|
|
349
|
+
* did install; a STUCK seat never installs the release at all. Different
|
|
350
|
+
* cause, different fix, different alert id — they can be true at once.
|
|
351
|
+
*/
|
|
352
|
+
function upgradeStuckRule(machine, cfg = {}, now = Date.now()) {
|
|
353
|
+
const u = machine && machine.upgrade;
|
|
354
|
+
if (!u || typeof u !== "object") return null;
|
|
355
|
+
// Strict: the producer (autoupdate.sh via collect.upgradeSummary) writes an
|
|
356
|
+
// integer, and both validate it. A string here means something else wrote
|
|
357
|
+
// this record, and a rule that coerces would be reporting on a shape it does
|
|
358
|
+
// not understand.
|
|
359
|
+
const n = Number.isInteger(u.failStreak) ? u.failStreak : 0;
|
|
360
|
+
if (!(n > 0)) return null;
|
|
361
|
+
const sev = severityFor(n, num(cfg.warnStreak, Infinity), num(cfg.critStreak, Infinity));
|
|
362
|
+
if (!sev) return null;
|
|
363
|
+
const target = typeof u.streakTarget === "string" && u.streakTarget ? u.streakTarget : (typeof u.to === "string" ? u.to : "@latest");
|
|
364
|
+
// A DURATION, not only a tally: four attempts since Monday and four attempts
|
|
365
|
+
// in the last hour are different seats with different urgency, and the count
|
|
366
|
+
// alone cannot tell them apart.
|
|
367
|
+
const sinceMs = typeof u.stuckSince === "string" ? Date.parse(u.stuckSince) : NaN;
|
|
368
|
+
const days = Number.isFinite(sinceMs) ? Math.floor((now - sinceMs) / 86400000) : null;
|
|
369
|
+
const forHow = days === null ? "" : days >= 1 ? ` over ${days} ${days === 1 ? "day" : "days"}` : " today";
|
|
370
|
+
const why = typeof u.reason === "string" && u.reason ? ` Last failure: ${u.reason.split(":")[0]}.` : "";
|
|
371
|
+
return {
|
|
372
|
+
id: "upgrade_stuck",
|
|
373
|
+
severity: sev,
|
|
374
|
+
kind: "rollout",
|
|
375
|
+
detail: `${n} consecutive upgrade attempts to ${target} have failed${forHow} — this seat is running ${typeof u.from === "string" && u.from ? u.from : "older code"} and will not arrive on its own.${why} It needs a person at the machine.`,
|
|
376
|
+
};
|
|
377
|
+
}
|
|
378
|
+
|
|
285
379
|
function subAgentsRule(status, cfg = {}) {
|
|
286
380
|
const v = num(status.subAgentsRunning);
|
|
287
381
|
const cap = num(cfg.infoCap, Infinity);
|