@cohortapp/agent-sdk 2.18.14 → 2.18.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/runbooks/fleet-rollout.md +45 -1
- package/lib/assurance/batch.mjs +353 -0
- package/lib/assurance/first-reply.mjs +423 -0
- package/lib/assurance/notice-voice.mjs +357 -0
- package/lib/assurance/plan-note.mjs +43 -0
- package/lib/assurance/room-budget.mjs +55 -6
- package/lib/comms/send-gate.mjs +59 -0
- package/lib/identity/persona.mjs +31 -2
- package/lib/org/inbound/hydrate.mjs +35 -1
- package/lib/org/inbound/project.mjs +3 -0
- package/lib/session/frontdoor.mjs +87 -8
- package/lib/session/handoffs.mjs +57 -0
- package/lib/session/inbox-claims.mjs +47 -2
- package/lib/session/revive.mjs +302 -5
- package/lib/telemetry/alerts.mjs +94 -0
- package/lib/telemetry/collect.mjs +250 -2
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +317 -14
- package/scripts/daemon/assurance.mjs +765 -45
- package/scripts/daemon/deliver.mjs +109 -0
- package/scripts/daemon/dispatcher.mjs +21 -3
- package/scripts/daemon/inbox-deferral.mjs +102 -9
- package/scripts/daemon/session-lock.mjs +41 -1
- package/scripts/fleet/rollout.mjs +292 -9
- package/scripts/hooks/pre-write-yaml-validate.mjs +63 -2
- package/scripts/local-triggers/autoupdate.sh +338 -20
|
@@ -43,7 +43,17 @@
|
|
|
43
43
|
# STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
|
|
44
44
|
# (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
|
|
45
45
|
# loop cannot straddle both samples by luck, and well inside the hourly
|
|
46
|
-
# cadence. A 25 s loop fails on the second sample and on uptime.
|
|
46
|
+
# cadence. A 25 s loop fails on the second sample and on uptime. A pid
|
|
47
|
+
# CHANGE within the window is not, on its own, proof of that loop: this
|
|
48
|
+
# script's own restart_daemon (or an operator's `launchctl kickstart -k`)
|
|
49
|
+
# changes the pid deliberately and can land inside the same window —
|
|
50
|
+
# 2.18.14 read that as a crash loop and rolled back a release that had
|
|
51
|
+
# done nothing wrong. daemon_last_exit() tells the two apart: a crash
|
|
52
|
+
# loop's reaped exit is a real non-zero every time; a deliberate
|
|
53
|
+
# kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
|
|
54
|
+
# real non-zero exit fails here on the pid change alone — a clean or
|
|
55
|
+
# unknown exit falls through to the uptime check, which still catches a
|
|
56
|
+
# genuinely unstable respawn (its pid is always younger than STABLE_S).
|
|
47
57
|
# (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
|
|
48
58
|
# lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
|
|
49
59
|
# ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
|
|
@@ -270,6 +280,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
|
|
|
270
280
|
HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
|
|
271
281
|
DAEMON_LABEL=""
|
|
272
282
|
SESSION_LABEL=""
|
|
283
|
+
# The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
|
|
284
|
+
# below — the on-host backstop for a daemon that is DOWN, which a live daemon
|
|
285
|
+
# can never observe about itself. Defined up here (not with SESSION_STALE_S,
|
|
286
|
+
# which lives in the not-up-to-date path) because revive_daemon fires on the
|
|
287
|
+
# UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
|
|
288
|
+
# 90s front-door observers.
|
|
289
|
+
LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
|
|
290
|
+
DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
|
|
291
|
+
# Gap between the two settle-window reads that prove the kickstarted daemon is
|
|
292
|
+
# one continuous process, not a come-up-then-crash loop. A one-shot read would
|
|
293
|
+
# pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
|
|
294
|
+
DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
|
|
273
295
|
|
|
274
296
|
daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
|
|
275
297
|
pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
|
|
@@ -355,14 +377,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
|
|
|
355
377
|
[ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
|
|
356
378
|
return 0
|
|
357
379
|
}
|
|
380
|
+
daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
|
|
381
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
|
|
382
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
383
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
|
|
384
|
+
}
|
|
358
385
|
is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
|
|
359
|
-
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
|
|
386
|
+
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
|
|
360
387
|
HEALTH_REASON=""
|
|
361
388
|
p1="$(daemon_pids)"
|
|
362
389
|
[ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
|
|
363
390
|
[ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
|
|
364
391
|
p2="$(daemon_pids)"
|
|
365
|
-
[ "$p1"
|
|
392
|
+
if [ "$p1" != "$p2" ]; then
|
|
393
|
+
# A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
|
|
394
|
+
# (this same script's own restart_daemon, or an operator's) changes the pid
|
|
395
|
+
# deliberately and can land inside this very window. 2.18.14 read that as a
|
|
396
|
+
# crash loop and rolled back a release that had done nothing wrong. The
|
|
397
|
+
# daemon's OWN last exit says which one happened: a crash-looping process
|
|
398
|
+
# exits non-zero every time launchd reaps it; a clean kickstart's reaped
|
|
399
|
+
# exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
|
|
400
|
+
# non-zero exit is a crash loop here; a clean or unknown exit is treated as
|
|
401
|
+
# an intended restart and falls through to the uptime check below, which
|
|
402
|
+
# still catches a genuinely unstable respawn (a crash-looping daemon's new
|
|
403
|
+
# pid is always younger than STABLE_S).
|
|
404
|
+
exit_code="$(daemon_last_exit)"
|
|
405
|
+
case "$exit_code" in
|
|
406
|
+
""|"-"|"0")
|
|
407
|
+
# A clean/unknown exit with NO new pid at all is not a restart in
|
|
408
|
+
# progress, it is the daemon gone — the "no daemon process" case, not a
|
|
409
|
+
# crash loop and not healthy.
|
|
410
|
+
[ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
|
|
411
|
+
log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
|
|
412
|
+
p1="$p2"
|
|
413
|
+
;;
|
|
414
|
+
*)
|
|
415
|
+
HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
|
|
416
|
+
return 1
|
|
417
|
+
;;
|
|
418
|
+
esac
|
|
419
|
+
fi
|
|
366
420
|
oldest=""
|
|
367
421
|
for pid in $p1; do
|
|
368
422
|
up="$(pid_uptime_s "$pid")"
|
|
@@ -450,19 +504,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
|
|
|
450
504
|
# Leave that one to session_reconciled, which already names it.
|
|
451
505
|
label_loaded "$SESSION_LABEL" || return 0
|
|
452
506
|
session_beating && return 0
|
|
453
|
-
|
|
507
|
+
# Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
|
|
508
|
+
# not enough to drop a session, because a supervisor resume gaps the beat for
|
|
509
|
+
# a moment. Confirm the door is STILL silent on a second read 60s later
|
|
510
|
+
# before kickstarting — if it beat again in between, that was a resume gap,
|
|
511
|
+
# not a wedge, and the session keeps its context.
|
|
512
|
+
log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
|
|
513
|
+
sleep "$SESSION_REVIVE_RECHECK_S"
|
|
514
|
+
session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
|
|
515
|
+
# Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
|
|
516
|
+
# note. The supervisor holds the `session` singleton (state/locks/process/
|
|
517
|
+
# session.pid) for its whole life and beats as a healthy resume progresses.
|
|
518
|
+
# We are past the FULL grace and two reads apart, so a resume gap is already
|
|
519
|
+
# excluded by time — a live holder here is a supervisor PARKED in its launch
|
|
520
|
+
# probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
|
|
521
|
+
# Consulting the lock tells a parked supervisor (override, logged) from a dead
|
|
522
|
+
# one (clean restart); it never blocks the parked case a hard skip would.
|
|
523
|
+
local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
|
|
524
|
+
[ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
|
|
525
|
+
if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
|
|
526
|
+
log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
|
|
527
|
+
fi
|
|
528
|
+
log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
|
|
454
529
|
launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
|
|
455
530
|
&& { log "front door $SESSION_LABEL restarted"; return 0; }
|
|
456
531
|
log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
|
|
457
532
|
return 1
|
|
458
533
|
}
|
|
459
534
|
|
|
535
|
+
daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
|
|
536
|
+
[ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
|
|
537
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
|
|
538
|
+
NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
|
|
539
|
+
END { if (!f) print "absent" }'
|
|
540
|
+
}
|
|
541
|
+
daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
|
|
542
|
+
node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
|
|
543
|
+
"$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
|
|
544
|
+
}
|
|
545
|
+
REVIVE_NOTE_REL="state/telemetry/revive-note.json"
|
|
546
|
+
write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
|
|
547
|
+
local dir="$AGENT_DIR/state/telemetry" tmp
|
|
548
|
+
mkdir -p "$dir" 2>/dev/null || return 0
|
|
549
|
+
tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
|
|
550
|
+
printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
|
|
551
|
+
"$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
|
|
552
|
+
&& mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
|
|
553
|
+
return 0
|
|
554
|
+
}
|
|
555
|
+
daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
|
|
556
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
557
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
|
|
558
|
+
}
|
|
559
|
+
daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
|
|
560
|
+
local pid start
|
|
561
|
+
pid="$(daemon_launchctl_pid)"
|
|
562
|
+
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
563
|
+
start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
|
|
564
|
+
printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
|
|
565
|
+
elif [ -n "$pid" ]; then
|
|
566
|
+
printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
|
|
567
|
+
else
|
|
568
|
+
printf '{"pid":null,"alive":false,"startTime":null}'
|
|
569
|
+
fi
|
|
570
|
+
}
|
|
571
|
+
revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
|
|
572
|
+
# is_healthy bails at the empty-pid check before daemon_last_exit (its label
|
|
573
|
+
# resolver) runs, so resolve them here — the down case is exactly the one
|
|
574
|
+
# where nothing upstream did.
|
|
575
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels
|
|
576
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
577
|
+
[ -f "$LIB_REVIVE" ] || return 0
|
|
578
|
+
# ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
|
|
579
|
+
# still has a pid — that is the health gate's business, and a kickstart would
|
|
580
|
+
# fight the very restart it is doing. pgrep-empty is Jacob's lane.
|
|
581
|
+
[ -z "$(daemon_pids)" ] || return 0
|
|
582
|
+
local st beat lpid palive verdict
|
|
583
|
+
st="$(daemon_launchctl_state)"
|
|
584
|
+
beat="$(daemon_beat_age_ms)"
|
|
585
|
+
# PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
|
|
586
|
+
# A listed pid is not a running daemon — verify it, and pass pidAlive so
|
|
587
|
+
# shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
|
|
588
|
+
# stamping it alive (which would deadlock: pgrep-empty says down, launchctl
|
|
589
|
+
# says pid, and nothing revives).
|
|
590
|
+
lpid="$(daemon_launchctl_pid)"
|
|
591
|
+
if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
|
|
592
|
+
verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
|
|
593
|
+
case "$verdict" in
|
|
594
|
+
revive) : ;;
|
|
595
|
+
*) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
|
|
596
|
+
esac
|
|
597
|
+
log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
|
|
598
|
+
launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
|
|
599
|
+
# ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
|
|
600
|
+
# passes a phantom pid AND a come-up-then-crash whose local beat it keeps
|
|
601
|
+
# stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
|
|
602
|
+
# live process (same pid, same start-time, alive on both) — the uptime
|
|
603
|
+
# continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
|
|
604
|
+
# hourly backstop does not round-trip the org just to confirm); continuity
|
|
605
|
+
# carries it, and the daemon's own presence publish is the fleet-visible proof.
|
|
606
|
+
local s1 s2 samples ok
|
|
607
|
+
s1="$(daemon_sample_json)"
|
|
608
|
+
sleep "$DAEMON_CONFIRM_SETTLE_S"
|
|
609
|
+
s2="$(daemon_sample_json)"
|
|
610
|
+
samples="[$s1,$s2]"
|
|
611
|
+
ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
|
|
612
|
+
if [ "$ok" = "ok" ]; then
|
|
613
|
+
log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
|
|
614
|
+
rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
|
|
615
|
+
return 0
|
|
616
|
+
fi
|
|
617
|
+
# ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
|
|
618
|
+
# this failure's log under the loop). One attempt, then hand up with what it
|
|
619
|
+
# saw: the launchctl list line and the tail of the daemon's own log.
|
|
620
|
+
local lc_line dlog_tail evidence
|
|
621
|
+
lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
|
|
622
|
+
dlog_tail="$(tail -n 3 "$DLOG_DIR/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
|
|
623
|
+
evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
|
|
624
|
+
log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
|
|
625
|
+
write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
|
|
626
|
+
return 1
|
|
627
|
+
}
|
|
460
628
|
session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
|
|
461
629
|
label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
|
|
462
630
|
pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
|
|
463
631
|
return 0
|
|
464
632
|
}
|
|
465
|
-
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
|
|
633
|
+
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
|
|
466
634
|
write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
|
|
467
635
|
local dir="$AGENT_DIR/state/session" tmp
|
|
468
636
|
mkdir -p "$dir" 2>/dev/null || return 0
|
|
@@ -481,18 +649,133 @@ write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell t
|
|
|
481
649
|
}
|
|
482
650
|
write_upgrade_notice(){ write_notice "$1" "$2"; }
|
|
483
651
|
LAST_JSON="$AGENT_DIR/state/autoupdate/last.json"
|
|
484
|
-
|
|
485
|
-
|
|
652
|
+
# How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK, and
|
|
653
|
+
# at what point it stops being a warning. See the `failStreak` block below.
|
|
654
|
+
STUCK_STREAK="${MAESTRO_AUTOUPDATE_STUCK_STREAK:-3}"
|
|
655
|
+
# ── failStreak — "this seat cannot get to @latest" is a FACT, not a discovery ─
|
|
656
|
+
# The fleet learned on 2026-09-25 that three seats had sat on 2.17.0 for days.
|
|
657
|
+
# Nothing was broken in the sense anything reports: the launchd job fired every
|
|
658
|
+
# hour, npm answered, the install ran, the health gate failed, the rollback
|
|
659
|
+
# worked, last.json recorded it honestly, and the beat carried it. Each hour was
|
|
660
|
+
# a correct, self-contained failure — and a correct failure repeated forty times
|
|
661
|
+
# is a different fact from a correct failure once. Only the REPETITION says the
|
|
662
|
+
# seat will never arrive on its own, and no single record can hold it, so the
|
|
663
|
+
# count lives in last.json and rides the beat the rest of that record already
|
|
664
|
+
# rides. There is no second channel here and there must not be one: the seats
|
|
665
|
+
# this is about are, by construction, the ones running the oldest code.
|
|
666
|
+
#
|
|
667
|
+
# WHAT COUNTS AS AN ATTEMPT — the whole honesty of the number:
|
|
668
|
+
# · An ATTEMPT is a record whose `to` differs from its `from`: the upgrade
|
|
669
|
+
# path actually installed (or tried to install) a different version. Those
|
|
670
|
+
# are the only records that move the streak.
|
|
671
|
+
# · A FAILED attempt (`ok:false`) increments it and stamps `stuckSince` with
|
|
672
|
+
# the first failure of the run of failures, so a reader gets a DURATION and
|
|
673
|
+
# not merely a tally — "4 attempts since Monday" and "4 attempts in the last
|
|
674
|
+
# hour" are different seats.
|
|
675
|
+
# · A SUCCEEDED attempt (`ok:true`) clears all three fields. Arrival is the
|
|
676
|
+
# only thing that resets it; a fleet-wide publish does not, because a seat
|
|
677
|
+
# that fails 2.18.12, 2.18.13 and 2.18.14 in turn has failed three times to
|
|
678
|
+
# do the one thing asked of it, and restarting the count on every release
|
|
679
|
+
# would guarantee the number never reaches any threshold.
|
|
680
|
+
# · Everything else PRESERVES the fields untouched. A `to == from` record is
|
|
681
|
+
# a report on the installed version, not an attempt to leave it, so it must
|
|
682
|
+
# not move a count of ATTEMPTS in either direction. (Every such call site
|
|
683
|
+
# today sits on the up-to-date branch and therefore takes the arrival path
|
|
684
|
+
# below instead — this rule governs the writer, for the next caller that
|
|
685
|
+
# does not.) And a run that SKIPS writes no record at all, which is the
|
|
686
|
+
# same preservation by a shorter route. THIS IS THE POINT OF (d): a seat that has not been ASKED —
|
|
687
|
+
# nothing newer published, or held for the day after its last failure — is
|
|
688
|
+
# not a stuck seat, and must never accumulate a streak for sitting still.
|
|
689
|
+
# Only a real, completed, failed attempt does.
|
|
690
|
+
# · WHICH RUNS ACTUALLY SKIP — the exact list, because a wrong one was
|
|
691
|
+
# written here first. The paths that reach `exit 0` with NO write_last are:
|
|
692
|
+
# the failed-target hold, the overlap lock, the kill-switch, and a failed
|
|
693
|
+
# `npm view`. "SIMPLY BEING UP TO DATE" IS NOT AMONG THEM and never was —
|
|
694
|
+
# the up-to-date branch is the daemon health gate, and it calls write_last
|
|
695
|
+
# on every one of its outcomes (healthy, unhealthy-current,
|
|
696
|
+
# ahead-of-registry, recovered, and the stale-daemon kickstart). The
|
|
697
|
+
# original text claimed otherwise; nothing checked it, and the streak
|
|
698
|
+
# arithmetic on that branch is not what the sentence implied.
|
|
699
|
+
# · ARRIVAL IS BEING AT @latest, NOT MERELY INSTALLING IT. Every write on the
|
|
700
|
+
# up-to-date branch passes `at_latest=true`, which clears all three fields
|
|
701
|
+
# — because that branch is only entered when `CUR >= LATEST`, i.e. there is
|
|
702
|
+
# nothing left for this seat to reach. That covers the case the increment
|
|
703
|
+
# rule alone gets wrong: if a bad release is YANKED and @latest rolls back
|
|
704
|
+
# to the version a seat already runs, the seat stops attempting anything,
|
|
705
|
+
# so nothing would ever reset it and `upgrade_stuck` would fire forever
|
|
706
|
+
# (critical past 6) on a seat that is in fact perfectly current. It also
|
|
707
|
+
# makes the stale-daemon kickstart record honest rather than accidental:
|
|
708
|
+
# that hop really is an arrival.
|
|
709
|
+
# The write is done in node rather than printf because the record now depends on
|
|
710
|
+
# the record before it; a shell that reads a file it is about to overwrite gets
|
|
711
|
+
# that wrong at exactly the moments it matters.
|
|
712
|
+
write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason, [$6 at_latest] — state/autoupdate/last.json (→ beat machine.upgrade)
|
|
713
|
+
local dir="$AGENT_DIR/state/autoupdate" tmp streak at_latest="${6:-false}"
|
|
486
714
|
mkdir -p "$dir" 2>/dev/null || return 0
|
|
487
715
|
tmp="$(mktemp "$dir/.last.XXXXXX" 2>/dev/null)" || return 0
|
|
488
|
-
|
|
489
|
-
|
|
716
|
+
streak="$(node -e '
|
|
717
|
+
const fs = require("fs");
|
|
718
|
+
const [file, tmp, from, to, okS, healthyS, reason, stuckStreakS, atLatestS] = process.argv.slice(1);
|
|
719
|
+
const ok = okS === "true";
|
|
720
|
+
const at = new Date().toISOString().replace(/\.\d{3}Z$/, "Z");
|
|
721
|
+
let prev = null;
|
|
722
|
+
try { prev = JSON.parse(fs.readFileSync(file, "utf8")); } catch {}
|
|
723
|
+
if (!prev || typeof prev !== "object") prev = {};
|
|
724
|
+
const prevStreak = Number.isInteger(prev.failStreak) && prev.failStreak > 0 ? prev.failStreak : 0;
|
|
725
|
+
const attempt = from !== to; // an upgrade was actually tried
|
|
726
|
+
const atLatest = atLatestS === "true"; // this seat is AT or past @latest — nothing left to reach
|
|
727
|
+
let failStreak, stuckSince, streakTarget;
|
|
728
|
+
if (atLatest || (attempt && ok)) { // arrival, by install OR because @latest came back to us (a yank)
|
|
729
|
+
failStreak = 0; stuckSince = ""; streakTarget = "";
|
|
730
|
+
} else if (!attempt) { // a report on the installed version — carry the count, do not touch it
|
|
731
|
+
failStreak = prevStreak;
|
|
732
|
+
stuckSince = typeof prev.stuckSince === "string" ? prev.stuckSince : "";
|
|
733
|
+
streakTarget = typeof prev.streakTarget === "string" ? prev.streakTarget : "";
|
|
734
|
+
} else {
|
|
735
|
+
failStreak = prevStreak + 1;
|
|
736
|
+
stuckSince = typeof prev.stuckSince === "string" && prev.stuckSince ? prev.stuckSince : at;
|
|
737
|
+
streakTarget = to;
|
|
738
|
+
}
|
|
739
|
+
const rec = { from, to, at, ok, healthy: healthyS === "true", reason };
|
|
740
|
+
if (failStreak > 0) {
|
|
741
|
+
rec.failStreak = failStreak;
|
|
742
|
+
if (stuckSince) rec.stuckSince = stuckSince;
|
|
743
|
+
if (streakTarget) rec.streakTarget = streakTarget;
|
|
744
|
+
}
|
|
745
|
+
fs.writeFileSync(tmp, JSON.stringify(rec, null, 2) + "\n");
|
|
746
|
+
// stdout is the shell'"'"'s only view of what was decided: "<streak> <target> <since>"
|
|
747
|
+
process.stdout.write(failStreak >= Number(stuckStreakS) && attempt && !ok ? `STUCK ${failStreak} ${streakTarget} ${stuckSince}` : "");
|
|
748
|
+
' "$LAST_JSON" "$tmp" "$1" "$2" "$3" "$4" "$5" "$STUCK_STREAK" "$at_latest" 2>/dev/null)" || streak=""
|
|
749
|
+
if [ -s "$tmp" ]; then
|
|
750
|
+
mv -f "$tmp" "$LAST_JSON"
|
|
751
|
+
else
|
|
752
|
+
# The node writer produced nothing. Before this branch existed the tmp was
|
|
753
|
+
# simply removed and last.json was LEFT AT ITS PREVIOUS VALUE — so a write
|
|
754
|
+
# failure did not lose a record, it published a STALE one, and the 24 h
|
|
755
|
+
# failed-target hold (which reads `to` and `reason` out of this file) then
|
|
756
|
+
# made its decision from a previous run's outcome while believing it was
|
|
757
|
+
# this run's. A missing record is fail-open by design here; a wrong one is
|
|
758
|
+
# not. So write the event itself with printf, which needs no interpreter,
|
|
759
|
+
# and say loudly that the count did not survive: the streak is the one
|
|
760
|
+
# field that cannot be reconstructed without reading the old file, and a
|
|
761
|
+
# silently reset counter is the failure this whole block exists to prevent.
|
|
762
|
+
log "WARN: write_last could not run node — recording the bare outcome without the failStreak. The count restarts from this run; see state/autoupdate/last.json."
|
|
763
|
+
printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
|
|
764
|
+
"$1" "$2" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$3" "$4" "$(printf '%s' "$5" | sed 's/[\\"]/\\&/g')" > "$LAST_JSON" 2>/dev/null || true
|
|
765
|
+
fi
|
|
490
766
|
rm -f "$tmp" 2>/dev/null
|
|
767
|
+
if [ -n "$streak" ]; then
|
|
768
|
+
set -- $streak
|
|
769
|
+
log "STUCK: $2 consecutive failed attempts to reach $3 (since $4). This seat will not arrive on its own — the org now carries machine.upgrade.failStreak=$2 and an \`upgrade_stuck\` alert on every beat. maestro doctor; then rm state/autoupdate/last.json to retry immediately."
|
|
770
|
+
fi
|
|
491
771
|
return 0
|
|
492
772
|
}
|
|
493
773
|
last_reason(){ # the reason field of last.json, or nothing
|
|
494
774
|
node -e 'try { process.stdout.write(String(JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).reason || "")); } catch {}' "$LAST_JSON" 2>/dev/null
|
|
495
775
|
}
|
|
776
|
+
last_streak(){ # the failStreak of last.json as a positive integer, or nothing
|
|
777
|
+
node -e 'try { const n = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).failStreak; if (Number.isInteger(n) && n > 0) process.stdout.write(String(n)); } catch {}' "$LAST_JSON" 2>/dev/null
|
|
778
|
+
}
|
|
496
779
|
# The version the RUNNING daemon started on. scripts/daemon/health.mjs resolves
|
|
497
780
|
# sdk_version once at load and writes it to state/dashboards/daemon-health.yaml
|
|
498
781
|
# with its pid, precisely so a reader can tell the installed package from the
|
|
@@ -595,13 +878,13 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
595
878
|
# the notice tells the session to restart itself onto the new code.
|
|
596
879
|
log "OK: daemon now on $CUR (was $STALE_FROM) — the manual upgrade is complete"
|
|
597
880
|
write_upgrade_notice "$STALE_FROM" "$CUR"
|
|
598
|
-
write_last "$STALE_FROM" "$CUR" true true ""
|
|
881
|
+
write_last "$STALE_FROM" "$CUR" true true "" true
|
|
599
882
|
elif [ -n "$STALE_FROM" ]; then
|
|
600
883
|
# Same version, newer files: the daemon is reconciled; the session is
|
|
601
884
|
# not told to restart (a touched or restored file must not cost it its
|
|
602
885
|
# context — the version did not change).
|
|
603
886
|
log "OK: daemon restarted onto the installed $CUR code; no session notice (same version)"
|
|
604
|
-
write_last "$CUR" "$CUR" true true ""
|
|
887
|
+
write_last "$CUR" "$CUR" true true "" true
|
|
605
888
|
elif [ -n "$AHEAD" ]; then
|
|
606
889
|
# The DAEMON is well; the FLEET is not. Recorded as ok/healthy — because
|
|
607
890
|
# it is — with the finding in `reason`, so the org learns that this seat
|
|
@@ -609,23 +892,49 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
609
892
|
# per episode: unlike an unhealthy daemon this does not clear itself, and
|
|
610
893
|
# a fact that ages out of one log file is how six days went by.
|
|
611
894
|
log "OK: healthy on $CUR, but AHEAD of the registry ($LATEST) — recorded for the org as ahead-of-registry"
|
|
612
|
-
write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST"
|
|
895
|
+
write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST" true
|
|
613
896
|
else
|
|
614
|
-
case "$PREV_REASON" in
|
|
897
|
+
case "$PREV_REASON" in
|
|
898
|
+
unhealthy-current*)
|
|
615
899
|
log "recovered: last.json no longer reports unhealthy-current"
|
|
616
|
-
write_last "$CUR" "$CUR" true true "" ;;
|
|
900
|
+
write_last "$CUR" "$CUR" true true "" true ;;
|
|
617
901
|
ahead-of-registry*)
|
|
618
902
|
log "recovered: $CUR is on the registry now (latest $LATEST) — last.json no longer reports ahead-of-registry"
|
|
619
|
-
write_last "$CUR" "$CUR" true true "" ;;
|
|
903
|
+
write_last "$CUR" "$CUR" true true "" true ;;
|
|
904
|
+
*)
|
|
905
|
+
# A healthy seat that is simply up to date writes NOTHING — there is
|
|
906
|
+
# nothing to report and an empty last.json is the honest state.
|
|
907
|
+
#
|
|
908
|
+
# The one exception is a standing failStreak, and it is the reason this
|
|
909
|
+
# branch exists at all. A streak only ever reset on ARRIVAL, i.e. on a
|
|
910
|
+
# successful hop. If a bad release is YANKED and @latest rolls back to
|
|
911
|
+
# the version this seat already runs, the seat stops attempting
|
|
912
|
+
# anything: every hour it takes this quiet path, writes nothing, and the
|
|
913
|
+
# count stands for ever — `upgrade_stuck` firing critical on a seat that
|
|
914
|
+
# is, in fact, exactly where the registry wants it. Being AT @latest is
|
|
915
|
+
# an arrival however you got there, so record one and say so.
|
|
916
|
+
STREAK="$(last_streak)"
|
|
917
|
+
if [ -n "$STREAK" ]; then
|
|
918
|
+
log "streak cleared: $CUR is @latest and this seat is healthy on it, but last.json still carried failStreak=$STREAK — @latest has come back to this version (a yank or a withdrawn release). Recording the arrival so the upgrade_stuck alert stops."
|
|
919
|
+
write_last "$CUR" "$CUR" true true "" true
|
|
920
|
+
fi ;;
|
|
620
921
|
esac
|
|
621
922
|
fi
|
|
622
923
|
else
|
|
924
|
+
# Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
|
|
925
|
+
# running process while the front door kept beating — the seat looked "up
|
|
926
|
+
# to date" every hour and was dark. The daemon cannot restart itself, so
|
|
927
|
+
# revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
|
|
928
|
+
# here and asserts the pid came back. Gated to the no-process case, so a
|
|
929
|
+
# crash-looping or beat-unhealthy daemon (which still has a pid) is left to
|
|
930
|
+
# the gate above, not kickstarted.
|
|
931
|
+
revive_daemon || true
|
|
623
932
|
log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
|
|
624
933
|
# The `unhealthy-current:` PREFIX is load-bearing (the episode check below
|
|
625
934
|
# and failed_hold_reason both match on it), so the ahead-of-registry fact
|
|
626
935
|
# is appended, never prepended — an unpublished build that is also sick is
|
|
627
936
|
# the worst case and must say both things.
|
|
628
|
-
write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}"
|
|
937
|
+
write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}" true
|
|
629
938
|
case "$PREV_REASON" in
|
|
630
939
|
unhealthy-current*) log "session already notified this episode (last.json was unhealthy-current); notice not rewritten" ;;
|
|
631
940
|
*) write_notice "$CUR" "$CUR" "unhealthy-current: $HEALTH_REASON" ;;
|
|
@@ -644,10 +953,19 @@ fi
|
|
|
644
953
|
# nothing.
|
|
645
954
|
FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
|
|
646
955
|
# How long a front door may go without beating before this run restarts it FOR
|
|
647
|
-
# it.
|
|
648
|
-
#
|
|
649
|
-
#
|
|
650
|
-
|
|
956
|
+
# it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
|
|
957
|
+
# a wedged front door is measured in the same window as a wedged daemon rather
|
|
958
|
+
# than the days it took to notice the last three. `revive_session` (this run,
|
|
959
|
+
# hourly) is the BACKSTOP; it does not need its own, looser number.
|
|
960
|
+
SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
|
|
961
|
+
# The daemon confirms a shut front door over TWO reads before it kickstarts
|
|
962
|
+
# (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
|
|
963
|
+
# 60s apart). `revive_session`, the hourly backstop, now does the same: a
|
|
964
|
+
# supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
|
|
965
|
+
# dying and the new one re-arming, and a single stale read landing inside that
|
|
966
|
+
# gap must not kickstart a session that is mid-resume. The two reads are 60s
|
|
967
|
+
# apart; the hourly backstop can afford the one wait.
|
|
968
|
+
SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
|
|
651
969
|
failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
|
|
652
970
|
node -e '
|
|
653
971
|
const [file, latest, holdS] = process.argv.slice(1);
|