@cohortapp/agent-sdk 2.18.14 → 2.18.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -43,7 +43,17 @@
43
43
  # STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
44
44
  # (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
45
45
  # loop cannot straddle both samples by luck, and well inside the hourly
46
- # cadence. A 25 s loop fails on the second sample and on uptime.
46
+ # cadence. A 25 s loop fails on the second sample and on uptime. A pid
47
+ # CHANGE within the window is not, on its own, proof of that loop: this
48
+ # script's own restart_daemon (or an operator's `launchctl kickstart -k`)
49
+ # changes the pid deliberately and can land inside the same window —
50
+ # 2.18.14 read that as a crash loop and rolled back a release that had
51
+ # done nothing wrong. daemon_last_exit() tells the two apart: a crash
52
+ # loop's reaped exit is a real non-zero every time; a deliberate
53
+ # kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
54
+ # real non-zero exit fails here on the pid change alone — a clean or
55
+ # unknown exit falls through to the uptime check, which still catches a
56
+ # genuinely unstable respawn (its pid is always younger than STABLE_S).
47
57
  # (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
48
58
  # lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
49
59
  # ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
@@ -270,6 +280,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
270
280
  HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
271
281
  DAEMON_LABEL=""
272
282
  SESSION_LABEL=""
283
+ # The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
284
+ # below — the on-host backstop for a daemon that is DOWN, which a live daemon
285
+ # can never observe about itself. Defined up here (not with SESSION_STALE_S,
286
+ # which lives in the not-up-to-date path) because revive_daemon fires on the
287
+ # UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
288
+ # 90s front-door observers.
289
+ LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
290
+ DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
291
+ # Gap between the two settle-window reads that prove the kickstarted daemon is
292
+ # one continuous process, not a come-up-then-crash loop. A one-shot read would
293
+ # pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
294
+ DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
273
295
 
274
296
  daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
275
297
  pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
@@ -355,14 +377,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
355
377
  [ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
356
378
  return 0
357
379
  }
380
+ daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
381
+ [ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
382
+ [ -n "$DAEMON_LABEL" ] || return 0
383
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
384
+ }
358
385
  is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
359
- local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
386
+ local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
360
387
  HEALTH_REASON=""
361
388
  p1="$(daemon_pids)"
362
389
  [ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
363
390
  [ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
364
391
  p2="$(daemon_pids)"
365
- [ "$p1" = "$p2" ] || { HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}) — crash loop"; return 1; }
392
+ if [ "$p1" != "$p2" ]; then
393
+ # A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
394
+ # (this same script's own restart_daemon, or an operator's) changes the pid
395
+ # deliberately and can land inside this very window. 2.18.14 read that as a
396
+ # crash loop and rolled back a release that had done nothing wrong. The
397
+ # daemon's OWN last exit says which one happened: a crash-looping process
398
+ # exits non-zero every time launchd reaps it; a clean kickstart's reaped
399
+ # exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
400
+ # non-zero exit is a crash loop here; a clean or unknown exit is treated as
401
+ # an intended restart and falls through to the uptime check below, which
402
+ # still catches a genuinely unstable respawn (a crash-looping daemon's new
403
+ # pid is always younger than STABLE_S).
404
+ exit_code="$(daemon_last_exit)"
405
+ case "$exit_code" in
406
+ ""|"-"|"0")
407
+ # A clean/unknown exit with NO new pid at all is not a restart in
408
+ # progress, it is the daemon gone — the "no daemon process" case, not a
409
+ # crash loop and not healthy.
410
+ [ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
411
+ log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
412
+ p1="$p2"
413
+ ;;
414
+ *)
415
+ HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
416
+ return 1
417
+ ;;
418
+ esac
419
+ fi
366
420
  oldest=""
367
421
  for pid in $p1; do
368
422
  up="$(pid_uptime_s "$pid")"
@@ -450,19 +504,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
450
504
  # Leave that one to session_reconciled, which already names it.
451
505
  label_loaded "$SESSION_LABEL" || return 0
452
506
  session_beating && return 0
453
- log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — kickstarting it rather than waiting for a session that cannot answer"
507
+ # Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
508
+ # not enough to drop a session, because a supervisor resume gaps the beat for
509
+ # a moment. Confirm the door is STILL silent on a second read 60s later
510
+ # before kickstarting — if it beat again in between, that was a resume gap,
511
+ # not a wedge, and the session keeps its context.
512
+ log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
513
+ sleep "$SESSION_REVIVE_RECHECK_S"
514
+ session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
515
+ # Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
516
+ # note. The supervisor holds the `session` singleton (state/locks/process/
517
+ # session.pid) for its whole life and beats as a healthy resume progresses.
518
+ # We are past the FULL grace and two reads apart, so a resume gap is already
519
+ # excluded by time — a live holder here is a supervisor PARKED in its launch
520
+ # probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
521
+ # Consulting the lock tells a parked supervisor (override, logged) from a dead
522
+ # one (clean restart); it never blocks the parked case a hard skip would.
523
+ local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
524
+ [ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
525
+ if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
526
+ log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
527
+ fi
528
+ log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
454
529
  launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
455
530
  && { log "front door $SESSION_LABEL restarted"; return 0; }
456
531
  log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
457
532
  return 1
458
533
  }
459
534
 
535
+ daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
536
+ [ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
537
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
538
+ NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
539
+ END { if (!f) print "absent" }'
540
+ }
541
+ daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
542
+ node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
543
+ "$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
544
+ }
545
+ REVIVE_NOTE_REL="state/telemetry/revive-note.json"
546
+ write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
547
+ local dir="$AGENT_DIR/state/telemetry" tmp
548
+ mkdir -p "$dir" 2>/dev/null || return 0
549
+ tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
550
+ printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
551
+ "$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
552
+ && mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
553
+ return 0
554
+ }
555
+ daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
556
+ [ -n "$DAEMON_LABEL" ] || return 0
557
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
558
+ }
559
+ daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
560
+ local pid start
561
+ pid="$(daemon_launchctl_pid)"
562
+ if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
563
+ start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
564
+ printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
565
+ elif [ -n "$pid" ]; then
566
+ printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
567
+ else
568
+ printf '{"pid":null,"alive":false,"startTime":null}'
569
+ fi
570
+ }
571
+ revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
572
+ # is_healthy bails at the empty-pid check before daemon_last_exit (its label
573
+ # resolver) runs, so resolve them here — the down case is exactly the one
574
+ # where nothing upstream did.
575
+ [ -n "$DAEMON_LABEL" ] || resolve_labels
576
+ [ -n "$DAEMON_LABEL" ] || return 0
577
+ [ -f "$LIB_REVIVE" ] || return 0
578
+ # ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
579
+ # still has a pid — that is the health gate's business, and a kickstart would
580
+ # fight the very restart it is doing. pgrep-empty is Jacob's lane.
581
+ [ -z "$(daemon_pids)" ] || return 0
582
+ local st beat lpid palive verdict
583
+ st="$(daemon_launchctl_state)"
584
+ beat="$(daemon_beat_age_ms)"
585
+ # PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
586
+ # A listed pid is not a running daemon — verify it, and pass pidAlive so
587
+ # shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
588
+ # stamping it alive (which would deadlock: pgrep-empty says down, launchctl
589
+ # says pid, and nothing revives).
590
+ lpid="$(daemon_launchctl_pid)"
591
+ if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
592
+ verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
593
+ case "$verdict" in
594
+ revive) : ;;
595
+ *) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
596
+ esac
597
+ log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
598
+ launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
599
+ # ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
600
+ # passes a phantom pid AND a come-up-then-crash whose local beat it keeps
601
+ # stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
602
+ # live process (same pid, same start-time, alive on both) — the uptime
603
+ # continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
604
+ # hourly backstop does not round-trip the org just to confirm); continuity
605
+ # carries it, and the daemon's own presence publish is the fleet-visible proof.
606
+ local s1 s2 samples ok
607
+ s1="$(daemon_sample_json)"
608
+ sleep "$DAEMON_CONFIRM_SETTLE_S"
609
+ s2="$(daemon_sample_json)"
610
+ samples="[$s1,$s2]"
611
+ ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
612
+ if [ "$ok" = "ok" ]; then
613
+ log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
614
+ rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
615
+ return 0
616
+ fi
617
+ # ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
618
+ # this failure's log under the loop). One attempt, then hand up with what it
619
+ # saw: the launchctl list line and the tail of the daemon's own log.
620
+ local lc_line dlog_tail evidence
621
+ lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
622
+ dlog_tail="$(tail -n 3 "$DLOG_DIR/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
623
+ evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
624
+ log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
625
+ write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
626
+ return 1
627
+ }
460
628
  session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
461
629
  label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
462
630
  pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
463
631
  return 0
464
632
  }
465
- json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
633
+ json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
466
634
  write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
467
635
  local dir="$AGENT_DIR/state/session" tmp
468
636
  mkdir -p "$dir" 2>/dev/null || return 0
@@ -481,18 +649,133 @@ write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell t
481
649
  }
482
650
  write_upgrade_notice(){ write_notice "$1" "$2"; }
483
651
  LAST_JSON="$AGENT_DIR/state/autoupdate/last.json"
484
- write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason — state/autoupdate/last.json (→ beat machine.upgrade)
485
- local dir="$AGENT_DIR/state/autoupdate" tmp
652
+ # How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK, and
653
+ # at what point it stops being a warning. See the `failStreak` block below.
654
+ STUCK_STREAK="${MAESTRO_AUTOUPDATE_STUCK_STREAK:-3}"
655
+ # ── failStreak — "this seat cannot get to @latest" is a FACT, not a discovery ─
656
+ # The fleet learned on 2026-09-25 that three seats had sat on 2.17.0 for days.
657
+ # Nothing was broken in the sense anything reports: the launchd job fired every
658
+ # hour, npm answered, the install ran, the health gate failed, the rollback
659
+ # worked, last.json recorded it honestly, and the beat carried it. Each hour was
660
+ # a correct, self-contained failure — and a correct failure repeated forty times
661
+ # is a different fact from a correct failure once. Only the REPETITION says the
662
+ # seat will never arrive on its own, and no single record can hold it, so the
663
+ # count lives in last.json and rides the beat the rest of that record already
664
+ # rides. There is no second channel here and there must not be one: the seats
665
+ # this is about are, by construction, the ones running the oldest code.
666
+ #
667
+ # WHAT COUNTS AS AN ATTEMPT — the whole honesty of the number:
668
+ # · An ATTEMPT is a record whose `to` differs from its `from`: the upgrade
669
+ # path actually installed (or tried to install) a different version. Those
670
+ # are the only records that move the streak.
671
+ # · A FAILED attempt (`ok:false`) increments it and stamps `stuckSince` with
672
+ # the first failure of the run of failures, so a reader gets a DURATION and
673
+ # not merely a tally — "4 attempts since Monday" and "4 attempts in the last
674
+ # hour" are different seats.
675
+ # · A SUCCEEDED attempt (`ok:true`) clears all three fields. Arrival is the
676
+ # only thing that resets it; a fleet-wide publish does not, because a seat
677
+ # that fails 2.18.12, 2.18.13 and 2.18.14 in turn has failed three times to
678
+ # do the one thing asked of it, and restarting the count on every release
679
+ # would guarantee the number never reaches any threshold.
680
+ # · Everything else PRESERVES the fields untouched. A `to == from` record is
681
+ # a report on the installed version, not an attempt to leave it, so it must
682
+ # not move a count of ATTEMPTS in either direction. (Every such call site
683
+ # today sits on the up-to-date branch and therefore takes the arrival path
684
+ # below instead — this rule governs the writer, for the next caller that
685
+ # does not.) And a run that SKIPS writes no record at all, which is the
686
+ # same preservation by a shorter route. THIS IS THE POINT OF (d): a seat that has not been ASKED —
687
+ # nothing newer published, or held for the day after its last failure — is
688
+ # not a stuck seat, and must never accumulate a streak for sitting still.
689
+ # Only a real, completed, failed attempt does.
690
+ # · WHICH RUNS ACTUALLY SKIP — the exact list, because a wrong one was
691
+ # written here first. The paths that reach `exit 0` with NO write_last are:
692
+ # the failed-target hold, the overlap lock, the kill-switch, and a failed
693
+ # `npm view`. "SIMPLY BEING UP TO DATE" IS NOT AMONG THEM and never was —
694
+ # the up-to-date branch is the daemon health gate, and it calls write_last
695
+ # on every one of its outcomes (healthy, unhealthy-current,
696
+ # ahead-of-registry, recovered, and the stale-daemon kickstart). The
697
+ # original text claimed otherwise; nothing checked it, and the streak
698
+ # arithmetic on that branch is not what the sentence implied.
699
+ # · ARRIVAL IS BEING AT @latest, NOT MERELY INSTALLING IT. Every write on the
700
+ # up-to-date branch passes `at_latest=true`, which clears all three fields
701
+ # — because that branch is only entered when `CUR >= LATEST`, i.e. there is
702
+ # nothing left for this seat to reach. That covers the case the increment
703
+ # rule alone gets wrong: if a bad release is YANKED and @latest rolls back
704
+ # to the version a seat already runs, the seat stops attempting anything,
705
+ # so nothing would ever reset it and `upgrade_stuck` would fire forever
706
+ # (critical past 6) on a seat that is in fact perfectly current. It also
707
+ # makes the stale-daemon kickstart record honest rather than accidental:
708
+ # that hop really is an arrival.
709
+ # The write is done in node rather than printf because the record now depends on
710
+ # the record before it; a shell that reads a file it is about to overwrite gets
711
+ # that wrong at exactly the moments it matters.
712
+ write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason, [$6 at_latest] — state/autoupdate/last.json (→ beat machine.upgrade)
713
+ local dir="$AGENT_DIR/state/autoupdate" tmp streak at_latest="${6:-false}"
486
714
  mkdir -p "$dir" 2>/dev/null || return 0
487
715
  tmp="$(mktemp "$dir/.last.XXXXXX" 2>/dev/null)" || return 0
488
- printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
489
- "$1" "$2" "$(date -u +%FT%TZ)" "$3" "$4" "$(json_str "$5")" > "$tmp" && mv -f "$tmp" "$LAST_JSON"
716
+ streak="$(node -e '
717
+ const fs = require("fs");
718
+ const [file, tmp, from, to, okS, healthyS, reason, stuckStreakS, atLatestS] = process.argv.slice(1);
719
+ const ok = okS === "true";
720
+ const at = new Date().toISOString().replace(/\.\d{3}Z$/, "Z");
721
+ let prev = null;
722
+ try { prev = JSON.parse(fs.readFileSync(file, "utf8")); } catch {}
723
+ if (!prev || typeof prev !== "object") prev = {};
724
+ const prevStreak = Number.isInteger(prev.failStreak) && prev.failStreak > 0 ? prev.failStreak : 0;
725
+ const attempt = from !== to; // an upgrade was actually tried
726
+ const atLatest = atLatestS === "true"; // this seat is AT or past @latest — nothing left to reach
727
+ let failStreak, stuckSince, streakTarget;
728
+ if (atLatest || (attempt && ok)) { // arrival, by install OR because @latest came back to us (a yank)
729
+ failStreak = 0; stuckSince = ""; streakTarget = "";
730
+ } else if (!attempt) { // a report on the installed version — carry the count, do not touch it
731
+ failStreak = prevStreak;
732
+ stuckSince = typeof prev.stuckSince === "string" ? prev.stuckSince : "";
733
+ streakTarget = typeof prev.streakTarget === "string" ? prev.streakTarget : "";
734
+ } else {
735
+ failStreak = prevStreak + 1;
736
+ stuckSince = typeof prev.stuckSince === "string" && prev.stuckSince ? prev.stuckSince : at;
737
+ streakTarget = to;
738
+ }
739
+ const rec = { from, to, at, ok, healthy: healthyS === "true", reason };
740
+ if (failStreak > 0) {
741
+ rec.failStreak = failStreak;
742
+ if (stuckSince) rec.stuckSince = stuckSince;
743
+ if (streakTarget) rec.streakTarget = streakTarget;
744
+ }
745
+ fs.writeFileSync(tmp, JSON.stringify(rec, null, 2) + "\n");
746
+ // stdout is the shell'"'"'s only view of what was decided: "<streak> <target> <since>"
747
+ process.stdout.write(failStreak >= Number(stuckStreakS) && attempt && !ok ? `STUCK ${failStreak} ${streakTarget} ${stuckSince}` : "");
748
+ ' "$LAST_JSON" "$tmp" "$1" "$2" "$3" "$4" "$5" "$STUCK_STREAK" "$at_latest" 2>/dev/null)" || streak=""
749
+ if [ -s "$tmp" ]; then
750
+ mv -f "$tmp" "$LAST_JSON"
751
+ else
752
+ # The node writer produced nothing. Before this branch existed the tmp was
753
+ # simply removed and last.json was LEFT AT ITS PREVIOUS VALUE — so a write
754
+ # failure did not lose a record, it published a STALE one, and the 24 h
755
+ # failed-target hold (which reads `to` and `reason` out of this file) then
756
+ # made its decision from a previous run's outcome while believing it was
757
+ # this run's. A missing record is fail-open by design here; a wrong one is
758
+ # not. So write the event itself with printf, which needs no interpreter,
759
+ # and say loudly that the count did not survive: the streak is the one
760
+ # field that cannot be reconstructed without reading the old file, and a
761
+ # silently reset counter is the failure this whole block exists to prevent.
762
+ log "WARN: write_last could not run node — recording the bare outcome without the failStreak. The count restarts from this run; see state/autoupdate/last.json."
763
+ printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
764
+ "$1" "$2" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$3" "$4" "$(printf '%s' "$5" | sed 's/[\\"]/\\&/g')" > "$LAST_JSON" 2>/dev/null || true
765
+ fi
490
766
  rm -f "$tmp" 2>/dev/null
767
+ if [ -n "$streak" ]; then
768
+ set -- $streak
769
+ log "STUCK: $2 consecutive failed attempts to reach $3 (since $4). This seat will not arrive on its own — the org now carries machine.upgrade.failStreak=$2 and an \`upgrade_stuck\` alert on every beat. maestro doctor; then rm state/autoupdate/last.json to retry immediately."
770
+ fi
491
771
  return 0
492
772
  }
493
773
  last_reason(){ # the reason field of last.json, or nothing
494
774
  node -e 'try { process.stdout.write(String(JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).reason || "")); } catch {}' "$LAST_JSON" 2>/dev/null
495
775
  }
776
+ last_streak(){ # the failStreak of last.json as a positive integer, or nothing
777
+ node -e 'try { const n = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).failStreak; if (Number.isInteger(n) && n > 0) process.stdout.write(String(n)); } catch {}' "$LAST_JSON" 2>/dev/null
778
+ }
496
779
  # The version the RUNNING daemon started on. scripts/daemon/health.mjs resolves
497
780
  # sdk_version once at load and writes it to state/dashboards/daemon-health.yaml
498
781
  # with its pid, precisely so a reader can tell the installed package from the
@@ -595,13 +878,13 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
595
878
  # the notice tells the session to restart itself onto the new code.
596
879
  log "OK: daemon now on $CUR (was $STALE_FROM) — the manual upgrade is complete"
597
880
  write_upgrade_notice "$STALE_FROM" "$CUR"
598
- write_last "$STALE_FROM" "$CUR" true true ""
881
+ write_last "$STALE_FROM" "$CUR" true true "" true
599
882
  elif [ -n "$STALE_FROM" ]; then
600
883
  # Same version, newer files: the daemon is reconciled; the session is
601
884
  # not told to restart (a touched or restored file must not cost it its
602
885
  # context — the version did not change).
603
886
  log "OK: daemon restarted onto the installed $CUR code; no session notice (same version)"
604
- write_last "$CUR" "$CUR" true true ""
887
+ write_last "$CUR" "$CUR" true true "" true
605
888
  elif [ -n "$AHEAD" ]; then
606
889
  # The DAEMON is well; the FLEET is not. Recorded as ok/healthy — because
607
890
  # it is — with the finding in `reason`, so the org learns that this seat
@@ -609,23 +892,49 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
609
892
  # per episode: unlike an unhealthy daemon this does not clear itself, and
610
893
  # a fact that ages out of one log file is how six days went by.
611
894
  log "OK: healthy on $CUR, but AHEAD of the registry ($LATEST) — recorded for the org as ahead-of-registry"
612
- write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST"
895
+ write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST" true
613
896
  else
614
- case "$PREV_REASON" in unhealthy-current*)
897
+ case "$PREV_REASON" in
898
+ unhealthy-current*)
615
899
  log "recovered: last.json no longer reports unhealthy-current"
616
- write_last "$CUR" "$CUR" true true "" ;;
900
+ write_last "$CUR" "$CUR" true true "" true ;;
617
901
  ahead-of-registry*)
618
902
  log "recovered: $CUR is on the registry now (latest $LATEST) — last.json no longer reports ahead-of-registry"
619
- write_last "$CUR" "$CUR" true true "" ;;
903
+ write_last "$CUR" "$CUR" true true "" true ;;
904
+ *)
905
+ # A healthy seat that is simply up to date writes NOTHING — there is
906
+ # nothing to report and an empty last.json is the honest state.
907
+ #
908
+ # The one exception is a standing failStreak, and it is the reason this
909
+ # branch exists at all. A streak only ever reset on ARRIVAL, i.e. on a
910
+ # successful hop. If a bad release is YANKED and @latest rolls back to
911
+ # the version this seat already runs, the seat stops attempting
912
+ # anything: every hour it takes this quiet path, writes nothing, and the
913
+ # count stands for ever — `upgrade_stuck` firing critical on a seat that
914
+ # is, in fact, exactly where the registry wants it. Being AT @latest is
915
+ # an arrival however you got there, so record one and say so.
916
+ STREAK="$(last_streak)"
917
+ if [ -n "$STREAK" ]; then
918
+ log "streak cleared: $CUR is @latest and this seat is healthy on it, but last.json still carried failStreak=$STREAK — @latest has come back to this version (a yank or a withdrawn release). Recording the arrival so the upgrade_stuck alert stops."
919
+ write_last "$CUR" "$CUR" true true "" true
920
+ fi ;;
620
921
  esac
621
922
  fi
622
923
  else
924
+ # Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
925
+ # running process while the front door kept beating — the seat looked "up
926
+ # to date" every hour and was dark. The daemon cannot restart itself, so
927
+ # revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
928
+ # here and asserts the pid came back. Gated to the no-process case, so a
929
+ # crash-looping or beat-unhealthy daemon (which still has a pid) is left to
930
+ # the gate above, not kickstarted.
931
+ revive_daemon || true
623
932
  log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
624
933
  # The `unhealthy-current:` PREFIX is load-bearing (the episode check below
625
934
  # and failed_hold_reason both match on it), so the ahead-of-registry fact
626
935
  # is appended, never prepended — an unpublished build that is also sick is
627
936
  # the worst case and must say both things.
628
- write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}"
937
+ write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}" true
629
938
  case "$PREV_REASON" in
630
939
  unhealthy-current*) log "session already notified this episode (last.json was unhealthy-current); notice not rewritten" ;;
631
940
  *) write_notice "$CUR" "$CUR" "unhealthy-current: $HEALTH_REASON" ;;
@@ -644,10 +953,19 @@ fi
644
953
  # nothing.
645
954
  FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
646
955
  # How long a front door may go without beating before this run restarts it FOR
647
- # it. Ten minutes: long enough that a busy session is never interrupted (the
648
- # heartbeat rides every tool use), short enough that a wedged one is measured in
649
- # minutes rather than the days it took to notice the last three.
650
- SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-600}"
956
+ # it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
957
+ # a wedged front door is measured in the same window as a wedged daemon rather
958
+ # than the days it took to notice the last three. `revive_session` (this run,
959
+ # hourly) is the BACKSTOP; it does not need its own, looser number.
960
+ SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
961
+ # The daemon confirms a shut front door over TWO reads before it kickstarts
962
+ # (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
963
+ # 60s apart). `revive_session`, the hourly backstop, now does the same: a
964
+ # supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
965
+ # dying and the new one re-arming, and a single stale read landing inside that
966
+ # gap must not kickstart a session that is mid-resume. The two reads are 60s
967
+ # apart; the hourly backstop can afford the one wait.
968
+ SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
651
969
  failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
652
970
  node -e '
653
971
  const [file, latest, holdS] = process.argv.slice(1);