@cohortapp/agent-sdk 2.18.15 → 2.18.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/org/inbound/hydrate.mjs +35 -1
- package/lib/org/inbound/project.mjs +3 -0
- package/lib/session/frontdoor.mjs +87 -8
- package/lib/session/handoffs.mjs +57 -0
- package/lib/session/inbox-claims.mjs +47 -2
- package/lib/session/revive.mjs +302 -5
- package/lib/telemetry/collect.mjs +224 -0
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +242 -9
- package/scripts/daemon/assurance.mjs +56 -1
- package/scripts/fleet/rollout.mjs +48 -4
- package/scripts/hooks/pre-write-yaml-validate.mjs +63 -2
- package/scripts/local-triggers/autoupdate.sh +194 -9
|
@@ -43,7 +43,17 @@
|
|
|
43
43
|
# STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
|
|
44
44
|
# (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
|
|
45
45
|
# loop cannot straddle both samples by luck, and well inside the hourly
|
|
46
|
-
# cadence. A 25 s loop fails on the second sample and on uptime.
|
|
46
|
+
# cadence. A 25 s loop fails on the second sample and on uptime. A pid
|
|
47
|
+
# CHANGE within the window is not, on its own, proof of that loop: this
|
|
48
|
+
# script's own restart_daemon (or an operator's `launchctl kickstart -k`)
|
|
49
|
+
# changes the pid deliberately and can land inside the same window —
|
|
50
|
+
# 2.18.14 read that as a crash loop and rolled back a release that had
|
|
51
|
+
# done nothing wrong. daemon_last_exit() tells the two apart: a crash
|
|
52
|
+
# loop's reaped exit is a real non-zero every time; a deliberate
|
|
53
|
+
# kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
|
|
54
|
+
# real non-zero exit fails here on the pid change alone — a clean or
|
|
55
|
+
# unknown exit falls through to the uptime check, which still catches a
|
|
56
|
+
# genuinely unstable respawn (its pid is always younger than STABLE_S).
|
|
47
57
|
# (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
|
|
48
58
|
# lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
|
|
49
59
|
# ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
|
|
@@ -270,6 +280,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
|
|
|
270
280
|
HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
|
|
271
281
|
DAEMON_LABEL=""
|
|
272
282
|
SESSION_LABEL=""
|
|
283
|
+
# The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
|
|
284
|
+
# below — the on-host backstop for a daemon that is DOWN, which a live daemon
|
|
285
|
+
# can never observe about itself. Defined up here (not with SESSION_STALE_S,
|
|
286
|
+
# which lives in the not-up-to-date path) because revive_daemon fires on the
|
|
287
|
+
# UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
|
|
288
|
+
# 90s front-door observers.
|
|
289
|
+
LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
|
|
290
|
+
DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
|
|
291
|
+
# Gap between the two settle-window reads that prove the kickstarted daemon is
|
|
292
|
+
# one continuous process, not a come-up-then-crash loop. A one-shot read would
|
|
293
|
+
# pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
|
|
294
|
+
DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
|
|
273
295
|
|
|
274
296
|
daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
|
|
275
297
|
pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
|
|
@@ -355,14 +377,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
|
|
|
355
377
|
[ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
|
|
356
378
|
return 0
|
|
357
379
|
}
|
|
380
|
+
daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
|
|
381
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
|
|
382
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
383
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
|
|
384
|
+
}
|
|
358
385
|
is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
|
|
359
|
-
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
|
|
386
|
+
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
|
|
360
387
|
HEALTH_REASON=""
|
|
361
388
|
p1="$(daemon_pids)"
|
|
362
389
|
[ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
|
|
363
390
|
[ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
|
|
364
391
|
p2="$(daemon_pids)"
|
|
365
|
-
[ "$p1"
|
|
392
|
+
if [ "$p1" != "$p2" ]; then
|
|
393
|
+
# A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
|
|
394
|
+
# (this same script's own restart_daemon, or an operator's) changes the pid
|
|
395
|
+
# deliberately and can land inside this very window. 2.18.14 read that as a
|
|
396
|
+
# crash loop and rolled back a release that had done nothing wrong. The
|
|
397
|
+
# daemon's OWN last exit says which one happened: a crash-looping process
|
|
398
|
+
# exits non-zero every time launchd reaps it; a clean kickstart's reaped
|
|
399
|
+
# exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
|
|
400
|
+
# non-zero exit is a crash loop here; a clean or unknown exit is treated as
|
|
401
|
+
# an intended restart and falls through to the uptime check below, which
|
|
402
|
+
# still catches a genuinely unstable respawn (a crash-looping daemon's new
|
|
403
|
+
# pid is always younger than STABLE_S).
|
|
404
|
+
exit_code="$(daemon_last_exit)"
|
|
405
|
+
case "$exit_code" in
|
|
406
|
+
""|"-"|"0")
|
|
407
|
+
# A clean/unknown exit with NO new pid at all is not a restart in
|
|
408
|
+
# progress, it is the daemon gone — the "no daemon process" case, not a
|
|
409
|
+
# crash loop and not healthy.
|
|
410
|
+
[ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
|
|
411
|
+
log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
|
|
412
|
+
p1="$p2"
|
|
413
|
+
;;
|
|
414
|
+
*)
|
|
415
|
+
HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
|
|
416
|
+
return 1
|
|
417
|
+
;;
|
|
418
|
+
esac
|
|
419
|
+
fi
|
|
366
420
|
oldest=""
|
|
367
421
|
for pid in $p1; do
|
|
368
422
|
up="$(pid_uptime_s "$pid")"
|
|
@@ -450,19 +504,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
|
|
|
450
504
|
# Leave that one to session_reconciled, which already names it.
|
|
451
505
|
label_loaded "$SESSION_LABEL" || return 0
|
|
452
506
|
session_beating && return 0
|
|
453
|
-
|
|
507
|
+
# Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
|
|
508
|
+
# not enough to drop a session, because a supervisor resume gaps the beat for
|
|
509
|
+
# a moment. Confirm the door is STILL silent on a second read 60s later
|
|
510
|
+
# before kickstarting — if it beat again in between, that was a resume gap,
|
|
511
|
+
# not a wedge, and the session keeps its context.
|
|
512
|
+
log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
|
|
513
|
+
sleep "$SESSION_REVIVE_RECHECK_S"
|
|
514
|
+
session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
|
|
515
|
+
# Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
|
|
516
|
+
# note. The supervisor holds the `session` singleton (state/locks/process/
|
|
517
|
+
# session.pid) for its whole life and beats as a healthy resume progresses.
|
|
518
|
+
# We are past the FULL grace and two reads apart, so a resume gap is already
|
|
519
|
+
# excluded by time — a live holder here is a supervisor PARKED in its launch
|
|
520
|
+
# probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
|
|
521
|
+
# Consulting the lock tells a parked supervisor (override, logged) from a dead
|
|
522
|
+
# one (clean restart); it never blocks the parked case a hard skip would.
|
|
523
|
+
local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
|
|
524
|
+
[ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
|
|
525
|
+
if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
|
|
526
|
+
log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
|
|
527
|
+
fi
|
|
528
|
+
log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
|
|
454
529
|
launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
|
|
455
530
|
&& { log "front door $SESSION_LABEL restarted"; return 0; }
|
|
456
531
|
log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
|
|
457
532
|
return 1
|
|
458
533
|
}
|
|
459
534
|
|
|
535
|
+
daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
|
|
536
|
+
[ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
|
|
537
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
|
|
538
|
+
NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
|
|
539
|
+
END { if (!f) print "absent" }'
|
|
540
|
+
}
|
|
541
|
+
daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
|
|
542
|
+
node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
|
|
543
|
+
"$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
|
|
544
|
+
}
|
|
545
|
+
REVIVE_NOTE_REL="state/telemetry/revive-note.json"
|
|
546
|
+
write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
|
|
547
|
+
local dir="$AGENT_DIR/state/telemetry" tmp
|
|
548
|
+
mkdir -p "$dir" 2>/dev/null || return 0
|
|
549
|
+
tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
|
|
550
|
+
printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
|
|
551
|
+
"$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
|
|
552
|
+
&& mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
|
|
553
|
+
return 0
|
|
554
|
+
}
|
|
555
|
+
daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
|
|
556
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
557
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
|
|
558
|
+
}
|
|
559
|
+
daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
|
|
560
|
+
local pid start
|
|
561
|
+
pid="$(daemon_launchctl_pid)"
|
|
562
|
+
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
563
|
+
start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
|
|
564
|
+
printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
|
|
565
|
+
elif [ -n "$pid" ]; then
|
|
566
|
+
printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
|
|
567
|
+
else
|
|
568
|
+
printf '{"pid":null,"alive":false,"startTime":null}'
|
|
569
|
+
fi
|
|
570
|
+
}
|
|
571
|
+
revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
|
|
572
|
+
# is_healthy bails at the empty-pid check before daemon_last_exit (its label
|
|
573
|
+
# resolver) runs, so resolve them here — the down case is exactly the one
|
|
574
|
+
# where nothing upstream did.
|
|
575
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels
|
|
576
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
577
|
+
[ -f "$LIB_REVIVE" ] || return 0
|
|
578
|
+
# ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
|
|
579
|
+
# still has a pid — that is the health gate's business, and a kickstart would
|
|
580
|
+
# fight the very restart it is doing. pgrep-empty is Jacob's lane.
|
|
581
|
+
[ -z "$(daemon_pids)" ] || return 0
|
|
582
|
+
local st beat lpid palive verdict
|
|
583
|
+
st="$(daemon_launchctl_state)"
|
|
584
|
+
beat="$(daemon_beat_age_ms)"
|
|
585
|
+
# PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
|
|
586
|
+
# A listed pid is not a running daemon — verify it, and pass pidAlive so
|
|
587
|
+
# shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
|
|
588
|
+
# stamping it alive (which would deadlock: pgrep-empty says down, launchctl
|
|
589
|
+
# says pid, and nothing revives).
|
|
590
|
+
lpid="$(daemon_launchctl_pid)"
|
|
591
|
+
if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
|
|
592
|
+
verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
|
|
593
|
+
case "$verdict" in
|
|
594
|
+
revive) : ;;
|
|
595
|
+
*) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
|
|
596
|
+
esac
|
|
597
|
+
log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
|
|
598
|
+
launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
|
|
599
|
+
# ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
|
|
600
|
+
# passes a phantom pid AND a come-up-then-crash whose local beat it keeps
|
|
601
|
+
# stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
|
|
602
|
+
# live process (same pid, same start-time, alive on both) — the uptime
|
|
603
|
+
# continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
|
|
604
|
+
# hourly backstop does not round-trip the org just to confirm); continuity
|
|
605
|
+
# carries it, and the daemon's own presence publish is the fleet-visible proof.
|
|
606
|
+
local s1 s2 samples ok
|
|
607
|
+
s1="$(daemon_sample_json)"
|
|
608
|
+
sleep "$DAEMON_CONFIRM_SETTLE_S"
|
|
609
|
+
s2="$(daemon_sample_json)"
|
|
610
|
+
samples="[$s1,$s2]"
|
|
611
|
+
ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
|
|
612
|
+
if [ "$ok" = "ok" ]; then
|
|
613
|
+
log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
|
|
614
|
+
rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
|
|
615
|
+
return 0
|
|
616
|
+
fi
|
|
617
|
+
# ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
|
|
618
|
+
# this failure's log under the loop). One attempt, then hand up with what it
|
|
619
|
+
# saw: the launchctl list line and the tail of the daemon's own log.
|
|
620
|
+
local lc_line dlog_tail evidence
|
|
621
|
+
lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
|
|
622
|
+
dlog_tail="$(tail -n 3 "$DLOG_DIR/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
|
|
623
|
+
evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
|
|
624
|
+
log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
|
|
625
|
+
write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
|
|
626
|
+
return 1
|
|
627
|
+
}
|
|
460
628
|
session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
|
|
461
629
|
label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
|
|
462
630
|
pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
|
|
463
631
|
return 0
|
|
464
632
|
}
|
|
465
|
-
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
|
|
633
|
+
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
|
|
466
634
|
write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
|
|
467
635
|
local dir="$AGENT_DIR/state/session" tmp
|
|
468
636
|
mkdir -p "$dir" 2>/dev/null || return 0
|
|
@@ -753,6 +921,14 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
753
921
|
esac
|
|
754
922
|
fi
|
|
755
923
|
else
|
|
924
|
+
# Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
|
|
925
|
+
# running process while the front door kept beating — the seat looked "up
|
|
926
|
+
# to date" every hour and was dark. The daemon cannot restart itself, so
|
|
927
|
+
# revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
|
|
928
|
+
# here and asserts the pid came back. Gated to the no-process case, so a
|
|
929
|
+
# crash-looping or beat-unhealthy daemon (which still has a pid) is left to
|
|
930
|
+
# the gate above, not kickstarted.
|
|
931
|
+
revive_daemon || true
|
|
756
932
|
log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
|
|
757
933
|
# The `unhealthy-current:` PREFIX is load-bearing (the episode check below
|
|
758
934
|
# and failed_hold_reason both match on it), so the ahead-of-registry fact
|
|
@@ -777,10 +953,19 @@ fi
|
|
|
777
953
|
# nothing.
|
|
778
954
|
FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
|
|
779
955
|
# How long a front door may go without beating before this run restarts it FOR
|
|
780
|
-
# it.
|
|
781
|
-
#
|
|
782
|
-
#
|
|
783
|
-
|
|
956
|
+
# it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
|
|
957
|
+
# a wedged front door is measured in the same window as a wedged daemon rather
|
|
958
|
+
# than the days it took to notice the last three. `revive_session` (this run,
|
|
959
|
+
# hourly) is the BACKSTOP; it does not need its own, looser number.
|
|
960
|
+
SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
|
|
961
|
+
# The daemon confirms a shut front door over TWO reads before it kickstarts
|
|
962
|
+
# (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
|
|
963
|
+
# 60s apart). `revive_session`, the hourly backstop, now does the same: a
|
|
964
|
+
# supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
|
|
965
|
+
# dying and the new one re-arming, and a single stale read landing inside that
|
|
966
|
+
# gap must not kickstart a session that is mid-resume. The two reads are 60s
|
|
967
|
+
# apart; the hourly backstop can afford the one wait.
|
|
968
|
+
SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
|
|
784
969
|
failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
|
|
785
970
|
node -e '
|
|
786
971
|
const [file, latest, holdS] = process.argv.slice(1);
|