@cohortapp/agent-sdk 2.18.15 → 2.18.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -43,7 +43,17 @@
43
43
  # STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
44
44
  # (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
45
45
  # loop cannot straddle both samples by luck, and well inside the hourly
46
- # cadence. A 25 s loop fails on the second sample and on uptime.
46
+ # cadence. A 25 s loop fails on the second sample and on uptime. A pid
47
+ # CHANGE within the window is not, on its own, proof of that loop: this
48
+ # script's own restart_daemon (or an operator's `launchctl kickstart -k`)
49
+ # changes the pid deliberately and can land inside the same window —
50
+ # 2.18.14 read that as a crash loop and rolled back a release that had
51
+ # done nothing wrong. daemon_last_exit() tells the two apart: a crash
52
+ # loop's reaped exit is a real non-zero every time; a deliberate
53
+ # kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
54
+ # real non-zero exit fails here on the pid change alone — a clean or
55
+ # unknown exit falls through to the uptime check, which still catches a
56
+ # genuinely unstable respawn (its pid is always younger than STABLE_S).
47
57
  # (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
48
58
  # lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
49
59
  # ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
@@ -270,6 +280,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
270
280
  HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
271
281
  DAEMON_LABEL=""
272
282
  SESSION_LABEL=""
283
+ # The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
284
+ # below — the on-host backstop for a daemon that is DOWN, which a live daemon
285
+ # can never observe about itself. Defined up here (not with SESSION_STALE_S,
286
+ # which lives in the not-up-to-date path) because revive_daemon fires on the
287
+ # UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
288
+ # 90s front-door observers.
289
+ LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
290
+ DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
291
+ # Gap between the two settle-window reads that prove the kickstarted daemon is
292
+ # one continuous process, not a come-up-then-crash loop. A one-shot read would
293
+ # pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
294
+ DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
273
295
 
274
296
  daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
275
297
  pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
@@ -355,14 +377,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
355
377
  [ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
356
378
  return 0
357
379
  }
380
+ daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
381
+ [ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
382
+ [ -n "$DAEMON_LABEL" ] || return 0
383
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
384
+ }
358
385
  is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
359
- local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
386
+ local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
360
387
  HEALTH_REASON=""
361
388
  p1="$(daemon_pids)"
362
389
  [ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
363
390
  [ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
364
391
  p2="$(daemon_pids)"
365
- [ "$p1" = "$p2" ] || { HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}) — crash loop"; return 1; }
392
+ if [ "$p1" != "$p2" ]; then
393
+ # A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
394
+ # (this same script's own restart_daemon, or an operator's) changes the pid
395
+ # deliberately and can land inside this very window. 2.18.14 read that as a
396
+ # crash loop and rolled back a release that had done nothing wrong. The
397
+ # daemon's OWN last exit says which one happened: a crash-looping process
398
+ # exits non-zero every time launchd reaps it; a clean kickstart's reaped
399
+ # exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
400
+ # non-zero exit is a crash loop here; a clean or unknown exit is treated as
401
+ # an intended restart and falls through to the uptime check below, which
402
+ # still catches a genuinely unstable respawn (a crash-looping daemon's new
403
+ # pid is always younger than STABLE_S).
404
+ exit_code="$(daemon_last_exit)"
405
+ case "$exit_code" in
406
+ ""|"-"|"0")
407
+ # A clean/unknown exit with NO new pid at all is not a restart in
408
+ # progress, it is the daemon gone — the "no daemon process" case, not a
409
+ # crash loop and not healthy.
410
+ [ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
411
+ log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
412
+ p1="$p2"
413
+ ;;
414
+ *)
415
+ HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
416
+ return 1
417
+ ;;
418
+ esac
419
+ fi
366
420
  oldest=""
367
421
  for pid in $p1; do
368
422
  up="$(pid_uptime_s "$pid")"
@@ -450,19 +504,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
450
504
  # Leave that one to session_reconciled, which already names it.
451
505
  label_loaded "$SESSION_LABEL" || return 0
452
506
  session_beating && return 0
453
- log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — kickstarting it rather than waiting for a session that cannot answer"
507
+ # Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
508
+ # not enough to drop a session, because a supervisor resume gaps the beat for
509
+ # a moment. Confirm the door is STILL silent on a second read 60s later
510
+ # before kickstarting — if it beat again in between, that was a resume gap,
511
+ # not a wedge, and the session keeps its context.
512
+ log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
513
+ sleep "$SESSION_REVIVE_RECHECK_S"
514
+ session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
515
+ # Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
516
+ # note. The supervisor holds the `session` singleton (state/locks/process/
517
+ # session.pid) for its whole life and beats as a healthy resume progresses.
518
+ # We are past the FULL grace and two reads apart, so a resume gap is already
519
+ # excluded by time — a live holder here is a supervisor PARKED in its launch
520
+ # probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
521
+ # Consulting the lock tells a parked supervisor (override, logged) from a dead
522
+ # one (clean restart); it never blocks the parked case a hard skip would.
523
+ local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
524
+ [ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
525
+ if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
526
+ log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
527
+ fi
528
+ log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
454
529
  launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
455
530
  && { log "front door $SESSION_LABEL restarted"; return 0; }
456
531
  log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
457
532
  return 1
458
533
  }
459
534
 
535
+ daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
536
+ [ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
537
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
538
+ NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
539
+ END { if (!f) print "absent" }'
540
+ }
541
+ daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
542
+ node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
543
+ "$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
544
+ }
545
+ REVIVE_NOTE_REL="state/telemetry/revive-note.json"
546
+ write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
547
+ local dir="$AGENT_DIR/state/telemetry" tmp
548
+ mkdir -p "$dir" 2>/dev/null || return 0
549
+ tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
550
+ printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
551
+ "$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
552
+ && mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
553
+ return 0
554
+ }
555
+ daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
556
+ [ -n "$DAEMON_LABEL" ] || return 0
557
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
558
+ }
559
+ daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
560
+ local pid start
561
+ pid="$(daemon_launchctl_pid)"
562
+ if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
563
+ start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
564
+ printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
565
+ elif [ -n "$pid" ]; then
566
+ printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
567
+ else
568
+ printf '{"pid":null,"alive":false,"startTime":null}'
569
+ fi
570
+ }
571
+ revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
572
+ # is_healthy bails at the empty-pid check before daemon_last_exit (its label
573
+ # resolver) runs, so resolve them here — the down case is exactly the one
574
+ # where nothing upstream did.
575
+ [ -n "$DAEMON_LABEL" ] || resolve_labels
576
+ [ -n "$DAEMON_LABEL" ] || return 0
577
+ [ -f "$LIB_REVIVE" ] || return 0
578
+ # ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
579
+ # still has a pid — that is the health gate's business, and a kickstart would
580
+ # fight the very restart it is doing. pgrep-empty is Jacob's lane.
581
+ [ -z "$(daemon_pids)" ] || return 0
582
+ local st beat lpid palive verdict
583
+ st="$(daemon_launchctl_state)"
584
+ beat="$(daemon_beat_age_ms)"
585
+ # PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
586
+ # A listed pid is not a running daemon — verify it, and pass pidAlive so
587
+ # shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
588
+ # stamping it alive (which would deadlock: pgrep-empty says down, launchctl
589
+ # says pid, and nothing revives).
590
+ lpid="$(daemon_launchctl_pid)"
591
+ if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
592
+ verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
593
+ case "$verdict" in
594
+ revive) : ;;
595
+ *) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
596
+ esac
597
+ log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
598
+ launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
599
+ # ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
600
+ # passes a phantom pid AND a come-up-then-crash whose local beat it keeps
601
+ # stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
602
+ # live process (same pid, same start-time, alive on both) — the uptime
603
+ # continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
604
+ # hourly backstop does not round-trip the org just to confirm); continuity
605
+ # carries it, and the daemon's own presence publish is the fleet-visible proof.
606
+ local s1 s2 samples ok
607
+ s1="$(daemon_sample_json)"
608
+ sleep "$DAEMON_CONFIRM_SETTLE_S"
609
+ s2="$(daemon_sample_json)"
610
+ samples="[$s1,$s2]"
611
+ ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
612
+ if [ "$ok" = "ok" ]; then
613
+ log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
614
+ rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
615
+ return 0
616
+ fi
617
+ # ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
618
+ # this failure's log under the loop). One attempt, then hand up with what it
619
+ # saw: the launchctl list line and the tail of the daemon's own log.
620
+ local lc_line dlog_tail evidence
621
+ lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
622
+ dlog_tail="$(tail -n 3 "$DLOG_DIR/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
623
+ evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
624
+ log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
625
+ write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
626
+ return 1
627
+ }
460
628
  session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
461
629
  label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
462
630
  pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
463
631
  return 0
464
632
  }
465
- json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
633
+ json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
466
634
  write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
467
635
  local dir="$AGENT_DIR/state/session" tmp
468
636
  mkdir -p "$dir" 2>/dev/null || return 0
@@ -753,6 +921,14 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
753
921
  esac
754
922
  fi
755
923
  else
924
+ # Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
925
+ # running process while the front door kept beating — the seat looked "up
926
+ # to date" every hour and was dark. The daemon cannot restart itself, so
927
+ # revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
928
+ # here and asserts the pid came back. Gated to the no-process case, so a
929
+ # crash-looping or beat-unhealthy daemon (which still has a pid) is left to
930
+ # the gate above, not kickstarted.
931
+ revive_daemon || true
756
932
  log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
757
933
  # The `unhealthy-current:` PREFIX is load-bearing (the episode check below
758
934
  # and failed_hold_reason both match on it), so the ahead-of-registry fact
@@ -777,10 +953,19 @@ fi
777
953
  # nothing.
778
954
  FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
779
955
  # How long a front door may go without beating before this run restarts it FOR
780
- # it. Ten minutes: long enough that a busy session is never interrupted (the
781
- # heartbeat rides every tool use), short enough that a wedged one is measured in
782
- # minutes rather than the days it took to notice the last three.
783
- SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-600}"
956
+ # it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
957
+ # a wedged front door is measured in the same window as a wedged daemon rather
958
+ # than the days it took to notice the last three. `revive_session` (this run,
959
+ # hourly) is the BACKSTOP; it does not need its own, looser number.
960
+ SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
961
+ # The daemon confirms a shut front door over TWO reads before it kickstarts
962
+ # (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
963
+ # 60s apart). `revive_session`, the hourly backstop, now does the same: a
964
+ # supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
965
+ # dying and the new one re-arming, and a single stale read landing inside that
966
+ # gap must not kickstart a session that is mid-resume. The two reads are 60s
967
+ # apart; the hourly backstop can afford the one wait.
968
+ SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
784
969
  failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
785
970
  node -e '
786
971
  const [file, latest, holdS] = process.argv.slice(1);