@cohortapp/agent-sdk 2.18.13 → 2.18.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/bin/maestro.mjs +38 -1
  2. package/docs/runbooks/fleet-rollout.md +58 -7
  3. package/docs/runbooks/recovery-and-failover.md +18 -0
  4. package/lib/assurance/batch.mjs +353 -0
  5. package/lib/assurance/first-reply.mjs +423 -0
  6. package/lib/assurance/notice-voice.mjs +357 -0
  7. package/lib/assurance/plan-note.mjs +43 -0
  8. package/lib/assurance/room-budget.mjs +55 -6
  9. package/lib/cadence-failure-class.mjs +245 -0
  10. package/lib/claude-bin.mjs +26 -7
  11. package/lib/cli/doctor-checks.mjs +149 -1
  12. package/lib/comms/send-gate.mjs +59 -0
  13. package/lib/diagnostics/alerts.mjs +33 -0
  14. package/lib/engine/agents/usage.mjs +45 -0
  15. package/lib/engine/budget.mjs +293 -29
  16. package/lib/engine/cli.mjs +54 -5
  17. package/lib/engine/loop.mjs +30 -0
  18. package/lib/engine/output/json.mjs +26 -0
  19. package/lib/engine/wire/errors.mjs +179 -0
  20. package/lib/engine/wire/search.mjs +44 -8
  21. package/lib/identity/persona.mjs +31 -2
  22. package/lib/org/quota.mjs +27 -0
  23. package/lib/session/config.mjs +4 -0
  24. package/lib/session/identity.mjs +71 -7
  25. package/lib/session/launch-failure.mjs +251 -0
  26. package/lib/session/resume-target.mjs +86 -0
  27. package/lib/telemetry/alerts.mjs +94 -0
  28. package/lib/telemetry/collect.mjs +155 -2
  29. package/lib/upgrade/pinned-drift.mjs +467 -0
  30. package/package.json +1 -1
  31. package/scaffold/config/alerts.yaml +7 -0
  32. package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
  33. package/scripts/ci/check.mjs +3 -0
  34. package/scripts/daemon/agent-daemon.mjs +75 -5
  35. package/scripts/daemon/assurance.mjs +709 -44
  36. package/scripts/daemon/cadence-consumer.mjs +281 -34
  37. package/scripts/daemon/deliver.mjs +109 -0
  38. package/scripts/daemon/dispatcher.mjs +21 -3
  39. package/scripts/daemon/inbox-deferral.mjs +102 -9
  40. package/scripts/daemon/session-lock.mjs +41 -1
  41. package/scripts/emergency-stop.sh +114 -13
  42. package/scripts/fleet/rollout.mjs +256 -10
  43. package/scripts/healthcheck.sh +131 -33
  44. package/scripts/local-triggers/autoupdate.sh +144 -11
  45. package/scripts/resume-operations.sh +101 -6
  46. package/scripts/session/supervisor.mjs +198 -5
@@ -481,18 +481,133 @@ write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell t
481
481
  }
482
482
  write_upgrade_notice(){ write_notice "$1" "$2"; }
483
483
  LAST_JSON="$AGENT_DIR/state/autoupdate/last.json"
484
- write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason — state/autoupdate/last.json (→ beat machine.upgrade)
485
- local dir="$AGENT_DIR/state/autoupdate" tmp
484
+ # How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK, and
485
+ # at what point it stops being a warning. See the `failStreak` block below.
486
+ STUCK_STREAK="${MAESTRO_AUTOUPDATE_STUCK_STREAK:-3}"
487
+ # ── failStreak — "this seat cannot get to @latest" is a FACT, not a discovery ─
488
+ # The fleet learned on 2026-09-25 that three seats had sat on 2.17.0 for days.
489
+ # Nothing was broken in the sense anything reports: the launchd job fired every
490
+ # hour, npm answered, the install ran, the health gate failed, the rollback
491
+ # worked, last.json recorded it honestly, and the beat carried it. Each hour was
492
+ # a correct, self-contained failure — and a correct failure repeated forty times
493
+ # is a different fact from a correct failure once. Only the REPETITION says the
494
+ # seat will never arrive on its own, and no single record can hold it, so the
495
+ # count lives in last.json and rides the beat the rest of that record already
496
+ # rides. There is no second channel here and there must not be one: the seats
497
+ # this is about are, by construction, the ones running the oldest code.
498
+ #
499
+ # WHAT COUNTS AS AN ATTEMPT — the whole honesty of the number:
500
+ # · An ATTEMPT is a record whose `to` differs from its `from`: the upgrade
501
+ # path actually installed (or tried to install) a different version. Those
502
+ # are the only records that move the streak.
503
+ # · A FAILED attempt (`ok:false`) increments it and stamps `stuckSince` with
504
+ # the first failure of the run of failures, so a reader gets a DURATION and
505
+ # not merely a tally — "4 attempts since Monday" and "4 attempts in the last
506
+ # hour" are different seats.
507
+ # · A SUCCEEDED attempt (`ok:true`) clears all three fields. Arrival is the
508
+ # only thing that resets it; a fleet-wide publish does not, because a seat
509
+ # that fails 2.18.12, 2.18.13 and 2.18.14 in turn has failed three times to
510
+ # do the one thing asked of it, and restarting the count on every release
511
+ # would guarantee the number never reaches any threshold.
512
+ # · Everything else PRESERVES the fields untouched. A `to == from` record is
513
+ # a report on the installed version, not an attempt to leave it, so it must
514
+ # not move a count of ATTEMPTS in either direction. (Every such call site
515
+ # today sits on the up-to-date branch and therefore takes the arrival path
516
+ # below instead — this rule governs the writer, for the next caller that
517
+ # does not.) And a run that SKIPS writes no record at all, which is the
518
+ # same preservation by a shorter route. THIS IS THE POINT OF (d): a seat that has not been ASKED —
519
+ # nothing newer published, or held for the day after its last failure — is
520
+ # not a stuck seat, and must never accumulate a streak for sitting still.
521
+ # Only a real, completed, failed attempt does.
522
+ # · WHICH RUNS ACTUALLY SKIP — the exact list, because a wrong one was
523
+ # written here first. The paths that reach `exit 0` with NO write_last are:
524
+ # the failed-target hold, the overlap lock, the kill-switch, and a failed
525
+ # `npm view`. "SIMPLY BEING UP TO DATE" IS NOT AMONG THEM and never was —
526
+ # the up-to-date branch is the daemon health gate, and it calls write_last
527
+ # on every one of its outcomes (healthy, unhealthy-current,
528
+ # ahead-of-registry, recovered, and the stale-daemon kickstart). The
529
+ # original text claimed otherwise; nothing checked it, and the streak
530
+ # arithmetic on that branch is not what the sentence implied.
531
+ # · ARRIVAL IS BEING AT @latest, NOT MERELY INSTALLING IT. Every write on the
532
+ # up-to-date branch passes `at_latest=true`, which clears all three fields
533
+ # — because that branch is only entered when `CUR >= LATEST`, i.e. there is
534
+ # nothing left for this seat to reach. That covers the case the increment
535
+ # rule alone gets wrong: if a bad release is YANKED and @latest rolls back
536
+ # to the version a seat already runs, the seat stops attempting anything,
537
+ # so nothing would ever reset it and `upgrade_stuck` would fire forever
538
+ # (critical past 6) on a seat that is in fact perfectly current. It also
539
+ # makes the stale-daemon kickstart record honest rather than accidental:
540
+ # that hop really is an arrival.
541
+ # The write is done in node rather than printf because the record now depends on
542
+ # the record before it; a shell that reads a file it is about to overwrite gets
543
+ # that wrong at exactly the moments it matters.
544
+ write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason, [$6 at_latest] — state/autoupdate/last.json (→ beat machine.upgrade)
545
+ local dir="$AGENT_DIR/state/autoupdate" tmp streak at_latest="${6:-false}"
486
546
  mkdir -p "$dir" 2>/dev/null || return 0
487
547
  tmp="$(mktemp "$dir/.last.XXXXXX" 2>/dev/null)" || return 0
488
- printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
489
- "$1" "$2" "$(date -u +%FT%TZ)" "$3" "$4" "$(json_str "$5")" > "$tmp" && mv -f "$tmp" "$LAST_JSON"
548
+ streak="$(node -e '
549
+ const fs = require("fs");
550
+ const [file, tmp, from, to, okS, healthyS, reason, stuckStreakS, atLatestS] = process.argv.slice(1);
551
+ const ok = okS === "true";
552
+ const at = new Date().toISOString().replace(/\.\d{3}Z$/, "Z");
553
+ let prev = null;
554
+ try { prev = JSON.parse(fs.readFileSync(file, "utf8")); } catch {}
555
+ if (!prev || typeof prev !== "object") prev = {};
556
+ const prevStreak = Number.isInteger(prev.failStreak) && prev.failStreak > 0 ? prev.failStreak : 0;
557
+ const attempt = from !== to; // an upgrade was actually tried
558
+ const atLatest = atLatestS === "true"; // this seat is AT or past @latest — nothing left to reach
559
+ let failStreak, stuckSince, streakTarget;
560
+ if (atLatest || (attempt && ok)) { // arrival, by install OR because @latest came back to us (a yank)
561
+ failStreak = 0; stuckSince = ""; streakTarget = "";
562
+ } else if (!attempt) { // a report on the installed version — carry the count, do not touch it
563
+ failStreak = prevStreak;
564
+ stuckSince = typeof prev.stuckSince === "string" ? prev.stuckSince : "";
565
+ streakTarget = typeof prev.streakTarget === "string" ? prev.streakTarget : "";
566
+ } else {
567
+ failStreak = prevStreak + 1;
568
+ stuckSince = typeof prev.stuckSince === "string" && prev.stuckSince ? prev.stuckSince : at;
569
+ streakTarget = to;
570
+ }
571
+ const rec = { from, to, at, ok, healthy: healthyS === "true", reason };
572
+ if (failStreak > 0) {
573
+ rec.failStreak = failStreak;
574
+ if (stuckSince) rec.stuckSince = stuckSince;
575
+ if (streakTarget) rec.streakTarget = streakTarget;
576
+ }
577
+ fs.writeFileSync(tmp, JSON.stringify(rec, null, 2) + "\n");
578
+ // stdout is the shell'"'"'s only view of what was decided: "<streak> <target> <since>"
579
+ process.stdout.write(failStreak >= Number(stuckStreakS) && attempt && !ok ? `STUCK ${failStreak} ${streakTarget} ${stuckSince}` : "");
580
+ ' "$LAST_JSON" "$tmp" "$1" "$2" "$3" "$4" "$5" "$STUCK_STREAK" "$at_latest" 2>/dev/null)" || streak=""
581
+ if [ -s "$tmp" ]; then
582
+ mv -f "$tmp" "$LAST_JSON"
583
+ else
584
+ # The node writer produced nothing. Before this branch existed the tmp was
585
+ # simply removed and last.json was LEFT AT ITS PREVIOUS VALUE — so a write
586
+ # failure did not lose a record, it published a STALE one, and the 24 h
587
+ # failed-target hold (which reads `to` and `reason` out of this file) then
588
+ # made its decision from a previous run's outcome while believing it was
589
+ # this run's. A missing record is fail-open by design here; a wrong one is
590
+ # not. So write the event itself with printf, which needs no interpreter,
591
+ # and say loudly that the count did not survive: the streak is the one
592
+ # field that cannot be reconstructed without reading the old file, and a
593
+ # silently reset counter is the failure this whole block exists to prevent.
594
+ log "WARN: write_last could not run node — recording the bare outcome without the failStreak. The count restarts from this run; see state/autoupdate/last.json."
595
+ printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
596
+ "$1" "$2" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$3" "$4" "$(printf '%s' "$5" | sed 's/[\\"]/\\&/g')" > "$LAST_JSON" 2>/dev/null || true
597
+ fi
490
598
  rm -f "$tmp" 2>/dev/null
599
+ if [ -n "$streak" ]; then
600
+ set -- $streak
601
+ log "STUCK: $2 consecutive failed attempts to reach $3 (since $4). This seat will not arrive on its own — the org now carries machine.upgrade.failStreak=$2 and an \`upgrade_stuck\` alert on every beat. maestro doctor; then rm state/autoupdate/last.json to retry immediately."
602
+ fi
491
603
  return 0
492
604
  }
493
605
  last_reason(){ # the reason field of last.json, or nothing
494
606
  node -e 'try { process.stdout.write(String(JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).reason || "")); } catch {}' "$LAST_JSON" 2>/dev/null
495
607
  }
608
+ last_streak(){ # the failStreak of last.json as a positive integer, or nothing
609
+ node -e 'try { const n = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).failStreak; if (Number.isInteger(n) && n > 0) process.stdout.write(String(n)); } catch {}' "$LAST_JSON" 2>/dev/null
610
+ }
496
611
  # The version the RUNNING daemon started on. scripts/daemon/health.mjs resolves
497
612
  # sdk_version once at load and writes it to state/dashboards/daemon-health.yaml
498
613
  # with its pid, precisely so a reader can tell the installed package from the
@@ -595,13 +710,13 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
595
710
  # the notice tells the session to restart itself onto the new code.
596
711
  log "OK: daemon now on $CUR (was $STALE_FROM) — the manual upgrade is complete"
597
712
  write_upgrade_notice "$STALE_FROM" "$CUR"
598
- write_last "$STALE_FROM" "$CUR" true true ""
713
+ write_last "$STALE_FROM" "$CUR" true true "" true
599
714
  elif [ -n "$STALE_FROM" ]; then
600
715
  # Same version, newer files: the daemon is reconciled; the session is
601
716
  # not told to restart (a touched or restored file must not cost it its
602
717
  # context — the version did not change).
603
718
  log "OK: daemon restarted onto the installed $CUR code; no session notice (same version)"
604
- write_last "$CUR" "$CUR" true true ""
719
+ write_last "$CUR" "$CUR" true true "" true
605
720
  elif [ -n "$AHEAD" ]; then
606
721
  # The DAEMON is well; the FLEET is not. Recorded as ok/healthy — because
607
722
  # it is — with the finding in `reason`, so the org learns that this seat
@@ -609,14 +724,32 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
609
724
  # per episode: unlike an unhealthy daemon this does not clear itself, and
610
725
  # a fact that ages out of one log file is how six days went by.
611
726
  log "OK: healthy on $CUR, but AHEAD of the registry ($LATEST) — recorded for the org as ahead-of-registry"
612
- write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST"
727
+ write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST" true
613
728
  else
614
- case "$PREV_REASON" in unhealthy-current*)
729
+ case "$PREV_REASON" in
730
+ unhealthy-current*)
615
731
  log "recovered: last.json no longer reports unhealthy-current"
616
- write_last "$CUR" "$CUR" true true "" ;;
732
+ write_last "$CUR" "$CUR" true true "" true ;;
617
733
  ahead-of-registry*)
618
734
  log "recovered: $CUR is on the registry now (latest $LATEST) — last.json no longer reports ahead-of-registry"
619
- write_last "$CUR" "$CUR" true true "" ;;
735
+ write_last "$CUR" "$CUR" true true "" true ;;
736
+ *)
737
+ # A healthy seat that is simply up to date writes NOTHING — there is
738
+ # nothing to report and an empty last.json is the honest state.
739
+ #
740
+ # The one exception is a standing failStreak, and it is the reason this
741
+ # branch exists at all. A streak only ever reset on ARRIVAL, i.e. on a
742
+ # successful hop. If a bad release is YANKED and @latest rolls back to
743
+ # the version this seat already runs, the seat stops attempting
744
+ # anything: every hour it takes this quiet path, writes nothing, and the
745
+ # count stands for ever — `upgrade_stuck` firing critical on a seat that
746
+ # is, in fact, exactly where the registry wants it. Being AT @latest is
747
+ # an arrival however you got there, so record one and say so.
748
+ STREAK="$(last_streak)"
749
+ if [ -n "$STREAK" ]; then
750
+ log "streak cleared: $CUR is @latest and this seat is healthy on it, but last.json still carried failStreak=$STREAK — @latest has come back to this version (a yank or a withdrawn release). Recording the arrival so the upgrade_stuck alert stops."
751
+ write_last "$CUR" "$CUR" true true "" true
752
+ fi ;;
620
753
  esac
621
754
  fi
622
755
  else
@@ -625,7 +758,7 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
625
758
  # and failed_hold_reason both match on it), so the ahead-of-registry fact
626
759
  # is appended, never prepended — an unpublished build that is also sick is
627
760
  # the worst case and must say both things.
628
- write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}"
761
+ write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}" true
629
762
  case "$PREV_REASON" in
630
763
  unhealthy-current*) log "session already notified this episode (last.json was unhealthy-current); notice not rewritten" ;;
631
764
  *) write_notice "$CUR" "$CUR" "unhealthy-current: $HEALTH_REASON" ;;
@@ -3,12 +3,34 @@
3
3
  # Usage: ./scripts/resume-operations.sh
4
4
  #
5
5
  # Reverses emergency-stop.sh:
6
- # 1. Verifies health check passes.
6
+ # 1. Verifies health check passes — EXCLUDING the stop flag itself.
7
7
  # 2. Removes the .emergency-stop flag.
8
8
  # 3. Reloads every installed `ai.maestro.<agent>-*` (and legacy
9
9
  # `ai.adaptic.<agent>-*`) launchd job.
10
10
  #
11
11
  # Agent first-name slug resolved from config/agent.json (SOT).
12
+ #
13
+ # THE DEADLOCK THIS SCRIPT USED TO HAVE. healthcheck.sh counts an active
14
+ # .emergency-stop flag as an ERROR; this script ran it bare and exited on any
15
+ # non-zero. So the one tool whose entire purpose is lifting the stop could
16
+ # never lift it — Eli Rosenberg's seat sat halted from 2026-09-24T08:16Z until
17
+ # the flag was removed by hand. Two independent faults produced that:
18
+ #
19
+ # (a) The stop flag was its own veto. Fixed by passing
20
+ # `--ignore-emergency-stop`, which demotes that ONE check to a note.
21
+ # It is a PUBLIC CLI flag — nothing refuses it to another caller, and
22
+ # nothing here pretends to (healthcheck.sh's header says so plainly).
23
+ # What makes it safe is scope, not exclusivity: it demotes exactly one
24
+ # check, every other check still gates the resume, and step 1b below
25
+ # VERIFIES that the healthcheck on this disk actually honoured it
26
+ # rather than assuming the two files are the same vintage.
27
+ # (b) DEGRADED was treated as failure. healthcheck exits 1 for warnings and
28
+ # 2 for errors; `if ! healthcheck` refused on both, so seven warnings
29
+ # about optional repos and an uninstalled Slack.app were enough to keep
30
+ # a seat halted. Warnings are "operational with limitations" by
31
+ # healthcheck's own contract — they are reported loudly and resumed.
32
+ # Only CRITICAL (exit 2) still refuses, and the refusal names which
33
+ # condition blocked it.
12
34
 
13
35
  set -e
14
36
 
@@ -46,19 +68,92 @@ if [ ! -f "$AGENT_DIR/.emergency-stop" ]; then
46
68
  launchctl load "$plist" 2>/dev/null && loaded=$((loaded + 1))
47
69
  fi
48
70
  done
49
- [ "$loaded" -gt 0 ] && echo "Loaded $loaded missing launchd job(s)"
71
+ # `if`, not `[ "$loaded" -gt 0 ] && echo ...` — but NOT for the reason
72
+ # first written here, which was wrong and would have seeded a wrong rule.
73
+ # `set -e` does not abort on the false short-circuit: bash exempts a
74
+ # failing command that is not the last in an && list. MEASURED:
75
+ # bash -c 'set -e; n=0; [ "$n" -gt 0 ] && echo hi; exit 0' → exit 0
76
+ # and the pre-fix script (which had `exit 0` on the next line) also
77
+ # exited 0 on a seat with nothing to load. The real hazard is narrower
78
+ # and has nothing to do with `set -e`: a false && list leaves status 1,
79
+ # so a script whose LAST command is one exits 1. `exit 0` follows today;
80
+ # the `if` keeps this line harmless if it ever stops following.
81
+ if [ "$loaded" -gt 0 ]; then echo "Loaded $loaded missing launchd job(s)"; fi
50
82
  exit 0
51
83
  fi
52
84
 
53
85
  echo "[$TIMESTAMP] RESUMING OPERATIONS (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
54
86
 
55
- # 1. Run health check first.
56
- echo "Running health check..."
57
- if ! "$SCRIPT_DIR/healthcheck.sh"; then
58
- echo "ERROR: Health check failed. Fix issues before resuming."
87
+ # 1. Run health check first — with the stop flag exempted, since lifting it is
88
+ # this script's entire job. Every OTHER check still gates the resume.
89
+ echo "Running health check (emergency-stop flag exempted)..."
90
+ set +e
91
+ HEALTH_OUTPUT=$("$SCRIPT_DIR/healthcheck.sh" --ignore-emergency-stop 2>&1)
92
+ HEALTH_STATUS=$?
93
+ set -e
94
+ echo "$HEALTH_OUTPUT"
95
+
96
+ # 1b. Prove the exemption was HONOURED before trusting the verdict.
97
+ #
98
+ # Fault (a)'s fix is version-coupled: it only works if the healthcheck.sh
99
+ # next to this script understands --ignore-emergency-stop. A seat that
100
+ # pins framework files (.maestroignore) can carry the new resume script
101
+ # beside a pre-fix healthcheck, which has no argument parsing at all — it
102
+ # swallows the flag, still counts the stop as an error, and prints no
103
+ # `Blocking:` line. MEASURED on that exact pairing: the deadlock is back,
104
+ # and the refusal reads "(healthcheck named no condition; see output
105
+ # above)" — worse than the original, because it names nothing to fix.
106
+ # The seats most likely to be halted are the ones most likely to be
107
+ # pinned, so this has to be detected and named rather than inferred.
108
+ #
109
+ # We are past the `[ -f .emergency-stop ]` guard, so a healthcheck that
110
+ # honoured the flag MUST have printed the NOTE line. Its absence means
111
+ # the exemption did not take: stale, missing, or not executable.
112
+ if ! printf '%s\n' "$HEALTH_OUTPUT" | grep -q '^\[NOTE\] Emergency stop flag' ||
113
+ printf '%s\n' "$HEALTH_OUTPUT" | grep -q '^Blocking:.*emergency-stop-flag'; then
114
+ echo ""
115
+ echo "REFUSING TO RESUME — scripts/healthcheck.sh is stale."
116
+ echo " --ignore-emergency-stop was ignored (healthcheck exit $HEALTH_STATUS): it did"
117
+ echo " not report '[NOTE] Emergency stop flag ... not counted', so its verdict still"
118
+ echo " counts the very flag this script exists to lift. Resuming on that verdict"
119
+ echo " would be resuming on a health check nobody read."
120
+ echo " Fix: run 'maestro upgrade' on this seat, then confirm .maestroignore is not"
121
+ echo " pinning scripts/healthcheck.sh, and that the file exists and is executable."
122
+ echo " The .emergency-stop flag has been LEFT IN PLACE."
123
+ echo "[$TIMESTAMP] Resume REFUSED — healthcheck.sh stale (exemption ignored, status $HEALTH_STATUS)" >> "$LOG_FILE"
59
124
  exit 1
60
125
  fi
61
126
 
127
+ if [ "$HEALTH_STATUS" -ge 2 ]; then
128
+ # Name the blocking condition. The operator should never have to re-run
129
+ # the health check with a filter to learn which of "1 errors" stopped them.
130
+ BLOCKING=$(printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^Blocking://p' | tr -s ' ' | sed 's/^ //')
131
+ [ -n "$BLOCKING" ] || BLOCKING="(healthcheck named no condition; see output above)"
132
+ echo ""
133
+ echo "REFUSING TO RESUME — health check is CRITICAL."
134
+ echo " Blocking condition(s): $BLOCKING"
135
+ # Re-list the failing lines under the refusal, RE-PREFIXED. The whole
136
+ # health output is echoed above, so the previous `grep '^\[FAIL\]'` here
137
+ # emitted bytes indistinguishable from it — deleting the line left the
138
+ # test that claimed to pin it green. The ` FAIL: ` prefix is produced
139
+ # only here, which is what makes that assertion load-bearing.
140
+ printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^\[FAIL\] / FAIL: /p'
141
+ echo " The .emergency-stop flag has been LEFT IN PLACE."
142
+ echo " Fix the condition(s) above, then re-run: $0"
143
+ {
144
+ echo "[$TIMESTAMP] Resume REFUSED — blocking: $BLOCKING"
145
+ } >> "$LOG_FILE"
146
+ exit 1
147
+ fi
148
+
149
+ if [ "$HEALTH_STATUS" -eq 1 ]; then
150
+ DEGRADED=$(printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^Warnings://p' | tr -s ' ' | sed 's/^ //')
151
+ echo ""
152
+ echo "NOTE: health check is DEGRADED — resuming anyway (warnings are not faults)."
153
+ echo " Warning condition(s): $DEGRADED"
154
+ echo "[$TIMESTAMP] Resume proceeding DEGRADED — warnings: $DEGRADED" >> "$LOG_FILE"
155
+ fi
156
+
62
157
  # 2. Remove stop flag.
63
158
  rm "$AGENT_DIR/.emergency-stop"
64
159
  echo "[$TIMESTAMP] Stop flag removed" >> "$LOG_FILE"
@@ -42,7 +42,7 @@
42
42
  * @module scripts/session/supervisor
43
43
  */
44
44
 
45
- import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync } from "node:fs";
45
+ import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync, readdirSync as fsReaddirSync } from "node:fs";
46
46
  import { randomUUID } from "node:crypto";
47
47
  import { join } from "node:path";
48
48
  import { spawn } from "node:child_process";
@@ -66,6 +66,11 @@ import {
66
66
  launchMode, recordLaunch, rotationDecision, rotateMainSession, BACKOFF_MS,
67
67
  } from "../../lib/session/identity.mjs";
68
68
  import { detectMux, chooseMux, buildMuxCommands, parseScreenList, parseExitFile, renderEnvFile } from "../../lib/session/launch-args.mjs";
69
+ import { resumeTargetState } from "../../lib/session/resume-target.mjs";
70
+ import {
71
+ classifyLaunchFailure, launchFailureSignature, parseLaunchFailures,
72
+ launchFailureStreak, escalationDecision, escalationHint, IDENTICAL_FAILURE_LIMIT,
73
+ } from "../../lib/session/launch-failure.mjs";
69
74
 
70
75
  /** Write a private (0600) file that must not already exist. The engine env file carries the seat token. */
71
76
  export function writePrivateFile(path, text) {
@@ -83,6 +88,17 @@ export const LOCK_NAME = "session";
83
88
  export const DEFAULT_POLL_MS = 2_000;
84
89
  /** The signals the supervisor owns (launchd sends SIGTERM to stop a job). */
85
90
  export const OWNED_SIGNALS = Object.freeze(["SIGTERM", "SIGINT"]);
91
+ /**
92
+ * How long after start the pane is captured ONCE, to keep whatever the runtime
93
+ * printed before it died. A resume against a missing conversation is over in
94
+ * about two seconds, taking its mux session (and its pane) with it — so the
95
+ * capture has to happen while it is still there. Best-effort by design: when
96
+ * it comes back empty, `resumeTargetState`'s preflight is what carries the
97
+ * verdict, and the post-mortem heuristics below carry the rest.
98
+ */
99
+ export const EARLY_PANE_MS = 1_500;
100
+ /** Consecutive identical launch failures before the seat escalates to the org. */
101
+ export const LAUNCH_FAILURE_LIMIT = IDENTICAL_FAILURE_LIMIT;
86
102
 
87
103
  /** Real signal wiring: `fn(signal)` on each owned signal. */
88
104
  function installSignalHandlers(fn) {
@@ -198,6 +214,15 @@ export async function runSupervisor(deps = {}) {
198
214
  seatSpawn: deps.seatSpawn || ((root) => resolveSeatSpawn(root)),
199
215
  writePrivateFile: deps.writePrivateFile || writePrivateFile,
200
216
  envFileId: deps.envFileId || randomUUID,
217
+ readdirSync: deps.readdirSync || fsReaddirSync,
218
+ launchFailureLimit: deps.launchFailureLimit ?? LAUNCH_FAILURE_LIMIT,
219
+ earlyPaneMs: deps.earlyPaneMs ?? EARLY_PANE_MS,
220
+ // The one-shot pane probe has its OWN timer seam, separate from the
221
+ // watchdog's `after`. They are different clocks doing different jobs — one
222
+ // fires once at a second and a half, the other recurs at the grace period —
223
+ // and collapsing them would make every watchdog test reason about a timer
224
+ // that has nothing to do with the watchdog.
225
+ afterPaneProbe: deps.afterPaneProbe || deps.after || defaultAfter,
201
226
  };
202
227
  const fsDeps = { readFileSync: d.readFileSync, writeJsonAtomic: d.writeJsonAtomic };
203
228
  const agentRoot = d.agentRoot;
@@ -271,7 +296,7 @@ export async function runSupervisor(deps = {}) {
271
296
  // 4. Identity + argv.
272
297
  let record = loadMainSession(agentRoot, fsDeps);
273
298
  if (!record || cfg.resume === false) record = newMainSession({ now: d.now, uuid: d.uuid });
274
- const mode = launchMode(record, cfg);
299
+ let mode = launchMode(record, cfg);
275
300
  let permissionArgs = d.permissionArgs;
276
301
  if (!permissionArgs) {
277
302
  permissionArgs = sessionPermissionArgs({ source: MAIN_SESSION_SOURCE, allowedTools: cfg.allowedTools });
@@ -288,6 +313,38 @@ export async function runSupervisor(deps = {}) {
288
313
  let seat = { engine: "claude", fields: {} };
289
314
  try { seat = (await d.seatSpawn(agentRoot)) || seat; } catch { seat = { engine: "claude", fields: {} }; }
290
315
  const engineCohort = seat.engine === "cohort";
316
+
317
+ // ── (a) A DEAD RESUME TARGET IS CAUGHT BEFORE THE LAUNCH, NOT AFTER IT ────
318
+ //
319
+ // James Kirkland's seat, measured 2026-09-25: `claude --resume <id>` for a
320
+ // conversation that no longer existed on that machine, exit 1 after 2 s,
321
+ // relaunched every ten minutes for DAYS. state/session/main-session.json held
322
+ // the dead id and nothing ever invalidated it.
323
+ //
324
+ // THE CHOICE, MADE EXPLICIT: invalidate the record AND fall back to a fresh
325
+ // id — both, in one write, here, before anything is spawned. Invalidating
326
+ // alone would leave the supervisor with no id (and `newMainSession` would
327
+ // mint one anyway, unrecorded); falling back alone would leave the dead id on
328
+ // disk for the next reader to resume again. `rotateMainSession` does both:
329
+ // the dead id survives as `rotatedFrom` — provenance, never a target.
330
+ //
331
+ // It runs on POSITIVE EVIDENCE ONLY (`resume-target.mjs` fails open): a
332
+ // wrong "missing" throws away a live transcript, a wrong "unknown" costs one
333
+ // launch and lands in the post-mortem that was already there.
334
+ if (mode === "resume") {
335
+ const target = resumeTargetState(
336
+ { homeDir: d.homeDir, projectDir: agentRoot, sessionId: record.sessionId, engine: seat.engine },
337
+ { readdirSync: d.readdirSync },
338
+ );
339
+ if (target.state === "missing") {
340
+ const dead = record.sessionId;
341
+ record = rotateMainSession(record, { now: d.now, uuid: d.uuid, reason: `resume target ${dead} is not on this machine` });
342
+ mode = launchMode(record, cfg);
343
+ const res = saveMainSession(agentRoot, record, fsDeps);
344
+ d.log(`resume target ${dead} is gone (${target.reason}) — invalidated it and starting a fresh session ${record.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
345
+ }
346
+ }
347
+
291
348
  const spec = buildSpawn({ lane: "main-session", bin: d.claudeBin || undefined, first, sessionId: record.sessionId, resumeMode: mode, permissions: permissionArgs, env: d.env, ...seat.fields });
292
349
  if (!spec.ok && engineCohort) {
293
350
  // An engine-cohort seat never falls back to claude (it would spend the seat's
@@ -348,8 +405,34 @@ export async function runSupervisor(deps = {}) {
348
405
  const restartsSoFar = carriedRestarts({ attention: priorAttention, heartbeat: priorHeartbeat });
349
406
  if (restartsSoFar > 0) d.log(`this launch carries ${restartsSoFar} silent restart(s) from the previous run`);
350
407
 
408
+ // ── (b) THE CARRIED LAUNCH-FAILURE STREAK ────────────────────────────────
409
+ //
410
+ // Read BEFORE the unlink below, for the same reason `carriedRestarts` is: a
411
+ // count held in memory is zero on every launch, because every launch is a
412
+ // fresh process. This one lives in its own file rather than on the attention
413
+ // record, because it has to survive the unlink that clears that record.
414
+ let priorFailures = null;
415
+ try { priorFailures = parseLaunchFailures(d.readFileSync(paths.launchFailuresFile, "utf8")); } catch { priorFailures = null; }
416
+
351
417
  try { d.unlinkSync(paths.lastExitFile); } catch { /* none from a previous run */ }
352
418
  try { d.unlinkSync(paths.attentionFile); } catch { /* none outstanding */ }
419
+
420
+ // A seat already past the limit re-states the escalation on EVERY launch.
421
+ // The note rides `machine.sessionNote` on the presence beat, and the unlink
422
+ // above just cleared it: without this the org would see the alert blink out
423
+ // each time the seat tried again, which reads as a seat that healed.
424
+ if (priorFailures) {
425
+ const carried = escalationDecision({ streak: priorFailures.streak, limit: d.launchFailureLimit, escalatedAt: priorFailures.escalatedAt });
426
+ if (carried.escalate) {
427
+ writeLaunchAttention({
428
+ d, paths, mux, muxName, now: d.now(),
429
+ streak: carried.streak, limit: carried.limit, signature: priorFailures.signature,
430
+ detail: `${priorFailures.signature} since ${priorFailures.firstAt || "an earlier launch"}`,
431
+ sessionId: record.sessionId, phase: "retrying",
432
+ });
433
+ d.log(`launch-failure escalation still standing: ${carried.reason}`);
434
+ }
435
+ }
353
436
  // 4c. An upgrade notice asking for a restart onto the version THIS launch
354
437
  // runs is honoured by this launch: retire it, so `maestro session
355
438
  // status` stops flagging a hop the session has completed (and a notice
@@ -380,6 +463,9 @@ export async function runSupervisor(deps = {}) {
380
463
  // almost certainly waiting on a dialog or a prompt nobody can see.
381
464
  const startedAt = Number(d.now());
382
465
  let launchFailed = false;
466
+ /** Whatever the mux client itself printed — `screen -D -m` reports a missing
467
+ * command here, and it is the one piece of evidence no pane capture can race. */
468
+ let startStdout = "";
383
469
  stopCmd = cmds.stop;
384
470
  // ── THE WATCHDOG ────────────────────────────────────────────────────────
385
471
  //
@@ -400,6 +486,10 @@ export async function runSupervisor(deps = {}) {
400
486
  let clears = 0;
401
487
  let restarts = restartsSoFar;
402
488
  let watchdogStopped = false;
489
+ /** The last non-empty pane the watchdog saw — evidence for the post-mortem. */
490
+ let lastPane = "";
491
+ /** Did the WATCHDOG end this session? A death we caused is never rotated on. */
492
+ let watchdogRestarted = false;
403
493
  let cancelPass = () => {};
404
494
  const paneFile = join(agentRoot, "state", "session", "pane-capture.txt");
405
495
 
@@ -412,6 +502,17 @@ export async function runSupervisor(deps = {}) {
412
502
  } catch { return ""; }
413
503
  };
414
504
 
505
+ // The pane is captured ONCE, early, and kept. A resume whose conversation is
506
+ // gone prints "No conversation found with session ID: <id>" and exits inside
507
+ // a couple of seconds, taking the mux session and its pane with it — so the
508
+ // only moment this evidence exists is while the doomed session is still up.
509
+ // Best-effort: an empty capture is not an absence of fault, it is an absence
510
+ // of evidence, and `classifyLaunchFailure` is written to say so.
511
+ let earlyPane = "";
512
+ d.afterPaneProbe(d.earlyPaneMs, () => {
513
+ capturePane().then((text) => { if (!earlyPane) earlyPane = paneTail(text); }).catch(() => {});
514
+ });
515
+
415
516
  const sendKeys = async (keys) => {
416
517
  for (const k of keys) {
417
518
  const cmd = sendKeyCommand(mux, muxName, k);
@@ -433,6 +534,7 @@ export async function runSupervisor(deps = {}) {
433
534
  if (!silence.silent) { schedulePass(); return; }
434
535
 
435
536
  const pane = paneTail(await capturePane());
537
+ if (pane) lastPane = pane;
436
538
  const seen = classifyPane(pane);
437
539
  const act = watchdogAction({ silent: true, kind: seen.kind, keys: seen.keys, clears, restarts });
438
540
 
@@ -459,6 +561,7 @@ export async function runSupervisor(deps = {}) {
459
561
  await sendKeys(seen.keys);
460
562
  } else if (act.act === "restart") {
461
563
  restarts = nextRestarts;
564
+ watchdogRestarted = true;
462
565
  d.log(`session ${muxName}: restarting (${restarts}) — the session is not answering and the screen cannot be cleared safely`);
463
566
  try { await d.execFile(cmds.stop.file, cmds.stop.args, { cwd: agentRoot, env: d.env }); } catch { /* the relaunch loop handles a dead mux */ }
464
567
  watchdogStopped = true; // the supervisor's own loop relaunches; do not fight it
@@ -485,6 +588,7 @@ export async function runSupervisor(deps = {}) {
485
588
  // env file; the mux client itself gets the supervisor's env, as for claude.
486
589
  if (envFile) d.writePrivateFile(envFile, renderEnvFile(spec.env));
487
590
  const r = await d.execFile(cmds.start.file, cmds.start.args, { cwd: agentRoot, env: d.env });
591
+ startStdout = `${String((r && r.stdout) || "")}${String((r && r.stderr) || "")}`.trim();
488
592
  if (r && r.code !== 0) { launchFailed = true; d.log(`${mux.kind} start exited ${r.code}${r.stderr ? `: ${String(r.stderr).trim()}` : ""}`); }
489
593
  else if (!cmds.blocking) await waitForProbe(cmds.waitProbe, d);
490
594
  } catch (err) {
@@ -506,11 +610,69 @@ export async function runSupervisor(deps = {}) {
506
610
  d.log(`session ${muxName} stopped on ${stopping} — id kept, exit ${EXIT_RELAUNCH}`);
507
611
  return EXIT_RELAUNCH;
508
612
  }
509
- const decision = rotationDecision({ record: launched, mode, exitCode, startedAt, endedAt, now: d.now });
613
+
614
+ // Did THIS launch ever report in? The single most useful fact about a run,
615
+ // and the one the ten-minute relaunch loop never asked: James's seat looked
616
+ // up for days and had not beaten once.
617
+ let endHeartbeat = null;
618
+ try { endHeartbeat = parseHeartbeat(d.readFileSync(paths.heartbeatFile, "utf8")); } catch { endHeartbeat = null; }
619
+ const endBeatMs = endHeartbeat && typeof endHeartbeat.ts === "string" ? Date.parse(endHeartbeat.ts) : NaN;
620
+ const beatSeen = Number.isFinite(endBeatMs) && endBeatMs >= startedAt;
621
+
622
+ // What the runtime itself said, if we caught it. `earlyPane` is the capture
623
+ // taken while a fast-failing session was still up; `lastPane` is the
624
+ // watchdog's. `startStdout` is whatever the mux client printed to us.
625
+ const evidence = [startStdout, earlyPane, lastPane].filter(Boolean).join("\n");
626
+ const failure = classifyLaunchFailure({ mode, exitCode, output: evidence, beatSeen });
627
+ if (failure.kind !== "none") d.log(`launch verdict: ${failure.kind}${failure.proven ? " (proven)" : ""} — ${failure.detail}`);
628
+
629
+ // ── (b) N IDENTICAL FAILURES ARE A FAULT, NOT N RETRIES ──────────────────
630
+ //
631
+ // The ten-minute backoff was the right instinct and the wrong outcome: every
632
+ // attempt failed identically and the seat stayed quiet about it for days,
633
+ // because the only record was a log on a machine nobody can reach. So the
634
+ // streak is counted across supervisor lifetimes and, at the limit, written
635
+ // where it LEAVES the machine — `state/session/attention.json` rides the
636
+ // presence beat as `machine.sessionNote`, which is the org's view of this
637
+ // seat. A watchdog restart is not counted here: that path has its own
638
+ // bounded budget (`carriedRestarts`) and counting it twice would escalate a
639
+ // fault the seat is already handling.
640
+ let escalation = { escalate: false, fresh: false, streak: 0, limit: d.launchFailureLimit, reason: "" };
641
+ if (!watchdogRestarted) {
642
+ const signature = launchFailureSignature({ mode, kind: failure.kind, exitCode });
643
+ const next = launchFailureStreak(priorFailures, { signature, at: new Date(Number(d.now())).toISOString() });
644
+ if (!next) {
645
+ // The launch worked. Evidence of work clears the record — the same reset
646
+ // rule `carriedRestarts` uses, and for the same reason.
647
+ try { d.unlinkSync(paths.launchFailuresFile); } catch { /* nothing carried */ }
648
+ } else {
649
+ escalation = escalationDecision({ streak: next.streak, limit: d.launchFailureLimit, escalatedAt: next.escalatedAt });
650
+ if (escalation.escalate && !next.escalatedAt) next.escalatedAt = new Date(Number(d.now())).toISOString();
651
+ try { d.writeJsonAtomic(paths.launchFailuresFile, next); } catch { /* the log line still says it */ }
652
+ d.log(`launch failure ${next.streak}× in a row (${signature}) — ${escalation.reason}`);
653
+ if (escalation.escalate) {
654
+ writeLaunchAttention({
655
+ d, paths, mux, muxName, now: d.now(),
656
+ streak: next.streak, limit: escalation.limit, signature,
657
+ detail: failure.detail, sessionId: launched.sessionId, phase: "failed",
658
+ });
659
+ if (escalation.fresh) d.log(`escalating to the org: ${escalationHint({ signature, streak: next.streak, limit: escalation.limit, detail: failure.detail, sessionId: launched.sessionId })}`);
660
+ }
661
+ }
662
+ }
663
+
664
+ const decision = rotationDecision({
665
+ record: launched, mode, exitCode, startedAt, endedAt, now: d.now,
666
+ resumeTargetMissing: failure.kind === "resume-target-missing",
667
+ beatSeen,
668
+ streakAtLimit: escalation.escalate,
669
+ selfStopped: watchdogRestarted,
670
+ });
510
671
  if (decision.action === "rotate") {
511
- const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid });
672
+ const why = decision.why || "the resume failed";
673
+ const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid, reason: why });
512
674
  const res = saveMainSession(agentRoot, rotated, fsDeps);
513
- d.log(`resume of ${launched.sessionId} failed inside the failure window — rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
675
+ d.log(`resume of ${launched.sessionId}: ${why}${decision.proven ? " (proven, so not rationed by the rotation budget)" : ""} — invalidated it and rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
514
676
  } else if (decision.action === "backoff") {
515
677
  d.log(`rotation budget spent for this hour — sleeping ${decision.sleepMs / 60000} min before relaunch`);
516
678
  await d.sleep(decision.sleepMs);
@@ -518,6 +680,37 @@ export async function runSupervisor(deps = {}) {
518
680
  return EXIT_RELAUNCH;
519
681
  }
520
682
 
683
+ /**
684
+ * Write the escalation where it LEAVES THE MACHINE.
685
+ *
686
+ * `state/session/attention.json` is read by `lib/telemetry/collect#sessionNote`
687
+ * and rides the presence beat to hq as `machine.sessionNote` — the only channel
688
+ * out of a seat that accepts no ssh. The record is shaped exactly like the
689
+ * watchdog's (`first-run#attentionRecord` fields) so the existing reader needs
690
+ * no special case; `restarts: 0` is honest and, per `carriedRestarts`, cannot
691
+ * steal budget from the watchdog.
692
+ *
693
+ * @param {object} a
694
+ */
695
+ export function writeLaunchAttention(a) {
696
+ const { d, paths, mux, muxName } = a;
697
+ const attach = mux && mux.kind === "tmux" ? `tmux attach -t =${muxName}` : `screen -r ${muxName}`;
698
+ const record = {
699
+ reason: "launch-failing",
700
+ since: new Date(Number(a.now)).toISOString(),
701
+ runMs: 0,
702
+ restarts: 0,
703
+ action: a.phase === "retrying" ? "wait" : "give-up",
704
+ attach,
705
+ signature: a.signature,
706
+ streak: a.streak,
707
+ limit: a.limit,
708
+ hint: escalationHint({ signature: a.signature, streak: a.streak, limit: a.limit, detail: a.detail, sessionId: a.sessionId }),
709
+ };
710
+ try { d.writeJsonAtomic(paths.attentionFile, record); } catch { /* the log line still says it */ }
711
+ return record;
712
+ }
713
+
521
714
  const isMain = (() => {
522
715
  try { return process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href; } catch { return false; }
523
716
  })();