@cohortapp/agent-sdk 2.18.13 → 2.18.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +58 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/assurance/batch.mjs +353 -0
- package/lib/assurance/first-reply.mjs +423 -0
- package/lib/assurance/notice-voice.mjs +357 -0
- package/lib/assurance/plan-note.mjs +43 -0
- package/lib/assurance/room-budget.mjs +55 -6
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/comms/send-gate.mjs +59 -0
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/identity/persona.mjs +31 -2
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/alerts.mjs +94 -0
- package/lib/telemetry/collect.mjs +155 -2
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/daemon/agent-daemon.mjs +75 -5
- package/scripts/daemon/assurance.mjs +709 -44
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/daemon/deliver.mjs +109 -0
- package/scripts/daemon/dispatcher.mjs +21 -3
- package/scripts/daemon/inbox-deferral.mjs +102 -9
- package/scripts/daemon/session-lock.mjs +41 -1
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +256 -10
- package/scripts/healthcheck.sh +131 -33
- package/scripts/local-triggers/autoupdate.sh +144 -11
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -481,18 +481,133 @@ write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell t
|
|
|
481
481
|
}
|
|
482
482
|
write_upgrade_notice(){ write_notice "$1" "$2"; }
|
|
483
483
|
LAST_JSON="$AGENT_DIR/state/autoupdate/last.json"
|
|
484
|
-
|
|
485
|
-
|
|
484
|
+
# How many CONSECUTIVE failed attempts to reach @latest make a seat STUCK, and
|
|
485
|
+
# at what point it stops being a warning. See the `failStreak` block below.
|
|
486
|
+
STUCK_STREAK="${MAESTRO_AUTOUPDATE_STUCK_STREAK:-3}"
|
|
487
|
+
# ── failStreak — "this seat cannot get to @latest" is a FACT, not a discovery ─
|
|
488
|
+
# The fleet learned on 2026-09-25 that three seats had sat on 2.17.0 for days.
|
|
489
|
+
# Nothing was broken in the sense anything reports: the launchd job fired every
|
|
490
|
+
# hour, npm answered, the install ran, the health gate failed, the rollback
|
|
491
|
+
# worked, last.json recorded it honestly, and the beat carried it. Each hour was
|
|
492
|
+
# a correct, self-contained failure — and a correct failure repeated forty times
|
|
493
|
+
# is a different fact from a correct failure once. Only the REPETITION says the
|
|
494
|
+
# seat will never arrive on its own, and no single record can hold it, so the
|
|
495
|
+
# count lives in last.json and rides the beat the rest of that record already
|
|
496
|
+
# rides. There is no second channel here and there must not be one: the seats
|
|
497
|
+
# this is about are, by construction, the ones running the oldest code.
|
|
498
|
+
#
|
|
499
|
+
# WHAT COUNTS AS AN ATTEMPT — the whole honesty of the number:
|
|
500
|
+
# · An ATTEMPT is a record whose `to` differs from its `from`: the upgrade
|
|
501
|
+
# path actually installed (or tried to install) a different version. Those
|
|
502
|
+
# are the only records that move the streak.
|
|
503
|
+
# · A FAILED attempt (`ok:false`) increments it and stamps `stuckSince` with
|
|
504
|
+
# the first failure of the run of failures, so a reader gets a DURATION and
|
|
505
|
+
# not merely a tally — "4 attempts since Monday" and "4 attempts in the last
|
|
506
|
+
# hour" are different seats.
|
|
507
|
+
# · A SUCCEEDED attempt (`ok:true`) clears all three fields. Arrival is the
|
|
508
|
+
# only thing that resets it; a fleet-wide publish does not, because a seat
|
|
509
|
+
# that fails 2.18.12, 2.18.13 and 2.18.14 in turn has failed three times to
|
|
510
|
+
# do the one thing asked of it, and restarting the count on every release
|
|
511
|
+
# would guarantee the number never reaches any threshold.
|
|
512
|
+
# · Everything else PRESERVES the fields untouched. A `to == from` record is
|
|
513
|
+
# a report on the installed version, not an attempt to leave it, so it must
|
|
514
|
+
# not move a count of ATTEMPTS in either direction. (Every such call site
|
|
515
|
+
# today sits on the up-to-date branch and therefore takes the arrival path
|
|
516
|
+
# below instead — this rule governs the writer, for the next caller that
|
|
517
|
+
# does not.) And a run that SKIPS writes no record at all, which is the
|
|
518
|
+
# same preservation by a shorter route. THIS IS THE POINT OF (d): a seat that has not been ASKED —
|
|
519
|
+
# nothing newer published, or held for the day after its last failure — is
|
|
520
|
+
# not a stuck seat, and must never accumulate a streak for sitting still.
|
|
521
|
+
# Only a real, completed, failed attempt does.
|
|
522
|
+
# · WHICH RUNS ACTUALLY SKIP — the exact list, because a wrong one was
|
|
523
|
+
# written here first. The paths that reach `exit 0` with NO write_last are:
|
|
524
|
+
# the failed-target hold, the overlap lock, the kill-switch, and a failed
|
|
525
|
+
# `npm view`. "SIMPLY BEING UP TO DATE" IS NOT AMONG THEM and never was —
|
|
526
|
+
# the up-to-date branch is the daemon health gate, and it calls write_last
|
|
527
|
+
# on every one of its outcomes (healthy, unhealthy-current,
|
|
528
|
+
# ahead-of-registry, recovered, and the stale-daemon kickstart). The
|
|
529
|
+
# original text claimed otherwise; nothing checked it, and the streak
|
|
530
|
+
# arithmetic on that branch is not what the sentence implied.
|
|
531
|
+
# · ARRIVAL IS BEING AT @latest, NOT MERELY INSTALLING IT. Every write on the
|
|
532
|
+
# up-to-date branch passes `at_latest=true`, which clears all three fields
|
|
533
|
+
# — because that branch is only entered when `CUR >= LATEST`, i.e. there is
|
|
534
|
+
# nothing left for this seat to reach. That covers the case the increment
|
|
535
|
+
# rule alone gets wrong: if a bad release is YANKED and @latest rolls back
|
|
536
|
+
# to the version a seat already runs, the seat stops attempting anything,
|
|
537
|
+
# so nothing would ever reset it and `upgrade_stuck` would fire forever
|
|
538
|
+
# (critical past 6) on a seat that is in fact perfectly current. It also
|
|
539
|
+
# makes the stale-daemon kickstart record honest rather than accidental:
|
|
540
|
+
# that hop really is an arrival.
|
|
541
|
+
# The write is done in node rather than printf because the record now depends on
|
|
542
|
+
# the record before it; a shell that reads a file it is about to overwrite gets
|
|
543
|
+
# that wrong at exactly the moments it matters.
|
|
544
|
+
write_last(){ # $1 from, $2 to, $3 ok, $4 healthy, $5 reason, [$6 at_latest] — state/autoupdate/last.json (→ beat machine.upgrade)
|
|
545
|
+
local dir="$AGENT_DIR/state/autoupdate" tmp streak at_latest="${6:-false}"
|
|
486
546
|
mkdir -p "$dir" 2>/dev/null || return 0
|
|
487
547
|
tmp="$(mktemp "$dir/.last.XXXXXX" 2>/dev/null)" || return 0
|
|
488
|
-
|
|
489
|
-
|
|
548
|
+
streak="$(node -e '
|
|
549
|
+
const fs = require("fs");
|
|
550
|
+
const [file, tmp, from, to, okS, healthyS, reason, stuckStreakS, atLatestS] = process.argv.slice(1);
|
|
551
|
+
const ok = okS === "true";
|
|
552
|
+
const at = new Date().toISOString().replace(/\.\d{3}Z$/, "Z");
|
|
553
|
+
let prev = null;
|
|
554
|
+
try { prev = JSON.parse(fs.readFileSync(file, "utf8")); } catch {}
|
|
555
|
+
if (!prev || typeof prev !== "object") prev = {};
|
|
556
|
+
const prevStreak = Number.isInteger(prev.failStreak) && prev.failStreak > 0 ? prev.failStreak : 0;
|
|
557
|
+
const attempt = from !== to; // an upgrade was actually tried
|
|
558
|
+
const atLatest = atLatestS === "true"; // this seat is AT or past @latest — nothing left to reach
|
|
559
|
+
let failStreak, stuckSince, streakTarget;
|
|
560
|
+
if (atLatest || (attempt && ok)) { // arrival, by install OR because @latest came back to us (a yank)
|
|
561
|
+
failStreak = 0; stuckSince = ""; streakTarget = "";
|
|
562
|
+
} else if (!attempt) { // a report on the installed version — carry the count, do not touch it
|
|
563
|
+
failStreak = prevStreak;
|
|
564
|
+
stuckSince = typeof prev.stuckSince === "string" ? prev.stuckSince : "";
|
|
565
|
+
streakTarget = typeof prev.streakTarget === "string" ? prev.streakTarget : "";
|
|
566
|
+
} else {
|
|
567
|
+
failStreak = prevStreak + 1;
|
|
568
|
+
stuckSince = typeof prev.stuckSince === "string" && prev.stuckSince ? prev.stuckSince : at;
|
|
569
|
+
streakTarget = to;
|
|
570
|
+
}
|
|
571
|
+
const rec = { from, to, at, ok, healthy: healthyS === "true", reason };
|
|
572
|
+
if (failStreak > 0) {
|
|
573
|
+
rec.failStreak = failStreak;
|
|
574
|
+
if (stuckSince) rec.stuckSince = stuckSince;
|
|
575
|
+
if (streakTarget) rec.streakTarget = streakTarget;
|
|
576
|
+
}
|
|
577
|
+
fs.writeFileSync(tmp, JSON.stringify(rec, null, 2) + "\n");
|
|
578
|
+
// stdout is the shell'"'"'s only view of what was decided: "<streak> <target> <since>"
|
|
579
|
+
process.stdout.write(failStreak >= Number(stuckStreakS) && attempt && !ok ? `STUCK ${failStreak} ${streakTarget} ${stuckSince}` : "");
|
|
580
|
+
' "$LAST_JSON" "$tmp" "$1" "$2" "$3" "$4" "$5" "$STUCK_STREAK" "$at_latest" 2>/dev/null)" || streak=""
|
|
581
|
+
if [ -s "$tmp" ]; then
|
|
582
|
+
mv -f "$tmp" "$LAST_JSON"
|
|
583
|
+
else
|
|
584
|
+
# The node writer produced nothing. Before this branch existed the tmp was
|
|
585
|
+
# simply removed and last.json was LEFT AT ITS PREVIOUS VALUE — so a write
|
|
586
|
+
# failure did not lose a record, it published a STALE one, and the 24 h
|
|
587
|
+
# failed-target hold (which reads `to` and `reason` out of this file) then
|
|
588
|
+
# made its decision from a previous run's outcome while believing it was
|
|
589
|
+
# this run's. A missing record is fail-open by design here; a wrong one is
|
|
590
|
+
# not. So write the event itself with printf, which needs no interpreter,
|
|
591
|
+
# and say loudly that the count did not survive: the streak is the one
|
|
592
|
+
# field that cannot be reconstructed without reading the old file, and a
|
|
593
|
+
# silently reset counter is the failure this whole block exists to prevent.
|
|
594
|
+
log "WARN: write_last could not run node — recording the bare outcome without the failStreak. The count restarts from this run; see state/autoupdate/last.json."
|
|
595
|
+
printf '{\n "from": "%s",\n "to": "%s",\n "at": "%s",\n "ok": %s,\n "healthy": %s,\n "reason": "%s"\n}\n' \
|
|
596
|
+
"$1" "$2" "$(date -u +%Y-%m-%dT%H:%M:%SZ)" "$3" "$4" "$(printf '%s' "$5" | sed 's/[\\"]/\\&/g')" > "$LAST_JSON" 2>/dev/null || true
|
|
597
|
+
fi
|
|
490
598
|
rm -f "$tmp" 2>/dev/null
|
|
599
|
+
if [ -n "$streak" ]; then
|
|
600
|
+
set -- $streak
|
|
601
|
+
log "STUCK: $2 consecutive failed attempts to reach $3 (since $4). This seat will not arrive on its own — the org now carries machine.upgrade.failStreak=$2 and an \`upgrade_stuck\` alert on every beat. maestro doctor; then rm state/autoupdate/last.json to retry immediately."
|
|
602
|
+
fi
|
|
491
603
|
return 0
|
|
492
604
|
}
|
|
493
605
|
last_reason(){ # the reason field of last.json, or nothing
|
|
494
606
|
node -e 'try { process.stdout.write(String(JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).reason || "")); } catch {}' "$LAST_JSON" 2>/dev/null
|
|
495
607
|
}
|
|
608
|
+
last_streak(){ # the failStreak of last.json as a positive integer, or nothing
|
|
609
|
+
node -e 'try { const n = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).failStreak; if (Number.isInteger(n) && n > 0) process.stdout.write(String(n)); } catch {}' "$LAST_JSON" 2>/dev/null
|
|
610
|
+
}
|
|
496
611
|
# The version the RUNNING daemon started on. scripts/daemon/health.mjs resolves
|
|
497
612
|
# sdk_version once at load and writes it to state/dashboards/daemon-health.yaml
|
|
498
613
|
# with its pid, precisely so a reader can tell the installed package from the
|
|
@@ -595,13 +710,13 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
595
710
|
# the notice tells the session to restart itself onto the new code.
|
|
596
711
|
log "OK: daemon now on $CUR (was $STALE_FROM) — the manual upgrade is complete"
|
|
597
712
|
write_upgrade_notice "$STALE_FROM" "$CUR"
|
|
598
|
-
write_last "$STALE_FROM" "$CUR" true true ""
|
|
713
|
+
write_last "$STALE_FROM" "$CUR" true true "" true
|
|
599
714
|
elif [ -n "$STALE_FROM" ]; then
|
|
600
715
|
# Same version, newer files: the daemon is reconciled; the session is
|
|
601
716
|
# not told to restart (a touched or restored file must not cost it its
|
|
602
717
|
# context — the version did not change).
|
|
603
718
|
log "OK: daemon restarted onto the installed $CUR code; no session notice (same version)"
|
|
604
|
-
write_last "$CUR" "$CUR" true true ""
|
|
719
|
+
write_last "$CUR" "$CUR" true true "" true
|
|
605
720
|
elif [ -n "$AHEAD" ]; then
|
|
606
721
|
# The DAEMON is well; the FLEET is not. Recorded as ok/healthy — because
|
|
607
722
|
# it is — with the finding in `reason`, so the org learns that this seat
|
|
@@ -609,14 +724,32 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
609
724
|
# per episode: unlike an unhealthy daemon this does not clear itself, and
|
|
610
725
|
# a fact that ages out of one log file is how six days went by.
|
|
611
726
|
log "OK: healthy on $CUR, but AHEAD of the registry ($LATEST) — recorded for the org as ahead-of-registry"
|
|
612
|
-
write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST"
|
|
727
|
+
write_last "$CUR" "$CUR" true true "ahead-of-registry: npm latest is $LATEST" true
|
|
613
728
|
else
|
|
614
|
-
case "$PREV_REASON" in
|
|
729
|
+
case "$PREV_REASON" in
|
|
730
|
+
unhealthy-current*)
|
|
615
731
|
log "recovered: last.json no longer reports unhealthy-current"
|
|
616
|
-
write_last "$CUR" "$CUR" true true "" ;;
|
|
732
|
+
write_last "$CUR" "$CUR" true true "" true ;;
|
|
617
733
|
ahead-of-registry*)
|
|
618
734
|
log "recovered: $CUR is on the registry now (latest $LATEST) — last.json no longer reports ahead-of-registry"
|
|
619
|
-
write_last "$CUR" "$CUR" true true "" ;;
|
|
735
|
+
write_last "$CUR" "$CUR" true true "" true ;;
|
|
736
|
+
*)
|
|
737
|
+
# A healthy seat that is simply up to date writes NOTHING — there is
|
|
738
|
+
# nothing to report and an empty last.json is the honest state.
|
|
739
|
+
#
|
|
740
|
+
# The one exception is a standing failStreak, and it is the reason this
|
|
741
|
+
# branch exists at all. A streak only ever reset on ARRIVAL, i.e. on a
|
|
742
|
+
# successful hop. If a bad release is YANKED and @latest rolls back to
|
|
743
|
+
# the version this seat already runs, the seat stops attempting
|
|
744
|
+
# anything: every hour it takes this quiet path, writes nothing, and the
|
|
745
|
+
# count stands for ever — `upgrade_stuck` firing critical on a seat that
|
|
746
|
+
# is, in fact, exactly where the registry wants it. Being AT @latest is
|
|
747
|
+
# an arrival however you got there, so record one and say so.
|
|
748
|
+
STREAK="$(last_streak)"
|
|
749
|
+
if [ -n "$STREAK" ]; then
|
|
750
|
+
log "streak cleared: $CUR is @latest and this seat is healthy on it, but last.json still carried failStreak=$STREAK — @latest has come back to this version (a yank or a withdrawn release). Recording the arrival so the upgrade_stuck alert stops."
|
|
751
|
+
write_last "$CUR" "$CUR" true true "" true
|
|
752
|
+
fi ;;
|
|
620
753
|
esac
|
|
621
754
|
fi
|
|
622
755
|
else
|
|
@@ -625,7 +758,7 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
625
758
|
# and failed_hold_reason both match on it), so the ahead-of-registry fact
|
|
626
759
|
# is appended, never prepended — an unpublished build that is also sick is
|
|
627
760
|
# the worst case and must say both things.
|
|
628
|
-
write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}"
|
|
761
|
+
write_last "$CUR" "$CUR" false false "unhealthy-current: $HEALTH_REASON${AHEAD:+ (and ahead-of-registry: npm latest is $LATEST)}" true
|
|
629
762
|
case "$PREV_REASON" in
|
|
630
763
|
unhealthy-current*) log "session already notified this episode (last.json was unhealthy-current); notice not rewritten" ;;
|
|
631
764
|
*) write_notice "$CUR" "$CUR" "unhealthy-current: $HEALTH_REASON" ;;
|
|
@@ -3,12 +3,34 @@
|
|
|
3
3
|
# Usage: ./scripts/resume-operations.sh
|
|
4
4
|
#
|
|
5
5
|
# Reverses emergency-stop.sh:
|
|
6
|
-
# 1. Verifies health check passes.
|
|
6
|
+
# 1. Verifies health check passes — EXCLUDING the stop flag itself.
|
|
7
7
|
# 2. Removes the .emergency-stop flag.
|
|
8
8
|
# 3. Reloads every installed `ai.maestro.<agent>-*` (and legacy
|
|
9
9
|
# `ai.adaptic.<agent>-*`) launchd job.
|
|
10
10
|
#
|
|
11
11
|
# Agent first-name slug resolved from config/agent.json (SOT).
|
|
12
|
+
#
|
|
13
|
+
# THE DEADLOCK THIS SCRIPT USED TO HAVE. healthcheck.sh counts an active
|
|
14
|
+
# .emergency-stop flag as an ERROR; this script ran it bare and exited on any
|
|
15
|
+
# non-zero. So the one tool whose entire purpose is lifting the stop could
|
|
16
|
+
# never lift it — Eli Rosenberg's seat sat halted from 2026-09-24T08:16Z until
|
|
17
|
+
# the flag was removed by hand. Two independent faults produced that:
|
|
18
|
+
#
|
|
19
|
+
# (a) The stop flag was its own veto. Fixed by passing
|
|
20
|
+
# `--ignore-emergency-stop`, which demotes that ONE check to a note.
|
|
21
|
+
# It is a PUBLIC CLI flag — nothing refuses it to another caller, and
|
|
22
|
+
# nothing here pretends to (healthcheck.sh's header says so plainly).
|
|
23
|
+
# What makes it safe is scope, not exclusivity: it demotes exactly one
|
|
24
|
+
# check, every other check still gates the resume, and step 1b below
|
|
25
|
+
# VERIFIES that the healthcheck on this disk actually honoured it
|
|
26
|
+
# rather than assuming the two files are the same vintage.
|
|
27
|
+
# (b) DEGRADED was treated as failure. healthcheck exits 1 for warnings and
|
|
28
|
+
# 2 for errors; `if ! healthcheck` refused on both, so seven warnings
|
|
29
|
+
# about optional repos and an uninstalled Slack.app were enough to keep
|
|
30
|
+
# a seat halted. Warnings are "operational with limitations" by
|
|
31
|
+
# healthcheck's own contract — they are reported loudly and resumed.
|
|
32
|
+
# Only CRITICAL (exit 2) still refuses, and the refusal names which
|
|
33
|
+
# condition blocked it.
|
|
12
34
|
|
|
13
35
|
set -e
|
|
14
36
|
|
|
@@ -46,19 +68,92 @@ if [ ! -f "$AGENT_DIR/.emergency-stop" ]; then
|
|
|
46
68
|
launchctl load "$plist" 2>/dev/null && loaded=$((loaded + 1))
|
|
47
69
|
fi
|
|
48
70
|
done
|
|
49
|
-
[ "$loaded" -gt 0 ] && echo
|
|
71
|
+
# `if`, not `[ "$loaded" -gt 0 ] && echo ...` — but NOT for the reason
|
|
72
|
+
# first written here, which was wrong and would have seeded a wrong rule.
|
|
73
|
+
# `set -e` does not abort on the false short-circuit: bash exempts a
|
|
74
|
+
# failing command that is not the last in an && list. MEASURED:
|
|
75
|
+
# bash -c 'set -e; n=0; [ "$n" -gt 0 ] && echo hi; exit 0' → exit 0
|
|
76
|
+
# and the pre-fix script (which had `exit 0` on the next line) also
|
|
77
|
+
# exited 0 on a seat with nothing to load. The real hazard is narrower
|
|
78
|
+
# and has nothing to do with `set -e`: a false && list leaves status 1,
|
|
79
|
+
# so a script whose LAST command is one exits 1. `exit 0` follows today;
|
|
80
|
+
# the `if` keeps this line harmless if it ever stops following.
|
|
81
|
+
if [ "$loaded" -gt 0 ]; then echo "Loaded $loaded missing launchd job(s)"; fi
|
|
50
82
|
exit 0
|
|
51
83
|
fi
|
|
52
84
|
|
|
53
85
|
echo "[$TIMESTAMP] RESUMING OPERATIONS (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
|
|
54
86
|
|
|
55
|
-
# 1. Run health check first
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
87
|
+
# 1. Run health check first — with the stop flag exempted, since lifting it is
|
|
88
|
+
# this script's entire job. Every OTHER check still gates the resume.
|
|
89
|
+
echo "Running health check (emergency-stop flag exempted)..."
|
|
90
|
+
set +e
|
|
91
|
+
HEALTH_OUTPUT=$("$SCRIPT_DIR/healthcheck.sh" --ignore-emergency-stop 2>&1)
|
|
92
|
+
HEALTH_STATUS=$?
|
|
93
|
+
set -e
|
|
94
|
+
echo "$HEALTH_OUTPUT"
|
|
95
|
+
|
|
96
|
+
# 1b. Prove the exemption was HONOURED before trusting the verdict.
|
|
97
|
+
#
|
|
98
|
+
# Fault (a)'s fix is version-coupled: it only works if the healthcheck.sh
|
|
99
|
+
# next to this script understands --ignore-emergency-stop. A seat that
|
|
100
|
+
# pins framework files (.maestroignore) can carry the new resume script
|
|
101
|
+
# beside a pre-fix healthcheck, which has no argument parsing at all — it
|
|
102
|
+
# swallows the flag, still counts the stop as an error, and prints no
|
|
103
|
+
# `Blocking:` line. MEASURED on that exact pairing: the deadlock is back,
|
|
104
|
+
# and the refusal reads "(healthcheck named no condition; see output
|
|
105
|
+
# above)" — worse than the original, because it names nothing to fix.
|
|
106
|
+
# The seats most likely to be halted are the ones most likely to be
|
|
107
|
+
# pinned, so this has to be detected and named rather than inferred.
|
|
108
|
+
#
|
|
109
|
+
# We are past the `[ -f .emergency-stop ]` guard, so a healthcheck that
|
|
110
|
+
# honoured the flag MUST have printed the NOTE line. Its absence means
|
|
111
|
+
# the exemption did not take: stale, missing, or not executable.
|
|
112
|
+
if ! printf '%s\n' "$HEALTH_OUTPUT" | grep -q '^\[NOTE\] Emergency stop flag' ||
|
|
113
|
+
printf '%s\n' "$HEALTH_OUTPUT" | grep -q '^Blocking:.*emergency-stop-flag'; then
|
|
114
|
+
echo ""
|
|
115
|
+
echo "REFUSING TO RESUME — scripts/healthcheck.sh is stale."
|
|
116
|
+
echo " --ignore-emergency-stop was ignored (healthcheck exit $HEALTH_STATUS): it did"
|
|
117
|
+
echo " not report '[NOTE] Emergency stop flag ... not counted', so its verdict still"
|
|
118
|
+
echo " counts the very flag this script exists to lift. Resuming on that verdict"
|
|
119
|
+
echo " would be resuming on a health check nobody read."
|
|
120
|
+
echo " Fix: run 'maestro upgrade' on this seat, then confirm .maestroignore is not"
|
|
121
|
+
echo " pinning scripts/healthcheck.sh, and that the file exists and is executable."
|
|
122
|
+
echo " The .emergency-stop flag has been LEFT IN PLACE."
|
|
123
|
+
echo "[$TIMESTAMP] Resume REFUSED — healthcheck.sh stale (exemption ignored, status $HEALTH_STATUS)" >> "$LOG_FILE"
|
|
59
124
|
exit 1
|
|
60
125
|
fi
|
|
61
126
|
|
|
127
|
+
if [ "$HEALTH_STATUS" -ge 2 ]; then
|
|
128
|
+
# Name the blocking condition. The operator should never have to re-run
|
|
129
|
+
# the health check with a filter to learn which of "1 errors" stopped them.
|
|
130
|
+
BLOCKING=$(printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^Blocking://p' | tr -s ' ' | sed 's/^ //')
|
|
131
|
+
[ -n "$BLOCKING" ] || BLOCKING="(healthcheck named no condition; see output above)"
|
|
132
|
+
echo ""
|
|
133
|
+
echo "REFUSING TO RESUME — health check is CRITICAL."
|
|
134
|
+
echo " Blocking condition(s): $BLOCKING"
|
|
135
|
+
# Re-list the failing lines under the refusal, RE-PREFIXED. The whole
|
|
136
|
+
# health output is echoed above, so the previous `grep '^\[FAIL\]'` here
|
|
137
|
+
# emitted bytes indistinguishable from it — deleting the line left the
|
|
138
|
+
# test that claimed to pin it green. The ` FAIL: ` prefix is produced
|
|
139
|
+
# only here, which is what makes that assertion load-bearing.
|
|
140
|
+
printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^\[FAIL\] / FAIL: /p'
|
|
141
|
+
echo " The .emergency-stop flag has been LEFT IN PLACE."
|
|
142
|
+
echo " Fix the condition(s) above, then re-run: $0"
|
|
143
|
+
{
|
|
144
|
+
echo "[$TIMESTAMP] Resume REFUSED — blocking: $BLOCKING"
|
|
145
|
+
} >> "$LOG_FILE"
|
|
146
|
+
exit 1
|
|
147
|
+
fi
|
|
148
|
+
|
|
149
|
+
if [ "$HEALTH_STATUS" -eq 1 ]; then
|
|
150
|
+
DEGRADED=$(printf '%s\n' "$HEALTH_OUTPUT" | sed -n 's/^Warnings://p' | tr -s ' ' | sed 's/^ //')
|
|
151
|
+
echo ""
|
|
152
|
+
echo "NOTE: health check is DEGRADED — resuming anyway (warnings are not faults)."
|
|
153
|
+
echo " Warning condition(s): $DEGRADED"
|
|
154
|
+
echo "[$TIMESTAMP] Resume proceeding DEGRADED — warnings: $DEGRADED" >> "$LOG_FILE"
|
|
155
|
+
fi
|
|
156
|
+
|
|
62
157
|
# 2. Remove stop flag.
|
|
63
158
|
rm "$AGENT_DIR/.emergency-stop"
|
|
64
159
|
echo "[$TIMESTAMP] Stop flag removed" >> "$LOG_FILE"
|
|
@@ -42,7 +42,7 @@
|
|
|
42
42
|
* @module scripts/session/supervisor
|
|
43
43
|
*/
|
|
44
44
|
|
|
45
|
-
import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync } from "node:fs";
|
|
45
|
+
import { existsSync as fsExistsSync, readFileSync as fsReadFileSync, mkdirSync as fsMkdirSync, unlinkSync as fsUnlinkSync, writeFileSync as fsWriteFileSync, chmodSync as fsChmodSync, readdirSync as fsReaddirSync } from "node:fs";
|
|
46
46
|
import { randomUUID } from "node:crypto";
|
|
47
47
|
import { join } from "node:path";
|
|
48
48
|
import { spawn } from "node:child_process";
|
|
@@ -66,6 +66,11 @@ import {
|
|
|
66
66
|
launchMode, recordLaunch, rotationDecision, rotateMainSession, BACKOFF_MS,
|
|
67
67
|
} from "../../lib/session/identity.mjs";
|
|
68
68
|
import { detectMux, chooseMux, buildMuxCommands, parseScreenList, parseExitFile, renderEnvFile } from "../../lib/session/launch-args.mjs";
|
|
69
|
+
import { resumeTargetState } from "../../lib/session/resume-target.mjs";
|
|
70
|
+
import {
|
|
71
|
+
classifyLaunchFailure, launchFailureSignature, parseLaunchFailures,
|
|
72
|
+
launchFailureStreak, escalationDecision, escalationHint, IDENTICAL_FAILURE_LIMIT,
|
|
73
|
+
} from "../../lib/session/launch-failure.mjs";
|
|
69
74
|
|
|
70
75
|
/** Write a private (0600) file that must not already exist. The engine env file carries the seat token. */
|
|
71
76
|
export function writePrivateFile(path, text) {
|
|
@@ -83,6 +88,17 @@ export const LOCK_NAME = "session";
|
|
|
83
88
|
export const DEFAULT_POLL_MS = 2_000;
|
|
84
89
|
/** The signals the supervisor owns (launchd sends SIGTERM to stop a job). */
|
|
85
90
|
export const OWNED_SIGNALS = Object.freeze(["SIGTERM", "SIGINT"]);
|
|
91
|
+
/**
|
|
92
|
+
* How long after start the pane is captured ONCE, to keep whatever the runtime
|
|
93
|
+
* printed before it died. A resume against a missing conversation is over in
|
|
94
|
+
* about two seconds, taking its mux session (and its pane) with it — so the
|
|
95
|
+
* capture has to happen while it is still there. Best-effort by design: when
|
|
96
|
+
* it comes back empty, `resumeTargetState`'s preflight is what carries the
|
|
97
|
+
* verdict, and the post-mortem heuristics below carry the rest.
|
|
98
|
+
*/
|
|
99
|
+
export const EARLY_PANE_MS = 1_500;
|
|
100
|
+
/** Consecutive identical launch failures before the seat escalates to the org. */
|
|
101
|
+
export const LAUNCH_FAILURE_LIMIT = IDENTICAL_FAILURE_LIMIT;
|
|
86
102
|
|
|
87
103
|
/** Real signal wiring: `fn(signal)` on each owned signal. */
|
|
88
104
|
function installSignalHandlers(fn) {
|
|
@@ -198,6 +214,15 @@ export async function runSupervisor(deps = {}) {
|
|
|
198
214
|
seatSpawn: deps.seatSpawn || ((root) => resolveSeatSpawn(root)),
|
|
199
215
|
writePrivateFile: deps.writePrivateFile || writePrivateFile,
|
|
200
216
|
envFileId: deps.envFileId || randomUUID,
|
|
217
|
+
readdirSync: deps.readdirSync || fsReaddirSync,
|
|
218
|
+
launchFailureLimit: deps.launchFailureLimit ?? LAUNCH_FAILURE_LIMIT,
|
|
219
|
+
earlyPaneMs: deps.earlyPaneMs ?? EARLY_PANE_MS,
|
|
220
|
+
// The one-shot pane probe has its OWN timer seam, separate from the
|
|
221
|
+
// watchdog's `after`. They are different clocks doing different jobs — one
|
|
222
|
+
// fires once at a second and a half, the other recurs at the grace period —
|
|
223
|
+
// and collapsing them would make every watchdog test reason about a timer
|
|
224
|
+
// that has nothing to do with the watchdog.
|
|
225
|
+
afterPaneProbe: deps.afterPaneProbe || deps.after || defaultAfter,
|
|
201
226
|
};
|
|
202
227
|
const fsDeps = { readFileSync: d.readFileSync, writeJsonAtomic: d.writeJsonAtomic };
|
|
203
228
|
const agentRoot = d.agentRoot;
|
|
@@ -271,7 +296,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
271
296
|
// 4. Identity + argv.
|
|
272
297
|
let record = loadMainSession(agentRoot, fsDeps);
|
|
273
298
|
if (!record || cfg.resume === false) record = newMainSession({ now: d.now, uuid: d.uuid });
|
|
274
|
-
|
|
299
|
+
let mode = launchMode(record, cfg);
|
|
275
300
|
let permissionArgs = d.permissionArgs;
|
|
276
301
|
if (!permissionArgs) {
|
|
277
302
|
permissionArgs = sessionPermissionArgs({ source: MAIN_SESSION_SOURCE, allowedTools: cfg.allowedTools });
|
|
@@ -288,6 +313,38 @@ export async function runSupervisor(deps = {}) {
|
|
|
288
313
|
let seat = { engine: "claude", fields: {} };
|
|
289
314
|
try { seat = (await d.seatSpawn(agentRoot)) || seat; } catch { seat = { engine: "claude", fields: {} }; }
|
|
290
315
|
const engineCohort = seat.engine === "cohort";
|
|
316
|
+
|
|
317
|
+
// ── (a) A DEAD RESUME TARGET IS CAUGHT BEFORE THE LAUNCH, NOT AFTER IT ────
|
|
318
|
+
//
|
|
319
|
+
// James Kirkland's seat, measured 2026-09-25: `claude --resume <id>` for a
|
|
320
|
+
// conversation that no longer existed on that machine, exit 1 after 2 s,
|
|
321
|
+
// relaunched every ten minutes for DAYS. state/session/main-session.json held
|
|
322
|
+
// the dead id and nothing ever invalidated it.
|
|
323
|
+
//
|
|
324
|
+
// THE CHOICE, MADE EXPLICIT: invalidate the record AND fall back to a fresh
|
|
325
|
+
// id — both, in one write, here, before anything is spawned. Invalidating
|
|
326
|
+
// alone would leave the supervisor with no id (and `newMainSession` would
|
|
327
|
+
// mint one anyway, unrecorded); falling back alone would leave the dead id on
|
|
328
|
+
// disk for the next reader to resume again. `rotateMainSession` does both:
|
|
329
|
+
// the dead id survives as `rotatedFrom` — provenance, never a target.
|
|
330
|
+
//
|
|
331
|
+
// It runs on POSITIVE EVIDENCE ONLY (`resume-target.mjs` fails open): a
|
|
332
|
+
// wrong "missing" throws away a live transcript, a wrong "unknown" costs one
|
|
333
|
+
// launch and lands in the post-mortem that was already there.
|
|
334
|
+
if (mode === "resume") {
|
|
335
|
+
const target = resumeTargetState(
|
|
336
|
+
{ homeDir: d.homeDir, projectDir: agentRoot, sessionId: record.sessionId, engine: seat.engine },
|
|
337
|
+
{ readdirSync: d.readdirSync },
|
|
338
|
+
);
|
|
339
|
+
if (target.state === "missing") {
|
|
340
|
+
const dead = record.sessionId;
|
|
341
|
+
record = rotateMainSession(record, { now: d.now, uuid: d.uuid, reason: `resume target ${dead} is not on this machine` });
|
|
342
|
+
mode = launchMode(record, cfg);
|
|
343
|
+
const res = saveMainSession(agentRoot, record, fsDeps);
|
|
344
|
+
d.log(`resume target ${dead} is gone (${target.reason}) — invalidated it and starting a fresh session ${record.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
291
348
|
const spec = buildSpawn({ lane: "main-session", bin: d.claudeBin || undefined, first, sessionId: record.sessionId, resumeMode: mode, permissions: permissionArgs, env: d.env, ...seat.fields });
|
|
292
349
|
if (!spec.ok && engineCohort) {
|
|
293
350
|
// An engine-cohort seat never falls back to claude (it would spend the seat's
|
|
@@ -348,8 +405,34 @@ export async function runSupervisor(deps = {}) {
|
|
|
348
405
|
const restartsSoFar = carriedRestarts({ attention: priorAttention, heartbeat: priorHeartbeat });
|
|
349
406
|
if (restartsSoFar > 0) d.log(`this launch carries ${restartsSoFar} silent restart(s) from the previous run`);
|
|
350
407
|
|
|
408
|
+
// ── (b) THE CARRIED LAUNCH-FAILURE STREAK ────────────────────────────────
|
|
409
|
+
//
|
|
410
|
+
// Read BEFORE the unlink below, for the same reason `carriedRestarts` is: a
|
|
411
|
+
// count held in memory is zero on every launch, because every launch is a
|
|
412
|
+
// fresh process. This one lives in its own file rather than on the attention
|
|
413
|
+
// record, because it has to survive the unlink that clears that record.
|
|
414
|
+
let priorFailures = null;
|
|
415
|
+
try { priorFailures = parseLaunchFailures(d.readFileSync(paths.launchFailuresFile, "utf8")); } catch { priorFailures = null; }
|
|
416
|
+
|
|
351
417
|
try { d.unlinkSync(paths.lastExitFile); } catch { /* none from a previous run */ }
|
|
352
418
|
try { d.unlinkSync(paths.attentionFile); } catch { /* none outstanding */ }
|
|
419
|
+
|
|
420
|
+
// A seat already past the limit re-states the escalation on EVERY launch.
|
|
421
|
+
// The note rides `machine.sessionNote` on the presence beat, and the unlink
|
|
422
|
+
// above just cleared it: without this the org would see the alert blink out
|
|
423
|
+
// each time the seat tried again, which reads as a seat that healed.
|
|
424
|
+
if (priorFailures) {
|
|
425
|
+
const carried = escalationDecision({ streak: priorFailures.streak, limit: d.launchFailureLimit, escalatedAt: priorFailures.escalatedAt });
|
|
426
|
+
if (carried.escalate) {
|
|
427
|
+
writeLaunchAttention({
|
|
428
|
+
d, paths, mux, muxName, now: d.now(),
|
|
429
|
+
streak: carried.streak, limit: carried.limit, signature: priorFailures.signature,
|
|
430
|
+
detail: `${priorFailures.signature} since ${priorFailures.firstAt || "an earlier launch"}`,
|
|
431
|
+
sessionId: record.sessionId, phase: "retrying",
|
|
432
|
+
});
|
|
433
|
+
d.log(`launch-failure escalation still standing: ${carried.reason}`);
|
|
434
|
+
}
|
|
435
|
+
}
|
|
353
436
|
// 4c. An upgrade notice asking for a restart onto the version THIS launch
|
|
354
437
|
// runs is honoured by this launch: retire it, so `maestro session
|
|
355
438
|
// status` stops flagging a hop the session has completed (and a notice
|
|
@@ -380,6 +463,9 @@ export async function runSupervisor(deps = {}) {
|
|
|
380
463
|
// almost certainly waiting on a dialog or a prompt nobody can see.
|
|
381
464
|
const startedAt = Number(d.now());
|
|
382
465
|
let launchFailed = false;
|
|
466
|
+
/** Whatever the mux client itself printed — `screen -D -m` reports a missing
|
|
467
|
+
* command here, and it is the one piece of evidence no pane capture can race. */
|
|
468
|
+
let startStdout = "";
|
|
383
469
|
stopCmd = cmds.stop;
|
|
384
470
|
// ── THE WATCHDOG ────────────────────────────────────────────────────────
|
|
385
471
|
//
|
|
@@ -400,6 +486,10 @@ export async function runSupervisor(deps = {}) {
|
|
|
400
486
|
let clears = 0;
|
|
401
487
|
let restarts = restartsSoFar;
|
|
402
488
|
let watchdogStopped = false;
|
|
489
|
+
/** The last non-empty pane the watchdog saw — evidence for the post-mortem. */
|
|
490
|
+
let lastPane = "";
|
|
491
|
+
/** Did the WATCHDOG end this session? A death we caused is never rotated on. */
|
|
492
|
+
let watchdogRestarted = false;
|
|
403
493
|
let cancelPass = () => {};
|
|
404
494
|
const paneFile = join(agentRoot, "state", "session", "pane-capture.txt");
|
|
405
495
|
|
|
@@ -412,6 +502,17 @@ export async function runSupervisor(deps = {}) {
|
|
|
412
502
|
} catch { return ""; }
|
|
413
503
|
};
|
|
414
504
|
|
|
505
|
+
// The pane is captured ONCE, early, and kept. A resume whose conversation is
|
|
506
|
+
// gone prints "No conversation found with session ID: <id>" and exits inside
|
|
507
|
+
// a couple of seconds, taking the mux session and its pane with it — so the
|
|
508
|
+
// only moment this evidence exists is while the doomed session is still up.
|
|
509
|
+
// Best-effort: an empty capture is not an absence of fault, it is an absence
|
|
510
|
+
// of evidence, and `classifyLaunchFailure` is written to say so.
|
|
511
|
+
let earlyPane = "";
|
|
512
|
+
d.afterPaneProbe(d.earlyPaneMs, () => {
|
|
513
|
+
capturePane().then((text) => { if (!earlyPane) earlyPane = paneTail(text); }).catch(() => {});
|
|
514
|
+
});
|
|
515
|
+
|
|
415
516
|
const sendKeys = async (keys) => {
|
|
416
517
|
for (const k of keys) {
|
|
417
518
|
const cmd = sendKeyCommand(mux, muxName, k);
|
|
@@ -433,6 +534,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
433
534
|
if (!silence.silent) { schedulePass(); return; }
|
|
434
535
|
|
|
435
536
|
const pane = paneTail(await capturePane());
|
|
537
|
+
if (pane) lastPane = pane;
|
|
436
538
|
const seen = classifyPane(pane);
|
|
437
539
|
const act = watchdogAction({ silent: true, kind: seen.kind, keys: seen.keys, clears, restarts });
|
|
438
540
|
|
|
@@ -459,6 +561,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
459
561
|
await sendKeys(seen.keys);
|
|
460
562
|
} else if (act.act === "restart") {
|
|
461
563
|
restarts = nextRestarts;
|
|
564
|
+
watchdogRestarted = true;
|
|
462
565
|
d.log(`session ${muxName}: restarting (${restarts}) — the session is not answering and the screen cannot be cleared safely`);
|
|
463
566
|
try { await d.execFile(cmds.stop.file, cmds.stop.args, { cwd: agentRoot, env: d.env }); } catch { /* the relaunch loop handles a dead mux */ }
|
|
464
567
|
watchdogStopped = true; // the supervisor's own loop relaunches; do not fight it
|
|
@@ -485,6 +588,7 @@ export async function runSupervisor(deps = {}) {
|
|
|
485
588
|
// env file; the mux client itself gets the supervisor's env, as for claude.
|
|
486
589
|
if (envFile) d.writePrivateFile(envFile, renderEnvFile(spec.env));
|
|
487
590
|
const r = await d.execFile(cmds.start.file, cmds.start.args, { cwd: agentRoot, env: d.env });
|
|
591
|
+
startStdout = `${String((r && r.stdout) || "")}${String((r && r.stderr) || "")}`.trim();
|
|
488
592
|
if (r && r.code !== 0) { launchFailed = true; d.log(`${mux.kind} start exited ${r.code}${r.stderr ? `: ${String(r.stderr).trim()}` : ""}`); }
|
|
489
593
|
else if (!cmds.blocking) await waitForProbe(cmds.waitProbe, d);
|
|
490
594
|
} catch (err) {
|
|
@@ -506,11 +610,69 @@ export async function runSupervisor(deps = {}) {
|
|
|
506
610
|
d.log(`session ${muxName} stopped on ${stopping} — id kept, exit ${EXIT_RELAUNCH}`);
|
|
507
611
|
return EXIT_RELAUNCH;
|
|
508
612
|
}
|
|
509
|
-
|
|
613
|
+
|
|
614
|
+
// Did THIS launch ever report in? The single most useful fact about a run,
|
|
615
|
+
// and the one the ten-minute relaunch loop never asked: James's seat looked
|
|
616
|
+
// up for days and had not beaten once.
|
|
617
|
+
let endHeartbeat = null;
|
|
618
|
+
try { endHeartbeat = parseHeartbeat(d.readFileSync(paths.heartbeatFile, "utf8")); } catch { endHeartbeat = null; }
|
|
619
|
+
const endBeatMs = endHeartbeat && typeof endHeartbeat.ts === "string" ? Date.parse(endHeartbeat.ts) : NaN;
|
|
620
|
+
const beatSeen = Number.isFinite(endBeatMs) && endBeatMs >= startedAt;
|
|
621
|
+
|
|
622
|
+
// What the runtime itself said, if we caught it. `earlyPane` is the capture
|
|
623
|
+
// taken while a fast-failing session was still up; `lastPane` is the
|
|
624
|
+
// watchdog's. `startStdout` is whatever the mux client printed to us.
|
|
625
|
+
const evidence = [startStdout, earlyPane, lastPane].filter(Boolean).join("\n");
|
|
626
|
+
const failure = classifyLaunchFailure({ mode, exitCode, output: evidence, beatSeen });
|
|
627
|
+
if (failure.kind !== "none") d.log(`launch verdict: ${failure.kind}${failure.proven ? " (proven)" : ""} — ${failure.detail}`);
|
|
628
|
+
|
|
629
|
+
// ── (b) N IDENTICAL FAILURES ARE A FAULT, NOT N RETRIES ──────────────────
|
|
630
|
+
//
|
|
631
|
+
// The ten-minute backoff was the right instinct and the wrong outcome: every
|
|
632
|
+
// attempt failed identically and the seat stayed quiet about it for days,
|
|
633
|
+
// because the only record was a log on a machine nobody can reach. So the
|
|
634
|
+
// streak is counted across supervisor lifetimes and, at the limit, written
|
|
635
|
+
// where it LEAVES the machine — `state/session/attention.json` rides the
|
|
636
|
+
// presence beat as `machine.sessionNote`, which is the org's view of this
|
|
637
|
+
// seat. A watchdog restart is not counted here: that path has its own
|
|
638
|
+
// bounded budget (`carriedRestarts`) and counting it twice would escalate a
|
|
639
|
+
// fault the seat is already handling.
|
|
640
|
+
let escalation = { escalate: false, fresh: false, streak: 0, limit: d.launchFailureLimit, reason: "" };
|
|
641
|
+
if (!watchdogRestarted) {
|
|
642
|
+
const signature = launchFailureSignature({ mode, kind: failure.kind, exitCode });
|
|
643
|
+
const next = launchFailureStreak(priorFailures, { signature, at: new Date(Number(d.now())).toISOString() });
|
|
644
|
+
if (!next) {
|
|
645
|
+
// The launch worked. Evidence of work clears the record — the same reset
|
|
646
|
+
// rule `carriedRestarts` uses, and for the same reason.
|
|
647
|
+
try { d.unlinkSync(paths.launchFailuresFile); } catch { /* nothing carried */ }
|
|
648
|
+
} else {
|
|
649
|
+
escalation = escalationDecision({ streak: next.streak, limit: d.launchFailureLimit, escalatedAt: next.escalatedAt });
|
|
650
|
+
if (escalation.escalate && !next.escalatedAt) next.escalatedAt = new Date(Number(d.now())).toISOString();
|
|
651
|
+
try { d.writeJsonAtomic(paths.launchFailuresFile, next); } catch { /* the log line still says it */ }
|
|
652
|
+
d.log(`launch failure ${next.streak}× in a row (${signature}) — ${escalation.reason}`);
|
|
653
|
+
if (escalation.escalate) {
|
|
654
|
+
writeLaunchAttention({
|
|
655
|
+
d, paths, mux, muxName, now: d.now(),
|
|
656
|
+
streak: next.streak, limit: escalation.limit, signature,
|
|
657
|
+
detail: failure.detail, sessionId: launched.sessionId, phase: "failed",
|
|
658
|
+
});
|
|
659
|
+
if (escalation.fresh) d.log(`escalating to the org: ${escalationHint({ signature, streak: next.streak, limit: escalation.limit, detail: failure.detail, sessionId: launched.sessionId })}`);
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
const decision = rotationDecision({
|
|
665
|
+
record: launched, mode, exitCode, startedAt, endedAt, now: d.now,
|
|
666
|
+
resumeTargetMissing: failure.kind === "resume-target-missing",
|
|
667
|
+
beatSeen,
|
|
668
|
+
streakAtLimit: escalation.escalate,
|
|
669
|
+
selfStopped: watchdogRestarted,
|
|
670
|
+
});
|
|
510
671
|
if (decision.action === "rotate") {
|
|
511
|
-
const
|
|
672
|
+
const why = decision.why || "the resume failed";
|
|
673
|
+
const rotated = rotateMainSession(launched, { now: d.now, uuid: d.uuid, reason: why });
|
|
512
674
|
const res = saveMainSession(agentRoot, rotated, fsDeps);
|
|
513
|
-
d.log(`resume of ${launched.sessionId}
|
|
675
|
+
d.log(`resume of ${launched.sessionId}: ${why}${decision.proven ? " (proven, so not rationed by the rotation budget)" : ""} — invalidated it and rotated to ${rotated.sessionId}${res.ok ? "" : ` (not saved: ${res.error})`}`);
|
|
514
676
|
} else if (decision.action === "backoff") {
|
|
515
677
|
d.log(`rotation budget spent for this hour — sleeping ${decision.sleepMs / 60000} min before relaunch`);
|
|
516
678
|
await d.sleep(decision.sleepMs);
|
|
@@ -518,6 +680,37 @@ export async function runSupervisor(deps = {}) {
|
|
|
518
680
|
return EXIT_RELAUNCH;
|
|
519
681
|
}
|
|
520
682
|
|
|
683
|
+
/**
|
|
684
|
+
* Write the escalation where it LEAVES THE MACHINE.
|
|
685
|
+
*
|
|
686
|
+
* `state/session/attention.json` is read by `lib/telemetry/collect#sessionNote`
|
|
687
|
+
* and rides the presence beat to hq as `machine.sessionNote` — the only channel
|
|
688
|
+
* out of a seat that accepts no ssh. The record is shaped exactly like the
|
|
689
|
+
* watchdog's (`first-run#attentionRecord` fields) so the existing reader needs
|
|
690
|
+
* no special case; `restarts: 0` is honest and, per `carriedRestarts`, cannot
|
|
691
|
+
* steal budget from the watchdog.
|
|
692
|
+
*
|
|
693
|
+
* @param {object} a
|
|
694
|
+
*/
|
|
695
|
+
export function writeLaunchAttention(a) {
|
|
696
|
+
const { d, paths, mux, muxName } = a;
|
|
697
|
+
const attach = mux && mux.kind === "tmux" ? `tmux attach -t =${muxName}` : `screen -r ${muxName}`;
|
|
698
|
+
const record = {
|
|
699
|
+
reason: "launch-failing",
|
|
700
|
+
since: new Date(Number(a.now)).toISOString(),
|
|
701
|
+
runMs: 0,
|
|
702
|
+
restarts: 0,
|
|
703
|
+
action: a.phase === "retrying" ? "wait" : "give-up",
|
|
704
|
+
attach,
|
|
705
|
+
signature: a.signature,
|
|
706
|
+
streak: a.streak,
|
|
707
|
+
limit: a.limit,
|
|
708
|
+
hint: escalationHint({ signature: a.signature, streak: a.streak, limit: a.limit, detail: a.detail, sessionId: a.sessionId }),
|
|
709
|
+
};
|
|
710
|
+
try { d.writeJsonAtomic(paths.attentionFile, record); } catch { /* the log line still says it */ }
|
|
711
|
+
return record;
|
|
712
|
+
}
|
|
713
|
+
|
|
521
714
|
const isMain = (() => {
|
|
522
715
|
try { return process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href; } catch { return false; }
|
|
523
716
|
})();
|