@cohortapp/agent-sdk 2.18.15 → 2.18.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/org/inbound/hydrate.mjs +35 -1
- package/lib/org/inbound/project.mjs +3 -0
- package/lib/session/frontdoor.mjs +105 -8
- package/lib/session/handoffs.mjs +57 -0
- package/lib/session/inbox-claims.mjs +106 -2
- package/lib/session/revive.mjs +302 -5
- package/lib/telemetry/collect.mjs +224 -0
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +262 -10
- package/scripts/daemon/assurance.mjs +56 -1
- package/scripts/daemon/deliver.mjs +80 -0
- package/scripts/fleet/rollout.mjs +48 -4
- package/scripts/hooks/pre-write-yaml-validate.mjs +63 -2
- package/scripts/local-triggers/autoupdate.sh +287 -22
|
@@ -17,7 +17,14 @@
|
|
|
17
17
|
* old_string replaced by new_string) and REFUSES — exit 2, a permissionDecision:"deny" JSON on
|
|
18
18
|
* stdout and the reason on stderr — when:
|
|
19
19
|
* - an Edit's old_string is absent from the file, or matches more than once without
|
|
20
|
-
* replace_all (the edit would land somewhere the writer did not look at);
|
|
20
|
+
* replace_all (the edit would land somewhere the writer did not look at); when that repeated
|
|
21
|
+
* match is an `id:` line, the refusal names the real cause — the file already carries a
|
|
22
|
+
* duplicate id and is unwritable via id-anchored edits until the collision is resolved;
|
|
23
|
+
* - the would-be top-level `items:` list (or a top-level sequence) would carry two rows sharing
|
|
24
|
+
* the same `id` value that was not already colliding on disk. A duplicate id slips past js-yaml
|
|
25
|
+
* (the rows are distinct list entries, not duplicate map keys) but makes every id-anchored edit
|
|
26
|
+
* ambiguous — the "collision then silently unwritable all day" failure — so it is refused loudly
|
|
27
|
+
* as a WHOLE FILE BLOCK at the write boundary;
|
|
21
28
|
* - the result does not parse as YAML (js-yaml, which also rejects duplicate keys);
|
|
22
29
|
* - a provenance-keyed value NEW in this write ends on a dangling connector. The delta is
|
|
23
30
|
* judged, not the whole file, so a pre-existing truncated row never blocks an unrelated
|
|
@@ -108,6 +115,39 @@ export function guardedRowCount(doc) {
|
|
|
108
115
|
return null;
|
|
109
116
|
}
|
|
110
117
|
|
|
118
|
+
/** The top-level list the id-uniqueness guard protects: `items:` or a top-level sequence, else null. */
|
|
119
|
+
function guardedList(doc) {
|
|
120
|
+
if (Array.isArray(doc)) return doc;
|
|
121
|
+
if (doc && typeof doc === "object" && Array.isArray(doc.items)) return doc.items;
|
|
122
|
+
return null;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* First `id` value that appears on two or more rows of the guarded top-level list, else null.
|
|
127
|
+
* A duplicate id makes every id-anchored edit ambiguous, so js-yaml lets it through (the rows are
|
|
128
|
+
* distinct list entries, not duplicate map keys) but the file becomes unwritable via id-anchored edits.
|
|
129
|
+
*/
|
|
130
|
+
export function duplicateItemId(doc) {
|
|
131
|
+
const list = guardedList(doc);
|
|
132
|
+
if (!list) return null;
|
|
133
|
+
const seen = new Set();
|
|
134
|
+
for (const row of list) {
|
|
135
|
+
if (!row || typeof row !== "object") continue;
|
|
136
|
+
const id = row.id;
|
|
137
|
+
if (typeof id !== "string" && typeof id !== "number") continue;
|
|
138
|
+
const key = String(id);
|
|
139
|
+
if (seen.has(key)) return key;
|
|
140
|
+
seen.add(key);
|
|
141
|
+
}
|
|
142
|
+
return null;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** If `oldStr` is anchored on an `id:` line, return the id value; else null. */
|
|
146
|
+
function repeatedIdAnchor(oldStr) {
|
|
147
|
+
const m = /(?:^|\n)\s*(?:-\s+)?id:\s*(.+?)\s*(?:\n|$)/.exec(oldStr);
|
|
148
|
+
return m ? m[1].replace(/^["']|["']$/g, "") : null;
|
|
149
|
+
}
|
|
150
|
+
|
|
111
151
|
/** Count non-overlapping occurrences of `needle` in `hay`. */
|
|
112
152
|
function countOccurrences(hay, needle) {
|
|
113
153
|
if (!needle) return 0;
|
|
@@ -138,7 +178,15 @@ export function wouldBeContent(toolName, ti, current) {
|
|
|
138
178
|
if (typeof oldStr !== "string" || oldStr === "") return { refuse: "Edit with an empty old_string" };
|
|
139
179
|
const n = countOccurrences(current, oldStr);
|
|
140
180
|
if (n === 0) return { refuse: "old_string is not present in the file on disk — the edit would not land where you looked" };
|
|
141
|
-
if (n > 1 && !ti.replace_all)
|
|
181
|
+
if (n > 1 && !ti.replace_all) {
|
|
182
|
+
// A repeated `id:` line means the file already carries a duplicate id — the collision, not the
|
|
183
|
+
// anchor, is why every id-anchored edit matches twice. Name that instead of the generic advice.
|
|
184
|
+
const dupId = repeatedIdAnchor(oldStr);
|
|
185
|
+
if (dupId !== null) {
|
|
186
|
+
return { refuse: `WHOLE FILE BLOCKED — the file already contains a duplicate id '${dupId}', so this id-anchored edit matches ${n} places and cannot land. The file is unwritable via id-anchored edits until the duplicate id is resolved (give one of the colliding rows a unique id).` };
|
|
187
|
+
}
|
|
188
|
+
return { refuse: `old_string matches ${n} places — anchor it to one row or pass replace_all` };
|
|
189
|
+
}
|
|
142
190
|
return { content: ti.replace_all ? current.split(oldStr).join(newStr) : current.replace(oldStr, () => newStr) };
|
|
143
191
|
}
|
|
144
192
|
if (toolName === "MultiEdit" && Array.isArray(ti.edits)) {
|
|
@@ -200,6 +248,19 @@ export function evaluate(payload, o = {}) {
|
|
|
200
248
|
try { prev = yaml.load(current); prevParsed = true; } catch { prevParsed = false; }
|
|
201
249
|
}
|
|
202
250
|
|
|
251
|
+
// Duplicate-id guard (id-uniqueness reserved at the write boundary). A second row sharing an id
|
|
252
|
+
// slips past js-yaml — the rows are distinct list entries, not duplicate map keys — but it makes
|
|
253
|
+
// every id-anchored edit ambiguous and the file effectively unwritable. Judged on the DELTA: a
|
|
254
|
+
// brand-new collision this write introduces always blocks; a pre-existing one the write leaves in
|
|
255
|
+
// place is surfaced by the id-anchored-Edit match-count path, not penalised twice here.
|
|
256
|
+
const nextDup = duplicateItemId(next);
|
|
257
|
+
if (nextDup !== null && (!prevParsed || duplicateItemId(prev) !== nextDup)) {
|
|
258
|
+
return {
|
|
259
|
+
decision: "deny",
|
|
260
|
+
reason: `${toolName} ${fp}: WHOLE FILE BLOCKED — two rows share id '${nextDup}'. A duplicate id makes every id-anchored edit ambiguous and the file unwritable. Give the new row a unique id before writing.`,
|
|
261
|
+
};
|
|
262
|
+
}
|
|
263
|
+
|
|
203
264
|
// Provenance truncation — judged on the DELTA so pre-existing rows never block a new write.
|
|
204
265
|
const before = new Set(prevParsed ? provenanceHits(prev).map((h) => `${h.id}\u0000${h.key}\u0000${h.value}`) : []);
|
|
205
266
|
const fresh = provenanceHits(next).filter((h) => !before.has(`${h.id}\u0000${h.key}\u0000${h.value}`));
|
|
@@ -43,7 +43,17 @@
|
|
|
43
43
|
# STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
|
|
44
44
|
# (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
|
|
45
45
|
# loop cannot straddle both samples by luck, and well inside the hourly
|
|
46
|
-
# cadence. A 25 s loop fails on the second sample and on uptime.
|
|
46
|
+
# cadence. A 25 s loop fails on the second sample and on uptime. A pid
|
|
47
|
+
# CHANGE within the window is not, on its own, proof of that loop: this
|
|
48
|
+
# script's own restart_daemon (or an operator's `launchctl kickstart -k`)
|
|
49
|
+
# changes the pid deliberately and can land inside the same window —
|
|
50
|
+
# 2.18.14 read that as a crash loop and rolled back a release that had
|
|
51
|
+
# done nothing wrong. daemon_last_exit() tells the two apart: a crash
|
|
52
|
+
# loop's reaped exit is a real non-zero every time; a deliberate
|
|
53
|
+
# kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
|
|
54
|
+
# real non-zero exit fails here on the pid change alone — a clean or
|
|
55
|
+
# unknown exit falls through to the uptime check, which still catches a
|
|
56
|
+
# genuinely unstable respawn (its pid is always younger than STABLE_S).
|
|
47
57
|
# (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
|
|
48
58
|
# lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
|
|
49
59
|
# ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
|
|
@@ -55,9 +65,20 @@
|
|
|
55
65
|
# and the running daemon's own boot marker — "[DAEMON] boot pid=<pid>"
|
|
56
66
|
# (maestro-daemon.mjs, from the release that added this gate), else the last "[daemon] Running" /
|
|
57
67
|
# "org-mesh connected" line. A crash earlier the same day that launchd
|
|
58
|
-
# (or an operator) already recovered from is history, not health.
|
|
59
|
-
#
|
|
60
|
-
#
|
|
68
|
+
# (or an operator) already recovered from is history, not health.
|
|
69
|
+
# The log is NOT always under $AGENT_DIR: launchd-wrapper.sh redirects it
|
|
70
|
+
# to an external SSD ($SSD_VOLUME/maestro/<agent>/logs/daemon) when one is
|
|
71
|
+
# mounted+writable. This scan resolves that same path (daemon_log_base,
|
|
72
|
+
# mirroring the wrapper's own SSD derivation) so a redirected seat is gated
|
|
73
|
+
# on its REAL log — not a fixed $AGENT_DIR path that never exists there,
|
|
74
|
+
# which silently disabled this whole check (and, on gates before the
|
|
75
|
+
# missing-log fail-open, rolled EVERY upgrade back on SSD seats). When even
|
|
76
|
+
# the resolved log cannot be read, the gate does NOT blindly fail-open:
|
|
77
|
+
# it falls back to the path-independent cadence heartbeat
|
|
78
|
+
# state/cadence-bus/health.json (lib/cadence-bus.mjs writeHealth, every
|
|
79
|
+
# ~15s, under the symlinked state/ tree) — fresh (< HEALTH_FRESH_S, 120s)
|
|
80
|
+
# proves the daemon loop is cycling → pass; stale/absent → no log AND no
|
|
81
|
+
# heartbeat is genuine unhealth → fail (checks (a)+(c) still corroborate).
|
|
61
82
|
# (c) A SERVER-ACKNOWLEDGED BEAT — state/org/last-beat.json {at, ok, code?,
|
|
62
83
|
# sdkVersion?}, which lib/org/mesh.mjs overwrites on EVERY presence beat
|
|
63
84
|
# (ok:true only when the org answered 2xx; a rejected or unreachable beat
|
|
@@ -137,7 +158,10 @@
|
|
|
137
158
|
# (default 600) and MAESTRO_AUTOUPDATE_HEALTH_WAIT (default 30), both seconds;
|
|
138
159
|
# MAESTRO_AUTOUPDATE_STABLE_S (default 90; MAESTRO_AUTOUPDATE_STABLE_GAP_S is the
|
|
139
160
|
# sleep between the two pid samples, default = STABLE_S, zeroed by the test) and
|
|
140
|
-
# MAESTRO_AUTOUPDATE_BEAT_FRESH_S (default 300);
|
|
161
|
+
# MAESTRO_AUTOUPDATE_BEAT_FRESH_S (default 300); MAESTRO_AUTOUPDATE_HEALTH_FRESH_S
|
|
162
|
+
# (default 120 — the cadence-heartbeat staleness bar for the log-absent fallback);
|
|
163
|
+
# MAESTRO_SSD_VOLUME (the daemon-log SSD volume, mirroring launchd-wrapper.sh; the
|
|
164
|
+
# test points it at a fixture dir); MAESTRO_AUTOUPDATE_RETRY_SLEEP
|
|
141
165
|
# (default 20 s, ×attempt);
|
|
142
166
|
# MAESTRO_AUTOUPDATE_LOCK_STALE_S (default 7200); MAESTRO_AUTOUPDATE_FAILED_HOLD_S
|
|
143
167
|
# (default 86400); MAESTRO_AUTOUPDATE_PATH_PREFIX (a stub-binary dir that wins
|
|
@@ -254,6 +278,42 @@ DAEMON_PATTERN="$AGENT_DIR/scripts/daemon/maestro-daemon.mjs"
|
|
|
254
278
|
# a pid-set change, i.e. "crash loop".
|
|
255
279
|
DAEMON_ARGV_RE="^[^ ]*/node $(printf '%s' "$DAEMON_PATTERN" | sed 's/[][\.*^$]/\\&/g')\$"
|
|
256
280
|
DLOG_DIR="$AGENT_DIR/logs/daemon"
|
|
281
|
+
# The daemon log does NOT always live under $AGENT_DIR. launchd-wrapper.sh
|
|
282
|
+
# redirects the daemon's stdout/stderr to an external SSD when one is mounted
|
|
283
|
+
# and writable — $SSD_VOLUME/maestro/<agent>/logs/daemon/ — falling back to
|
|
284
|
+
# $AGENT_DIR/logs/daemon only when that write is denied. is_healthy reads the
|
|
285
|
+
# daemon log for its fatal scan (b); keyed to a fixed $AGENT_DIR path it reads a
|
|
286
|
+
# file that never exists on a redirected seat, which SILENTLY DISABLES the fatal
|
|
287
|
+
# scan there (and, on gates older than the missing-log fail-open, rolled EVERY
|
|
288
|
+
# upgrade back — 2026-09 SSD seats). Resolve the SSD dir the SAME way the wrapper
|
|
289
|
+
# does (its own MAESTRO_SSD_VOLUME / /Volumes/*-SSD derivation) — this is the
|
|
290
|
+
# real path, not a widened guess — and, as a path-independent backstop robust to
|
|
291
|
+
# ANY redirect, cross-check the cadence heartbeat state/cadence-bus/health.json
|
|
292
|
+
# (written under the symlinked state/ tree every ~15s by lib/cadence-bus.mjs
|
|
293
|
+
# writeHealth) whenever the resolved log still cannot be read.
|
|
294
|
+
ssd_daemon_log_dir(){ # echo the SSD daemon-log dir the wrapper would have opened, or nothing when no SSD is mounted (mirrors launchd-wrapper.sh)
|
|
295
|
+
local vol name v
|
|
296
|
+
vol="${MAESTRO_SSD_VOLUME:-}"
|
|
297
|
+
if [ -z "$vol" ]; then
|
|
298
|
+
for v in /Volumes/*-SSD /Volumes/*SSD* /Volumes/maestro-data; do
|
|
299
|
+
if [ -d "$v" ] && [ "$v" != "/Volumes/Macintosh HD" ]; then vol="$v"; break; fi
|
|
300
|
+
done
|
|
301
|
+
fi
|
|
302
|
+
[ -n "$vol" ] && [ -d "$vol" ] || return 0
|
|
303
|
+
name="$(basename "$AGENT_DIR" | sed 's/-ai$//')"
|
|
304
|
+
echo "$vol/maestro/$name/logs/daemon"
|
|
305
|
+
}
|
|
306
|
+
DLOG_DIR_SSD="$(ssd_daemon_log_dir)"
|
|
307
|
+
daemon_log_base(){ # $1 = YYYY-MM-DD → the dir holding that day's daemon log: the SSD copy when the wrapper wrote it there, else $AGENT_DIR. Resolves by where the file IS (the wrapper picks per-boot and can fall back), so a redirected AND a fallback seat both read the right file.
|
|
308
|
+
if [ -n "$DLOG_DIR_SSD" ] && [ -f "$DLOG_DIR_SSD/daemon-$1.log" ]; then echo "$DLOG_DIR_SSD"; else echo "$DLOG_DIR"; fi
|
|
309
|
+
}
|
|
310
|
+
# The cadence heartbeat: state/cadence-bus/health.json {version,ts,pid,...},
|
|
311
|
+
# overwritten every ~15s by the consumer (lib/cadence-bus.mjs writeHealth). It
|
|
312
|
+
# lives under the symlinked state/ tree, so it is readable at a fixed path no
|
|
313
|
+
# matter where the daemon LOG went — the path-independent liveness signal the
|
|
314
|
+
# gate falls back to when the resolved daemon log cannot be read.
|
|
315
|
+
HEALTH_JSON="$AGENT_DIR/state/cadence-bus/health.json"
|
|
316
|
+
HEALTH_FRESH_S="${MAESTRO_AUTOUPDATE_HEALTH_FRESH_S:-120}" # 8× the 15s heartbeat cadence, well under the 300s org-offline bar
|
|
257
317
|
HEALTH_WAIT="${MAESTRO_AUTOUPDATE_HEALTH_WAIT:-30}"
|
|
258
318
|
STABLE_S="${MAESTRO_AUTOUPDATE_STABLE_S:-90}"
|
|
259
319
|
STABLE_GAP_S="${MAESTRO_AUTOUPDATE_STABLE_GAP_S:-$STABLE_S}" # the gap between the two pid samples; the test zeroes it (its stub changes per CALL)
|
|
@@ -270,6 +330,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
|
|
|
270
330
|
HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
|
|
271
331
|
DAEMON_LABEL=""
|
|
272
332
|
SESSION_LABEL=""
|
|
333
|
+
# The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
|
|
334
|
+
# below — the on-host backstop for a daemon that is DOWN, which a live daemon
|
|
335
|
+
# can never observe about itself. Defined up here (not with SESSION_STALE_S,
|
|
336
|
+
# which lives in the not-up-to-date path) because revive_daemon fires on the
|
|
337
|
+
# UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
|
|
338
|
+
# 90s front-door observers.
|
|
339
|
+
LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
|
|
340
|
+
DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
|
|
341
|
+
# Gap between the two settle-window reads that prove the kickstarted daemon is
|
|
342
|
+
# one continuous process, not a come-up-then-crash loop. A one-shot read would
|
|
343
|
+
# pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
|
|
344
|
+
DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
|
|
273
345
|
|
|
274
346
|
daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
|
|
275
347
|
pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
|
|
@@ -289,8 +361,22 @@ pid_uptime_s(){ # $1 = pid → seconds since it started, "" when ps cannot say
|
|
|
289
361
|
day_of_epoch(){ # $1 = epoch seconds → YYYY-MM-DD in local time (the wrapper's `date +%Y-%m-%d` basis)
|
|
290
362
|
date -r "$1" +%Y-%m-%d 2>/dev/null || date -d "@$1" +%Y-%m-%d 2>/dev/null
|
|
291
363
|
}
|
|
292
|
-
daemon_log_for(){ # $1 = uptime seconds → the log file the wrapper opened for THIS incarnation (named for its start day)
|
|
293
|
-
|
|
364
|
+
daemon_log_for(){ # $1 = uptime seconds → the log file the wrapper opened for THIS incarnation (named for its start day, in the dir it actually writes — SSD or $AGENT_DIR)
|
|
365
|
+
local day; day="$(day_of_epoch $(( $(date +%s) - $1 )))"
|
|
366
|
+
echo "$(daemon_log_base "$day")/daemon-$day.log"
|
|
367
|
+
}
|
|
368
|
+
cadence_heartbeat_stale(){ # prints WHY the cadence heartbeat is not a fresh proof of a live daemon loop, or nothing when it is (state/cadence-bus/health.json, path-independent)
|
|
369
|
+
node -e '
|
|
370
|
+
const [file, freshS] = process.argv.slice(1);
|
|
371
|
+
const out = (s) => process.stdout.write(s);
|
|
372
|
+
let j;
|
|
373
|
+
try { j = JSON.parse(require("fs").readFileSync(file, "utf8")); }
|
|
374
|
+
catch (e) { out(`daemon log unreadable at the resolved path and no cadence heartbeat (${e && e.code === "ENOENT" ? "state/cadence-bus/health.json absent" : "state/cadence-bus/health.json unreadable"})`); process.exit(0); }
|
|
375
|
+
const at = Date.parse(j && j.ts);
|
|
376
|
+
if (!Number.isFinite(at)) { out("daemon log unreadable at the resolved path and cadence health.json carries no parseable ts"); process.exit(0); }
|
|
377
|
+
const ageS = Math.round((Date.now() - at) / 1000);
|
|
378
|
+
if (ageS > Number(freshS)) out(`daemon log unreadable at the resolved path and the cadence heartbeat is ${ageS}s old (> ${freshS}s) — the daemon loop is not cycling`);
|
|
379
|
+
' "$HEALTH_JSON" "$HEALTH_FRESH_S" 2>/dev/null
|
|
294
380
|
}
|
|
295
381
|
first_fatal_after(){ # $1 = log file, $2 = line offset (lines up to it predate this daemon) → the first fatal line, or nothing
|
|
296
382
|
[ -f "$1" ] || return 0
|
|
@@ -355,14 +441,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
|
|
|
355
441
|
[ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
|
|
356
442
|
return 0
|
|
357
443
|
}
|
|
444
|
+
daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
|
|
445
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
|
|
446
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
447
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
|
|
448
|
+
}
|
|
358
449
|
is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
|
|
359
|
-
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
|
|
450
|
+
local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
|
|
360
451
|
HEALTH_REASON=""
|
|
361
452
|
p1="$(daemon_pids)"
|
|
362
453
|
[ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
|
|
363
454
|
[ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
|
|
364
455
|
p2="$(daemon_pids)"
|
|
365
|
-
[ "$p1"
|
|
456
|
+
if [ "$p1" != "$p2" ]; then
|
|
457
|
+
# A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
|
|
458
|
+
# (this same script's own restart_daemon, or an operator's) changes the pid
|
|
459
|
+
# deliberately and can land inside this very window. 2.18.14 read that as a
|
|
460
|
+
# crash loop and rolled back a release that had done nothing wrong. The
|
|
461
|
+
# daemon's OWN last exit says which one happened: a crash-looping process
|
|
462
|
+
# exits non-zero every time launchd reaps it; a clean kickstart's reaped
|
|
463
|
+
# exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
|
|
464
|
+
# non-zero exit is a crash loop here; a clean or unknown exit is treated as
|
|
465
|
+
# an intended restart and falls through to the uptime check below, which
|
|
466
|
+
# still catches a genuinely unstable respawn (a crash-looping daemon's new
|
|
467
|
+
# pid is always younger than STABLE_S).
|
|
468
|
+
exit_code="$(daemon_last_exit)"
|
|
469
|
+
case "$exit_code" in
|
|
470
|
+
""|"-"|"0")
|
|
471
|
+
# A clean/unknown exit with NO new pid at all is not a restart in
|
|
472
|
+
# progress, it is the daemon gone — the "no daemon process" case, not a
|
|
473
|
+
# crash loop and not healthy.
|
|
474
|
+
[ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
|
|
475
|
+
log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
|
|
476
|
+
p1="$p2"
|
|
477
|
+
;;
|
|
478
|
+
*)
|
|
479
|
+
HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
|
|
480
|
+
return 1
|
|
481
|
+
;;
|
|
482
|
+
esac
|
|
483
|
+
fi
|
|
366
484
|
oldest=""
|
|
367
485
|
for pid in $p1; do
|
|
368
486
|
up="$(pid_uptime_s "$pid")"
|
|
@@ -370,11 +488,26 @@ is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
|
|
|
370
488
|
[ -n "$oldest" ] && [ "$oldest" -ge "$up" ] || { oldest="$up"; oldest_pid="$pid"; }
|
|
371
489
|
done
|
|
372
490
|
dlog="$(daemon_log_for "$oldest")"
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
491
|
+
if [ -f "$dlog" ]; then
|
|
492
|
+
# The daemon log is readable (at $AGENT_DIR or the SSD path the wrapper
|
|
493
|
+
# redirected it to): scan it for a fatal line since this daemon booted.
|
|
494
|
+
offset=0; [ "$dlog" = "$PRE_LOG" ] && offset="$PRE_LOG_LINES"
|
|
495
|
+
boot="$(boot_offset "$dlog" "$oldest_pid")"
|
|
496
|
+
[ "$boot" -gt "$offset" ] && offset="$boot"
|
|
497
|
+
fatal="$(first_fatal_after "$dlog" "$offset")"
|
|
498
|
+
[ -z "$fatal" ] || { HEALTH_REASON="fatal in $(basename "$dlog") since this daemon booted (scanned from line $(( offset + 1 ))): $(printf '%s' "$fatal" | tr -d '"\\' | cut -c1-160)"; return 1; }
|
|
499
|
+
else
|
|
500
|
+
# The daemon log is not readable even at the SSD-resolved path (a redirect
|
|
501
|
+
# this gate could not resolve, or a brand-new start-day file the daemon has
|
|
502
|
+
# not written yet). Do NOT fail-open blindly (that silently drops check (b)
|
|
503
|
+
# on every such seat) and do NOT roll a healthy release back on a path
|
|
504
|
+
# artefact. Fall back to the path-independent cadence heartbeat: fresh →
|
|
505
|
+
# the daemon's main loop is demonstrably cycling, so the missing log is a
|
|
506
|
+
# path artefact, not death → pass; stale/absent → no log AND no heartbeat is
|
|
507
|
+
# genuine unhealth → fail.
|
|
508
|
+
why="$(cadence_heartbeat_stale)"
|
|
509
|
+
[ -z "$why" ] || { HEALTH_REASON="$why"; return 1; }
|
|
510
|
+
fi
|
|
378
511
|
if org_enrolled; then
|
|
379
512
|
why="$(beat_not_acknowledged $(( $(date +%s) - oldest - 2 )) "$(installed_version)")"
|
|
380
513
|
[ -z "$why" ] || { HEALTH_REASON="$why"; return 1; }
|
|
@@ -409,8 +542,9 @@ label_loaded(){ # $1 = label — measured in the gui domain (right from ssh too)
|
|
|
409
542
|
restart_daemon(){
|
|
410
543
|
resolve_labels
|
|
411
544
|
# Mark "since the restart" for the fatal scan: the new process appends to
|
|
412
|
-
# today's file (the wrapper names it for its start day), after these lines
|
|
413
|
-
|
|
545
|
+
# today's file (the wrapper names it for its start day), after these lines —
|
|
546
|
+
# in the dir the wrapper actually writes (SSD when redirected, else $AGENT_DIR).
|
|
547
|
+
PRE_LOG="$(daemon_log_base "$(date +%Y-%m-%d)")/daemon-$(date +%Y-%m-%d).log"
|
|
414
548
|
PRE_LOG_LINES=0
|
|
415
549
|
[ -f "$PRE_LOG" ] && PRE_LOG_LINES="$(wc -l < "$PRE_LOG" 2>/dev/null | tr -d ' ')"
|
|
416
550
|
PRE_LOG_LINES="${PRE_LOG_LINES:-0}"
|
|
@@ -450,19 +584,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
|
|
|
450
584
|
# Leave that one to session_reconciled, which already names it.
|
|
451
585
|
label_loaded "$SESSION_LABEL" || return 0
|
|
452
586
|
session_beating && return 0
|
|
453
|
-
|
|
587
|
+
# Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
|
|
588
|
+
# not enough to drop a session, because a supervisor resume gaps the beat for
|
|
589
|
+
# a moment. Confirm the door is STILL silent on a second read 60s later
|
|
590
|
+
# before kickstarting — if it beat again in between, that was a resume gap,
|
|
591
|
+
# not a wedge, and the session keeps its context.
|
|
592
|
+
log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
|
|
593
|
+
sleep "$SESSION_REVIVE_RECHECK_S"
|
|
594
|
+
session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
|
|
595
|
+
# Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
|
|
596
|
+
# note. The supervisor holds the `session` singleton (state/locks/process/
|
|
597
|
+
# session.pid) for its whole life and beats as a healthy resume progresses.
|
|
598
|
+
# We are past the FULL grace and two reads apart, so a resume gap is already
|
|
599
|
+
# excluded by time — a live holder here is a supervisor PARKED in its launch
|
|
600
|
+
# probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
|
|
601
|
+
# Consulting the lock tells a parked supervisor (override, logged) from a dead
|
|
602
|
+
# one (clean restart); it never blocks the parked case a hard skip would.
|
|
603
|
+
local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
|
|
604
|
+
[ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
|
|
605
|
+
if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
|
|
606
|
+
log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
|
|
607
|
+
fi
|
|
608
|
+
log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
|
|
454
609
|
launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
|
|
455
610
|
&& { log "front door $SESSION_LABEL restarted"; return 0; }
|
|
456
611
|
log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
|
|
457
612
|
return 1
|
|
458
613
|
}
|
|
459
614
|
|
|
615
|
+
daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
|
|
616
|
+
[ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
|
|
617
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
|
|
618
|
+
NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
|
|
619
|
+
END { if (!f) print "absent" }'
|
|
620
|
+
}
|
|
621
|
+
daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
|
|
622
|
+
node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
|
|
623
|
+
"$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
|
|
624
|
+
}
|
|
625
|
+
REVIVE_NOTE_REL="state/telemetry/revive-note.json"
|
|
626
|
+
write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
|
|
627
|
+
local dir="$AGENT_DIR/state/telemetry" tmp
|
|
628
|
+
mkdir -p "$dir" 2>/dev/null || return 0
|
|
629
|
+
tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
|
|
630
|
+
printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
|
|
631
|
+
"$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
|
|
632
|
+
&& mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
|
|
633
|
+
return 0
|
|
634
|
+
}
|
|
635
|
+
daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
|
|
636
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
637
|
+
launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
|
|
638
|
+
}
|
|
639
|
+
daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
|
|
640
|
+
local pid start
|
|
641
|
+
pid="$(daemon_launchctl_pid)"
|
|
642
|
+
if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
|
|
643
|
+
start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
|
|
644
|
+
printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
|
|
645
|
+
elif [ -n "$pid" ]; then
|
|
646
|
+
printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
|
|
647
|
+
else
|
|
648
|
+
printf '{"pid":null,"alive":false,"startTime":null}'
|
|
649
|
+
fi
|
|
650
|
+
}
|
|
651
|
+
revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
|
|
652
|
+
# is_healthy bails at the empty-pid check before daemon_last_exit (its label
|
|
653
|
+
# resolver) runs, so resolve them here — the down case is exactly the one
|
|
654
|
+
# where nothing upstream did.
|
|
655
|
+
[ -n "$DAEMON_LABEL" ] || resolve_labels
|
|
656
|
+
[ -n "$DAEMON_LABEL" ] || return 0
|
|
657
|
+
[ -f "$LIB_REVIVE" ] || return 0
|
|
658
|
+
# ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
|
|
659
|
+
# still has a pid — that is the health gate's business, and a kickstart would
|
|
660
|
+
# fight the very restart it is doing. pgrep-empty is Jacob's lane.
|
|
661
|
+
[ -z "$(daemon_pids)" ] || return 0
|
|
662
|
+
local st beat lpid palive verdict
|
|
663
|
+
st="$(daemon_launchctl_state)"
|
|
664
|
+
beat="$(daemon_beat_age_ms)"
|
|
665
|
+
# PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
|
|
666
|
+
# A listed pid is not a running daemon — verify it, and pass pidAlive so
|
|
667
|
+
# shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
|
|
668
|
+
# stamping it alive (which would deadlock: pgrep-empty says down, launchctl
|
|
669
|
+
# says pid, and nothing revives).
|
|
670
|
+
lpid="$(daemon_launchctl_pid)"
|
|
671
|
+
if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
|
|
672
|
+
verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
|
|
673
|
+
case "$verdict" in
|
|
674
|
+
revive) : ;;
|
|
675
|
+
*) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
|
|
676
|
+
esac
|
|
677
|
+
log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
|
|
678
|
+
launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
|
|
679
|
+
# ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
|
|
680
|
+
# passes a phantom pid AND a come-up-then-crash whose local beat it keeps
|
|
681
|
+
# stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
|
|
682
|
+
# live process (same pid, same start-time, alive on both) — the uptime
|
|
683
|
+
# continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
|
|
684
|
+
# hourly backstop does not round-trip the org just to confirm); continuity
|
|
685
|
+
# carries it, and the daemon's own presence publish is the fleet-visible proof.
|
|
686
|
+
local s1 s2 samples ok
|
|
687
|
+
s1="$(daemon_sample_json)"
|
|
688
|
+
sleep "$DAEMON_CONFIRM_SETTLE_S"
|
|
689
|
+
s2="$(daemon_sample_json)"
|
|
690
|
+
samples="[$s1,$s2]"
|
|
691
|
+
ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
|
|
692
|
+
if [ "$ok" = "ok" ]; then
|
|
693
|
+
log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
|
|
694
|
+
rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
|
|
695
|
+
return 0
|
|
696
|
+
fi
|
|
697
|
+
# ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
|
|
698
|
+
# this failure's log under the loop). One attempt, then hand up with what it
|
|
699
|
+
# saw: the launchctl list line and the tail of the daemon's own log.
|
|
700
|
+
local lc_line dlog_tail evidence
|
|
701
|
+
lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
|
|
702
|
+
dlog_tail="$(tail -n 3 "$(daemon_log_base "$(date +%Y-%m-%d)")/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
|
|
703
|
+
evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
|
|
704
|
+
log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
|
|
705
|
+
write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
|
|
706
|
+
return 1
|
|
707
|
+
}
|
|
460
708
|
session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
|
|
461
709
|
label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
|
|
462
710
|
pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
|
|
463
711
|
return 0
|
|
464
712
|
}
|
|
465
|
-
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
|
|
713
|
+
json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
|
|
466
714
|
write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
|
|
467
715
|
local dir="$AGENT_DIR/state/session" tmp
|
|
468
716
|
mkdir -p "$dir" 2>/dev/null || return 0
|
|
@@ -753,6 +1001,14 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
|
|
|
753
1001
|
esac
|
|
754
1002
|
fi
|
|
755
1003
|
else
|
|
1004
|
+
# Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
|
|
1005
|
+
# running process while the front door kept beating — the seat looked "up
|
|
1006
|
+
# to date" every hour and was dark. The daemon cannot restart itself, so
|
|
1007
|
+
# revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
|
|
1008
|
+
# here and asserts the pid came back. Gated to the no-process case, so a
|
|
1009
|
+
# crash-looping or beat-unhealthy daemon (which still has a pid) is left to
|
|
1010
|
+
# the gate above, not kickstarted.
|
|
1011
|
+
revive_daemon || true
|
|
756
1012
|
log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
|
|
757
1013
|
# The `unhealthy-current:` PREFIX is load-bearing (the episode check below
|
|
758
1014
|
# and failed_hold_reason both match on it), so the ahead-of-registry fact
|
|
@@ -777,10 +1033,19 @@ fi
|
|
|
777
1033
|
# nothing.
|
|
778
1034
|
FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
|
|
779
1035
|
# How long a front door may go without beating before this run restarts it FOR
|
|
780
|
-
# it.
|
|
781
|
-
#
|
|
782
|
-
#
|
|
783
|
-
|
|
1036
|
+
# it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
|
|
1037
|
+
# a wedged front door is measured in the same window as a wedged daemon rather
|
|
1038
|
+
# than the days it took to notice the last three. `revive_session` (this run,
|
|
1039
|
+
# hourly) is the BACKSTOP; it does not need its own, looser number.
|
|
1040
|
+
SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
|
|
1041
|
+
# The daemon confirms a shut front door over TWO reads before it kickstarts
|
|
1042
|
+
# (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
|
|
1043
|
+
# 60s apart). `revive_session`, the hourly backstop, now does the same: a
|
|
1044
|
+
# supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
|
|
1045
|
+
# dying and the new one re-arming, and a single stale read landing inside that
|
|
1046
|
+
# gap must not kickstart a session that is mid-resume. The two reads are 60s
|
|
1047
|
+
# apart; the hourly backstop can afford the one wait.
|
|
1048
|
+
SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
|
|
784
1049
|
failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
|
|
785
1050
|
node -e '
|
|
786
1051
|
const [file, latest, holdS] = process.argv.slice(1);
|