@cohortapp/agent-sdk 2.18.15 → 2.18.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,7 +17,14 @@
17
17
  * old_string replaced by new_string) and REFUSES — exit 2, a permissionDecision:"deny" JSON on
18
18
  * stdout and the reason on stderr — when:
19
19
  * - an Edit's old_string is absent from the file, or matches more than once without
20
- * replace_all (the edit would land somewhere the writer did not look at);
20
+ * replace_all (the edit would land somewhere the writer did not look at); when that repeated
21
+ * match is an `id:` line, the refusal names the real cause — the file already carries a
22
+ * duplicate id and is unwritable via id-anchored edits until the collision is resolved;
23
+ * - the would-be top-level `items:` list (or a top-level sequence) would carry two rows sharing
24
+ * the same `id` value that was not already colliding on disk. A duplicate id slips past js-yaml
25
+ * (the rows are distinct list entries, not duplicate map keys) but makes every id-anchored edit
26
+ * ambiguous — the "collision then silently unwritable all day" failure — so it is refused loudly
27
+ * as a WHOLE FILE BLOCK at the write boundary;
21
28
  * - the result does not parse as YAML (js-yaml, which also rejects duplicate keys);
22
29
  * - a provenance-keyed value NEW in this write ends on a dangling connector. The delta is
23
30
  * judged, not the whole file, so a pre-existing truncated row never blocks an unrelated
@@ -108,6 +115,39 @@ export function guardedRowCount(doc) {
108
115
  return null;
109
116
  }
110
117
 
118
+ /** The top-level list the id-uniqueness guard protects: `items:` or a top-level sequence, else null. */
119
+ function guardedList(doc) {
120
+ if (Array.isArray(doc)) return doc;
121
+ if (doc && typeof doc === "object" && Array.isArray(doc.items)) return doc.items;
122
+ return null;
123
+ }
124
+
125
+ /**
126
+ * First `id` value that appears on two or more rows of the guarded top-level list, else null.
127
+ * A duplicate id makes every id-anchored edit ambiguous, so js-yaml lets it through (the rows are
128
+ * distinct list entries, not duplicate map keys) but the file becomes unwritable via id-anchored edits.
129
+ */
130
+ export function duplicateItemId(doc) {
131
+ const list = guardedList(doc);
132
+ if (!list) return null;
133
+ const seen = new Set();
134
+ for (const row of list) {
135
+ if (!row || typeof row !== "object") continue;
136
+ const id = row.id;
137
+ if (typeof id !== "string" && typeof id !== "number") continue;
138
+ const key = String(id);
139
+ if (seen.has(key)) return key;
140
+ seen.add(key);
141
+ }
142
+ return null;
143
+ }
144
+
145
+ /** If `oldStr` is anchored on an `id:` line, return the id value; else null. */
146
+ function repeatedIdAnchor(oldStr) {
147
+ const m = /(?:^|\n)\s*(?:-\s+)?id:\s*(.+?)\s*(?:\n|$)/.exec(oldStr);
148
+ return m ? m[1].replace(/^["']|["']$/g, "") : null;
149
+ }
150
+
111
151
  /** Count non-overlapping occurrences of `needle` in `hay`. */
112
152
  function countOccurrences(hay, needle) {
113
153
  if (!needle) return 0;
@@ -138,7 +178,15 @@ export function wouldBeContent(toolName, ti, current) {
138
178
  if (typeof oldStr !== "string" || oldStr === "") return { refuse: "Edit with an empty old_string" };
139
179
  const n = countOccurrences(current, oldStr);
140
180
  if (n === 0) return { refuse: "old_string is not present in the file on disk — the edit would not land where you looked" };
141
- if (n > 1 && !ti.replace_all) return { refuse: `old_string matches ${n} places — anchor it to one row or pass replace_all` };
181
+ if (n > 1 && !ti.replace_all) {
182
+ // A repeated `id:` line means the file already carries a duplicate id — the collision, not the
183
+ // anchor, is why every id-anchored edit matches twice. Name that instead of the generic advice.
184
+ const dupId = repeatedIdAnchor(oldStr);
185
+ if (dupId !== null) {
186
+ return { refuse: `WHOLE FILE BLOCKED — the file already contains a duplicate id '${dupId}', so this id-anchored edit matches ${n} places and cannot land. The file is unwritable via id-anchored edits until the duplicate id is resolved (give one of the colliding rows a unique id).` };
187
+ }
188
+ return { refuse: `old_string matches ${n} places — anchor it to one row or pass replace_all` };
189
+ }
142
190
  return { content: ti.replace_all ? current.split(oldStr).join(newStr) : current.replace(oldStr, () => newStr) };
143
191
  }
144
192
  if (toolName === "MultiEdit" && Array.isArray(ti.edits)) {
@@ -200,6 +248,19 @@ export function evaluate(payload, o = {}) {
200
248
  try { prev = yaml.load(current); prevParsed = true; } catch { prevParsed = false; }
201
249
  }
202
250
 
251
+ // Duplicate-id guard (id-uniqueness reserved at the write boundary). A second row sharing an id
252
+ // slips past js-yaml — the rows are distinct list entries, not duplicate map keys — but it makes
253
+ // every id-anchored edit ambiguous and the file effectively unwritable. Judged on the DELTA: a
254
+ // brand-new collision this write introduces always blocks; a pre-existing one the write leaves in
255
+ // place is surfaced by the id-anchored-Edit match-count path, not penalised twice here.
256
+ const nextDup = duplicateItemId(next);
257
+ if (nextDup !== null && (!prevParsed || duplicateItemId(prev) !== nextDup)) {
258
+ return {
259
+ decision: "deny",
260
+ reason: `${toolName} ${fp}: WHOLE FILE BLOCKED — two rows share id '${nextDup}'. A duplicate id makes every id-anchored edit ambiguous and the file unwritable. Give the new row a unique id before writing.`,
261
+ };
262
+ }
263
+
203
264
  // Provenance truncation — judged on the DELTA so pre-existing rows never block a new write.
204
265
  const before = new Set(prevParsed ? provenanceHits(prev).map((h) => `${h.id}\u0000${h.key}\u0000${h.value}`) : []);
205
266
  const fresh = provenanceHits(next).filter((h) => !before.has(`${h.id}\u0000${h.key}\u0000${h.value}`));
@@ -43,7 +43,17 @@
43
43
  # STABLE_S apart AND every pid's `ps etime` is ≥ STABLE_S. STABLE_S is 90 s
44
44
  # (MAESTRO_AUTOUPDATE_STABLE_S): 3× the observed ~25 s crash interval, so a
45
45
  # loop cannot straddle both samples by luck, and well inside the hourly
46
- # cadence. A 25 s loop fails on the second sample and on uptime.
46
+ # cadence. A 25 s loop fails on the second sample and on uptime. A pid
47
+ # CHANGE within the window is not, on its own, proof of that loop: this
48
+ # script's own restart_daemon (or an operator's `launchctl kickstart -k`)
49
+ # changes the pid deliberately and can land inside the same window —
50
+ # 2.18.14 read that as a crash loop and rolled back a release that had
51
+ # done nothing wrong. daemon_last_exit() tells the two apart: a crash
52
+ # loop's reaped exit is a real non-zero every time; a deliberate
53
+ # kickstart's is 0 (or "-"/unknown, for a launchd too old to say). Only a
54
+ # real non-zero exit fails here on the pid change alone — a clean or
55
+ # unknown exit falls through to the uptime check, which still catches a
56
+ # genuinely unstable respawn (its pid is always younger than STABLE_S).
47
57
  # (b) NO FATAL LINE in the daemon's own log since THIS daemon booted: the
48
58
  # lines lib/util/unhandled.mjs and agent-daemon.mjs write on the way down
49
59
  # ("[DAEMON] uncaughtException:", "[daemon] Fatal:"), the wrapper's
@@ -55,9 +65,20 @@
55
65
  # and the running daemon's own boot marker — "[DAEMON] boot pid=<pid>"
56
66
  # (maestro-daemon.mjs, from the release that added this gate), else the last "[daemon] Running" /
57
67
  # "org-mesh connected" line. A crash earlier the same day that launchd
58
- # (or an operator) already recovered from is history, not health. A
59
- # missing log file has no fatal lines (fail-open here — (a) and (c)
60
- # carry the gate).
68
+ # (or an operator) already recovered from is history, not health.
69
+ # The log is NOT always under $AGENT_DIR: launchd-wrapper.sh redirects it
70
+ # to an external SSD ($SSD_VOLUME/maestro/<agent>/logs/daemon) when one is
71
+ # mounted+writable. This scan resolves that same path (daemon_log_base,
72
+ # mirroring the wrapper's own SSD derivation) so a redirected seat is gated
73
+ # on its REAL log — not a fixed $AGENT_DIR path that never exists there,
74
+ # which silently disabled this whole check (and, on gates before the
75
+ # missing-log fail-open, rolled EVERY upgrade back on SSD seats). When even
76
+ # the resolved log cannot be read, the gate does NOT blindly fail-open:
77
+ # it falls back to the path-independent cadence heartbeat
78
+ # state/cadence-bus/health.json (lib/cadence-bus.mjs writeHealth, every
79
+ # ~15s, under the symlinked state/ tree) — fresh (< HEALTH_FRESH_S, 120s)
80
+ # proves the daemon loop is cycling → pass; stale/absent → no log AND no
81
+ # heartbeat is genuine unhealth → fail (checks (a)+(c) still corroborate).
61
82
  # (c) A SERVER-ACKNOWLEDGED BEAT — state/org/last-beat.json {at, ok, code?,
62
83
  # sdkVersion?}, which lib/org/mesh.mjs overwrites on EVERY presence beat
63
84
  # (ok:true only when the org answered 2xx; a rejected or unreachable beat
@@ -137,7 +158,10 @@
137
158
  # (default 600) and MAESTRO_AUTOUPDATE_HEALTH_WAIT (default 30), both seconds;
138
159
  # MAESTRO_AUTOUPDATE_STABLE_S (default 90; MAESTRO_AUTOUPDATE_STABLE_GAP_S is the
139
160
  # sleep between the two pid samples, default = STABLE_S, zeroed by the test) and
140
- # MAESTRO_AUTOUPDATE_BEAT_FRESH_S (default 300); MAESTRO_AUTOUPDATE_RETRY_SLEEP
161
+ # MAESTRO_AUTOUPDATE_BEAT_FRESH_S (default 300); MAESTRO_AUTOUPDATE_HEALTH_FRESH_S
162
+ # (default 120 — the cadence-heartbeat staleness bar for the log-absent fallback);
163
+ # MAESTRO_SSD_VOLUME (the daemon-log SSD volume, mirroring launchd-wrapper.sh; the
164
+ # test points it at a fixture dir); MAESTRO_AUTOUPDATE_RETRY_SLEEP
141
165
  # (default 20 s, ×attempt);
142
166
  # MAESTRO_AUTOUPDATE_LOCK_STALE_S (default 7200); MAESTRO_AUTOUPDATE_FAILED_HOLD_S
143
167
  # (default 86400); MAESTRO_AUTOUPDATE_PATH_PREFIX (a stub-binary dir that wins
@@ -254,6 +278,42 @@ DAEMON_PATTERN="$AGENT_DIR/scripts/daemon/maestro-daemon.mjs"
254
278
  # a pid-set change, i.e. "crash loop".
255
279
  DAEMON_ARGV_RE="^[^ ]*/node $(printf '%s' "$DAEMON_PATTERN" | sed 's/[][\.*^$]/\\&/g')\$"
256
280
  DLOG_DIR="$AGENT_DIR/logs/daemon"
281
+ # The daemon log does NOT always live under $AGENT_DIR. launchd-wrapper.sh
282
+ # redirects the daemon's stdout/stderr to an external SSD when one is mounted
283
+ # and writable — $SSD_VOLUME/maestro/<agent>/logs/daemon/ — falling back to
284
+ # $AGENT_DIR/logs/daemon only when that write is denied. is_healthy reads the
285
+ # daemon log for its fatal scan (b); keyed to a fixed $AGENT_DIR path it reads a
286
+ # file that never exists on a redirected seat, which SILENTLY DISABLES the fatal
287
+ # scan there (and, on gates older than the missing-log fail-open, rolled EVERY
288
+ # upgrade back — 2026-09 SSD seats). Resolve the SSD dir the SAME way the wrapper
289
+ # does (its own MAESTRO_SSD_VOLUME / /Volumes/*-SSD derivation) — this is the
290
+ # real path, not a widened guess — and, as a path-independent backstop robust to
291
+ # ANY redirect, cross-check the cadence heartbeat state/cadence-bus/health.json
292
+ # (written under the symlinked state/ tree every ~15s by lib/cadence-bus.mjs
293
+ # writeHealth) whenever the resolved log still cannot be read.
294
+ ssd_daemon_log_dir(){ # echo the SSD daemon-log dir the wrapper would have opened, or nothing when no SSD is mounted (mirrors launchd-wrapper.sh)
295
+ local vol name v
296
+ vol="${MAESTRO_SSD_VOLUME:-}"
297
+ if [ -z "$vol" ]; then
298
+ for v in /Volumes/*-SSD /Volumes/*SSD* /Volumes/maestro-data; do
299
+ if [ -d "$v" ] && [ "$v" != "/Volumes/Macintosh HD" ]; then vol="$v"; break; fi
300
+ done
301
+ fi
302
+ [ -n "$vol" ] && [ -d "$vol" ] || return 0
303
+ name="$(basename "$AGENT_DIR" | sed 's/-ai$//')"
304
+ echo "$vol/maestro/$name/logs/daemon"
305
+ }
306
+ DLOG_DIR_SSD="$(ssd_daemon_log_dir)"
307
+ daemon_log_base(){ # $1 = YYYY-MM-DD → the dir holding that day's daemon log: the SSD copy when the wrapper wrote it there, else $AGENT_DIR. Resolves by where the file IS (the wrapper picks per-boot and can fall back), so a redirected AND a fallback seat both read the right file.
308
+ if [ -n "$DLOG_DIR_SSD" ] && [ -f "$DLOG_DIR_SSD/daemon-$1.log" ]; then echo "$DLOG_DIR_SSD"; else echo "$DLOG_DIR"; fi
309
+ }
310
+ # The cadence heartbeat: state/cadence-bus/health.json {version,ts,pid,...},
311
+ # overwritten every ~15s by the consumer (lib/cadence-bus.mjs writeHealth). It
312
+ # lives under the symlinked state/ tree, so it is readable at a fixed path no
313
+ # matter where the daemon LOG went — the path-independent liveness signal the
314
+ # gate falls back to when the resolved daemon log cannot be read.
315
+ HEALTH_JSON="$AGENT_DIR/state/cadence-bus/health.json"
316
+ HEALTH_FRESH_S="${MAESTRO_AUTOUPDATE_HEALTH_FRESH_S:-120}" # 8× the 15s heartbeat cadence, well under the 300s org-offline bar
257
317
  HEALTH_WAIT="${MAESTRO_AUTOUPDATE_HEALTH_WAIT:-30}"
258
318
  STABLE_S="${MAESTRO_AUTOUPDATE_STABLE_S:-90}"
259
319
  STABLE_GAP_S="${MAESTRO_AUTOUPDATE_STABLE_GAP_S:-$STABLE_S}" # the gap between the two pid samples; the test zeroes it (its stub changes per CALL)
@@ -270,6 +330,18 @@ FATAL_RE='\[DAEMON\] uncaughtException:|\[daemon\] Fatal:|\[wrapper\] FATAL:|ERR
270
330
  HEALTH_REASON="" # set by is_healthy on failure; a phrase, no quotes
271
331
  DAEMON_LABEL=""
272
332
  SESSION_LABEL=""
333
+ # The pure daemon-revive decision (lib/session/revive.mjs) run by revive_daemon
334
+ # below — the on-host backstop for a daemon that is DOWN, which a live daemon
335
+ # can never observe about itself. Defined up here (not with SESSION_STALE_S,
336
+ # which lives in the not-up-to-date path) because revive_daemon fires on the
337
+ # UP-TO-DATE unhealthy branch too. DAEMON_STALE_S mirrors SESSION_STALE_S / the
338
+ # 90s front-door observers.
339
+ LIB_REVIVE="$AGENT_DIR/lib/session/revive.mjs"
340
+ DAEMON_STALE_S="${MAESTRO_DAEMON_STALE_S:-90}"
341
+ # Gap between the two settle-window reads that prove the kickstarted daemon is
342
+ # one continuous process, not a come-up-then-crash loop. A one-shot read would
343
+ # pass a phantom pid AND a crash loop whose local beat file it keeps re-stamping.
344
+ DAEMON_CONFIRM_SETTLE_S="${MAESTRO_DAEMON_CONFIRM_SETTLE_S:-5}"
273
345
 
274
346
  daemon_pids(){ # every pid whose argv IS this agent's daemon entry (see DAEMON_ARGV_RE), sorted, space-joined ("" when none)
275
347
  pgrep -f "$DAEMON_ARGV_RE" 2>/dev/null | sort -n | tr '\n' ' ' | sed 's/ *$//'
@@ -289,8 +361,22 @@ pid_uptime_s(){ # $1 = pid → seconds since it started, "" when ps cannot say
289
361
  day_of_epoch(){ # $1 = epoch seconds → YYYY-MM-DD in local time (the wrapper's `date +%Y-%m-%d` basis)
290
362
  date -r "$1" +%Y-%m-%d 2>/dev/null || date -d "@$1" +%Y-%m-%d 2>/dev/null
291
363
  }
292
- daemon_log_for(){ # $1 = uptime seconds → the log file the wrapper opened for THIS incarnation (named for its start day)
293
- echo "$DLOG_DIR/daemon-$(day_of_epoch $(( $(date +%s) - $1 ))).log"
364
+ daemon_log_for(){ # $1 = uptime seconds → the log file the wrapper opened for THIS incarnation (named for its start day, in the dir it actually writes — SSD or $AGENT_DIR)
365
+ local day; day="$(day_of_epoch $(( $(date +%s) - $1 )))"
366
+ echo "$(daemon_log_base "$day")/daemon-$day.log"
367
+ }
368
+ cadence_heartbeat_stale(){ # prints WHY the cadence heartbeat is not a fresh proof of a live daemon loop, or nothing when it is (state/cadence-bus/health.json, path-independent)
369
+ node -e '
370
+ const [file, freshS] = process.argv.slice(1);
371
+ const out = (s) => process.stdout.write(s);
372
+ let j;
373
+ try { j = JSON.parse(require("fs").readFileSync(file, "utf8")); }
374
+ catch (e) { out(`daemon log unreadable at the resolved path and no cadence heartbeat (${e && e.code === "ENOENT" ? "state/cadence-bus/health.json absent" : "state/cadence-bus/health.json unreadable"})`); process.exit(0); }
375
+ const at = Date.parse(j && j.ts);
376
+ if (!Number.isFinite(at)) { out("daemon log unreadable at the resolved path and cadence health.json carries no parseable ts"); process.exit(0); }
377
+ const ageS = Math.round((Date.now() - at) / 1000);
378
+ if (ageS > Number(freshS)) out(`daemon log unreadable at the resolved path and the cadence heartbeat is ${ageS}s old (> ${freshS}s) — the daemon loop is not cycling`);
379
+ ' "$HEALTH_JSON" "$HEALTH_FRESH_S" 2>/dev/null
294
380
  }
295
381
  first_fatal_after(){ # $1 = log file, $2 = line offset (lines up to it predate this daemon) → the first fatal line, or nothing
296
382
  [ -f "$1" ] || return 0
@@ -355,14 +441,46 @@ sibling_job_audit(){ # log-only: every OTHER ai.maestro.<first>-* job whose LAST
355
441
  [ -n "$bad" ] && log "sibling jobs with a non-zero last exit (not a gate reason; see logs/launchd and logs/polling): $bad"
356
442
  return 0
357
443
  }
444
+ daemon_last_exit(){ # the DAEMON_LABEL's last exit status (column 2 of `launchctl list`, the same column sibling_job_audit reads), or nothing when the label is unknown or not in the list
445
+ [ -n "$DAEMON_LABEL" ] || resolve_labels # is_healthy on the up-to-date/not-stale path never called restart_daemon, so labels may still be unresolved here
446
+ [ -n "$DAEMON_LABEL" ] || return 0
447
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print $2; exit }'
448
+ }
358
449
  is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
359
- local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot
450
+ local p1 p2 pid up oldest oldest_pid dlog fatal why offset boot exit_code
360
451
  HEALTH_REASON=""
361
452
  p1="$(daemon_pids)"
362
453
  [ -n "$p1" ] || { HEALTH_REASON="no daemon process for $DAEMON_PATTERN"; return 1; }
363
454
  [ "$STABLE_GAP_S" -gt 0 ] && sleep "$STABLE_GAP_S"
364
455
  p2="$(daemon_pids)"
365
- [ "$p1" = "$p2" ] || { HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}) — crash loop"; return 1; }
456
+ if [ "$p1" != "$p2" ]; then
457
+ # A pid change alone is not proof of a crash loop: `launchctl kickstart -k`
458
+ # (this same script's own restart_daemon, or an operator's) changes the pid
459
+ # deliberately and can land inside this very window. 2.18.14 read that as a
460
+ # crash loop and rolled back a release that had done nothing wrong. The
461
+ # daemon's OWN last exit says which one happened: a crash-looping process
462
+ # exits non-zero every time launchd reaps it; a clean kickstart's reaped
463
+ # exit is 0 (or "-"/unknown, for a launchd too old to say). Only a REAL
464
+ # non-zero exit is a crash loop here; a clean or unknown exit is treated as
465
+ # an intended restart and falls through to the uptime check below, which
466
+ # still catches a genuinely unstable respawn (a crash-looping daemon's new
467
+ # pid is always younger than STABLE_S).
468
+ exit_code="$(daemon_last_exit)"
469
+ case "$exit_code" in
470
+ ""|"-"|"0")
471
+ # A clean/unknown exit with NO new pid at all is not a restart in
472
+ # progress, it is the daemon gone — the "no daemon process" case, not a
473
+ # crash loop and not healthy.
474
+ [ -n "$p2" ] || { HEALTH_REASON="daemon pid $p1 exited (last exit ${exit_code:-unknown}) and no replacement is running"; return 1; }
475
+ log "daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2}) but its last exit was ${exit_code:-unknown} — treating as an intended restart (deliberate kickstart), not a crash loop; checking the new pid's uptime"
476
+ p1="$p2"
477
+ ;;
478
+ *)
479
+ HEALTH_REASON="daemon pid changed within ${STABLE_GAP_S}s (${p1} -> ${p2:-none}), last exit ${exit_code} — crash loop"
480
+ return 1
481
+ ;;
482
+ esac
483
+ fi
366
484
  oldest=""
367
485
  for pid in $p1; do
368
486
  up="$(pid_uptime_s "$pid")"
@@ -370,11 +488,26 @@ is_healthy(){ # (a)+(b)+(c) above. Sets HEALTH_REASON on failure.
370
488
  [ -n "$oldest" ] && [ "$oldest" -ge "$up" ] || { oldest="$up"; oldest_pid="$pid"; }
371
489
  done
372
490
  dlog="$(daemon_log_for "$oldest")"
373
- offset=0; [ "$dlog" = "$PRE_LOG" ] && offset="$PRE_LOG_LINES"
374
- boot="$(boot_offset "$dlog" "$oldest_pid")"
375
- [ "$boot" -gt "$offset" ] && offset="$boot"
376
- fatal="$(first_fatal_after "$dlog" "$offset")"
377
- [ -z "$fatal" ] || { HEALTH_REASON="fatal in $(basename "$dlog") since this daemon booted (scanned from line $(( offset + 1 ))): $(printf '%s' "$fatal" | tr -d '"\\' | cut -c1-160)"; return 1; }
491
+ if [ -f "$dlog" ]; then
492
+ # The daemon log is readable (at $AGENT_DIR or the SSD path the wrapper
493
+ # redirected it to): scan it for a fatal line since this daemon booted.
494
+ offset=0; [ "$dlog" = "$PRE_LOG" ] && offset="$PRE_LOG_LINES"
495
+ boot="$(boot_offset "$dlog" "$oldest_pid")"
496
+ [ "$boot" -gt "$offset" ] && offset="$boot"
497
+ fatal="$(first_fatal_after "$dlog" "$offset")"
498
+ [ -z "$fatal" ] || { HEALTH_REASON="fatal in $(basename "$dlog") since this daemon booted (scanned from line $(( offset + 1 ))): $(printf '%s' "$fatal" | tr -d '"\\' | cut -c1-160)"; return 1; }
499
+ else
500
+ # The daemon log is not readable even at the SSD-resolved path (a redirect
501
+ # this gate could not resolve, or a brand-new start-day file the daemon has
502
+ # not written yet). Do NOT fail-open blindly (that silently drops check (b)
503
+ # on every such seat) and do NOT roll a healthy release back on a path
504
+ # artefact. Fall back to the path-independent cadence heartbeat: fresh →
505
+ # the daemon's main loop is demonstrably cycling, so the missing log is a
506
+ # path artefact, not death → pass; stale/absent → no log AND no heartbeat is
507
+ # genuine unhealth → fail.
508
+ why="$(cadence_heartbeat_stale)"
509
+ [ -z "$why" ] || { HEALTH_REASON="$why"; return 1; }
510
+ fi
378
511
  if org_enrolled; then
379
512
  why="$(beat_not_acknowledged $(( $(date +%s) - oldest - 2 )) "$(installed_version)")"
380
513
  [ -z "$why" ] || { HEALTH_REASON="$why"; return 1; }
@@ -409,8 +542,9 @@ label_loaded(){ # $1 = label — measured in the gui domain (right from ssh too)
409
542
  restart_daemon(){
410
543
  resolve_labels
411
544
  # Mark "since the restart" for the fatal scan: the new process appends to
412
- # today's file (the wrapper names it for its start day), after these lines.
413
- PRE_LOG="$DLOG_DIR/daemon-$(date +%Y-%m-%d).log"
545
+ # today's file (the wrapper names it for its start day), after these lines —
546
+ # in the dir the wrapper actually writes (SSD when redirected, else $AGENT_DIR).
547
+ PRE_LOG="$(daemon_log_base "$(date +%Y-%m-%d)")/daemon-$(date +%Y-%m-%d).log"
414
548
  PRE_LOG_LINES=0
415
549
  [ -f "$PRE_LOG" ] && PRE_LOG_LINES="$(wc -l < "$PRE_LOG" 2>/dev/null | tr -d ' ')"
416
550
  PRE_LOG_LINES="${PRE_LOG_LINES:-0}"
@@ -450,19 +584,133 @@ revive_session(){ # a wedged front door cannot restart itself; restart it FOR it
450
584
  # Leave that one to session_reconciled, which already names it.
451
585
  label_loaded "$SESSION_LABEL" || return 0
452
586
  session_beating && return 0
453
- log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — kickstarting it rather than waiting for a session that cannot answer"
587
+ # Two-read hysteresis (see SESSION_REVIVE_RECHECK_S): the first stale read is
588
+ # not enough to drop a session, because a supervisor resume gaps the beat for
589
+ # a moment. Confirm the door is STILL silent on a second read 60s later
590
+ # before kickstarting — if it beat again in between, that was a resume gap,
591
+ # not a wedge, and the session keeps its context.
592
+ log "front door $SESSION_LABEL is loaded but not beating (> ${SESSION_STALE_S}s) — confirming over a second read ${SESSION_REVIVE_RECHECK_S}s apart before kickstarting"
593
+ sleep "$SESSION_REVIVE_RECHECK_S"
594
+ session_beating && { log "front door $SESSION_LABEL beat again on the second read — a resume gap, not a wedge; leaving it"; return 0; }
595
+ # Supervisor coordination via the resume loop's OWN lock (Jacob), not a soft
596
+ # note. The supervisor holds the `session` singleton (state/locks/process/
597
+ # session.pid) for its whole life and beats as a healthy resume progresses.
598
+ # We are past the FULL grace and two reads apart, so a resume gap is already
599
+ # excluded by time — a live holder here is a supervisor PARKED in its launch
600
+ # probe (the 3.5-day / 15-day case), for which the hard kickstart is right.
601
+ # Consulting the lock tells a parked supervisor (override, logged) from a dead
602
+ # one (clean restart); it never blocks the parked case a hard skip would.
603
+ local slock="$AGENT_DIR/state/locks/process/session.pid" spid=""
604
+ [ -f "$slock" ] && spid="$(head -n1 "$slock" 2>/dev/null | tr -dc '0-9')"
605
+ if [ -n "$spid" ] && kill -0 "$spid" 2>/dev/null; then
606
+ log "front door $SESSION_LABEL dark past grace while a supervisor (pid $spid) still holds the session lock — parked in its launch probe; hard-kickstarting over it"
607
+ fi
608
+ log "front door $SESSION_LABEL not beating on two reads ${SESSION_REVIVE_RECHECK_S}s apart — kickstarting it rather than waiting for a session that cannot answer"
454
609
  launchctl kickstart -k "gui/$(id -u)/$SESSION_LABEL" >> "$LOG" 2>&1 \
455
610
  && { log "front door $SESSION_LABEL restarted"; return 0; }
456
611
  log "WARN: could not kickstart $SESSION_LABEL (maestro session restart --force)"
457
612
  return 1
458
613
  }
459
614
 
615
+ daemon_launchctl_state(){ # "pid" | "dash" | "absent" for DAEMON_LABEL, read from column 1 (PID) of `launchctl list` — a number is a live pid, "-" is loaded-with-no-pid, missing is absent
616
+ [ -n "$DAEMON_LABEL" ] || { printf 'absent'; return 0; }
617
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" '
618
+ NR>1 && $3==l { print ($1 ~ /^[0-9]+$/ ? "pid" : "dash"); f=1; exit }
619
+ END { if (!f) print "absent" }'
620
+ }
621
+ daemon_beat_age_ms(){ # age of the daemon dashboard beat file in ms; a large number when it is absent (never a small/fresh value)
622
+ node -e 'try { const s = require("fs").statSync(process.argv[1]); process.stdout.write(String(Math.max(0, Date.now() - s.mtimeMs))); } catch { process.stdout.write("999999999"); }' \
623
+ "$AGENT_DIR/state/dashboards/daemon-health.yaml" 2>/dev/null || printf '999999999'
624
+ }
625
+ REVIVE_NOTE_REL="state/telemetry/revive-note.json"
626
+ write_revive_note(){ # $1=reason $2=target $3=detail — a failed revive must land ON THE BEAT (collect.mjs → machine.reviveNote), not only in this log; a person reading the fleet sees the failed escalation
627
+ local dir="$AGENT_DIR/state/telemetry" tmp
628
+ mkdir -p "$dir" 2>/dev/null || return 0
629
+ tmp="$(mktemp "$dir/.revive-note.XXXXXX" 2>/dev/null)" || return 0
630
+ printf '{\n "reason": "%s",\n "target": "%s",\n "detail": "%s",\n "at": "%s"\n}\n' \
631
+ "$(json_str "$1")" "$(json_str "$2")" "$(json_str "$3")" "$(date -u +%FT%TZ)" > "$tmp" \
632
+ && mv -f "$tmp" "$dir/$(basename "$REVIVE_NOTE_REL")"
633
+ return 0
634
+ }
635
+ daemon_launchctl_pid(){ # the numeric pid launchctl lists for DAEMON_LABEL (empty when it lists a dash or is absent)
636
+ [ -n "$DAEMON_LABEL" ] || return 0
637
+ launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { if ($1 ~ /^[0-9]+$/) print $1; exit }'
638
+ }
639
+ daemon_sample_json(){ # one settle-window observation: {"pid":N|null,"alive":bool,"startTime":"…"|null}. A LISTED pid is not a RUNNING one — kill -0 verifies; ps -o lstart fingerprints the instance so a relaunch (crash loop) shows a different start-time.
640
+ local pid start
641
+ pid="$(daemon_launchctl_pid)"
642
+ if [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null; then
643
+ start="$(ps -o lstart= -p "$pid" 2>/dev/null | tr -d '\n"')"
644
+ printf '{"pid":%s,"alive":true,"startTime":"%s"}' "$pid" "$start"
645
+ elif [ -n "$pid" ]; then
646
+ printf '{"pid":%s,"alive":false,"startTime":null}' "$pid" # phantom: launchctl lists it, ps says dead
647
+ else
648
+ printf '{"pid":null,"alive":false,"startTime":null}'
649
+ fi
650
+ }
651
+ revive_daemon(){ # the daemon-DOWN backstop. A daemon cannot restart ITSELF — a process alive to ask shows a live pid — so this hourly job is the on-host actor for shouldReviveDaemon (lib/session/revive.mjs). Jacob 2026-09-25: heartbeat.json ticking, no overdue handoff, daemon dead; visible on-host ONLY here.
652
+ # is_healthy bails at the empty-pid check before daemon_last_exit (its label
653
+ # resolver) runs, so resolve them here — the down case is exactly the one
654
+ # where nothing upstream did.
655
+ [ -n "$DAEMON_LABEL" ] || resolve_labels
656
+ [ -n "$DAEMON_LABEL" ] || return 0
657
+ [ -f "$LIB_REVIVE" ] || return 0
658
+ # ONLY the no-process case. A crash-looping, young, or beat-unhealthy daemon
659
+ # still has a pid — that is the health gate's business, and a kickstart would
660
+ # fight the very restart it is doing. pgrep-empty is Jacob's lane.
661
+ [ -z "$(daemon_pids)" ] || return 0
662
+ local st beat lpid palive verdict
663
+ st="$(daemon_launchctl_state)"
664
+ beat="$(daemon_beat_age_ms)"
665
+ # PHANTOM PID (Hannah): launchctl can list a pid whose process is dead.
666
+ # A listed pid is not a running daemon — verify it, and pass pidAlive so
667
+ # shouldReviveDaemon demotes a phantom to the dash gate instead of rubber-
668
+ # stamping it alive (which would deadlock: pgrep-empty says down, launchctl
669
+ # says pid, and nothing revives).
670
+ lpid="$(daemon_launchctl_pid)"
671
+ if [ -n "$lpid" ] && kill -0 "$lpid" 2>/dev/null; then palive="true"; else palive="false"; fi
672
+ verdict="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.shouldReviveDaemon({ launchctl: process.argv[2], beatAgeMs: Number(process.argv[3]), staleMs: Number(process.argv[4]), pidAlive: process.argv[5] === "true" }); process.stdout.write(v.revive ? "revive" : ("no:" + v.reason));' "$LIB_REVIVE" "$st" "$beat" "$((DAEMON_STALE_S * 1000))" "$palive" 2>/dev/null)"
673
+ case "$verdict" in
674
+ revive) : ;;
675
+ *) log "daemon $DAEMON_LABEL down-check: launchctl=$st pidAlive=$palive beat=${beat}ms — not reviving (${verdict:-no})"; return 0 ;;
676
+ esac
677
+ log "daemon $DAEMON_LABEL is loaded but down (launchctl=$st pidAlive=$palive, daemon beat ${beat}ms stale) — kickstarting it; a daemon cannot restart itself, so the hourly backstop does"
678
+ launchctl kickstart -k "gui/$UID_/$DAEMON_LABEL" >> "$LOG" 2>&1 || { log "WARN: could not kickstart $DAEMON_LABEL"; write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "kickstart returned nonzero"; return 1; }
679
+ # ASSERT across a SETTLE WINDOW (Jacob, incident-validated): a one-shot read
680
+ # passes a phantom pid AND a come-up-then-crash whose local beat it keeps
681
+ # stamping. Two reads DAEMON_CONFIRM_SETTLE_S apart prove one continuous,
682
+ # live process (same pid, same start-time, alive on both) — the uptime
683
+ # continuity a crash loop cannot fake. serverBeatFresh is left unset here (an
684
+ # hourly backstop does not round-trip the org just to confirm); continuity
685
+ # carries it, and the daemon's own presence publish is the fleet-visible proof.
686
+ local s1 s2 samples ok
687
+ s1="$(daemon_sample_json)"
688
+ sleep "$DAEMON_CONFIRM_SETTLE_S"
689
+ s2="$(daemon_sample_json)"
690
+ samples="[$s1,$s2]"
691
+ ok="$(node --input-type=module -e 'import { pathToFileURL } from "node:url"; const m = await import(pathToFileURL(process.argv[1]).href); const v = m.confirmRevived({ samples: JSON.parse(process.argv[2]), staleMs: Number(process.argv[3]) }); process.stdout.write(v.ok ? "ok" : v.reason);' "$LIB_REVIVE" "$samples" "$((DAEMON_STALE_S * 1000))" 2>/dev/null)"
692
+ if [ "$ok" = "ok" ]; then
693
+ log "daemon $DAEMON_LABEL restarted and holding one continuous live pid across ${DAEMON_CONFIRM_SETTLE_S}s — the backstop revived it"
694
+ rm -f "$AGENT_DIR/$REVIVE_NOTE_REL" 2>/dev/null # recovered — clear the beat's failed-revive note
695
+ return 0
696
+ fi
697
+ # ESCALATE WITH EVIDENCE, and STOP — no blind second kickstart (it would bury
698
+ # this failure's log under the loop). One attempt, then hand up with what it
699
+ # saw: the launchctl list line and the tail of the daemon's own log.
700
+ local lc_line dlog_tail evidence
701
+ lc_line="$(launchctl list 2>/dev/null | awk -v l="$DAEMON_LABEL" 'NR>1 && $3==l { print; exit }')"
702
+ dlog_tail="$(tail -n 3 "$(daemon_log_base "$(date +%Y-%m-%d)")/daemon-$(date +%Y-%m-%d).log" 2>/dev/null | tr '\n' '|')"
703
+ evidence="assert=${ok:-unknown} samples=${samples} launchctl=[${lc_line:-none}] dlog=[${dlog_tail:-none}]"
704
+ log "daemon $DAEMON_LABEL kickstart did NOT revive it (${ok:-unknown}) — a helper's exit code is not a revive; this seat needs a person (failure to escalate). ${evidence}"
705
+ write_revive_note "daemon-revive-failed" "$DAEMON_LABEL" "$evidence"
706
+ return 1
707
+ }
460
708
  session_reconciled(){ # only asked when a -session plist was generated; never a rollback reason
461
709
  label_loaded "$SESSION_LABEL" || { log "reconcile-failed: session job $SESSION_LABEL is not loaded (maestro session start)"; return 1; }
462
710
  pgrep -f "$AGENT_DIR/scripts/session/supervisor" >/dev/null 2>&1 || { log "reconcile-failed: $SESSION_LABEL is loaded but no supervisor process is alive for $AGENT_DIR (maestro session status)"; return 1; }
463
711
  return 0
464
712
  }
465
- json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n' ' '; } # a value safe inside "…" in the JSON files below
713
+ json_str(){ printf '%s' "$1" | tr -d '"\\' | tr '\n\t\r' ' ' | tr -d '\000-\037'; } # a value safe inside "…" in the JSON files below — quotes/backslashes stripped, and EVERY control char (tabs from `launchctl list` columns, CR, others) folded to a space or removed, so an evidence string built from raw tool output never breaks JSON.parse
466
714
  write_notice(){ # $1 = from, $2 = to, [$3 = reason → healthy:false] — tell the front-door session; never restart it here
467
715
  local dir="$AGENT_DIR/state/session" tmp
468
716
  mkdir -p "$dir" 2>/dev/null || return 0
@@ -753,6 +1001,14 @@ if [ "$CUR" = "$LATEST" ] || [ "$(printf '%s\n%s\n' "$CUR" "$LATEST" | sort -V |
753
1001
  esac
754
1002
  fi
755
1003
  else
1004
+ # Jacob's daemon-down lane: up to date, but the daemon is loaded with NO
1005
+ # running process while the front door kept beating — the seat looked "up
1006
+ # to date" every hour and was dark. The daemon cannot restart itself, so
1007
+ # revive_daemon (the on-host actor for shouldReviveDaemon) kickstarts it
1008
+ # here and asserts the pid came back. Gated to the no-process case, so a
1009
+ # crash-looping or beat-unhealthy daemon (which still has a pid) is left to
1010
+ # the gate above, not kickstarted.
1011
+ revive_daemon || true
756
1012
  log "UNHEALTHY on $CUR${STALE_FROM:+ (kickstarted from $STALE_FROM)}: $HEALTH_REASON — no upgrade to roll back; the org cannot see this seat. maestro doctor"
757
1013
  # The `unhealthy-current:` PREFIX is load-bearing (the episode check below
758
1014
  # and failed_hold_reason both match on it), so the ahead-of-registry fact
@@ -777,10 +1033,19 @@ fi
777
1033
  # nothing.
778
1034
  FAILED_HOLD_S="${MAESTRO_AUTOUPDATE_FAILED_HOLD_S:-86400}"
779
1035
  # How long a front door may go without beating before this run restarts it FOR
780
- # it. Ten minutes: long enough that a busy session is never interrupted (the
781
- # heartbeat rides every tool use), short enough that a wedged one is measured in
782
- # minutes rather than the days it took to notice the last three.
783
- SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-600}"
1036
+ # it. 90s — the same bar as STABLE_S, the other 90s observer in this gate — so
1037
+ # a wedged front door is measured in the same window as a wedged daemon rather
1038
+ # than the days it took to notice the last three. `revive_session` (this run,
1039
+ # hourly) is the BACKSTOP; it does not need its own, looser number.
1040
+ SESSION_STALE_S="${MAESTRO_SESSION_STALE_S:-90}"
1041
+ # The daemon confirms a shut front door over TWO reads before it kickstarts
1042
+ # (lib/session/revive.mjs DEFAULT_REVIVE_CONFIRM_READS, REVIVE_CHECK_INTERVAL_MS
1043
+ # 60s apart). `revive_session`, the hourly backstop, now does the same: a
1044
+ # supervisor resume (`claude --resume`) gaps the heartbeat between the old feed
1045
+ # dying and the new one re-arming, and a single stale read landing inside that
1046
+ # gap must not kickstart a session that is mid-resume. The two reads are 60s
1047
+ # apart; the hourly backstop can afford the one wait.
1048
+ SESSION_REVIVE_RECHECK_S="${MAESTRO_SESSION_REVIVE_RECHECK_S:-60}"
784
1049
  failed_hold_reason(){ # prints "<reason> at <at>" when LATEST failed health within FAILED_HOLD_S
785
1050
  node -e '
786
1051
  const [file, latest, holdS] = process.argv.slice(1);