@cohortapp/agent-sdk 2.18.13 → 2.18.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -96,6 +96,13 @@ import { writeHandoff, listHandoffs, expireHandoffs, pruneHandoffs, handoffPaths
96
96
  import { writeFileAtomic } from "../../lib/fs-atomic.mjs";
97
97
  import { resolveClaudeBin as sharedResolveClaude } from "../../lib/claude-bin.mjs";
98
98
  import { getCadenceDef } from "./cadence-handlers.mjs";
99
+ // PERMANENT vs TRANSIENT (lane/permanent-errors). A retry loop that never
100
+ // inspects its error class is how two seats burned down: a cadence whose prompt
101
+ // file does not exist retried at poll speed forever (measured 16 requeues per
102
+ // 30 s on Eli Rosenberg's seat, 2026-09-24). The taxonomy is pure and lives on
103
+ // its own so the policy can be argued with without a daemon.
104
+ import { classifyCadenceFailure, isPermanent, PERMANENT_MAX_ATTEMPTS } from "../../lib/cadence-failure-class.mjs";
105
+ import { bump as bumpCounter } from "../../lib/diagnostics/counters.mjs";
99
106
  import { obligationAllowedUnderPosture } from "../../lib/plan/compile.mjs";
100
107
  import { isHumanLaneCadence } from "../../lib/cadences.mjs";
101
108
  import { sessionPermissionArgs } from "../../lib/session-permissions.mjs";
@@ -636,6 +643,12 @@ export function startConsumer(opts = {}) {
636
643
  const budgetEscalateMs = opts.budgetEscalateMs ?? DEFAULT_BUDGET_ESCALATE_MS;
637
644
  const maxSpawnMs = opts.maxSpawnMs ?? DEFAULT_SPAWN_TIMEOUT_MS;
638
645
  const spawnSession = opts.spawnSession || realSpawnSession;
646
+ // Cadence registry lookup. Injectable because the config-driven half
647
+ // (`config/.cadence-registry.json`) is read once and memoised at module
648
+ // scope, which a hermetic test cannot re-point — and the permanent-error
649
+ // path is reached precisely through a CONFIGURED cadence whose prompt is
650
+ // absent, so it has to be reachable in a test.
651
+ const cadenceDef = typeof opts.getCadenceDef === "function" ? opts.getCadenceDef : getCadenceDef;
639
652
  const userLogger = opts.logger;
640
653
  // Test / tuning hooks for the reliability layer.
641
654
  const backoffSchedule = opts.backoffSchedule || BACKOFF_SCHEDULE_MS;
@@ -725,13 +738,64 @@ export function startConsumer(opts = {}) {
725
738
  * `renderCadencePromptBody`, so the handed-off prompt is byte-identical to
726
739
  * what a sub-session would have been given, parallelism directive included —
727
740
  * and land it under state/session/handoffs/prompts/<tickId>.md.
741
+ *
742
+ * WHY IT TAGS ITS OWN FAILURES. This function does two unrelated things: it
743
+ * READS the cadence's prompt (whose failure says the spawn cannot work
744
+ * either — the spawn opens the same path) and it WRITES a rendered copy into
745
+ * the handoff dir (whose failure says nothing about the spawn at all). The
746
+ * caller has to tell them apart, and the first attempt at that inferred it
747
+ * from the message — `msg.includes(promptPath)`.
748
+ *
749
+ * That inference is WRONG, measured 2026-09-25: `readFileSync` on a
750
+ * DIRECTORY throws exactly `EISDIR: illegal operation on a directory, read`,
751
+ * with no path in the text. A directory at the configured prompt path
752
+ * therefore passed the `existsSync` pre-check, threw here, was read as "not
753
+ * about the prompt", classified transient, and fell through to a spawn — the
754
+ * commonest unreadable case, and precisely the one the PROMPT_UNREADABLE code
755
+ * exists for. Node puts the path in the message for ENOENT and EACCES and
756
+ * not for EISDIR; no caller should have to know which.
757
+ *
758
+ * So the phase is not inferred, it is STAMPED, at the only two places that
759
+ * know it: `err.cadencePhase` is `"prompt"` on the read and `"handoff"` on
760
+ * everything after it. `err.cadenceErrno` carries `err.code` alongside, since
761
+ * the errno is the contract and the message is not.
728
762
  */
729
763
  function renderHandoffPrompt(tickId, promptPath) {
730
764
  const fullPrompt = join(agentRoot, promptPath);
731
- const body = renderCadencePromptBody(agentRoot, readFileSync(fullPrompt, "utf-8"));
732
- const out = join(handoffPaths(agentRoot).prompts, `${tickId}.md`);
733
- writeFileAtomic(out, body);
734
- return out;
765
+ let source;
766
+ try {
767
+ source = readFileSync(fullPrompt, "utf-8");
768
+ } catch (err) {
769
+ throw stampPhase(err, "prompt");
770
+ }
771
+ // Everything past the read is about OUR output, not the cadence's input:
772
+ // a render bug or an unwritable handoff dir leaves the spawn perfectly able
773
+ // to run, so neither may be read as a permanent prompt fault.
774
+ try {
775
+ const body = renderCadencePromptBody(agentRoot, source);
776
+ const out = join(handoffPaths(agentRoot).prompts, `${tickId}.md`);
777
+ writeFileAtomic(out, body);
778
+ return out;
779
+ } catch (err) {
780
+ throw stampPhase(err, "handoff");
781
+ }
782
+ }
783
+
784
+ /**
785
+ * Stamp an error with the phase it came from, so a catch does not have to
786
+ * guess. Non-destructive: an already-stamped error keeps its innermost phase
787
+ * (the read is nested inside nothing, but this keeps re-throws honest), and a
788
+ * non-object throw is wrapped rather than dropped.
789
+ */
790
+ function stampPhase(err, phase) {
791
+ if (!err || typeof err !== "object") {
792
+ const wrapped = new Error(String(err ?? "unknown error"));
793
+ wrapped.cadencePhase = phase;
794
+ return wrapped;
795
+ }
796
+ if (!err.cadencePhase) err.cadencePhase = phase;
797
+ if (!err.cadenceErrno && typeof err.code === "string") err.cadenceErrno = err.code;
798
+ return err;
735
799
  }
736
800
 
737
801
  /**
@@ -759,7 +823,7 @@ export function startConsumer(opts = {}) {
759
823
  // predicate both places — the daemon and the plan cannot disagree about
760
824
  // what is suspended.
761
825
  if (posture && cadence) {
762
- const def = getCadenceDef(cadence) || {};
826
+ const def = cadenceDef(cadence) || {};
763
827
  const verdict = obligationAllowedUnderPosture(
764
828
  {
765
829
  kind: "SCHEDULE",
@@ -824,6 +888,13 @@ export function startConsumer(opts = {}) {
824
888
  handed_off: 0,
825
889
  handoff_timeouts: 0,
826
890
  deferred: 0,
891
+ // A failure the classifier called PERMANENT — the cause cannot change by
892
+ // waiting, so the tick was dead-lettered instead of retried. Rides the
893
+ // heartbeat (writeHealth) so `maestro cadence status` and the beat carry it
894
+ // without a new channel; `last_permanent_failure` names the cadence and the
895
+ // code so the reader does not have to go back to the log to find out which.
896
+ permanent_failures: 0,
897
+ last_permanent_failure: null,
827
898
  last_event_id: null,
828
899
  last_decision: null,
829
900
  };
@@ -887,6 +958,99 @@ export function startConsumer(opts = {}) {
887
958
  return { ok: false, decision: "deferred", reason };
888
959
  }
889
960
 
961
+ /**
962
+ * A failure the classifier calls PERMANENT: stop retrying, record it
963
+ * durably, and make it visible to a person.
964
+ *
965
+ * THE FAULT THIS CLOSES (Eli Rosenberg's seat, 2026-09-24). A cadence whose
966
+ * configured prompt file was not on disk requeued forever — measured at 16
967
+ * per 30 seconds — because nothing in the retry path ever asked whether the
968
+ * cause could change by waiting. It cannot: a file that does not exist will
969
+ * not exist 30 seconds later.
970
+ *
971
+ * Three things happen, and all three matter:
972
+ *
973
+ * 1. STOP. `failTick` with `min(this consumer's budget,
974
+ * PERMANENT_MAX_ATTEMPTS)` — the permanent cap lowers a budget, never
975
+ * raises one. Two attempts, then dlq/: not because the cause can never be
976
+ * repaired (someone can drop the prompt onto the seat), but because it
977
+ * cannot be repaired BY WAITING.
978
+ *
979
+ * STATED HONESTLY (corrected 2026-09-25). `failTick`'s first outcome for
980
+ * attempt 1 is a REQUEUE, so what this buys is "one extra attempt, then
981
+ * stop" — not "the cadence's own schedule is the retry window", which is
982
+ * what an earlier draft of this comment claimed. The schedule only
983
+ * becomes the window once the last attempt reaches dlq/. Spacing the one
984
+ * extra attempt is a separate act, and it is done here:
985
+ * `holdCadenceForBackoff` moves the per-cadence gate forward so the retry
986
+ * cannot land in the same drain that produced the first failure. The
987
+ * measured before/after is 16 requeues per 30 s, unbounded → 2 attempts
988
+ * one backoff interval apart, then a durable stop.
989
+ *
990
+ * The budget is the smaller half of this; the bigger half is (2) and (3),
991
+ * and that a permanent fault no longer falls through to a second code
992
+ * path that opens the same missing file.
993
+ * 2. RECORD. The dlq/ file is the durable record and now carries
994
+ * `permanent:<code>` in `last_error`, so the reason survives a restart
995
+ * and a log rotation. A durable counter (`cadence.permanent_failure`)
996
+ * goes to logs/diagnostics/counters/<day>.jsonl alongside it.
997
+ * 3. BE VISIBLE. The counter is what the seat's EXISTING alert lane reads
998
+ * (lib/diagnostics/alerts.mjs `cadence_permanently_failing`, delivered by
999
+ * the alerts cadence through config/alerts.yaml's webhook, or logged
1000
+ * when there is none) — no new channel. And `stats.permanent_failures`
1001
+ * rides the heartbeat, so the seat's health file and everything
1002
+ * downstream of it say so without anyone tailing a log.
1003
+ *
1004
+ * Never throws: the counter bump is best-effort by contract and failTick is
1005
+ * already guarded.
1006
+ *
1007
+ * @param {object} event the claimed tick
1008
+ * @param {object} verdict a CadenceFailureVerdict (class "permanent")
1009
+ * @param {string} detail the raw error text, for the dlq record + log
1010
+ * @param {string} stage log stage naming WHERE it was caught
1011
+ */
1012
+ function failPermanently(event, verdict, detail, stage) {
1013
+ // The permanent cap LOWERS a budget, never raises one: if this consumer is
1014
+ // already stricter than the cap, its own number wins.
1015
+ const budget = Math.min(maxAttempts, verdict.maxAttempts ?? PERMANENT_MAX_ATTEMPTS);
1016
+ const reason = `permanent:${verdict.code} — ${verdict.reason}${detail ? `: ${detail}` : ""}`;
1017
+ log({
1018
+ level: "error",
1019
+ stage,
1020
+ permanent: true,
1021
+ code: verdict.code,
1022
+ id: event.id,
1023
+ cadence: event.cadence,
1024
+ error: detail || verdict.reason,
1025
+ max_attempts: budget,
1026
+ note: "cause cannot change by waiting — one more attempt after the cadence backoff, then dead-lettered",
1027
+ });
1028
+ try {
1029
+ bumpCounter("cadence.permanent_failure", { cadence: event.cadence, code: verdict.code }, { agentRoot });
1030
+ } catch { /* a counter must never crash the path it observes */ }
1031
+ stats.permanent_failures += 1;
1032
+ stats.last_permanent_failure = {
1033
+ cadence: event.cadence,
1034
+ code: verdict.code,
1035
+ at: new Date().toISOString(),
1036
+ };
1037
+ stats.last_decision = "permanent-failure";
1038
+ // Space the one remaining attempt. Without this the requeue below is
1039
+ // re-claimed by the very next drain pass — "bounded" but still at poll
1040
+ // speed, which is the behaviour this lane exists to remove.
1041
+ const heldUntil = holdCadenceForBackoff(event.cadence);
1042
+ const outcome = failTick(agentRoot, event.id, reason, { maxAttempts: budget });
1043
+ if (outcome?.destination === "dlq") stats.dlq += 1;
1044
+ else {
1045
+ stats.retries += 1;
1046
+ log({ level: "warn", stage: "permanent_retry_held", id: event.id, cadence: event.cadence, code: verdict.code, retry_at: new Date(heldUntil).toISOString() });
1047
+ }
1048
+ // The heartbeat is the visible surface; write it now rather than waiting up
1049
+ // to `heartbeatMs` for a reader to learn a cadence just died.
1050
+ heartbeat();
1051
+ return { ok: false, decision: outcome?.destination === "dlq" ? "dlq-permanent" : "failed-permanent", code: verdict.code };
1052
+ }
1053
+
890
1054
  // Ledger hygiene runs from the sweep at most this often.
891
1055
  const HANDOFF_PRUNE_EVERY_MS = 60 * 60_000;
892
1056
  let lastHandoffPruneAt = 0;
@@ -989,20 +1153,47 @@ export function startConsumer(opts = {}) {
989
1153
  s.failures += 1;
990
1154
  // Exponential back-off honouring the (test-overridable) schedule.
991
1155
  const idx = Math.min(s.failures, backoffSchedule.length - 1);
992
- s.nextAllowedAt = Date.now() + backoffSchedule[idx];
1156
+ s.nextAllowedAt = nowMs() + backoffSchedule[idx];
993
1157
  if (s.failures >= circuitThreshold) {
994
- s.openUntil = Date.now() + circuitDurationMs;
1158
+ s.openUntil = nowMs() + circuitDurationMs;
995
1159
  log({ level: "error", stage: "circuit_opened", cadence, failures: s.failures, open_until: new Date(s.openUntil).toISOString() });
996
1160
  writeCircuitFile();
997
1161
  }
998
1162
  }
999
1163
 
1164
+ /**
1165
+ * Hold a cadence off the spawn path for one backoff interval WITHOUT
1166
+ * advancing it toward an open circuit.
1167
+ *
1168
+ * WHY IT IS SEPARATE FROM `recordSubsessionFailure`. A permanent failure is
1169
+ * not a flaky sub-session: counting it toward `circuitThreshold` would trip a
1170
+ * breaker whose whole job is to ride out a bad patch, for a cause that has no
1171
+ * patch to ride out. What it DOES need is spacing — `PERMANENT_MAX_ATTEMPTS`
1172
+ * is a count, not a delay, and `failPermanently`'s first outcome is a requeue,
1173
+ * so without this the "one extra attempt" lands in the very same drain that
1174
+ * produced the first. `escalate` consults `isCadenceAllowed` before it reaches
1175
+ * the prompt at all, so moving `nextAllowedAt` forward is all it takes for the
1176
+ * extra attempt to cost a real interval rather than a millisecond.
1177
+ *
1178
+ * Idempotent-ish and monotonic: never moves the hold EARLIER, so a circuit or
1179
+ * a longer transient backoff already in force keeps its own deadline.
1180
+ */
1181
+ function holdCadenceForBackoff(cadence) {
1182
+ const s = getCadenceState(cadence);
1183
+ // The first rung is deliberately 0 ("retry immediately once"); a permanent
1184
+ // fault has already proved it needs a gap, so take the first NON-zero rung.
1185
+ const step = backoffSchedule.find((ms) => ms > 0) ?? 0;
1186
+ const until = nowMs() + step;
1187
+ if (until > s.nextAllowedAt) s.nextAllowedAt = until;
1188
+ return s.nextAllowedAt;
1189
+ }
1190
+
1000
1191
  function writeCircuitFile() {
1001
1192
  // Persist the open-circuit snapshot so doctor + the operator can see
1002
1193
  // which cadences are currently held back without scraping logs.
1003
1194
  const open = {};
1004
1195
  for (const [cad, s] of cadenceState.entries()) {
1005
- if (s.openUntil > Date.now()) {
1196
+ if (s.openUntil > nowMs()) {
1006
1197
  open[cad] = { failures: s.failures, open_until: new Date(s.openUntil).toISOString() };
1007
1198
  }
1008
1199
  }
@@ -1020,7 +1211,7 @@ export function startConsumer(opts = {}) {
1020
1211
 
1021
1212
  function isCadenceAllowed(cadence) {
1022
1213
  const s = getCadenceState(cadence);
1023
- const now = Date.now();
1214
+ const now = nowMs();
1024
1215
  if (s.openUntil > now) return { allowed: false, reason: "circuit-open", retry_at: s.openUntil };
1025
1216
  if (s.nextAllowedAt > now) return { allowed: false, reason: "backoff", retry_at: s.nextAllowedAt };
1026
1217
  // Circuit closes automatically when openUntil passes.
@@ -1107,7 +1298,7 @@ export function startConsumer(opts = {}) {
1107
1298
  }
1108
1299
  }
1109
1300
 
1110
- const def = getCadenceDef(event.cadence);
1301
+ const def = cadenceDef(event.cadence);
1111
1302
  let promptPath = def?.prompt;
1112
1303
  if (!promptPath) {
1113
1304
  // Unknown cadence — try the conventional location.
@@ -1121,6 +1312,15 @@ export function startConsumer(opts = {}) {
1121
1312
  stats.dlq += 1;
1122
1313
  return { ok: false, decision: "dlq-no-prompt" };
1123
1314
  }
1315
+ } else if (!existsSync(join(agentRoot, promptPath))) {
1316
+ // THE ELI FAULT, caught before it can loop. The `!promptPath` branch above
1317
+ // has checked the conventional path since it was written; a prompt named
1318
+ // by CONFIG was never checked at all — it was handed straight to the
1319
+ // handoff renderer and then, when that threw, to the spawn, which opens
1320
+ // the same path. Both fail, neither is terminal, and the tick comes back
1321
+ // on the bus 30 seconds later. Check it once, here, and classify.
1322
+ const verdict = classifyCadenceFailure({ phase: "prompt", error: `prompt not found: ${promptPath}` });
1323
+ return failPermanently(event, verdict, `configured prompt missing: ${promptPath}`, "escalate_prompt_missing");
1124
1324
  }
1125
1325
 
1126
1326
  // Per-cadence in-flight guard (F9): never the same cadence twice at once —
@@ -1139,31 +1339,69 @@ export function startConsumer(opts = {}) {
1139
1339
  const fd = frontDoorState();
1140
1340
  const verdict = shouldHandOffTick({ ...fd, mode: def?.mode, metadata: event.metadata });
1141
1341
  if (verdict.handOff) {
1342
+ // The PROMPT read is its own step, with its own catch. It used to sit
1343
+ // inside the handoff try/catch, so a prompt that could not be read was
1344
+ // indistinguishable from a handoff file that could not be written — and
1345
+ // the shared catch fell through to the spawn, which opens the same
1346
+ // prompt path. That is the busy loop: `handoff_failed_spawning_instead`
1347
+ // 16 times per 30 seconds on Eli Rosenberg's seat, 2026-09-24.
1348
+ // Split apart, each failure gets the answer it deserves: a permanent
1349
+ // prompt fault stops here; an unwritable handoff still falls through,
1350
+ // because the spawn genuinely might work.
1351
+ let rendered = null;
1142
1352
  try {
1143
- const rendered = renderHandoffPrompt(event.id, promptPath);
1144
- const h = writeHandoff(agentRoot, {
1145
- tickId: event.id,
1146
- cadence: event.cadence,
1147
- mode: def?.mode || "escalate",
1148
- promptPath: rendered,
1149
- metadata: { ...(event.metadata || {}), sourcePrompt: promptPath },
1150
- }, { now: nowMs(), deadlineMs: handoffDeadlineMs });
1151
- if (!h.ok) throw new Error(h.error || "handoff write failed");
1152
- completeTick(agentRoot, event.id, {
1153
- decision: "handed-to-session",
1154
- cadence: event.cadence,
1155
- prompt: promptPath,
1156
- handoff: h.path,
1157
- deadline_at: h.handoff && h.handoff.deadlineAt,
1158
- });
1159
- stats.handed_off += 1;
1160
- stats.last_decision = "handed-to-session";
1161
- log({ level: "info", stage: "handed_to_session", id: event.id, cadence: event.cadence, prompt: rendered, deadline_at: h.handoff && h.handoff.deadlineAt });
1162
- return { ok: true, decision: "handed-to-session" };
1353
+ rendered = renderHandoffPrompt(event.id, promptPath);
1163
1354
  } catch (err) {
1164
- // A handoff we could not write is not a reason to lose the tick —
1165
- // fall through to the legacy spawn, loudly.
1166
- log({ level: "warn", stage: "handoff_failed_spawning_instead", id: event.id, cadence: event.cadence, error: err && err.message });
1355
+ // renderHandoffPrompt stamps which half failed (see its doc): the
1356
+ // READ of the cadence prompt, or everything after it. Only the read
1357
+ // says anything about whether a spawn would work — the spawn opens
1358
+ // the same path — so only `cadencePhase === "prompt"` may reach a
1359
+ // permanent verdict. An unwritable handoff dir falls through to the
1360
+ // spawn like any other handoff fault.
1361
+ //
1362
+ // This used to be inferred from `msg.includes(promptPath)`, which is
1363
+ // false for EISDIR (Node omits the path), so the commonest unreadable
1364
+ // case — a DIRECTORY at the prompt path — fell through and spawned.
1365
+ // Never classify a path fault by whether the message quotes the path.
1366
+ const msg = (err && err.message) || "";
1367
+ const promptVerdict = classifyCadenceFailure({
1368
+ phase: err?.cadencePhase === "prompt" ? "prompt" : "spawn",
1369
+ error: msg,
1370
+ errno: err?.cadenceErrno,
1371
+ });
1372
+ if (isPermanent(promptVerdict)) {
1373
+ return failPermanently(event, promptVerdict, msg, "handoff_prompt_failed_permanently");
1374
+ }
1375
+ log({ level: "warn", stage: "handoff_render_failed_spawning_instead", id: event.id, cadence: event.cadence, phase: err?.cadencePhase || null, error: msg });
1376
+ }
1377
+ if (rendered) {
1378
+ try {
1379
+ const h = writeHandoff(agentRoot, {
1380
+ tickId: event.id,
1381
+ cadence: event.cadence,
1382
+ mode: def?.mode || "escalate",
1383
+ promptPath: rendered,
1384
+ metadata: { ...(event.metadata || {}), sourcePrompt: promptPath },
1385
+ }, { now: nowMs(), deadlineMs: handoffDeadlineMs });
1386
+ if (!h.ok) throw new Error(h.error || "handoff write failed");
1387
+ completeTick(agentRoot, event.id, {
1388
+ decision: "handed-to-session",
1389
+ cadence: event.cadence,
1390
+ prompt: promptPath,
1391
+ handoff: h.path,
1392
+ deadline_at: h.handoff && h.handoff.deadlineAt,
1393
+ });
1394
+ stats.handed_off += 1;
1395
+ stats.last_decision = "handed-to-session";
1396
+ log({ level: "info", stage: "handed_to_session", id: event.id, cadence: event.cadence, prompt: rendered, deadline_at: h.handoff && h.handoff.deadlineAt });
1397
+ return { ok: true, decision: "handed-to-session" };
1398
+ } catch (err) {
1399
+ // A handoff FILE we could not write is not a reason to lose the tick,
1400
+ // and it says nothing about whether the spawn would work — fall
1401
+ // through to the legacy spawn, loudly, exactly as before. The prompt
1402
+ // half of this, which DOES say so, is handled above.
1403
+ log({ level: "warn", stage: "handoff_failed_spawning_instead", id: event.id, cadence: event.cadence, error: err && err.message });
1404
+ }
1167
1405
  }
1168
1406
  }
1169
1407
  }
@@ -1292,6 +1530,15 @@ export function startConsumer(opts = {}) {
1292
1530
  stats.spawn_failures += 1;
1293
1531
  recordSubsessionFailure(event.cadence);
1294
1532
  const reason = result.error || (stderrTail ? `exit ${result.exit_code}: ${stderrTail}` : `exit ${result.exit_code}`);
1533
+ // Can this possibly succeed if we try again? `realSpawnSession` answers
1534
+ // -2 (prompt not on disk) / -3 (prompt unreadable) for causes that do not
1535
+ // move; those get the classifier's bounded budget and a durable,
1536
+ // visible record. Everything else — a timeout, a non-zero exit, a crash —
1537
+ // is transient and unchanged.
1538
+ const spawnVerdict = classifyCadenceFailure({ phase: "spawn", exitCode: result.exit_code, error: reason });
1539
+ if (isPermanent(spawnVerdict)) {
1540
+ return failPermanently(event, spawnVerdict, reason, "subsession_failed_permanently");
1541
+ }
1295
1542
  const outcome = failTick(agentRoot, event.id, reason, { maxAttempts });
1296
1543
  if (outcome?.destination === "dlq") stats.dlq += 1;
1297
1544
  else stats.retries += 1;
@@ -1304,7 +1551,7 @@ export function startConsumer(opts = {}) {
1304
1551
  stats.received += 1;
1305
1552
  stats.last_event_id = event.id;
1306
1553
 
1307
- const def = getCadenceDef(event.cadence);
1554
+ const def = cadenceDef(event.cadence);
1308
1555
  if (def?.mode === "inline" && typeof def.handler === "function") {
1309
1556
  try {
1310
1557
  const out = await def.handler({ event, agentRoot, log });
@@ -1,13 +1,42 @@
1
1
  #!/bin/bash
2
2
  # Emergency Stop — Immediately halts all Maestro agent operations.
3
- # Usage: ./scripts/emergency-stop.sh
3
+ # Usage: ./scripts/emergency-stop.sh [--dry-run] [--help]
4
4
  #
5
5
  # This script is the kill switch for all autonomous operations:
6
6
  # 1. Drops .emergency-stop flag (every workflow / cadence consumer / enqueue
7
7
  # script honours this on the next tick).
8
8
  # 2. Unloads every installed `ai.maestro.<agent>-*` (and legacy
9
9
  # `ai.adaptic.<agent>-*`) launchd job.
10
- # 3. Kills running Claude Code subagent processes.
10
+ # 3. Kills running Claude Code subagent processes (EXCEPT this process and
11
+ # its ancestors — see step 3).
12
+ #
13
+ # --dry-run
14
+ # Report the kill set instead of signalling it. The flag and the launchd
15
+ # unload still happen; only the signals are withheld — and BOTH the closing
16
+ # banner and the log line say "DRY RUN — NO PROCESSES SIGNALLED", so a dry
17
+ # run can never be mistaken for a halt. It exists so the self-sparing in
18
+ # step 3 is testable without a suite that kills the operator's own session
19
+ # to prove that it does not.
20
+ #
21
+ # WHY A FLAG AND NOT AN ENV VAR. This used to be read from the ambient
22
+ # environment as MAESTRO_EMERGENCY_STOP_DRY_RUN, while step 4 printed
23
+ # "EMERGENCY STOP COMPLETE — All operations halted" unconditionally. A stray
24
+ # `export`, a line in .env, or an EnvironmentVariables entry in a plist
25
+ # would therefore turn the kill switch into a no-op — every claude process
26
+ # surviving — while the banner and the log both asserted the halt had
27
+ # succeeded. That is the same fault class this file was being repaired for:
28
+ # something that looks like it is handling the case and is not. On a
29
+ # break-glass control the escape hatch must be typed at the call site, once,
30
+ # deliberately. The env var is NOT consulted; setting it does nothing.
31
+ #
32
+ # SCOPE OF THE KILL (step 3), stated rather than hidden: every process of THIS
33
+ # user whose command line matches `claude`, minus this process and its
34
+ # ancestors. That is machine-wide, not agent-scoped — an unrelated Claude
35
+ # session of yours in another directory WILL be terminated (21 processes
36
+ # matched on the host where this was measured). Nothing here can narrow it
37
+ # honestly, because a Claude Code session's argv does not carry the agent dir;
38
+ # so the set is printed before it is signalled, and --dry-run shows it without
39
+ # signalling anything.
11
40
  #
12
41
  # Plist resolution: the agent's first-name slug is read from config/agent.json
13
42
  # (SOT) so unload targets the correct labels; falls back to the directory
@@ -21,6 +50,25 @@ LOG_FILE="$AGENT_DIR/logs/emergency-stop.log"
21
50
  TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
22
51
  mkdir -p "$(dirname "$LOG_FILE")" 2>/dev/null || true
23
52
 
53
+ # Argument parsing runs BEFORE anything is halted: a typo must refuse loudly,
54
+ # not drop the flag and then exit.
55
+ DRY_RUN=0
56
+ while [ $# -gt 0 ]; do
57
+ case "$1" in
58
+ --dry-run) DRY_RUN=1 ;;
59
+ -h | --help)
60
+ sed -n '2,/^$/p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
61
+ exit 0
62
+ ;;
63
+ *)
64
+ echo "emergency-stop: unknown argument: $1" >&2
65
+ echo "Usage: emergency-stop.sh [--dry-run]" >&2
66
+ exit 2
67
+ ;;
68
+ esac
69
+ shift
70
+ done
71
+
24
72
  # Resolve agent first-name slug from SOT (config/agent.json) so the unload
25
73
  # loop targets the right launchd labels. Falls back to the basename of the
26
74
  # agent directory (stripping -ai suffix).
@@ -39,7 +87,11 @@ LAUNCH_AGENTS_DIR="$HOME/Library/LaunchAgents"
39
87
  # (deployed agents whose plists predate the rename). The loops guard with [ -f ].
40
88
  PLIST_GLOB="$LAUNCH_AGENTS_DIR/ai.maestro.${AGENT_FIRST}-*.plist $LAUNCH_AGENTS_DIR/ai.adaptic.${AGENT_FIRST}-*.plist"
41
89
 
42
- echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
90
+ if [ "$DRY_RUN" -eq 1 ]; then
91
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN INITIATED (agent=$AGENT_FIRST) — no process will be signalled" | tee -a "$LOG_FILE"
92
+ else
93
+ echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
94
+ fi
43
95
 
44
96
  # 1. Drop the stop flag FIRST so any in-flight work sees it on next tick.
45
97
  echo "$TIMESTAMP" > "$AGENT_DIR/.emergency-stop"
@@ -57,23 +109,72 @@ for plist in $PLIST_GLOB; do
57
109
  done
58
110
  echo "[$TIMESTAMP] Unloaded $unloaded launchd job(s)" >> "$LOG_FILE"
59
111
 
60
- # 3. Kill running Claude Code subagent processes. Filter to processes that
61
- # have AGENT_ROOT or this agent's directory in their cwd to avoid
62
- # killing unrelated claude sessions the operator may have running.
112
+ # 3. Kill running Claude Code subagent processes.
113
+ #
114
+ # THE MIRROR-IMAGE FAULT THIS FIXES: `pgrep -f claude` matches the process
115
+ # that is RUNNING THIS SCRIPT whenever an agent session invokes it — which
116
+ # is the normal way it gets invoked. The stop then killed its own caller
117
+ # mid-flight, so step 4 never ran and the halt was never logged complete:
118
+ # a script that destroys the condition it needs in order to finish, the
119
+ # same shape as resume-operations.sh refusing to lift its own stop flag.
120
+ # The comment here also claimed a cwd filter that did not exist.
121
+ #
122
+ # So: build the kill set, then subtract this process and every ancestor of
123
+ # it. Everything else matching `claude` is still terminated — the stop is
124
+ # still a kill switch, it just no longer includes the hand on the switch.
125
+ # See SCOPE OF THE KILL in the header: "everything else" really does mean
126
+ # every matching process of this user on this machine, so it is printed.
63
127
  echo "Stopping Claude Code agent processes..."
64
- # pgrep -lf is more selective than pkill -f
65
- pids=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
66
- if [ -n "$pids" ]; then
128
+
129
+ # PIDs to spare: this shell and its whole ancestor chain (the launching
130
+ # session, its shell, launchd). Walk up via ppid until PID 1.
131
+ SPARE=" $$ "
132
+ _p=$$
133
+ while [ -n "$_p" ] && [ "$_p" -gt 1 ]; do
134
+ _p=$(ps -o ppid= -p "$_p" 2>/dev/null | tr -d ' ')
135
+ [ -n "$_p" ] || break
136
+ SPARE="$SPARE$_p "
137
+ done
138
+
139
+ # claude_targets — matching PIDs (this user only) minus the spare set.
140
+ claude_targets() {
141
+ local out=""
142
+ local pid
143
+ for pid in $(pgrep -u "$(id -u)" -f "claude" 2>/dev/null || true); do
144
+ case "$SPARE" in
145
+ *" $pid "*) continue ;;
146
+ esac
147
+ out="$out$pid "
148
+ done
149
+ printf '%s' "$out"
150
+ }
151
+
152
+ pids=$(claude_targets)
153
+ if [ "$DRY_RUN" -eq 1 ]; then
154
+ echo "[DRY-RUN] sparing:$SPARE"
155
+ echo "[DRY-RUN] would terminate: ${pids:-<none>}"
156
+ echo "[$TIMESTAMP] DRY RUN — would terminate: ${pids:-<none>}" >> "$LOG_FILE"
157
+ elif [ -n "$pids" ]; then
158
+ # Name the set before signalling it — this is machine-wide for this user.
159
+ echo "Terminating (every matching process of this user, not just this agent): $pids"
67
160
  # Send SIGTERM first, give 3s, then SIGKILL stragglers.
68
161
  kill -TERM $pids 2>/dev/null || true
69
162
  sleep 3
70
- still=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
71
- [ -n "$still" ] && kill -KILL $still 2>/dev/null || true
163
+ still=$(claude_targets)
164
+ if [ -n "$still" ]; then kill -KILL $still 2>/dev/null || true; fi
72
165
  echo "[$TIMESTAMP] Claude processes terminated ($pids)" >> "$LOG_FILE"
166
+ else
167
+ echo "[$TIMESTAMP] No Claude processes to terminate (self/ancestors spared)" >> "$LOG_FILE"
73
168
  fi
74
169
 
75
- # 4. Log completion.
76
- echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
170
+ # 4. Log completion. A dry run says so in BOTH places — the banner an operator
171
+ # reads and the log an incident review reads — because the previous version
172
+ # printed the halt banner either way.
173
+ if [ "$DRY_RUN" -eq 1 ]; then
174
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN — NO PROCESSES SIGNALLED (flag set, launchd unloaded)" | tee -a "$LOG_FILE"
175
+ else
176
+ echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
177
+ fi
77
178
  echo ""
78
179
  echo "To resume operations:"
79
180
  echo " ./scripts/resume-operations.sh"
@@ -262,13 +262,20 @@ export function gateLine(r) {
262
262
  }
263
263
 
264
264
  /**
265
- * A note about paths the GATES themselves dirtied. `npm test` appends to a
266
- * tracked runtime ledger (.claude-flow/policy/state.json), so a second run in a
267
- * row would refuse on a file the first run wrote. The clean-tree gate is NOT
265
+ * A note about paths the GATES themselves dirtied: a second run in a row would
266
+ * otherwise refuse on a file the FIRST run wrote. The clean-tree gate is NOT
268
267
  * weakened for this — publishing a tree you cannot describe stays a refusal —
269
268
  * the run just says which paths are the gates' own leavings so the operator
270
269
  * reverts them instead of hunting them. Pure.
271
270
  *
271
+ * ~~"`npm test` appends to a tracked runtime ledger
272
+ * (.claude-flow/policy/state.json)"~~ — struck 2026-09-25: that file is the
273
+ * reason this function exists, and it is no longer tracked (it is a per-machine
274
+ * receipt chain, so it never should have been; `.gitignore` says why). This
275
+ * function stays because the SHAPE recurs — the next gate that writes into the
276
+ * tree will do the same thing — and because naming the paths beats hunting
277
+ * them. When it returns null for a whole rollout, that is the expected reading.
278
+ *
272
279
  * @param {string[]} before dirty paths before the gates ran
273
280
  * @param {string[]} after dirty paths after
274
281
  * @returns {string|null}