@cohortapp/agent-sdk 2.18.13 → 2.18.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +14 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/collect.mjs +129 -0
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +10 -3
- package/scripts/healthcheck.sh +131 -33
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -96,6 +96,13 @@ import { writeHandoff, listHandoffs, expireHandoffs, pruneHandoffs, handoffPaths
|
|
|
96
96
|
import { writeFileAtomic } from "../../lib/fs-atomic.mjs";
|
|
97
97
|
import { resolveClaudeBin as sharedResolveClaude } from "../../lib/claude-bin.mjs";
|
|
98
98
|
import { getCadenceDef } from "./cadence-handlers.mjs";
|
|
99
|
+
// PERMANENT vs TRANSIENT (lane/permanent-errors). A retry loop that never
|
|
100
|
+
// inspects its error class is how two seats burned down: a cadence whose prompt
|
|
101
|
+
// file does not exist retried at poll speed forever (measured 16 requeues per
|
|
102
|
+
// 30 s on Eli Rosenberg's seat, 2026-09-24). The taxonomy is pure and lives on
|
|
103
|
+
// its own so the policy can be argued with without a daemon.
|
|
104
|
+
import { classifyCadenceFailure, isPermanent, PERMANENT_MAX_ATTEMPTS } from "../../lib/cadence-failure-class.mjs";
|
|
105
|
+
import { bump as bumpCounter } from "../../lib/diagnostics/counters.mjs";
|
|
99
106
|
import { obligationAllowedUnderPosture } from "../../lib/plan/compile.mjs";
|
|
100
107
|
import { isHumanLaneCadence } from "../../lib/cadences.mjs";
|
|
101
108
|
import { sessionPermissionArgs } from "../../lib/session-permissions.mjs";
|
|
@@ -636,6 +643,12 @@ export function startConsumer(opts = {}) {
|
|
|
636
643
|
const budgetEscalateMs = opts.budgetEscalateMs ?? DEFAULT_BUDGET_ESCALATE_MS;
|
|
637
644
|
const maxSpawnMs = opts.maxSpawnMs ?? DEFAULT_SPAWN_TIMEOUT_MS;
|
|
638
645
|
const spawnSession = opts.spawnSession || realSpawnSession;
|
|
646
|
+
// Cadence registry lookup. Injectable because the config-driven half
|
|
647
|
+
// (`config/.cadence-registry.json`) is read once and memoised at module
|
|
648
|
+
// scope, which a hermetic test cannot re-point — and the permanent-error
|
|
649
|
+
// path is reached precisely through a CONFIGURED cadence whose prompt is
|
|
650
|
+
// absent, so it has to be reachable in a test.
|
|
651
|
+
const cadenceDef = typeof opts.getCadenceDef === "function" ? opts.getCadenceDef : getCadenceDef;
|
|
639
652
|
const userLogger = opts.logger;
|
|
640
653
|
// Test / tuning hooks for the reliability layer.
|
|
641
654
|
const backoffSchedule = opts.backoffSchedule || BACKOFF_SCHEDULE_MS;
|
|
@@ -725,13 +738,64 @@ export function startConsumer(opts = {}) {
|
|
|
725
738
|
* `renderCadencePromptBody`, so the handed-off prompt is byte-identical to
|
|
726
739
|
* what a sub-session would have been given, parallelism directive included —
|
|
727
740
|
* and land it under state/session/handoffs/prompts/<tickId>.md.
|
|
741
|
+
*
|
|
742
|
+
* WHY IT TAGS ITS OWN FAILURES. This function does two unrelated things: it
|
|
743
|
+
* READS the cadence's prompt (whose failure says the spawn cannot work
|
|
744
|
+
* either — the spawn opens the same path) and it WRITES a rendered copy into
|
|
745
|
+
* the handoff dir (whose failure says nothing about the spawn at all). The
|
|
746
|
+
* caller has to tell them apart, and the first attempt at that inferred it
|
|
747
|
+
* from the message — `msg.includes(promptPath)`.
|
|
748
|
+
*
|
|
749
|
+
* That inference is WRONG, measured 2026-09-25: `readFileSync` on a
|
|
750
|
+
* DIRECTORY throws exactly `EISDIR: illegal operation on a directory, read`,
|
|
751
|
+
* with no path in the text. A directory at the configured prompt path
|
|
752
|
+
* therefore passed the `existsSync` pre-check, threw here, was read as "not
|
|
753
|
+
* about the prompt", classified transient, and fell through to a spawn — the
|
|
754
|
+
* commonest unreadable case, and precisely the one the PROMPT_UNREADABLE code
|
|
755
|
+
* exists for. Node puts the path in the message for ENOENT and EACCES and
|
|
756
|
+
* not for EISDIR; no caller should have to know which.
|
|
757
|
+
*
|
|
758
|
+
* So the phase is not inferred, it is STAMPED, at the only two places that
|
|
759
|
+
* know it: `err.cadencePhase` is `"prompt"` on the read and `"handoff"` on
|
|
760
|
+
* everything after it. `err.cadenceErrno` carries `err.code` alongside, since
|
|
761
|
+
* the errno is the contract and the message is not.
|
|
728
762
|
*/
|
|
729
763
|
function renderHandoffPrompt(tickId, promptPath) {
|
|
730
764
|
const fullPrompt = join(agentRoot, promptPath);
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
765
|
+
let source;
|
|
766
|
+
try {
|
|
767
|
+
source = readFileSync(fullPrompt, "utf-8");
|
|
768
|
+
} catch (err) {
|
|
769
|
+
throw stampPhase(err, "prompt");
|
|
770
|
+
}
|
|
771
|
+
// Everything past the read is about OUR output, not the cadence's input:
|
|
772
|
+
// a render bug or an unwritable handoff dir leaves the spawn perfectly able
|
|
773
|
+
// to run, so neither may be read as a permanent prompt fault.
|
|
774
|
+
try {
|
|
775
|
+
const body = renderCadencePromptBody(agentRoot, source);
|
|
776
|
+
const out = join(handoffPaths(agentRoot).prompts, `${tickId}.md`);
|
|
777
|
+
writeFileAtomic(out, body);
|
|
778
|
+
return out;
|
|
779
|
+
} catch (err) {
|
|
780
|
+
throw stampPhase(err, "handoff");
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
/**
|
|
785
|
+
* Stamp an error with the phase it came from, so a catch does not have to
|
|
786
|
+
* guess. Non-destructive: an already-stamped error keeps its innermost phase
|
|
787
|
+
* (the read is nested inside nothing, but this keeps re-throws honest), and a
|
|
788
|
+
* non-object throw is wrapped rather than dropped.
|
|
789
|
+
*/
|
|
790
|
+
function stampPhase(err, phase) {
|
|
791
|
+
if (!err || typeof err !== "object") {
|
|
792
|
+
const wrapped = new Error(String(err ?? "unknown error"));
|
|
793
|
+
wrapped.cadencePhase = phase;
|
|
794
|
+
return wrapped;
|
|
795
|
+
}
|
|
796
|
+
if (!err.cadencePhase) err.cadencePhase = phase;
|
|
797
|
+
if (!err.cadenceErrno && typeof err.code === "string") err.cadenceErrno = err.code;
|
|
798
|
+
return err;
|
|
735
799
|
}
|
|
736
800
|
|
|
737
801
|
/**
|
|
@@ -759,7 +823,7 @@ export function startConsumer(opts = {}) {
|
|
|
759
823
|
// predicate both places — the daemon and the plan cannot disagree about
|
|
760
824
|
// what is suspended.
|
|
761
825
|
if (posture && cadence) {
|
|
762
|
-
const def =
|
|
826
|
+
const def = cadenceDef(cadence) || {};
|
|
763
827
|
const verdict = obligationAllowedUnderPosture(
|
|
764
828
|
{
|
|
765
829
|
kind: "SCHEDULE",
|
|
@@ -824,6 +888,13 @@ export function startConsumer(opts = {}) {
|
|
|
824
888
|
handed_off: 0,
|
|
825
889
|
handoff_timeouts: 0,
|
|
826
890
|
deferred: 0,
|
|
891
|
+
// A failure the classifier called PERMANENT — the cause cannot change by
|
|
892
|
+
// waiting, so the tick was dead-lettered instead of retried. Rides the
|
|
893
|
+
// heartbeat (writeHealth) so `maestro cadence status` and the beat carry it
|
|
894
|
+
// without a new channel; `last_permanent_failure` names the cadence and the
|
|
895
|
+
// code so the reader does not have to go back to the log to find out which.
|
|
896
|
+
permanent_failures: 0,
|
|
897
|
+
last_permanent_failure: null,
|
|
827
898
|
last_event_id: null,
|
|
828
899
|
last_decision: null,
|
|
829
900
|
};
|
|
@@ -887,6 +958,99 @@ export function startConsumer(opts = {}) {
|
|
|
887
958
|
return { ok: false, decision: "deferred", reason };
|
|
888
959
|
}
|
|
889
960
|
|
|
961
|
+
/**
|
|
962
|
+
* A failure the classifier calls PERMANENT: stop retrying, record it
|
|
963
|
+
* durably, and make it visible to a person.
|
|
964
|
+
*
|
|
965
|
+
* THE FAULT THIS CLOSES (Eli Rosenberg's seat, 2026-09-24). A cadence whose
|
|
966
|
+
* configured prompt file was not on disk requeued forever — measured at 16
|
|
967
|
+
* per 30 seconds — because nothing in the retry path ever asked whether the
|
|
968
|
+
* cause could change by waiting. It cannot: a file that does not exist will
|
|
969
|
+
* not exist 30 seconds later.
|
|
970
|
+
*
|
|
971
|
+
* Three things happen, and all three matter:
|
|
972
|
+
*
|
|
973
|
+
* 1. STOP. `failTick` with `min(this consumer's budget,
|
|
974
|
+
* PERMANENT_MAX_ATTEMPTS)` — the permanent cap lowers a budget, never
|
|
975
|
+
* raises one. Two attempts, then dlq/: not because the cause can never be
|
|
976
|
+
* repaired (someone can drop the prompt onto the seat), but because it
|
|
977
|
+
* cannot be repaired BY WAITING.
|
|
978
|
+
*
|
|
979
|
+
* STATED HONESTLY (corrected 2026-09-25). `failTick`'s first outcome for
|
|
980
|
+
* attempt 1 is a REQUEUE, so what this buys is "one extra attempt, then
|
|
981
|
+
* stop" — not "the cadence's own schedule is the retry window", which is
|
|
982
|
+
* what an earlier draft of this comment claimed. The schedule only
|
|
983
|
+
* becomes the window once the last attempt reaches dlq/. Spacing the one
|
|
984
|
+
* extra attempt is a separate act, and it is done here:
|
|
985
|
+
* `holdCadenceForBackoff` moves the per-cadence gate forward so the retry
|
|
986
|
+
* cannot land in the same drain that produced the first failure. The
|
|
987
|
+
* measured before/after is 16 requeues per 30 s, unbounded → 2 attempts
|
|
988
|
+
* one backoff interval apart, then a durable stop.
|
|
989
|
+
*
|
|
990
|
+
* The budget is the smaller half of this; the bigger half is (2) and (3),
|
|
991
|
+
* and that a permanent fault no longer falls through to a second code
|
|
992
|
+
* path that opens the same missing file.
|
|
993
|
+
* 2. RECORD. The dlq/ file is the durable record and now carries
|
|
994
|
+
* `permanent:<code>` in `last_error`, so the reason survives a restart
|
|
995
|
+
* and a log rotation. A durable counter (`cadence.permanent_failure`)
|
|
996
|
+
* goes to logs/diagnostics/counters/<day>.jsonl alongside it.
|
|
997
|
+
* 3. BE VISIBLE. The counter is what the seat's EXISTING alert lane reads
|
|
998
|
+
* (lib/diagnostics/alerts.mjs `cadence_permanently_failing`, delivered by
|
|
999
|
+
* the alerts cadence through config/alerts.yaml's webhook, or logged
|
|
1000
|
+
* when there is none) — no new channel. And `stats.permanent_failures`
|
|
1001
|
+
* rides the heartbeat, so the seat's health file and everything
|
|
1002
|
+
* downstream of it say so without anyone tailing a log.
|
|
1003
|
+
*
|
|
1004
|
+
* Never throws: the counter bump is best-effort by contract and failTick is
|
|
1005
|
+
* already guarded.
|
|
1006
|
+
*
|
|
1007
|
+
* @param {object} event the claimed tick
|
|
1008
|
+
* @param {object} verdict a CadenceFailureVerdict (class "permanent")
|
|
1009
|
+
* @param {string} detail the raw error text, for the dlq record + log
|
|
1010
|
+
* @param {string} stage log stage naming WHERE it was caught
|
|
1011
|
+
*/
|
|
1012
|
+
function failPermanently(event, verdict, detail, stage) {
|
|
1013
|
+
// The permanent cap LOWERS a budget, never raises one: if this consumer is
|
|
1014
|
+
// already stricter than the cap, its own number wins.
|
|
1015
|
+
const budget = Math.min(maxAttempts, verdict.maxAttempts ?? PERMANENT_MAX_ATTEMPTS);
|
|
1016
|
+
const reason = `permanent:${verdict.code} — ${verdict.reason}${detail ? `: ${detail}` : ""}`;
|
|
1017
|
+
log({
|
|
1018
|
+
level: "error",
|
|
1019
|
+
stage,
|
|
1020
|
+
permanent: true,
|
|
1021
|
+
code: verdict.code,
|
|
1022
|
+
id: event.id,
|
|
1023
|
+
cadence: event.cadence,
|
|
1024
|
+
error: detail || verdict.reason,
|
|
1025
|
+
max_attempts: budget,
|
|
1026
|
+
note: "cause cannot change by waiting — one more attempt after the cadence backoff, then dead-lettered",
|
|
1027
|
+
});
|
|
1028
|
+
try {
|
|
1029
|
+
bumpCounter("cadence.permanent_failure", { cadence: event.cadence, code: verdict.code }, { agentRoot });
|
|
1030
|
+
} catch { /* a counter must never crash the path it observes */ }
|
|
1031
|
+
stats.permanent_failures += 1;
|
|
1032
|
+
stats.last_permanent_failure = {
|
|
1033
|
+
cadence: event.cadence,
|
|
1034
|
+
code: verdict.code,
|
|
1035
|
+
at: new Date().toISOString(),
|
|
1036
|
+
};
|
|
1037
|
+
stats.last_decision = "permanent-failure";
|
|
1038
|
+
// Space the one remaining attempt. Without this the requeue below is
|
|
1039
|
+
// re-claimed by the very next drain pass — "bounded" but still at poll
|
|
1040
|
+
// speed, which is the behaviour this lane exists to remove.
|
|
1041
|
+
const heldUntil = holdCadenceForBackoff(event.cadence);
|
|
1042
|
+
const outcome = failTick(agentRoot, event.id, reason, { maxAttempts: budget });
|
|
1043
|
+
if (outcome?.destination === "dlq") stats.dlq += 1;
|
|
1044
|
+
else {
|
|
1045
|
+
stats.retries += 1;
|
|
1046
|
+
log({ level: "warn", stage: "permanent_retry_held", id: event.id, cadence: event.cadence, code: verdict.code, retry_at: new Date(heldUntil).toISOString() });
|
|
1047
|
+
}
|
|
1048
|
+
// The heartbeat is the visible surface; write it now rather than waiting up
|
|
1049
|
+
// to `heartbeatMs` for a reader to learn a cadence just died.
|
|
1050
|
+
heartbeat();
|
|
1051
|
+
return { ok: false, decision: outcome?.destination === "dlq" ? "dlq-permanent" : "failed-permanent", code: verdict.code };
|
|
1052
|
+
}
|
|
1053
|
+
|
|
890
1054
|
// Ledger hygiene runs from the sweep at most this often.
|
|
891
1055
|
const HANDOFF_PRUNE_EVERY_MS = 60 * 60_000;
|
|
892
1056
|
let lastHandoffPruneAt = 0;
|
|
@@ -989,20 +1153,47 @@ export function startConsumer(opts = {}) {
|
|
|
989
1153
|
s.failures += 1;
|
|
990
1154
|
// Exponential back-off honouring the (test-overridable) schedule.
|
|
991
1155
|
const idx = Math.min(s.failures, backoffSchedule.length - 1);
|
|
992
|
-
s.nextAllowedAt =
|
|
1156
|
+
s.nextAllowedAt = nowMs() + backoffSchedule[idx];
|
|
993
1157
|
if (s.failures >= circuitThreshold) {
|
|
994
|
-
s.openUntil =
|
|
1158
|
+
s.openUntil = nowMs() + circuitDurationMs;
|
|
995
1159
|
log({ level: "error", stage: "circuit_opened", cadence, failures: s.failures, open_until: new Date(s.openUntil).toISOString() });
|
|
996
1160
|
writeCircuitFile();
|
|
997
1161
|
}
|
|
998
1162
|
}
|
|
999
1163
|
|
|
1164
|
+
/**
|
|
1165
|
+
* Hold a cadence off the spawn path for one backoff interval WITHOUT
|
|
1166
|
+
* advancing it toward an open circuit.
|
|
1167
|
+
*
|
|
1168
|
+
* WHY IT IS SEPARATE FROM `recordSubsessionFailure`. A permanent failure is
|
|
1169
|
+
* not a flaky sub-session: counting it toward `circuitThreshold` would trip a
|
|
1170
|
+
* breaker whose whole job is to ride out a bad patch, for a cause that has no
|
|
1171
|
+
* patch to ride out. What it DOES need is spacing — `PERMANENT_MAX_ATTEMPTS`
|
|
1172
|
+
* is a count, not a delay, and `failPermanently`'s first outcome is a requeue,
|
|
1173
|
+
* so without this the "one extra attempt" lands in the very same drain that
|
|
1174
|
+
* produced the first. `escalate` consults `isCadenceAllowed` before it reaches
|
|
1175
|
+
* the prompt at all, so moving `nextAllowedAt` forward is all it takes for the
|
|
1176
|
+
* extra attempt to cost a real interval rather than a millisecond.
|
|
1177
|
+
*
|
|
1178
|
+
* Idempotent-ish and monotonic: never moves the hold EARLIER, so a circuit or
|
|
1179
|
+
* a longer transient backoff already in force keeps its own deadline.
|
|
1180
|
+
*/
|
|
1181
|
+
function holdCadenceForBackoff(cadence) {
|
|
1182
|
+
const s = getCadenceState(cadence);
|
|
1183
|
+
// The first rung is deliberately 0 ("retry immediately once"); a permanent
|
|
1184
|
+
// fault has already proved it needs a gap, so take the first NON-zero rung.
|
|
1185
|
+
const step = backoffSchedule.find((ms) => ms > 0) ?? 0;
|
|
1186
|
+
const until = nowMs() + step;
|
|
1187
|
+
if (until > s.nextAllowedAt) s.nextAllowedAt = until;
|
|
1188
|
+
return s.nextAllowedAt;
|
|
1189
|
+
}
|
|
1190
|
+
|
|
1000
1191
|
function writeCircuitFile() {
|
|
1001
1192
|
// Persist the open-circuit snapshot so doctor + the operator can see
|
|
1002
1193
|
// which cadences are currently held back without scraping logs.
|
|
1003
1194
|
const open = {};
|
|
1004
1195
|
for (const [cad, s] of cadenceState.entries()) {
|
|
1005
|
-
if (s.openUntil >
|
|
1196
|
+
if (s.openUntil > nowMs()) {
|
|
1006
1197
|
open[cad] = { failures: s.failures, open_until: new Date(s.openUntil).toISOString() };
|
|
1007
1198
|
}
|
|
1008
1199
|
}
|
|
@@ -1020,7 +1211,7 @@ export function startConsumer(opts = {}) {
|
|
|
1020
1211
|
|
|
1021
1212
|
function isCadenceAllowed(cadence) {
|
|
1022
1213
|
const s = getCadenceState(cadence);
|
|
1023
|
-
const now =
|
|
1214
|
+
const now = nowMs();
|
|
1024
1215
|
if (s.openUntil > now) return { allowed: false, reason: "circuit-open", retry_at: s.openUntil };
|
|
1025
1216
|
if (s.nextAllowedAt > now) return { allowed: false, reason: "backoff", retry_at: s.nextAllowedAt };
|
|
1026
1217
|
// Circuit closes automatically when openUntil passes.
|
|
@@ -1107,7 +1298,7 @@ export function startConsumer(opts = {}) {
|
|
|
1107
1298
|
}
|
|
1108
1299
|
}
|
|
1109
1300
|
|
|
1110
|
-
const def =
|
|
1301
|
+
const def = cadenceDef(event.cadence);
|
|
1111
1302
|
let promptPath = def?.prompt;
|
|
1112
1303
|
if (!promptPath) {
|
|
1113
1304
|
// Unknown cadence — try the conventional location.
|
|
@@ -1121,6 +1312,15 @@ export function startConsumer(opts = {}) {
|
|
|
1121
1312
|
stats.dlq += 1;
|
|
1122
1313
|
return { ok: false, decision: "dlq-no-prompt" };
|
|
1123
1314
|
}
|
|
1315
|
+
} else if (!existsSync(join(agentRoot, promptPath))) {
|
|
1316
|
+
// THE ELI FAULT, caught before it can loop. The `!promptPath` branch above
|
|
1317
|
+
// has checked the conventional path since it was written; a prompt named
|
|
1318
|
+
// by CONFIG was never checked at all — it was handed straight to the
|
|
1319
|
+
// handoff renderer and then, when that threw, to the spawn, which opens
|
|
1320
|
+
// the same path. Both fail, neither is terminal, and the tick comes back
|
|
1321
|
+
// on the bus 30 seconds later. Check it once, here, and classify.
|
|
1322
|
+
const verdict = classifyCadenceFailure({ phase: "prompt", error: `prompt not found: ${promptPath}` });
|
|
1323
|
+
return failPermanently(event, verdict, `configured prompt missing: ${promptPath}`, "escalate_prompt_missing");
|
|
1124
1324
|
}
|
|
1125
1325
|
|
|
1126
1326
|
// Per-cadence in-flight guard (F9): never the same cadence twice at once —
|
|
@@ -1139,31 +1339,69 @@ export function startConsumer(opts = {}) {
|
|
|
1139
1339
|
const fd = frontDoorState();
|
|
1140
1340
|
const verdict = shouldHandOffTick({ ...fd, mode: def?.mode, metadata: event.metadata });
|
|
1141
1341
|
if (verdict.handOff) {
|
|
1342
|
+
// The PROMPT read is its own step, with its own catch. It used to sit
|
|
1343
|
+
// inside the handoff try/catch, so a prompt that could not be read was
|
|
1344
|
+
// indistinguishable from a handoff file that could not be written — and
|
|
1345
|
+
// the shared catch fell through to the spawn, which opens the same
|
|
1346
|
+
// prompt path. That is the busy loop: `handoff_failed_spawning_instead`
|
|
1347
|
+
// 16 times per 30 seconds on Eli Rosenberg's seat, 2026-09-24.
|
|
1348
|
+
// Split apart, each failure gets the answer it deserves: a permanent
|
|
1349
|
+
// prompt fault stops here; an unwritable handoff still falls through,
|
|
1350
|
+
// because the spawn genuinely might work.
|
|
1351
|
+
let rendered = null;
|
|
1142
1352
|
try {
|
|
1143
|
-
|
|
1144
|
-
const h = writeHandoff(agentRoot, {
|
|
1145
|
-
tickId: event.id,
|
|
1146
|
-
cadence: event.cadence,
|
|
1147
|
-
mode: def?.mode || "escalate",
|
|
1148
|
-
promptPath: rendered,
|
|
1149
|
-
metadata: { ...(event.metadata || {}), sourcePrompt: promptPath },
|
|
1150
|
-
}, { now: nowMs(), deadlineMs: handoffDeadlineMs });
|
|
1151
|
-
if (!h.ok) throw new Error(h.error || "handoff write failed");
|
|
1152
|
-
completeTick(agentRoot, event.id, {
|
|
1153
|
-
decision: "handed-to-session",
|
|
1154
|
-
cadence: event.cadence,
|
|
1155
|
-
prompt: promptPath,
|
|
1156
|
-
handoff: h.path,
|
|
1157
|
-
deadline_at: h.handoff && h.handoff.deadlineAt,
|
|
1158
|
-
});
|
|
1159
|
-
stats.handed_off += 1;
|
|
1160
|
-
stats.last_decision = "handed-to-session";
|
|
1161
|
-
log({ level: "info", stage: "handed_to_session", id: event.id, cadence: event.cadence, prompt: rendered, deadline_at: h.handoff && h.handoff.deadlineAt });
|
|
1162
|
-
return { ok: true, decision: "handed-to-session" };
|
|
1353
|
+
rendered = renderHandoffPrompt(event.id, promptPath);
|
|
1163
1354
|
} catch (err) {
|
|
1164
|
-
//
|
|
1165
|
-
//
|
|
1166
|
-
|
|
1355
|
+
// renderHandoffPrompt stamps which half failed (see its doc): the
|
|
1356
|
+
// READ of the cadence prompt, or everything after it. Only the read
|
|
1357
|
+
// says anything about whether a spawn would work — the spawn opens
|
|
1358
|
+
// the same path — so only `cadencePhase === "prompt"` may reach a
|
|
1359
|
+
// permanent verdict. An unwritable handoff dir falls through to the
|
|
1360
|
+
// spawn like any other handoff fault.
|
|
1361
|
+
//
|
|
1362
|
+
// This used to be inferred from `msg.includes(promptPath)`, which is
|
|
1363
|
+
// false for EISDIR (Node omits the path), so the commonest unreadable
|
|
1364
|
+
// case — a DIRECTORY at the prompt path — fell through and spawned.
|
|
1365
|
+
// Never classify a path fault by whether the message quotes the path.
|
|
1366
|
+
const msg = (err && err.message) || "";
|
|
1367
|
+
const promptVerdict = classifyCadenceFailure({
|
|
1368
|
+
phase: err?.cadencePhase === "prompt" ? "prompt" : "spawn",
|
|
1369
|
+
error: msg,
|
|
1370
|
+
errno: err?.cadenceErrno,
|
|
1371
|
+
});
|
|
1372
|
+
if (isPermanent(promptVerdict)) {
|
|
1373
|
+
return failPermanently(event, promptVerdict, msg, "handoff_prompt_failed_permanently");
|
|
1374
|
+
}
|
|
1375
|
+
log({ level: "warn", stage: "handoff_render_failed_spawning_instead", id: event.id, cadence: event.cadence, phase: err?.cadencePhase || null, error: msg });
|
|
1376
|
+
}
|
|
1377
|
+
if (rendered) {
|
|
1378
|
+
try {
|
|
1379
|
+
const h = writeHandoff(agentRoot, {
|
|
1380
|
+
tickId: event.id,
|
|
1381
|
+
cadence: event.cadence,
|
|
1382
|
+
mode: def?.mode || "escalate",
|
|
1383
|
+
promptPath: rendered,
|
|
1384
|
+
metadata: { ...(event.metadata || {}), sourcePrompt: promptPath },
|
|
1385
|
+
}, { now: nowMs(), deadlineMs: handoffDeadlineMs });
|
|
1386
|
+
if (!h.ok) throw new Error(h.error || "handoff write failed");
|
|
1387
|
+
completeTick(agentRoot, event.id, {
|
|
1388
|
+
decision: "handed-to-session",
|
|
1389
|
+
cadence: event.cadence,
|
|
1390
|
+
prompt: promptPath,
|
|
1391
|
+
handoff: h.path,
|
|
1392
|
+
deadline_at: h.handoff && h.handoff.deadlineAt,
|
|
1393
|
+
});
|
|
1394
|
+
stats.handed_off += 1;
|
|
1395
|
+
stats.last_decision = "handed-to-session";
|
|
1396
|
+
log({ level: "info", stage: "handed_to_session", id: event.id, cadence: event.cadence, prompt: rendered, deadline_at: h.handoff && h.handoff.deadlineAt });
|
|
1397
|
+
return { ok: true, decision: "handed-to-session" };
|
|
1398
|
+
} catch (err) {
|
|
1399
|
+
// A handoff FILE we could not write is not a reason to lose the tick,
|
|
1400
|
+
// and it says nothing about whether the spawn would work — fall
|
|
1401
|
+
// through to the legacy spawn, loudly, exactly as before. The prompt
|
|
1402
|
+
// half of this, which DOES say so, is handled above.
|
|
1403
|
+
log({ level: "warn", stage: "handoff_failed_spawning_instead", id: event.id, cadence: event.cadence, error: err && err.message });
|
|
1404
|
+
}
|
|
1167
1405
|
}
|
|
1168
1406
|
}
|
|
1169
1407
|
}
|
|
@@ -1292,6 +1530,15 @@ export function startConsumer(opts = {}) {
|
|
|
1292
1530
|
stats.spawn_failures += 1;
|
|
1293
1531
|
recordSubsessionFailure(event.cadence);
|
|
1294
1532
|
const reason = result.error || (stderrTail ? `exit ${result.exit_code}: ${stderrTail}` : `exit ${result.exit_code}`);
|
|
1533
|
+
// Can this possibly succeed if we try again? `realSpawnSession` answers
|
|
1534
|
+
// -2 (prompt not on disk) / -3 (prompt unreadable) for causes that do not
|
|
1535
|
+
// move; those get the classifier's bounded budget and a durable,
|
|
1536
|
+
// visible record. Everything else — a timeout, a non-zero exit, a crash —
|
|
1537
|
+
// is transient and unchanged.
|
|
1538
|
+
const spawnVerdict = classifyCadenceFailure({ phase: "spawn", exitCode: result.exit_code, error: reason });
|
|
1539
|
+
if (isPermanent(spawnVerdict)) {
|
|
1540
|
+
return failPermanently(event, spawnVerdict, reason, "subsession_failed_permanently");
|
|
1541
|
+
}
|
|
1295
1542
|
const outcome = failTick(agentRoot, event.id, reason, { maxAttempts });
|
|
1296
1543
|
if (outcome?.destination === "dlq") stats.dlq += 1;
|
|
1297
1544
|
else stats.retries += 1;
|
|
@@ -1304,7 +1551,7 @@ export function startConsumer(opts = {}) {
|
|
|
1304
1551
|
stats.received += 1;
|
|
1305
1552
|
stats.last_event_id = event.id;
|
|
1306
1553
|
|
|
1307
|
-
const def =
|
|
1554
|
+
const def = cadenceDef(event.cadence);
|
|
1308
1555
|
if (def?.mode === "inline" && typeof def.handler === "function") {
|
|
1309
1556
|
try {
|
|
1310
1557
|
const out = await def.handler({ event, agentRoot, log });
|
|
@@ -1,13 +1,42 @@
|
|
|
1
1
|
#!/bin/bash
|
|
2
2
|
# Emergency Stop — Immediately halts all Maestro agent operations.
|
|
3
|
-
# Usage: ./scripts/emergency-stop.sh
|
|
3
|
+
# Usage: ./scripts/emergency-stop.sh [--dry-run] [--help]
|
|
4
4
|
#
|
|
5
5
|
# This script is the kill switch for all autonomous operations:
|
|
6
6
|
# 1. Drops .emergency-stop flag (every workflow / cadence consumer / enqueue
|
|
7
7
|
# script honours this on the next tick).
|
|
8
8
|
# 2. Unloads every installed `ai.maestro.<agent>-*` (and legacy
|
|
9
9
|
# `ai.adaptic.<agent>-*`) launchd job.
|
|
10
|
-
# 3. Kills running Claude Code subagent processes
|
|
10
|
+
# 3. Kills running Claude Code subagent processes (EXCEPT this process and
|
|
11
|
+
# its ancestors — see step 3).
|
|
12
|
+
#
|
|
13
|
+
# --dry-run
|
|
14
|
+
# Report the kill set instead of signalling it. The flag and the launchd
|
|
15
|
+
# unload still happen; only the signals are withheld — and BOTH the closing
|
|
16
|
+
# banner and the log line say "DRY RUN — NO PROCESSES SIGNALLED", so a dry
|
|
17
|
+
# run can never be mistaken for a halt. It exists so the self-sparing in
|
|
18
|
+
# step 3 is testable without a suite that kills the operator's own session
|
|
19
|
+
# to prove that it does not.
|
|
20
|
+
#
|
|
21
|
+
# WHY A FLAG AND NOT AN ENV VAR. This used to be read from the ambient
|
|
22
|
+
# environment as MAESTRO_EMERGENCY_STOP_DRY_RUN, while step 4 printed
|
|
23
|
+
# "EMERGENCY STOP COMPLETE — All operations halted" unconditionally. A stray
|
|
24
|
+
# `export`, a line in .env, or an EnvironmentVariables entry in a plist
|
|
25
|
+
# would therefore turn the kill switch into a no-op — every claude process
|
|
26
|
+
# surviving — while the banner and the log both asserted the halt had
|
|
27
|
+
# succeeded. That is the same fault class this file was being repaired for:
|
|
28
|
+
# something that looks like it is handling the case and is not. On a
|
|
29
|
+
# break-glass control the escape hatch must be typed at the call site, once,
|
|
30
|
+
# deliberately. The env var is NOT consulted; setting it does nothing.
|
|
31
|
+
#
|
|
32
|
+
# SCOPE OF THE KILL (step 3), stated rather than hidden: every process of THIS
|
|
33
|
+
# user whose command line matches `claude`, minus this process and its
|
|
34
|
+
# ancestors. That is machine-wide, not agent-scoped — an unrelated Claude
|
|
35
|
+
# session of yours in another directory WILL be terminated (21 processes
|
|
36
|
+
# matched on the host where this was measured). Nothing here can narrow it
|
|
37
|
+
# honestly, because a Claude Code session's argv does not carry the agent dir;
|
|
38
|
+
# so the set is printed before it is signalled, and --dry-run shows it without
|
|
39
|
+
# signalling anything.
|
|
11
40
|
#
|
|
12
41
|
# Plist resolution: the agent's first-name slug is read from config/agent.json
|
|
13
42
|
# (SOT) so unload targets the correct labels; falls back to the directory
|
|
@@ -21,6 +50,25 @@ LOG_FILE="$AGENT_DIR/logs/emergency-stop.log"
|
|
|
21
50
|
TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
|
|
22
51
|
mkdir -p "$(dirname "$LOG_FILE")" 2>/dev/null || true
|
|
23
52
|
|
|
53
|
+
# Argument parsing runs BEFORE anything is halted: a typo must refuse loudly,
|
|
54
|
+
# not drop the flag and then exit.
|
|
55
|
+
DRY_RUN=0
|
|
56
|
+
while [ $# -gt 0 ]; do
|
|
57
|
+
case "$1" in
|
|
58
|
+
--dry-run) DRY_RUN=1 ;;
|
|
59
|
+
-h | --help)
|
|
60
|
+
sed -n '2,/^$/p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
|
|
61
|
+
exit 0
|
|
62
|
+
;;
|
|
63
|
+
*)
|
|
64
|
+
echo "emergency-stop: unknown argument: $1" >&2
|
|
65
|
+
echo "Usage: emergency-stop.sh [--dry-run]" >&2
|
|
66
|
+
exit 2
|
|
67
|
+
;;
|
|
68
|
+
esac
|
|
69
|
+
shift
|
|
70
|
+
done
|
|
71
|
+
|
|
24
72
|
# Resolve agent first-name slug from SOT (config/agent.json) so the unload
|
|
25
73
|
# loop targets the right launchd labels. Falls back to the basename of the
|
|
26
74
|
# agent directory (stripping -ai suffix).
|
|
@@ -39,7 +87,11 @@ LAUNCH_AGENTS_DIR="$HOME/Library/LaunchAgents"
|
|
|
39
87
|
# (deployed agents whose plists predate the rename). The loops guard with [ -f ].
|
|
40
88
|
PLIST_GLOB="$LAUNCH_AGENTS_DIR/ai.maestro.${AGENT_FIRST}-*.plist $LAUNCH_AGENTS_DIR/ai.adaptic.${AGENT_FIRST}-*.plist"
|
|
41
89
|
|
|
42
|
-
|
|
90
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
91
|
+
echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN INITIATED (agent=$AGENT_FIRST) — no process will be signalled" | tee -a "$LOG_FILE"
|
|
92
|
+
else
|
|
93
|
+
echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
|
|
94
|
+
fi
|
|
43
95
|
|
|
44
96
|
# 1. Drop the stop flag FIRST so any in-flight work sees it on next tick.
|
|
45
97
|
echo "$TIMESTAMP" > "$AGENT_DIR/.emergency-stop"
|
|
@@ -57,23 +109,72 @@ for plist in $PLIST_GLOB; do
|
|
|
57
109
|
done
|
|
58
110
|
echo "[$TIMESTAMP] Unloaded $unloaded launchd job(s)" >> "$LOG_FILE"
|
|
59
111
|
|
|
60
|
-
# 3. Kill running Claude Code subagent processes.
|
|
61
|
-
#
|
|
62
|
-
#
|
|
112
|
+
# 3. Kill running Claude Code subagent processes.
|
|
113
|
+
#
|
|
114
|
+
# THE MIRROR-IMAGE FAULT THIS FIXES: `pgrep -f claude` matches the process
|
|
115
|
+
# that is RUNNING THIS SCRIPT whenever an agent session invokes it — which
|
|
116
|
+
# is the normal way it gets invoked. The stop then killed its own caller
|
|
117
|
+
# mid-flight, so step 4 never ran and the halt was never logged complete:
|
|
118
|
+
# a script that destroys the condition it needs in order to finish, the
|
|
119
|
+
# same shape as resume-operations.sh refusing to lift its own stop flag.
|
|
120
|
+
# The comment here also claimed a cwd filter that did not exist.
|
|
121
|
+
#
|
|
122
|
+
# So: build the kill set, then subtract this process and every ancestor of
|
|
123
|
+
# it. Everything else matching `claude` is still terminated — the stop is
|
|
124
|
+
# still a kill switch, it just no longer includes the hand on the switch.
|
|
125
|
+
# See SCOPE OF THE KILL in the header: "everything else" really does mean
|
|
126
|
+
# every matching process of this user on this machine, so it is printed.
|
|
63
127
|
echo "Stopping Claude Code agent processes..."
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
128
|
+
|
|
129
|
+
# PIDs to spare: this shell and its whole ancestor chain (the launching
|
|
130
|
+
# session, its shell, launchd). Walk up via ppid until PID 1.
|
|
131
|
+
SPARE=" $$ "
|
|
132
|
+
_p=$$
|
|
133
|
+
while [ -n "$_p" ] && [ "$_p" -gt 1 ]; do
|
|
134
|
+
_p=$(ps -o ppid= -p "$_p" 2>/dev/null | tr -d ' ')
|
|
135
|
+
[ -n "$_p" ] || break
|
|
136
|
+
SPARE="$SPARE$_p "
|
|
137
|
+
done
|
|
138
|
+
|
|
139
|
+
# claude_targets — matching PIDs (this user only) minus the spare set.
|
|
140
|
+
claude_targets() {
|
|
141
|
+
local out=""
|
|
142
|
+
local pid
|
|
143
|
+
for pid in $(pgrep -u "$(id -u)" -f "claude" 2>/dev/null || true); do
|
|
144
|
+
case "$SPARE" in
|
|
145
|
+
*" $pid "*) continue ;;
|
|
146
|
+
esac
|
|
147
|
+
out="$out$pid "
|
|
148
|
+
done
|
|
149
|
+
printf '%s' "$out"
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
pids=$(claude_targets)
|
|
153
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
154
|
+
echo "[DRY-RUN] sparing:$SPARE"
|
|
155
|
+
echo "[DRY-RUN] would terminate: ${pids:-<none>}"
|
|
156
|
+
echo "[$TIMESTAMP] DRY RUN — would terminate: ${pids:-<none>}" >> "$LOG_FILE"
|
|
157
|
+
elif [ -n "$pids" ]; then
|
|
158
|
+
# Name the set before signalling it — this is machine-wide for this user.
|
|
159
|
+
echo "Terminating (every matching process of this user, not just this agent): $pids"
|
|
67
160
|
# Send SIGTERM first, give 3s, then SIGKILL stragglers.
|
|
68
161
|
kill -TERM $pids 2>/dev/null || true
|
|
69
162
|
sleep 3
|
|
70
|
-
still=$(
|
|
71
|
-
[ -n "$still" ]
|
|
163
|
+
still=$(claude_targets)
|
|
164
|
+
if [ -n "$still" ]; then kill -KILL $still 2>/dev/null || true; fi
|
|
72
165
|
echo "[$TIMESTAMP] Claude processes terminated ($pids)" >> "$LOG_FILE"
|
|
166
|
+
else
|
|
167
|
+
echo "[$TIMESTAMP] No Claude processes to terminate (self/ancestors spared)" >> "$LOG_FILE"
|
|
73
168
|
fi
|
|
74
169
|
|
|
75
|
-
# 4. Log completion.
|
|
76
|
-
|
|
170
|
+
# 4. Log completion. A dry run says so in BOTH places — the banner an operator
|
|
171
|
+
# reads and the log an incident review reads — because the previous version
|
|
172
|
+
# printed the halt banner either way.
|
|
173
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
174
|
+
echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN — NO PROCESSES SIGNALLED (flag set, launchd unloaded)" | tee -a "$LOG_FILE"
|
|
175
|
+
else
|
|
176
|
+
echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
|
|
177
|
+
fi
|
|
77
178
|
echo ""
|
|
78
179
|
echo "To resume operations:"
|
|
79
180
|
echo " ./scripts/resume-operations.sh"
|
|
@@ -262,13 +262,20 @@ export function gateLine(r) {
|
|
|
262
262
|
}
|
|
263
263
|
|
|
264
264
|
/**
|
|
265
|
-
* A note about paths the GATES themselves dirtied
|
|
266
|
-
*
|
|
267
|
-
* row would refuse on a file the first run wrote. The clean-tree gate is NOT
|
|
265
|
+
* A note about paths the GATES themselves dirtied: a second run in a row would
|
|
266
|
+
* otherwise refuse on a file the FIRST run wrote. The clean-tree gate is NOT
|
|
268
267
|
* weakened for this — publishing a tree you cannot describe stays a refusal —
|
|
269
268
|
* the run just says which paths are the gates' own leavings so the operator
|
|
270
269
|
* reverts them instead of hunting them. Pure.
|
|
271
270
|
*
|
|
271
|
+
* ~~"`npm test` appends to a tracked runtime ledger
|
|
272
|
+
* (.claude-flow/policy/state.json)"~~ — struck 2026-09-25: that file is the
|
|
273
|
+
* reason this function exists, and it is no longer tracked (it is a per-machine
|
|
274
|
+
* receipt chain, so it never should have been; `.gitignore` says why). This
|
|
275
|
+
* function stays because the SHAPE recurs — the next gate that writes into the
|
|
276
|
+
* tree will do the same thing — and because naming the paths beats hunting
|
|
277
|
+
* them. When it returns null for a whole rollout, that is the expected reading.
|
|
278
|
+
*
|
|
272
279
|
* @param {string[]} before dirty paths before the gates ran
|
|
273
280
|
* @param {string[]} after dirty paths after
|
|
274
281
|
* @returns {string|null}
|