@cohortapp/agent-sdk 2.18.12 → 2.18.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/bin/maestro.mjs +38 -1
  2. package/docs/runbooks/fleet-rollout.md +14 -7
  3. package/docs/runbooks/recovery-and-failover.md +18 -0
  4. package/lib/cadence-failure-class.mjs +245 -0
  5. package/lib/claude-bin.mjs +26 -7
  6. package/lib/cli/doctor-checks.mjs +149 -1
  7. package/lib/comms/send-gate.mjs +6 -4
  8. package/lib/diagnostics/alerts.mjs +33 -0
  9. package/lib/engine/agents/usage.mjs +45 -0
  10. package/lib/engine/budget.mjs +293 -29
  11. package/lib/engine/cli.mjs +54 -5
  12. package/lib/engine/loop.mjs +30 -0
  13. package/lib/engine/output/json.mjs +26 -0
  14. package/lib/engine/wire/errors.mjs +179 -0
  15. package/lib/engine/wire/search.mjs +44 -8
  16. package/lib/identity/claude-md.mjs +107 -0
  17. package/lib/identity/disclosure-instructions.mjs +148 -0
  18. package/lib/identity/disclosure-scrub.mjs +207 -0
  19. package/lib/identity/persona.mjs +141 -6
  20. package/lib/org/inbound/conversation-frame.mjs +289 -0
  21. package/lib/org/inbound/directedness.mjs +27 -7
  22. package/lib/org/quota.mjs +27 -0
  23. package/lib/session/config.mjs +4 -0
  24. package/lib/session/identity.mjs +71 -7
  25. package/lib/session/launch-failure.mjs +251 -0
  26. package/lib/session/resume-target.mjs +86 -0
  27. package/lib/telemetry/collect.mjs +129 -0
  28. package/lib/upgrade/pinned-drift.mjs +467 -0
  29. package/package.json +1 -1
  30. package/plugins/maestro-skills/skills/persona-discipline.md +24 -2
  31. package/scaffold/config/alerts.yaml +7 -0
  32. package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
  33. package/scripts/ci/check.mjs +3 -0
  34. package/scripts/ci/run-tests.mjs +16 -2
  35. package/scripts/daemon/cadence-consumer.mjs +281 -34
  36. package/scripts/daemon/context-compiler.mjs +9 -1
  37. package/scripts/daemon/prompt-builder.mjs +219 -137
  38. package/scripts/daemon/responder.mjs +226 -26
  39. package/scripts/emergency-stop.sh +114 -13
  40. package/scripts/fleet/rollout.mjs +10 -3
  41. package/scripts/healthcheck.sh +131 -33
  42. package/scripts/resume-operations.sh +101 -6
  43. package/scripts/session/supervisor.mjs +198 -5
@@ -74,9 +74,53 @@ import { generateAck } from "./assurance.mjs";
74
74
  // must be large; attach genuinely verbose content rather than dumping it). The
75
75
  // SAME constant leads the full-session prompt in prompt-builder.mjs, so a reactive
76
76
  // quick reply and a full inbox/backlog session shape outbound prose identically.
77
- // This path never renders the persona block, which is why the doctrine is a
78
- // separately-exported constant rather than part of voiceRules().
79
- import { MESSAGE_CRAFT } from "../../lib/identity/persona.mjs";
77
+ // This path renders the seat persona through renderSeatPersona() rather than
78
+ // the full voiceRules() block, which is why the doctrine is a separately-
79
+ // exported constant rather than part of voiceRules().
80
+ //
81
+ // SELF_PRESENTATION travels with it, and for a sharper reason. The preamble this
82
+ // path loads is scraped out of the SEAT's CLAUDE.md — a file `maestro upgrade`
83
+ // deliberately never touches, so a seat scaffolded before the "How you write and
84
+ // carry yourself" section existed gets NO instruction about self-presentation on
85
+ // this plane at all. A model with no such instruction falls back to the
86
+ // generic-assistant register it ships with, which is how quick replies came to
87
+ // open with a rendering of the `identity_line` template in
88
+ // policies/ai-disclosure.yaml inside the organisation's own channels — a line
89
+ // owed to an external first contact and to nobody else. Shipping the rules from
90
+ // the repo is what lets the fix reach a seat through `maestro upgrade`, once
91
+ // this is released — nothing here changes a seat that is still on an older SDK.
92
+ //
93
+ // Adding the rule is only half of it. The preamble below is SEAT-LOCAL PROSE,
94
+ // and some seats were hand-edited to carry the opener as a standing rule back
95
+ // when the send gate demanded it. A counter-instruction does not delete a
96
+ // contradiction, and the seat's copy arrives first, so the scrub strips the
97
+ // stale instruction out of the scraped text before the model ever sees it. The
98
+ // truthfulness invariant is exempt from the scrub by construction — see
99
+ // lib/identity/disclosure-scrub.mjs.
100
+ import { MESSAGE_CRAFT, SELF_PRESENTATION, renderSeatPersona } from "../../lib/identity/persona.mjs";
101
+ import { scrubSeatText } from "../../lib/identity/disclosure-scrub.mjs";
102
+ // The CLAUDE.md preamble helpers — one definition, shared with prompt-builder.
103
+ import {
104
+ scrapeClaudeMdSections as _scrape,
105
+ stripScaffoldSentinels as _stripSentinels,
106
+ preambleTargets,
107
+ } from "../../lib/identity/claude-md.mjs";
108
+ // A seat CLAUDE.md line ordering a self-introduction reaches this prompt and the
109
+ // repo cannot see it. The check therefore runs at RUNTIME over the composed
110
+ // preamble — see lib/identity/disclosure-instructions.mjs.
111
+ import { stripDisclosureInstructionsAndWarn } from "../../lib/identity/disclosure-instructions.mjs";
112
+ // The conversation frame. A reply into a live thread used to arrive as a bare
113
+ // transcript with no statement that the agent is IN it — see
114
+ // lib/org/inbound/conversation-frame.mjs for the rendered prompt that proved it.
115
+ import {
116
+ conversationFrame,
117
+ dedupeThreadAgainstHistory,
118
+ frameRecipientClass,
119
+ markOwnTurns,
120
+ } from "../../lib/org/inbound/conversation-frame.mjs";
121
+ // Internal vs external, by the SAME predicate the send gate uses, so the frame
122
+ // and the gate can never disagree about who the agent is talking to.
123
+ import { classifyRecipient } from "../../lib/comms/send-gate.mjs";
80
124
  // A human's roll-call to the whole room. The reply-shaping block (≤ 5 lines,
81
125
  // about myself only, no @mentions, no ack) lives with the vocabulary in
82
126
  // broadcast.mjs so the session path (prompt-builder) says the same thing this
@@ -561,28 +605,95 @@ function logInteraction(item, responseText, o = {}) {
561
605
  // Identity preamble (cached)
562
606
  // ---------------------------------------------------------------------------
563
607
 
608
+ // The CLAUDE.md scrape and the scaffold-sentinel strip live in
609
+ // lib/identity/claude-md.mjs. They used to be defined HERE, which is how the
610
+ // full-session plane (prompt-builder.mjs) kept its own unfixed copy of the same
611
+ // scrape: one plane got the fix, the other did not, and the other is the tier a
612
+ // substantive reply escalates to. Re-exported for the tests that render this
613
+ // plane, so there is one definition and two callers.
614
+ export { scrapeClaudeMdSections, stripScaffoldSentinels } from "../../lib/identity/claude-md.mjs";
615
+
616
+ /**
617
+ * The quick reply's identity + house-style preamble.
618
+ *
619
+ * IDENTITY COMES FROM config/agent.json, NOT FROM CLAUDE.md.
620
+ *
621
+ * This used to scrape `## Identity` out of the seat's CLAUDE.md and hand it to
622
+ * the model as-is. On every seat in the fleet that section is the scaffold
623
+ * template — `You are ` + "`{{agent.fullName}}`, `{{agent.title}}` at" +
624
+ * " `{{agent.company}}`…" — followed by "If those tokens are still unresolved,
625
+ * your identity has not been configured yet — run `maestro setup`". Nothing
626
+ * substitutes those tokens on this path: the responder is a `claude --print`
627
+ * call that is never shown config/agent.json. So EVERY reactive reply the fleet
628
+ * sent opened with an identity block that named nobody and then told the model
629
+ * its identity was unconfigured — which lib/identity/persona.mjs documents, in
630
+ * its own header, as the exact condition that makes an agent introduce and sign
631
+ * itself with a rendering of the policy's identity_line.
632
+ *
633
+ * `renderSeatPersona` is the renderer the FULL-SESSION path has used since that
634
+ * header was written. It resolves the real name, title, company, standing and
635
+ * reporting line from config/agent.json, and it carries `voiceRules()` —
636
+ * including "Do not introduce, describe or sign yourself as an assistant, a bot,
637
+ * an AI helper, or as 'working on behalf of' someone" AND the untouched
638
+ * truthfulness invariant. Using it here is what puts the two planes on one
639
+ * identity instead of one resolved and one template.
640
+ *
641
+ * The CLAUDE.md scrape is kept for the house-style sections the persona block
642
+ * does not cover, with `## Identity` DROPPED whenever the persona rendered —
643
+ * two identity blocks, one of them unresolved, is worse than either alone. It is
644
+ * RESTORED when nothing rendered, because a prompt that says "run maestro setup"
645
+ * is the honest rendering of a seat nobody has set up
646
+ * (lib/identity/claude-md.mjs#preambleTargets; pinned by
647
+ * responder-unconfigured-seat.test.mjs).
648
+ *
649
+ * AND THE SCRAPE IS FILTERED. It is seat-authored prose, and a seat file saying
650
+ * "On first message in any thread, introduce yourself as <the identity_line>."
651
+ * reaches this prompt verbatim, survives every upgrade,
652
+ * and defeats everything above — while being invisible from this repo, so no
653
+ * test that renders the scaffold can catch it. The check therefore runs at
654
+ * RUNTIME on the text about to go to the model, and warns once per line naming
655
+ * the file to edit (lib/identity/disclosure-instructions.mjs). It is narrow on
656
+ * purpose and keeps a line that FORBIDS the behaviour; the truthfulness
657
+ * invariant is untouched.
658
+ */
564
659
  function loadPreamble() {
565
660
  if (cachedPreamble) return cachedPreamble;
661
+ const persona = renderSeatPersona(AGENT_REPO_DIR);
662
+ const claudeMd = join(AGENT_REPO_DIR, "CLAUDE.md");
663
+ let scraped = "";
566
664
  try {
567
- const raw = readFileSync(join(AGENT_REPO_DIR, "CLAUDE.md"), "utf-8");
568
- // Extract just the identity and communication rules
569
- const lines = raw.split("\n");
570
- const sections = [];
571
- let capturing = false;
572
- const targets = ["## Identity", "## Company Context", "## Communication Rules"];
573
- for (const line of lines) {
574
- if (targets.some((h) => line.startsWith(h))) { capturing = true; sections.push(line); continue; }
575
- if (capturing && /^## [A-Z]/.test(line) && !targets.some((h) => line.startsWith(h))) { capturing = false; continue; }
576
- if (capturing) sections.push(line);
577
- }
578
- cachedPreamble = sections.join("\n").trim();
579
- if (cachedPreamble.length < 100) cachedPreamble = FALLBACK_PREAMBLE;
665
+ const raw = readFileSync(claudeMd, "utf-8");
666
+ const targets = preambleTargets(Boolean(persona), ["## Company Context", "## Communication Rules"]);
667
+ scraped = _stripSentinels(_scrape(raw, targets));
580
668
  } catch {
581
- cachedPreamble = FALLBACK_PREAMBLE;
669
+ scraped = "";
582
670
  }
671
+ // The seat file is the one input this repo cannot see. A line in it ordering a
672
+ // self-introduction survives every upgrade and defeats everything above, so it
673
+ // is dropped here, on the text about to go to the model, and warned about.
674
+ scraped = stripDisclosureInstructionsAndWarn(scraped, claudeMd).trim();
675
+ // Scrubbed as ONE string, and deliberately including the persona: the persona
676
+ // is rendered from config/agent.json, which is seat-local too, so a
677
+ // disclosure sentence stuffed into `persona`/`background`/`bio` renders into
678
+ // this prompt exactly as a CLAUDE.md line would. Scrubbing the composed text
679
+ // covers both seat inputs with one pass — and it happens BEFORE the length
680
+ // floor below, so a seat whose identity section is mostly a disclosure
681
+ // instruction falls through to the framework fallback rather than shipping a
682
+ // two-line stump. The truthfulness invariant is exempt by construction; see
683
+ // lib/identity/disclosure-scrub.mjs.
684
+ const composed = scrubSeatText([persona, scraped].filter(Boolean).join("\n\n").trim());
685
+ // The floor applies to the WHOLE preamble, not to the scrape alone: a seat
686
+ // with a resolved persona and an empty CLAUDE.md is configured, and must not
687
+ // be thrown back to the fallback that knows none of it.
688
+ cachedPreamble = composed.length < 100 ? FALLBACK_PREAMBLE : composed;
583
689
  return cachedPreamble;
584
690
  }
585
691
 
692
+ /** For tests: drop the cached preamble so a fresh seat dir is re-read. */
693
+ export function _resetPreambleCache() {
694
+ cachedPreamble = null;
695
+ }
696
+
586
697
  function buildFallbackPreamble() {
587
698
  const a = loadAgent();
588
699
  const principalName = a.principal?.fullName || "the principal";
@@ -597,11 +708,34 @@ const FALLBACK_PREAMBLE = buildFallbackPreamble();
597
708
  // Load user profile for context
598
709
  // ---------------------------------------------------------------------------
599
710
 
711
+ /**
712
+ * The seat's own notes about a sender, injected into the SYSTEM prompt.
713
+ *
714
+ * Filtered like the CLAUDE.md scrape, and for the same reason: this is
715
+ * seat-authored prose in an instruction position, written on the seat's own
716
+ * disk where the repo cannot see it. A profile saying "introduce yourself as an
717
+ * AI assistant when writing to this person" would defeat every other half of
718
+ * the fix for exactly one correspondent — the hardest version of the bug to
719
+ * reproduce. See lib/identity/disclosure-instructions.mjs.
720
+ *
721
+ * @param {string} sender
722
+ * @returns {string|null}
723
+ */
600
724
  function loadUserProfile(sender) {
601
725
  try {
602
726
  const profileName = sender.replace(/\s+/g, "-").toLowerCase();
603
727
  const path = join(AGENT_REPO_DIR, "memory", "profiles", "users", `${profileName}.yaml`);
604
- return readFileSync(path, "utf-8");
728
+ // Both filters, for the same reason the CLAUDE.md preamble gets both: these
729
+ // files are seat-local, written by the agent itself over months, and land in
730
+ // the system prompt under a "Sender profile:" heading — i.e. as instruction,
731
+ // not as quoted data. `scrubSeatText` removes a rendered identity TEMPLATE;
732
+ // `stripDisclosureInstructionsAndWarn` removes an INSTRUCTION to open with
733
+ // one and names the file on stderr. A profile that recorded "always
734
+ // introduces itself to this person as an AI assistant" would otherwise
735
+ // re-arm the behaviour for that one sender — the hardest version of this
736
+ // bug to reproduce — no matter what the framework rules say.
737
+ const raw = readFileSync(path, "utf-8");
738
+ return stripDisclosureInstructionsAndWarn(scrubSeatText(raw), path) || null;
605
739
  } catch {
606
740
  return null;
607
741
  }
@@ -844,23 +978,65 @@ export async function loadCurrentWork(item, deps = {}) {
844
978
  * persona needs: describe it as your own work, and never quote a session name
845
979
  * or id (they are internal — `session_status` says the same).
846
980
  */
847
- export function buildQuickReplyUserContent({ item = {}, classResult = {}, conversationHistory = null, currentWork = "", rollCall = false } = {}) {
981
+ export function buildQuickReplyUserContent({
982
+ item = {},
983
+ classResult = {},
984
+ conversationHistory = null,
985
+ currentWork = "",
986
+ rollCall = false,
987
+ agentName = "",
988
+ recipientClass = "internal",
989
+ } = {}) {
990
+ // A roll-call's thread context is the room's other answers — not evidence
991
+ // about this seat, and exactly what the reply must not summarise.
992
+ const rawThread = !rollCall && item.thread_context ? String(item.thread_context) : "";
993
+ // The two transcripts overlapped almost completely on a mid-thread channel
994
+ // reply: the same turns arrived twice, worth up to COHORT_HISTORY_LIMIT ×
995
+ // COHORT_HISTORY_CHARS, and read as two different records of one exchange.
996
+ const thread = dedupeThreadAgainstHistory(conversationHistory, rawThread);
997
+ const history = markOwnTurns(conversationHistory || "", agentName);
998
+ const threadMarked = markOwnTurns(thread, agentName);
999
+
1000
+ // The frame is what tells the model it is CONTINUING something. Spent only
1001
+ // when there is a conversation to frame, and paid for by the dedupe above.
1002
+ const frame = conversationFrame({
1003
+ item,
1004
+ agentName,
1005
+ recipientClass,
1006
+ history,
1007
+ threadContext: threadMarked,
1008
+ });
1009
+
848
1010
  return [
1011
+ frame || null,
1012
+ frame ? "" : null,
849
1013
  `From: ${item.sender} (${item.sender_privilege || "unknown"})`,
850
1014
  `Via: ${item.service} / ${item.channel}`,
851
1015
  item.subject ? `Subject: ${item.subject}` : null,
852
- conversationHistory ? `\nRecent conversation history:\n${conversationHistory}` : null,
1016
+ history ? `\nRecent conversation history:\n${history}` : null,
853
1017
  currentWork
854
1018
  ? `\n${currentWork}\n(If asked what you are doing, answer from the lines above — as your own work, in your own words, board item ids included where a person could look one up. Never quote session names or ids.)`
855
1019
  : null,
856
1020
  `\nCurrent message:\n${item.content || "(empty)"}`,
857
- // A roll-call's thread context is the room's other answers — not evidence
858
- // about this seat, and exactly what the reply must not summarise.
859
- !rollCall && item.thread_context ? `\nThread context:\n${item.thread_context}` : null,
1021
+ threadMarked ? `\nEarlier in this thread:\n${threadMarked}` : null,
860
1022
  `\nClassification: ${classResult.summary}`,
861
- ].filter(Boolean).join("\n");
1023
+ ].filter((l) => l !== null).join("\n");
862
1024
  }
863
1025
 
1026
+ /**
1027
+ * Says, in the prompt, what the assembly order already implies: everything above
1028
+ * this point on this plane is SEAT-LOCAL (the seat's CLAUDE.md sections and the
1029
+ * per-sender profile), and the framework rules below outrank it.
1030
+ *
1031
+ * lib/identity/disclosure-scrub already removes the one seat-local instruction
1032
+ * known to have caused harm. This sentence covers the variants the scrub's shape
1033
+ * rule does not recognise: a model that is handed two conflicting instructions
1034
+ * and no ordering between them resolves the conflict by position, and the seat's
1035
+ * copy is first.
1036
+ */
1037
+ const SEAT_TEXT_PRECEDENCE =
1038
+ "The sections above are this seat's own local notes. The rules that follow are the framework's, they apply to every member of this organisation, and where the two conflict the rules below win.";
1039
+
864
1040
  /**
865
1041
  * The real, CLI-backed answer generator. `deps` are test seams only —
866
1042
  * `{runCLI, currentWorkBlock, forbidsTextFor, loadConversationHistory}` — so
@@ -877,14 +1053,19 @@ export async function realGenerateResponse(item, classResult, deps = {}) {
877
1053
 
878
1054
  const systemPrompt = `${preamble}
879
1055
 
880
- You are generating a direct response to this message. Be concise and actionable.
1056
+ You are writing the next turn in a conversation you are already part of. Be concise and actionable.
881
1057
  If it's a question, answer it. If it's a request, confirm and describe what you'll do or have done.
882
1058
  If it's informational, acknowledge appropriately.
1059
+ Open with the substance. Do not greet, do not introduce or describe yourself, and do not restate your role, your reporting line or what you are — the people you are writing to already know, and none of them asked.
883
1060
 
884
1061
  Keep responses focused — 1-4 sentences for simple items, up to a short paragraph for more nuanced ones.
885
1062
  Match the sender's tone and urgency level.
886
1063
  ${profile ? `\nSender profile:\n${profile}` : ""}
887
1064
  ${rollCall ? `\n${collectiveInstructions({ intent: itemCollectiveIntent(item, isRollCallItem), respondents: item.respondents })}\n` : ""}
1065
+ ${SEAT_TEXT_PRECEDENCE}
1066
+
1067
+ ${SELF_PRESENTATION}
1068
+
888
1069
  ${MESSAGE_CRAFT}`;
889
1070
 
890
1071
  // A roll-call is answered from what this seat KNOWS about itself — its main
@@ -897,7 +1078,26 @@ ${MESSAGE_CRAFT}`;
897
1078
  const conversationHistory = rollCall ? null : await (deps.loadConversationHistory || loadConversationHistory)(item);
898
1079
  const currentWork = await loadCurrentWork(item, deps);
899
1080
 
900
- const userContent = buildQuickReplyUserContent({ item, classResult, conversationHistory, currentWork, rollCall });
1081
+ // The seat's own name, so its prior turns in the transcript can be marked as
1082
+ // ITS OWN. Without it the model reads lines it authored as a third party's,
1083
+ // which is a conversation it has not joined — and a model joining a
1084
+ // conversation cold introduces itself.
1085
+ const agentName = (() => {
1086
+ const a = loadAgent();
1087
+ const n = (a.fullName || "").trim();
1088
+ if (n && !/^unconfigured\b/i.test(n)) return n;
1089
+ const f = (a.firstName || "").trim();
1090
+ return f && !/^unconfigured\b/i.test(f) ? f : "";
1091
+ })();
1092
+ // Origin first, then the send gate's own predicate — see
1093
+ // conversation-frame.mjs#frameRecipientClass for why a bare
1094
+ // `classifyRecipient(service, channel_id || channel)` put colleagues in the
1095
+ // agent's own #risk-compliance channel on the "external" branch, silently.
1096
+ const recipientClass = frameRecipientClass(item, classifyRecipient);
1097
+
1098
+ const userContent = buildQuickReplyUserContent({
1099
+ item, classResult, conversationHistory, currentWork, rollCall, agentName, recipientClass,
1100
+ });
901
1101
 
902
1102
  // (b2) Session-router decision. Compute the routing key from a daemon→router
903
1103
  // adapter view of the item. If the item can't be keyed (unknown service,
@@ -1,13 +1,42 @@
1
1
  #!/bin/bash
2
2
  # Emergency Stop — Immediately halts all Maestro agent operations.
3
- # Usage: ./scripts/emergency-stop.sh
3
+ # Usage: ./scripts/emergency-stop.sh [--dry-run] [--help]
4
4
  #
5
5
  # This script is the kill switch for all autonomous operations:
6
6
  # 1. Drops .emergency-stop flag (every workflow / cadence consumer / enqueue
7
7
  # script honours this on the next tick).
8
8
  # 2. Unloads every installed `ai.maestro.<agent>-*` (and legacy
9
9
  # `ai.adaptic.<agent>-*`) launchd job.
10
- # 3. Kills running Claude Code subagent processes.
10
+ # 3. Kills running Claude Code subagent processes (EXCEPT this process and
11
+ # its ancestors — see step 3).
12
+ #
13
+ # --dry-run
14
+ # Report the kill set instead of signalling it. The flag and the launchd
15
+ # unload still happen; only the signals are withheld — and BOTH the closing
16
+ # banner and the log line say "DRY RUN — NO PROCESSES SIGNALLED", so a dry
17
+ # run can never be mistaken for a halt. It exists so the self-sparing in
18
+ # step 3 is testable without a suite that kills the operator's own session
19
+ # to prove that it does not.
20
+ #
21
+ # WHY A FLAG AND NOT AN ENV VAR. This used to be read from the ambient
22
+ # environment as MAESTRO_EMERGENCY_STOP_DRY_RUN, while step 4 printed
23
+ # "EMERGENCY STOP COMPLETE — All operations halted" unconditionally. A stray
24
+ # `export`, a line in .env, or an EnvironmentVariables entry in a plist
25
+ # would therefore turn the kill switch into a no-op — every claude process
26
+ # surviving — while the banner and the log both asserted the halt had
27
+ # succeeded. That is the same fault class this file was being repaired for:
28
+ # something that looks like it is handling the case and is not. On a
29
+ # break-glass control the escape hatch must be typed at the call site, once,
30
+ # deliberately. The env var is NOT consulted; setting it does nothing.
31
+ #
32
+ # SCOPE OF THE KILL (step 3), stated rather than hidden: every process of THIS
33
+ # user whose command line matches `claude`, minus this process and its
34
+ # ancestors. That is machine-wide, not agent-scoped — an unrelated Claude
35
+ # session of yours in another directory WILL be terminated (21 processes
36
+ # matched on the host where this was measured). Nothing here can narrow it
37
+ # honestly, because a Claude Code session's argv does not carry the agent dir;
38
+ # so the set is printed before it is signalled, and --dry-run shows it without
39
+ # signalling anything.
11
40
  #
12
41
  # Plist resolution: the agent's first-name slug is read from config/agent.json
13
42
  # (SOT) so unload targets the correct labels; falls back to the directory
@@ -21,6 +50,25 @@ LOG_FILE="$AGENT_DIR/logs/emergency-stop.log"
21
50
  TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
22
51
  mkdir -p "$(dirname "$LOG_FILE")" 2>/dev/null || true
23
52
 
53
+ # Argument parsing runs BEFORE anything is halted: a typo must refuse loudly,
54
+ # not drop the flag and then exit.
55
+ DRY_RUN=0
56
+ while [ $# -gt 0 ]; do
57
+ case "$1" in
58
+ --dry-run) DRY_RUN=1 ;;
59
+ -h | --help)
60
+ sed -n '2,/^$/p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
61
+ exit 0
62
+ ;;
63
+ *)
64
+ echo "emergency-stop: unknown argument: $1" >&2
65
+ echo "Usage: emergency-stop.sh [--dry-run]" >&2
66
+ exit 2
67
+ ;;
68
+ esac
69
+ shift
70
+ done
71
+
24
72
  # Resolve agent first-name slug from SOT (config/agent.json) so the unload
25
73
  # loop targets the right launchd labels. Falls back to the basename of the
26
74
  # agent directory (stripping -ai suffix).
@@ -39,7 +87,11 @@ LAUNCH_AGENTS_DIR="$HOME/Library/LaunchAgents"
39
87
  # (deployed agents whose plists predate the rename). The loops guard with [ -f ].
40
88
  PLIST_GLOB="$LAUNCH_AGENTS_DIR/ai.maestro.${AGENT_FIRST}-*.plist $LAUNCH_AGENTS_DIR/ai.adaptic.${AGENT_FIRST}-*.plist"
41
89
 
42
- echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
90
+ if [ "$DRY_RUN" -eq 1 ]; then
91
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN INITIATED (agent=$AGENT_FIRST) — no process will be signalled" | tee -a "$LOG_FILE"
92
+ else
93
+ echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
94
+ fi
43
95
 
44
96
  # 1. Drop the stop flag FIRST so any in-flight work sees it on next tick.
45
97
  echo "$TIMESTAMP" > "$AGENT_DIR/.emergency-stop"
@@ -57,23 +109,72 @@ for plist in $PLIST_GLOB; do
57
109
  done
58
110
  echo "[$TIMESTAMP] Unloaded $unloaded launchd job(s)" >> "$LOG_FILE"
59
111
 
60
- # 3. Kill running Claude Code subagent processes. Filter to processes that
61
- # have AGENT_ROOT or this agent's directory in their cwd to avoid
62
- # killing unrelated claude sessions the operator may have running.
112
+ # 3. Kill running Claude Code subagent processes.
113
+ #
114
+ # THE MIRROR-IMAGE FAULT THIS FIXES: `pgrep -f claude` matches the process
115
+ # that is RUNNING THIS SCRIPT whenever an agent session invokes it — which
116
+ # is the normal way it gets invoked. The stop then killed its own caller
117
+ # mid-flight, so step 4 never ran and the halt was never logged complete:
118
+ # a script that destroys the condition it needs in order to finish, the
119
+ # same shape as resume-operations.sh refusing to lift its own stop flag.
120
+ # The comment here also claimed a cwd filter that did not exist.
121
+ #
122
+ # So: build the kill set, then subtract this process and every ancestor of
123
+ # it. Everything else matching `claude` is still terminated — the stop is
124
+ # still a kill switch, it just no longer includes the hand on the switch.
125
+ # See SCOPE OF THE KILL in the header: "everything else" really does mean
126
+ # every matching process of this user on this machine, so it is printed.
63
127
  echo "Stopping Claude Code agent processes..."
64
- # pgrep -lf is more selective than pkill -f
65
- pids=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
66
- if [ -n "$pids" ]; then
128
+
129
+ # PIDs to spare: this shell and its whole ancestor chain (the launching
130
+ # session, its shell, launchd). Walk up via ppid until PID 1.
131
+ SPARE=" $$ "
132
+ _p=$$
133
+ while [ -n "$_p" ] && [ "$_p" -gt 1 ]; do
134
+ _p=$(ps -o ppid= -p "$_p" 2>/dev/null | tr -d ' ')
135
+ [ -n "$_p" ] || break
136
+ SPARE="$SPARE$_p "
137
+ done
138
+
139
+ # claude_targets — matching PIDs (this user only) minus the spare set.
140
+ claude_targets() {
141
+ local out=""
142
+ local pid
143
+ for pid in $(pgrep -u "$(id -u)" -f "claude" 2>/dev/null || true); do
144
+ case "$SPARE" in
145
+ *" $pid "*) continue ;;
146
+ esac
147
+ out="$out$pid "
148
+ done
149
+ printf '%s' "$out"
150
+ }
151
+
152
+ pids=$(claude_targets)
153
+ if [ "$DRY_RUN" -eq 1 ]; then
154
+ echo "[DRY-RUN] sparing:$SPARE"
155
+ echo "[DRY-RUN] would terminate: ${pids:-<none>}"
156
+ echo "[$TIMESTAMP] DRY RUN — would terminate: ${pids:-<none>}" >> "$LOG_FILE"
157
+ elif [ -n "$pids" ]; then
158
+ # Name the set before signalling it — this is machine-wide for this user.
159
+ echo "Terminating (every matching process of this user, not just this agent): $pids"
67
160
  # Send SIGTERM first, give 3s, then SIGKILL stragglers.
68
161
  kill -TERM $pids 2>/dev/null || true
69
162
  sleep 3
70
- still=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
71
- [ -n "$still" ] && kill -KILL $still 2>/dev/null || true
163
+ still=$(claude_targets)
164
+ if [ -n "$still" ]; then kill -KILL $still 2>/dev/null || true; fi
72
165
  echo "[$TIMESTAMP] Claude processes terminated ($pids)" >> "$LOG_FILE"
166
+ else
167
+ echo "[$TIMESTAMP] No Claude processes to terminate (self/ancestors spared)" >> "$LOG_FILE"
73
168
  fi
74
169
 
75
- # 4. Log completion.
76
- echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
170
+ # 4. Log completion. A dry run says so in BOTH places — the banner an operator
171
+ # reads and the log an incident review reads — because the previous version
172
+ # printed the halt banner either way.
173
+ if [ "$DRY_RUN" -eq 1 ]; then
174
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN — NO PROCESSES SIGNALLED (flag set, launchd unloaded)" | tee -a "$LOG_FILE"
175
+ else
176
+ echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
177
+ fi
77
178
  echo ""
78
179
  echo "To resume operations:"
79
180
  echo " ./scripts/resume-operations.sh"
@@ -262,13 +262,20 @@ export function gateLine(r) {
262
262
  }
263
263
 
264
264
  /**
265
- * A note about paths the GATES themselves dirtied. `npm test` appends to a
266
- * tracked runtime ledger (.claude-flow/policy/state.json), so a second run in a
267
- * row would refuse on a file the first run wrote. The clean-tree gate is NOT
265
+ * A note about paths the GATES themselves dirtied: a second run in a row would
266
+ * otherwise refuse on a file the FIRST run wrote. The clean-tree gate is NOT
268
267
  * weakened for this — publishing a tree you cannot describe stays a refusal —
269
268
  * the run just says which paths are the gates' own leavings so the operator
270
269
  * reverts them instead of hunting them. Pure.
271
270
  *
271
+ * ~~"`npm test` appends to a tracked runtime ledger
272
+ * (.claude-flow/policy/state.json)"~~ — struck 2026-09-25: that file is the
273
+ * reason this function exists, and it is no longer tracked (it is a per-machine
274
+ * receipt chain, so it never should have been; `.gitignore` says why). This
275
+ * function stays because the SHAPE recurs — the next gate that writes into the
276
+ * tree will do the same thing — and because naming the paths beats hunting
277
+ * them. When it returns null for a whole rollout, that is the expected reading.
278
+ *
272
279
  * @param {string[]} before dirty paths before the gates ran
273
280
  * @param {string[]} after dirty paths after
274
281
  * @returns {string|null}