@cohortapp/agent-sdk 2.18.13 → 2.18.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/maestro.mjs +38 -1
- package/docs/runbooks/fleet-rollout.md +58 -7
- package/docs/runbooks/recovery-and-failover.md +18 -0
- package/lib/assurance/batch.mjs +353 -0
- package/lib/assurance/first-reply.mjs +423 -0
- package/lib/assurance/notice-voice.mjs +357 -0
- package/lib/assurance/plan-note.mjs +43 -0
- package/lib/assurance/room-budget.mjs +55 -6
- package/lib/cadence-failure-class.mjs +245 -0
- package/lib/claude-bin.mjs +26 -7
- package/lib/cli/doctor-checks.mjs +149 -1
- package/lib/comms/send-gate.mjs +59 -0
- package/lib/diagnostics/alerts.mjs +33 -0
- package/lib/engine/agents/usage.mjs +45 -0
- package/lib/engine/budget.mjs +293 -29
- package/lib/engine/cli.mjs +54 -5
- package/lib/engine/loop.mjs +30 -0
- package/lib/engine/output/json.mjs +26 -0
- package/lib/engine/wire/errors.mjs +179 -0
- package/lib/engine/wire/search.mjs +44 -8
- package/lib/identity/persona.mjs +31 -2
- package/lib/org/quota.mjs +27 -0
- package/lib/session/config.mjs +4 -0
- package/lib/session/identity.mjs +71 -7
- package/lib/session/launch-failure.mjs +251 -0
- package/lib/session/resume-target.mjs +86 -0
- package/lib/telemetry/alerts.mjs +94 -0
- package/lib/telemetry/collect.mjs +155 -2
- package/lib/upgrade/pinned-drift.mjs +467 -0
- package/package.json +1 -1
- package/scaffold/config/alerts.yaml +7 -0
- package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
- package/scripts/ci/check.mjs +3 -0
- package/scripts/daemon/agent-daemon.mjs +75 -5
- package/scripts/daemon/assurance.mjs +709 -44
- package/scripts/daemon/cadence-consumer.mjs +281 -34
- package/scripts/daemon/deliver.mjs +109 -0
- package/scripts/daemon/dispatcher.mjs +21 -3
- package/scripts/daemon/inbox-deferral.mjs +102 -9
- package/scripts/daemon/session-lock.mjs +41 -1
- package/scripts/emergency-stop.sh +114 -13
- package/scripts/fleet/rollout.mjs +256 -10
- package/scripts/healthcheck.sh +131 -33
- package/scripts/local-triggers/autoupdate.sh +144 -11
- package/scripts/resume-operations.sh +101 -6
- package/scripts/session/supervisor.mjs +198 -5
|
@@ -18,11 +18,26 @@
|
|
|
18
18
|
* every `.deferred` file in any service inbox dir whose body
|
|
19
19
|
* references `channel`. If exactly one is found, renames it back
|
|
20
20
|
* to its original name (next poll picks it up). If N>1 are found,
|
|
21
|
-
* keeps only the LATEST (by timestamp)
|
|
22
|
-
* marks the others `.processed-bundled` for
|
|
23
|
-
* the
|
|
24
|
-
*
|
|
25
|
-
*
|
|
21
|
+
* keeps only the LATEST (by timestamp) of each BUNDLING GROUP,
|
|
22
|
+
* promotes those, and marks the others `.processed-bundled` for
|
|
23
|
+
* the audit trail.
|
|
24
|
+
*
|
|
25
|
+
* WHAT A BUNDLING GROUP IS, AND WHY IT IS NOT JUST "THE SURFACE".
|
|
26
|
+
* The collapse is only sound when the promoted item genuinely
|
|
27
|
+
* carries what is being folded away. That used to be asserted
|
|
28
|
+
* unconditionally — "the latest item's `thread_context` already
|
|
29
|
+
* contains the prior messages" — and it is true of a Slack
|
|
30
|
+
* channel or DM item and FALSE of a Cohort space post, which
|
|
31
|
+
* `lib/org/inbound` never hydrates a thread context for. Fold a
|
|
32
|
+
* second person's question into one of those and it is not
|
|
33
|
+
* bundled, it is deleted: marked `.processed-bundled` and never
|
|
34
|
+
* answered by anyone.
|
|
35
|
+
*
|
|
36
|
+
* So a group is (surface, sender) — a person's own flurry always
|
|
37
|
+
* collapses to their latest word — UNLESS the surface's newest
|
|
38
|
+
* item carries `thread_context`, in which case the whole surface
|
|
39
|
+
* collapses as before, because the session really will see every
|
|
40
|
+
* message it stands for.
|
|
26
41
|
*
|
|
27
42
|
* The file format is the YAML emitted by the slack/gmail/calendar
|
|
28
43
|
* pollers (a flat top-level object with quoted scalar fields, plus an
|
|
@@ -155,6 +170,42 @@ function readScalar(body, field) {
|
|
|
155
170
|
return null;
|
|
156
171
|
}
|
|
157
172
|
|
|
173
|
+
/**
|
|
174
|
+
* Does this file carry a TOP-LEVEL key at all, block scalar or not?
|
|
175
|
+
*
|
|
176
|
+
* `readScalar` answers "what is this field's single-line value", which is the
|
|
177
|
+
* wrong question for `thread_context`: the pollers write it as a block scalar
|
|
178
|
+
* (`thread_context: |`), so its value never lives on the key's own line. All
|
|
179
|
+
* the bundling decision needs to know is whether the field is PRESENT — the
|
|
180
|
+
* pollers emit it only when there is history to emit (`if (item.thread_context)`
|
|
181
|
+
* in scripts/poller/utils.mjs), so presence is exactly "this item carries the
|
|
182
|
+
* conversation around it".
|
|
183
|
+
*
|
|
184
|
+
* @param {string} body
|
|
185
|
+
* @param {string} field
|
|
186
|
+
* @returns {boolean}
|
|
187
|
+
*/
|
|
188
|
+
function hasTopLevelKey(body, field) {
|
|
189
|
+
if (typeof body !== "string") return false;
|
|
190
|
+
// TWO MECHANISMS, ONE PROPERTY, and the redundancy is stated rather than
|
|
191
|
+
// pretended away: the pattern is anchored at column 0, so an indented line
|
|
192
|
+
// cannot match it and the `continue` below is a BELT, not the mechanism.
|
|
193
|
+
// Deleting the `continue` alone changes no behaviour — a mutation of it
|
|
194
|
+
// leaves the suite green, and that is honest rather than a hole. What the
|
|
195
|
+
// suite does refuse is the realistic future edit: loosening this anchor to
|
|
196
|
+
// `^\s*` (the shape `readScalar` had before audit L7) with nothing else in
|
|
197
|
+
// the way. That is pinned as a BEHAVIOUR, in inbox-deferral.test.mjs — "L7:
|
|
198
|
+
// body line `thread_context:` in a block scalar does not fake carrying
|
|
199
|
+
// history" — because the property worth pinning is that a colleague's
|
|
200
|
+
// question survives, not which of two lines delivered it.
|
|
201
|
+
const re = new RegExp(`^${escapeRegExp(field)}\\s*:`);
|
|
202
|
+
for (const line of body.split("\n")) {
|
|
203
|
+
if (/^\s/.test(line)) continue; // indented — belongs to a block, not top level
|
|
204
|
+
if (re.test(line)) return true;
|
|
205
|
+
}
|
|
206
|
+
return false;
|
|
207
|
+
}
|
|
208
|
+
|
|
158
209
|
/**
|
|
159
210
|
* Promote every `.deferred` item targeting `channel` back into the
|
|
160
211
|
* live inbox. Bundles bursts: if multiple deferred items exist for
|
|
@@ -219,6 +270,26 @@ export function promoteDeferred(channel, agentRoot) {
|
|
|
219
270
|
readScalar(body, "thread_id") ||
|
|
220
271
|
readScalar(body, "scope_id") ||
|
|
221
272
|
"",
|
|
273
|
+
// WHO SENT IT, and WHETHER THE ITEM CARRIES ITS CONVERSATION.
|
|
274
|
+
//
|
|
275
|
+
// These two decide whether bundling is a bundle or a deletion. The
|
|
276
|
+
// latest-wins collapse is justified by one sentence in this module's
|
|
277
|
+
// header — "the latest item's thread_context already contains the
|
|
278
|
+
// prior messages as conversation history, so Claude sees everything" —
|
|
279
|
+
// and that sentence is TRUE OF SOME ITEMS AND FALSE OF OTHERS. Slack
|
|
280
|
+
// channel and DM items carry `thread_context` (slack-poller.mjs:434,
|
|
281
|
+
// :558); a Cohort space post does NOT, because `lib/org/inbound`
|
|
282
|
+
// never hydrates one for an untreaded space message. Bundling a
|
|
283
|
+
// second person's ask into a Cohort item that carries no history is
|
|
284
|
+
// not folding a duplicate away, it is marking their question
|
|
285
|
+
// `.processed-bundled` and never answering it.
|
|
286
|
+
//
|
|
287
|
+
// So: the collapse now requires evidence. Same sender is always safe
|
|
288
|
+
// (a person's own flurry — their newest message is their latest word
|
|
289
|
+
// either way). A DIFFERENT sender may only be folded into an item that
|
|
290
|
+
// actually carries the room's history.
|
|
291
|
+
sender: (readScalar(body, "sender") || "").trim().toLowerCase(),
|
|
292
|
+
carriesHistory: hasTopLevelKey(body, "thread_context"),
|
|
222
293
|
});
|
|
223
294
|
}
|
|
224
295
|
if (matches.length === 0) continue;
|
|
@@ -226,13 +297,35 @@ export function promoteDeferred(channel, agentRoot) {
|
|
|
226
297
|
// Group by surface within the channel, then bundle latest-wins WITHIN each
|
|
227
298
|
// group. Distinct surfaces each promote their own latest; none is dropped
|
|
228
299
|
// just because a newer message landed on a different surface in the room.
|
|
229
|
-
const
|
|
300
|
+
const surfaces = new Map();
|
|
230
301
|
for (const m of matches) {
|
|
231
|
-
if (!
|
|
232
|
-
|
|
302
|
+
if (!surfaces.has(m.surface)) surfaces.set(m.surface, []);
|
|
303
|
+
surfaces.get(m.surface).push(m);
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
// SPLIT A SURFACE BY SENDER UNLESS ITS WINNER CARRIES THE HISTORY.
|
|
307
|
+
// With history present this is byte-for-byte the old behaviour: one group
|
|
308
|
+
// per surface, latest wins, everything else bundled. Without it, each
|
|
309
|
+
// sender keeps their own latest and promotes it — so two people's asks
|
|
310
|
+
// serialise into two sessions rather than one of them disappearing.
|
|
311
|
+
const groups = [];
|
|
312
|
+
for (const group of surfaces.values()) {
|
|
313
|
+
group.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
|
|
314
|
+
const winner = group[group.length - 1];
|
|
315
|
+
const senders = new Set(group.map((m) => m.sender));
|
|
316
|
+
if (senders.size <= 1 || winner.carriesHistory) {
|
|
317
|
+
groups.push(group);
|
|
318
|
+
continue;
|
|
319
|
+
}
|
|
320
|
+
const bySender = new Map();
|
|
321
|
+
for (const m of group) {
|
|
322
|
+
if (!bySender.has(m.sender)) bySender.set(m.sender, []);
|
|
323
|
+
bySender.get(m.sender).push(m);
|
|
324
|
+
}
|
|
325
|
+
for (const g of bySender.values()) groups.push(g);
|
|
233
326
|
}
|
|
234
327
|
|
|
235
|
-
for (const group of groups
|
|
328
|
+
for (const group of groups) {
|
|
236
329
|
// Latest-wins: lex-sort ISO timestamps, take the most recent.
|
|
237
330
|
group.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
|
|
238
331
|
const latest = group[group.length - 1];
|
|
@@ -311,6 +311,37 @@ export function checkRecentlySent(channel, threadTs, type) {
|
|
|
311
311
|
return { allowed: true };
|
|
312
312
|
}
|
|
313
313
|
|
|
314
|
+
/**
|
|
315
|
+
* PURE. The ONE definition of which conversation a thread lock covers.
|
|
316
|
+
*
|
|
317
|
+
* WHY THIS EXISTS AS A FUNCTION. The normalisation used to be written twice —
|
|
318
|
+
* once where the lock is taken (agent-daemon, as an inline ternary) and once
|
|
319
|
+
* where it is released (`releaseThreadLock`, as a different inline rule) — and
|
|
320
|
+
* the two did not agree. The acquiring side turned a DM into `dm-channel` from
|
|
321
|
+
* `item.is_dm`; the releasing side only recognised a DM by a channel id
|
|
322
|
+
* starting with "D", which is a SLACK id shape. A Cohort DM therefore took a
|
|
323
|
+
* `dm-channel` lock and released nothing, and the room stayed locked for the
|
|
324
|
+
* full sixty-minute TTL with every later message deferred behind it. Two
|
|
325
|
+
* copies of a key derivation is one copy too many; this is now the only one.
|
|
326
|
+
*
|
|
327
|
+
* @param {object} o
|
|
328
|
+
* @param {string} o.channel
|
|
329
|
+
* @param {string} [o.threadId] the item's own thread id, if it has one
|
|
330
|
+
* @param {boolean} [o.isDm] the daemon's own DM verdict (item.is_dm)
|
|
331
|
+
* @param {boolean} [o.channelMainLock=true] false restores the pre-2026-09-25
|
|
332
|
+
* behaviour where an untreaded channel post took no lock at all
|
|
333
|
+
* @returns {string|null} the lock key, or null when this item takes no lock
|
|
334
|
+
*/
|
|
335
|
+
export function threadLockKey(o = {}) {
|
|
336
|
+
const channel = o.channel;
|
|
337
|
+
if (!channel) return null;
|
|
338
|
+
// A Slack DM id ("D…") is a DM whatever the caller believed.
|
|
339
|
+
if (String(channel).startsWith("D")) return "dm-channel";
|
|
340
|
+
if (o.threadId) return String(o.threadId);
|
|
341
|
+
if (o.isDm === true) return "dm-channel";
|
|
342
|
+
return o.channelMainLock === false ? null : "channel-main";
|
|
343
|
+
}
|
|
344
|
+
|
|
314
345
|
/**
|
|
315
346
|
* Check if a session was already dispatched for the same thread recently.
|
|
316
347
|
* Prevents multiple sessions from responding to the same thread when
|
|
@@ -516,7 +547,16 @@ export function releaseThreadLock(channel, threadTs) {
|
|
|
516
547
|
// the first was skipped with "thread_dedup: DM channel already
|
|
517
548
|
// dispatched 1200s ago").
|
|
518
549
|
if (channel.startsWith("D")) threadTs = "dm-channel";
|
|
519
|
-
|
|
550
|
+
// A caller that hands us nothing is asking us to guess, and the only guess
|
|
551
|
+
// that mirrors the acquiring side is `threadLockKey`'s: an untreaded, non-DM
|
|
552
|
+
// channel post locks the room's main feed. Unlinking a lock that was never
|
|
553
|
+
// taken (the off-switch case, or a genuinely lockless item) is a no-op under
|
|
554
|
+
// the catch below, so guessing here can only ever free a lock, never wedge
|
|
555
|
+
// one. The reverse — returning early, as this did — left every
|
|
556
|
+
// `channel-main` lock to expire on its sixty-minute TTL, which is the whole
|
|
557
|
+
// channel gagged for an hour after one message.
|
|
558
|
+
if (!threadTs) threadTs = threadLockKey({ channel, threadId: null, isDm: false });
|
|
559
|
+
if (!threadTs) return;
|
|
520
560
|
const safeKey = sanitiseItemId(`thread-${channel}-${threadTs}`);
|
|
521
561
|
const lockPath = join(LOCKS_DIR, `${safeKey}.lock`);
|
|
522
562
|
try {
|
|
@@ -1,13 +1,42 @@
|
|
|
1
1
|
#!/bin/bash
|
|
2
2
|
# Emergency Stop — Immediately halts all Maestro agent operations.
|
|
3
|
-
# Usage: ./scripts/emergency-stop.sh
|
|
3
|
+
# Usage: ./scripts/emergency-stop.sh [--dry-run] [--help]
|
|
4
4
|
#
|
|
5
5
|
# This script is the kill switch for all autonomous operations:
|
|
6
6
|
# 1. Drops .emergency-stop flag (every workflow / cadence consumer / enqueue
|
|
7
7
|
# script honours this on the next tick).
|
|
8
8
|
# 2. Unloads every installed `ai.maestro.<agent>-*` (and legacy
|
|
9
9
|
# `ai.adaptic.<agent>-*`) launchd job.
|
|
10
|
-
# 3. Kills running Claude Code subagent processes
|
|
10
|
+
# 3. Kills running Claude Code subagent processes (EXCEPT this process and
|
|
11
|
+
# its ancestors — see step 3).
|
|
12
|
+
#
|
|
13
|
+
# --dry-run
|
|
14
|
+
# Report the kill set instead of signalling it. The flag and the launchd
|
|
15
|
+
# unload still happen; only the signals are withheld — and BOTH the closing
|
|
16
|
+
# banner and the log line say "DRY RUN — NO PROCESSES SIGNALLED", so a dry
|
|
17
|
+
# run can never be mistaken for a halt. It exists so the self-sparing in
|
|
18
|
+
# step 3 is testable without a suite that kills the operator's own session
|
|
19
|
+
# to prove that it does not.
|
|
20
|
+
#
|
|
21
|
+
# WHY A FLAG AND NOT AN ENV VAR. This used to be read from the ambient
|
|
22
|
+
# environment as MAESTRO_EMERGENCY_STOP_DRY_RUN, while step 4 printed
|
|
23
|
+
# "EMERGENCY STOP COMPLETE — All operations halted" unconditionally. A stray
|
|
24
|
+
# `export`, a line in .env, or an EnvironmentVariables entry in a plist
|
|
25
|
+
# would therefore turn the kill switch into a no-op — every claude process
|
|
26
|
+
# surviving — while the banner and the log both asserted the halt had
|
|
27
|
+
# succeeded. That is the same fault class this file was being repaired for:
|
|
28
|
+
# something that looks like it is handling the case and is not. On a
|
|
29
|
+
# break-glass control the escape hatch must be typed at the call site, once,
|
|
30
|
+
# deliberately. The env var is NOT consulted; setting it does nothing.
|
|
31
|
+
#
|
|
32
|
+
# SCOPE OF THE KILL (step 3), stated rather than hidden: every process of THIS
|
|
33
|
+
# user whose command line matches `claude`, minus this process and its
|
|
34
|
+
# ancestors. That is machine-wide, not agent-scoped — an unrelated Claude
|
|
35
|
+
# session of yours in another directory WILL be terminated (21 processes
|
|
36
|
+
# matched on the host where this was measured). Nothing here can narrow it
|
|
37
|
+
# honestly, because a Claude Code session's argv does not carry the agent dir;
|
|
38
|
+
# so the set is printed before it is signalled, and --dry-run shows it without
|
|
39
|
+
# signalling anything.
|
|
11
40
|
#
|
|
12
41
|
# Plist resolution: the agent's first-name slug is read from config/agent.json
|
|
13
42
|
# (SOT) so unload targets the correct labels; falls back to the directory
|
|
@@ -21,6 +50,25 @@ LOG_FILE="$AGENT_DIR/logs/emergency-stop.log"
|
|
|
21
50
|
TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
|
|
22
51
|
mkdir -p "$(dirname "$LOG_FILE")" 2>/dev/null || true
|
|
23
52
|
|
|
53
|
+
# Argument parsing runs BEFORE anything is halted: a typo must refuse loudly,
|
|
54
|
+
# not drop the flag and then exit.
|
|
55
|
+
DRY_RUN=0
|
|
56
|
+
while [ $# -gt 0 ]; do
|
|
57
|
+
case "$1" in
|
|
58
|
+
--dry-run) DRY_RUN=1 ;;
|
|
59
|
+
-h | --help)
|
|
60
|
+
sed -n '2,/^$/p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
|
|
61
|
+
exit 0
|
|
62
|
+
;;
|
|
63
|
+
*)
|
|
64
|
+
echo "emergency-stop: unknown argument: $1" >&2
|
|
65
|
+
echo "Usage: emergency-stop.sh [--dry-run]" >&2
|
|
66
|
+
exit 2
|
|
67
|
+
;;
|
|
68
|
+
esac
|
|
69
|
+
shift
|
|
70
|
+
done
|
|
71
|
+
|
|
24
72
|
# Resolve agent first-name slug from SOT (config/agent.json) so the unload
|
|
25
73
|
# loop targets the right launchd labels. Falls back to the basename of the
|
|
26
74
|
# agent directory (stripping -ai suffix).
|
|
@@ -39,7 +87,11 @@ LAUNCH_AGENTS_DIR="$HOME/Library/LaunchAgents"
|
|
|
39
87
|
# (deployed agents whose plists predate the rename). The loops guard with [ -f ].
|
|
40
88
|
PLIST_GLOB="$LAUNCH_AGENTS_DIR/ai.maestro.${AGENT_FIRST}-*.plist $LAUNCH_AGENTS_DIR/ai.adaptic.${AGENT_FIRST}-*.plist"
|
|
41
89
|
|
|
42
|
-
|
|
90
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
91
|
+
echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN INITIATED (agent=$AGENT_FIRST) — no process will be signalled" | tee -a "$LOG_FILE"
|
|
92
|
+
else
|
|
93
|
+
echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
|
|
94
|
+
fi
|
|
43
95
|
|
|
44
96
|
# 1. Drop the stop flag FIRST so any in-flight work sees it on next tick.
|
|
45
97
|
echo "$TIMESTAMP" > "$AGENT_DIR/.emergency-stop"
|
|
@@ -57,23 +109,72 @@ for plist in $PLIST_GLOB; do
|
|
|
57
109
|
done
|
|
58
110
|
echo "[$TIMESTAMP] Unloaded $unloaded launchd job(s)" >> "$LOG_FILE"
|
|
59
111
|
|
|
60
|
-
# 3. Kill running Claude Code subagent processes.
|
|
61
|
-
#
|
|
62
|
-
#
|
|
112
|
+
# 3. Kill running Claude Code subagent processes.
|
|
113
|
+
#
|
|
114
|
+
# THE MIRROR-IMAGE FAULT THIS FIXES: `pgrep -f claude` matches the process
|
|
115
|
+
# that is RUNNING THIS SCRIPT whenever an agent session invokes it — which
|
|
116
|
+
# is the normal way it gets invoked. The stop then killed its own caller
|
|
117
|
+
# mid-flight, so step 4 never ran and the halt was never logged complete:
|
|
118
|
+
# a script that destroys the condition it needs in order to finish, the
|
|
119
|
+
# same shape as resume-operations.sh refusing to lift its own stop flag.
|
|
120
|
+
# The comment here also claimed a cwd filter that did not exist.
|
|
121
|
+
#
|
|
122
|
+
# So: build the kill set, then subtract this process and every ancestor of
|
|
123
|
+
# it. Everything else matching `claude` is still terminated — the stop is
|
|
124
|
+
# still a kill switch, it just no longer includes the hand on the switch.
|
|
125
|
+
# See SCOPE OF THE KILL in the header: "everything else" really does mean
|
|
126
|
+
# every matching process of this user on this machine, so it is printed.
|
|
63
127
|
echo "Stopping Claude Code agent processes..."
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
128
|
+
|
|
129
|
+
# PIDs to spare: this shell and its whole ancestor chain (the launching
|
|
130
|
+
# session, its shell, launchd). Walk up via ppid until PID 1.
|
|
131
|
+
SPARE=" $$ "
|
|
132
|
+
_p=$$
|
|
133
|
+
while [ -n "$_p" ] && [ "$_p" -gt 1 ]; do
|
|
134
|
+
_p=$(ps -o ppid= -p "$_p" 2>/dev/null | tr -d ' ')
|
|
135
|
+
[ -n "$_p" ] || break
|
|
136
|
+
SPARE="$SPARE$_p "
|
|
137
|
+
done
|
|
138
|
+
|
|
139
|
+
# claude_targets — matching PIDs (this user only) minus the spare set.
|
|
140
|
+
claude_targets() {
|
|
141
|
+
local out=""
|
|
142
|
+
local pid
|
|
143
|
+
for pid in $(pgrep -u "$(id -u)" -f "claude" 2>/dev/null || true); do
|
|
144
|
+
case "$SPARE" in
|
|
145
|
+
*" $pid "*) continue ;;
|
|
146
|
+
esac
|
|
147
|
+
out="$out$pid "
|
|
148
|
+
done
|
|
149
|
+
printf '%s' "$out"
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
pids=$(claude_targets)
|
|
153
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
154
|
+
echo "[DRY-RUN] sparing:$SPARE"
|
|
155
|
+
echo "[DRY-RUN] would terminate: ${pids:-<none>}"
|
|
156
|
+
echo "[$TIMESTAMP] DRY RUN — would terminate: ${pids:-<none>}" >> "$LOG_FILE"
|
|
157
|
+
elif [ -n "$pids" ]; then
|
|
158
|
+
# Name the set before signalling it — this is machine-wide for this user.
|
|
159
|
+
echo "Terminating (every matching process of this user, not just this agent): $pids"
|
|
67
160
|
# Send SIGTERM first, give 3s, then SIGKILL stragglers.
|
|
68
161
|
kill -TERM $pids 2>/dev/null || true
|
|
69
162
|
sleep 3
|
|
70
|
-
still=$(
|
|
71
|
-
[ -n "$still" ]
|
|
163
|
+
still=$(claude_targets)
|
|
164
|
+
if [ -n "$still" ]; then kill -KILL $still 2>/dev/null || true; fi
|
|
72
165
|
echo "[$TIMESTAMP] Claude processes terminated ($pids)" >> "$LOG_FILE"
|
|
166
|
+
else
|
|
167
|
+
echo "[$TIMESTAMP] No Claude processes to terminate (self/ancestors spared)" >> "$LOG_FILE"
|
|
73
168
|
fi
|
|
74
169
|
|
|
75
|
-
# 4. Log completion.
|
|
76
|
-
|
|
170
|
+
# 4. Log completion. A dry run says so in BOTH places — the banner an operator
|
|
171
|
+
# reads and the log an incident review reads — because the previous version
|
|
172
|
+
# printed the halt banner either way.
|
|
173
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
174
|
+
echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN — NO PROCESSES SIGNALLED (flag set, launchd unloaded)" | tee -a "$LOG_FILE"
|
|
175
|
+
else
|
|
176
|
+
echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
|
|
177
|
+
fi
|
|
77
178
|
echo ""
|
|
78
179
|
echo "To resume operations:"
|
|
79
180
|
echo " ./scripts/resume-operations.sh"
|