@cohortapp/agent-sdk 2.18.13 → 2.18.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/bin/maestro.mjs +38 -1
  2. package/docs/runbooks/fleet-rollout.md +58 -7
  3. package/docs/runbooks/recovery-and-failover.md +18 -0
  4. package/lib/assurance/batch.mjs +353 -0
  5. package/lib/assurance/first-reply.mjs +423 -0
  6. package/lib/assurance/notice-voice.mjs +357 -0
  7. package/lib/assurance/plan-note.mjs +43 -0
  8. package/lib/assurance/room-budget.mjs +55 -6
  9. package/lib/cadence-failure-class.mjs +245 -0
  10. package/lib/claude-bin.mjs +26 -7
  11. package/lib/cli/doctor-checks.mjs +149 -1
  12. package/lib/comms/send-gate.mjs +59 -0
  13. package/lib/diagnostics/alerts.mjs +33 -0
  14. package/lib/engine/agents/usage.mjs +45 -0
  15. package/lib/engine/budget.mjs +293 -29
  16. package/lib/engine/cli.mjs +54 -5
  17. package/lib/engine/loop.mjs +30 -0
  18. package/lib/engine/output/json.mjs +26 -0
  19. package/lib/engine/wire/errors.mjs +179 -0
  20. package/lib/engine/wire/search.mjs +44 -8
  21. package/lib/identity/persona.mjs +31 -2
  22. package/lib/org/quota.mjs +27 -0
  23. package/lib/session/config.mjs +4 -0
  24. package/lib/session/identity.mjs +71 -7
  25. package/lib/session/launch-failure.mjs +251 -0
  26. package/lib/session/resume-target.mjs +86 -0
  27. package/lib/telemetry/alerts.mjs +94 -0
  28. package/lib/telemetry/collect.mjs +155 -2
  29. package/lib/upgrade/pinned-drift.mjs +467 -0
  30. package/package.json +1 -1
  31. package/scaffold/config/alerts.yaml +7 -0
  32. package/scripts/ci/check-cadence-prompts-exist.mjs +96 -0
  33. package/scripts/ci/check.mjs +3 -0
  34. package/scripts/daemon/agent-daemon.mjs +75 -5
  35. package/scripts/daemon/assurance.mjs +709 -44
  36. package/scripts/daemon/cadence-consumer.mjs +281 -34
  37. package/scripts/daemon/deliver.mjs +109 -0
  38. package/scripts/daemon/dispatcher.mjs +21 -3
  39. package/scripts/daemon/inbox-deferral.mjs +102 -9
  40. package/scripts/daemon/session-lock.mjs +41 -1
  41. package/scripts/emergency-stop.sh +114 -13
  42. package/scripts/fleet/rollout.mjs +256 -10
  43. package/scripts/healthcheck.sh +131 -33
  44. package/scripts/local-triggers/autoupdate.sh +144 -11
  45. package/scripts/resume-operations.sh +101 -6
  46. package/scripts/session/supervisor.mjs +198 -5
@@ -18,11 +18,26 @@
18
18
  * every `.deferred` file in any service inbox dir whose body
19
19
  * references `channel`. If exactly one is found, renames it back
20
20
  * to its original name (next poll picks it up). If N>1 are found,
21
- * keeps only the LATEST (by timestamp), promotes that one, and
22
- * marks the others `.processed-bundled` for the audit trail —
23
- * the latest item's `thread_context` already contains the prior
24
- * messages as conversation history, so Claude sees everything
25
- * and can compose a single coherent reply.
21
+ * keeps only the LATEST (by timestamp) of each BUNDLING GROUP,
22
+ * promotes those, and marks the others `.processed-bundled` for
23
+ * the audit trail.
24
+ *
25
+ * WHAT A BUNDLING GROUP IS, AND WHY IT IS NOT JUST "THE SURFACE".
26
+ * The collapse is only sound when the promoted item genuinely
27
+ * carries what is being folded away. That used to be asserted
28
+ * unconditionally — "the latest item's `thread_context` already
29
+ * contains the prior messages" — and it is true of a Slack
30
+ * channel or DM item and FALSE of a Cohort space post, which
31
+ * `lib/org/inbound` never hydrates a thread context for. Fold a
32
+ * second person's question into one of those and it is not
33
+ * bundled, it is deleted: marked `.processed-bundled` and never
34
+ * answered by anyone.
35
+ *
36
+ * So a group is (surface, sender) — a person's own flurry always
37
+ * collapses to their latest word — UNLESS the surface's newest
38
+ * item carries `thread_context`, in which case the whole surface
39
+ * collapses as before, because the session really will see every
40
+ * message it stands for.
26
41
  *
27
42
  * The file format is the YAML emitted by the slack/gmail/calendar
28
43
  * pollers (a flat top-level object with quoted scalar fields, plus an
@@ -155,6 +170,42 @@ function readScalar(body, field) {
155
170
  return null;
156
171
  }
157
172
 
173
+ /**
174
+ * Does this file carry a TOP-LEVEL key at all, block scalar or not?
175
+ *
176
+ * `readScalar` answers "what is this field's single-line value", which is the
177
+ * wrong question for `thread_context`: the pollers write it as a block scalar
178
+ * (`thread_context: |`), so its value never lives on the key's own line. All
179
+ * the bundling decision needs to know is whether the field is PRESENT — the
180
+ * pollers emit it only when there is history to emit (`if (item.thread_context)`
181
+ * in scripts/poller/utils.mjs), so presence is exactly "this item carries the
182
+ * conversation around it".
183
+ *
184
+ * @param {string} body
185
+ * @param {string} field
186
+ * @returns {boolean}
187
+ */
188
+ function hasTopLevelKey(body, field) {
189
+ if (typeof body !== "string") return false;
190
+ // TWO MECHANISMS, ONE PROPERTY, and the redundancy is stated rather than
191
+ // pretended away: the pattern is anchored at column 0, so an indented line
192
+ // cannot match it and the `continue` below is a BELT, not the mechanism.
193
+ // Deleting the `continue` alone changes no behaviour — a mutation of it
194
+ // leaves the suite green, and that is honest rather than a hole. What the
195
+ // suite does refuse is the realistic future edit: loosening this anchor to
196
+ // `^\s*` (the shape `readScalar` had before audit L7) with nothing else in
197
+ // the way. That is pinned as a BEHAVIOUR, in inbox-deferral.test.mjs — "L7:
198
+ // body line `thread_context:` in a block scalar does not fake carrying
199
+ // history" — because the property worth pinning is that a colleague's
200
+ // question survives, not which of two lines delivered it.
201
+ const re = new RegExp(`^${escapeRegExp(field)}\\s*:`);
202
+ for (const line of body.split("\n")) {
203
+ if (/^\s/.test(line)) continue; // indented — belongs to a block, not top level
204
+ if (re.test(line)) return true;
205
+ }
206
+ return false;
207
+ }
208
+
158
209
  /**
159
210
  * Promote every `.deferred` item targeting `channel` back into the
160
211
  * live inbox. Bundles bursts: if multiple deferred items exist for
@@ -219,6 +270,26 @@ export function promoteDeferred(channel, agentRoot) {
219
270
  readScalar(body, "thread_id") ||
220
271
  readScalar(body, "scope_id") ||
221
272
  "",
273
+ // WHO SENT IT, and WHETHER THE ITEM CARRIES ITS CONVERSATION.
274
+ //
275
+ // These two decide whether bundling is a bundle or a deletion. The
276
+ // latest-wins collapse is justified by one sentence in this module's
277
+ // header — "the latest item's thread_context already contains the
278
+ // prior messages as conversation history, so Claude sees everything" —
279
+ // and that sentence is TRUE OF SOME ITEMS AND FALSE OF OTHERS. Slack
280
+ // channel and DM items carry `thread_context` (slack-poller.mjs:434,
281
+ // :558); a Cohort space post does NOT, because `lib/org/inbound`
282
+ // never hydrates one for an untreaded space message. Bundling a
283
+ // second person's ask into a Cohort item that carries no history is
284
+ // not folding a duplicate away, it is marking their question
285
+ // `.processed-bundled` and never answering it.
286
+ //
287
+ // So: the collapse now requires evidence. Same sender is always safe
288
+ // (a person's own flurry — their newest message is their latest word
289
+ // either way). A DIFFERENT sender may only be folded into an item that
290
+ // actually carries the room's history.
291
+ sender: (readScalar(body, "sender") || "").trim().toLowerCase(),
292
+ carriesHistory: hasTopLevelKey(body, "thread_context"),
222
293
  });
223
294
  }
224
295
  if (matches.length === 0) continue;
@@ -226,13 +297,35 @@ export function promoteDeferred(channel, agentRoot) {
226
297
  // Group by surface within the channel, then bundle latest-wins WITHIN each
227
298
  // group. Distinct surfaces each promote their own latest; none is dropped
228
299
  // just because a newer message landed on a different surface in the room.
229
- const groups = new Map();
300
+ const surfaces = new Map();
230
301
  for (const m of matches) {
231
- if (!groups.has(m.surface)) groups.set(m.surface, []);
232
- groups.get(m.surface).push(m);
302
+ if (!surfaces.has(m.surface)) surfaces.set(m.surface, []);
303
+ surfaces.get(m.surface).push(m);
304
+ }
305
+
306
+ // SPLIT A SURFACE BY SENDER UNLESS ITS WINNER CARRIES THE HISTORY.
307
+ // With history present this is byte-for-byte the old behaviour: one group
308
+ // per surface, latest wins, everything else bundled. Without it, each
309
+ // sender keeps their own latest and promotes it — so two people's asks
310
+ // serialise into two sessions rather than one of them disappearing.
311
+ const groups = [];
312
+ for (const group of surfaces.values()) {
313
+ group.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
314
+ const winner = group[group.length - 1];
315
+ const senders = new Set(group.map((m) => m.sender));
316
+ if (senders.size <= 1 || winner.carriesHistory) {
317
+ groups.push(group);
318
+ continue;
319
+ }
320
+ const bySender = new Map();
321
+ for (const m of group) {
322
+ if (!bySender.has(m.sender)) bySender.set(m.sender, []);
323
+ bySender.get(m.sender).push(m);
324
+ }
325
+ for (const g of bySender.values()) groups.push(g);
233
326
  }
234
327
 
235
- for (const group of groups.values()) {
328
+ for (const group of groups) {
236
329
  // Latest-wins: lex-sort ISO timestamps, take the most recent.
237
330
  group.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
238
331
  const latest = group[group.length - 1];
@@ -311,6 +311,37 @@ export function checkRecentlySent(channel, threadTs, type) {
311
311
  return { allowed: true };
312
312
  }
313
313
 
314
+ /**
315
+ * PURE. The ONE definition of which conversation a thread lock covers.
316
+ *
317
+ * WHY THIS EXISTS AS A FUNCTION. The normalisation used to be written twice —
318
+ * once where the lock is taken (agent-daemon, as an inline ternary) and once
319
+ * where it is released (`releaseThreadLock`, as a different inline rule) — and
320
+ * the two did not agree. The acquiring side turned a DM into `dm-channel` from
321
+ * `item.is_dm`; the releasing side only recognised a DM by a channel id
322
+ * starting with "D", which is a SLACK id shape. A Cohort DM therefore took a
323
+ * `dm-channel` lock and released nothing, and the room stayed locked for the
324
+ * full sixty-minute TTL with every later message deferred behind it. Two
325
+ * copies of a key derivation is one copy too many; this is now the only one.
326
+ *
327
+ * @param {object} o
328
+ * @param {string} o.channel
329
+ * @param {string} [o.threadId] the item's own thread id, if it has one
330
+ * @param {boolean} [o.isDm] the daemon's own DM verdict (item.is_dm)
331
+ * @param {boolean} [o.channelMainLock=true] false restores the pre-2026-09-25
332
+ * behaviour where an untreaded channel post took no lock at all
333
+ * @returns {string|null} the lock key, or null when this item takes no lock
334
+ */
335
+ export function threadLockKey(o = {}) {
336
+ const channel = o.channel;
337
+ if (!channel) return null;
338
+ // A Slack DM id ("D…") is a DM whatever the caller believed.
339
+ if (String(channel).startsWith("D")) return "dm-channel";
340
+ if (o.threadId) return String(o.threadId);
341
+ if (o.isDm === true) return "dm-channel";
342
+ return o.channelMainLock === false ? null : "channel-main";
343
+ }
344
+
314
345
  /**
315
346
  * Check if a session was already dispatched for the same thread recently.
316
347
  * Prevents multiple sessions from responding to the same thread when
@@ -516,7 +547,16 @@ export function releaseThreadLock(channel, threadTs) {
516
547
  // the first was skipped with "thread_dedup: DM channel already
517
548
  // dispatched 1200s ago").
518
549
  if (channel.startsWith("D")) threadTs = "dm-channel";
519
- if (!threadTs) return; // non-DM channel without a thread — no lock to release
550
+ // A caller that hands us nothing is asking us to guess, and the only guess
551
+ // that mirrors the acquiring side is `threadLockKey`'s: an untreaded, non-DM
552
+ // channel post locks the room's main feed. Unlinking a lock that was never
553
+ // taken (the off-switch case, or a genuinely lockless item) is a no-op under
554
+ // the catch below, so guessing here can only ever free a lock, never wedge
555
+ // one. The reverse — returning early, as this did — left every
556
+ // `channel-main` lock to expire on its sixty-minute TTL, which is the whole
557
+ // channel gagged for an hour after one message.
558
+ if (!threadTs) threadTs = threadLockKey({ channel, threadId: null, isDm: false });
559
+ if (!threadTs) return;
520
560
  const safeKey = sanitiseItemId(`thread-${channel}-${threadTs}`);
521
561
  const lockPath = join(LOCKS_DIR, `${safeKey}.lock`);
522
562
  try {
@@ -1,13 +1,42 @@
1
1
  #!/bin/bash
2
2
  # Emergency Stop — Immediately halts all Maestro agent operations.
3
- # Usage: ./scripts/emergency-stop.sh
3
+ # Usage: ./scripts/emergency-stop.sh [--dry-run] [--help]
4
4
  #
5
5
  # This script is the kill switch for all autonomous operations:
6
6
  # 1. Drops .emergency-stop flag (every workflow / cadence consumer / enqueue
7
7
  # script honours this on the next tick).
8
8
  # 2. Unloads every installed `ai.maestro.<agent>-*` (and legacy
9
9
  # `ai.adaptic.<agent>-*`) launchd job.
10
- # 3. Kills running Claude Code subagent processes.
10
+ # 3. Kills running Claude Code subagent processes (EXCEPT this process and
11
+ # its ancestors — see step 3).
12
+ #
13
+ # --dry-run
14
+ # Report the kill set instead of signalling it. The flag and the launchd
15
+ # unload still happen; only the signals are withheld — and BOTH the closing
16
+ # banner and the log line say "DRY RUN — NO PROCESSES SIGNALLED", so a dry
17
+ # run can never be mistaken for a halt. It exists so the self-sparing in
18
+ # step 3 is testable without a suite that kills the operator's own session
19
+ # to prove that it does not.
20
+ #
21
+ # WHY A FLAG AND NOT AN ENV VAR. This used to be read from the ambient
22
+ # environment as MAESTRO_EMERGENCY_STOP_DRY_RUN, while step 4 printed
23
+ # "EMERGENCY STOP COMPLETE — All operations halted" unconditionally. A stray
24
+ # `export`, a line in .env, or an EnvironmentVariables entry in a plist
25
+ # would therefore turn the kill switch into a no-op — every claude process
26
+ # surviving — while the banner and the log both asserted the halt had
27
+ # succeeded. That is the same fault class this file was being repaired for:
28
+ # something that looks like it is handling the case and is not. On a
29
+ # break-glass control the escape hatch must be typed at the call site, once,
30
+ # deliberately. The env var is NOT consulted; setting it does nothing.
31
+ #
32
+ # SCOPE OF THE KILL (step 3), stated rather than hidden: every process of THIS
33
+ # user whose command line matches `claude`, minus this process and its
34
+ # ancestors. That is machine-wide, not agent-scoped — an unrelated Claude
35
+ # session of yours in another directory WILL be terminated (21 processes
36
+ # matched on the host where this was measured). Nothing here can narrow it
37
+ # honestly, because a Claude Code session's argv does not carry the agent dir;
38
+ # so the set is printed before it is signalled, and --dry-run shows it without
39
+ # signalling anything.
11
40
  #
12
41
  # Plist resolution: the agent's first-name slug is read from config/agent.json
13
42
  # (SOT) so unload targets the correct labels; falls back to the directory
@@ -21,6 +50,25 @@ LOG_FILE="$AGENT_DIR/logs/emergency-stop.log"
21
50
  TIMESTAMP=$(date -u +"%Y-%m-%dT%H:%M:%SZ")
22
51
  mkdir -p "$(dirname "$LOG_FILE")" 2>/dev/null || true
23
52
 
53
+ # Argument parsing runs BEFORE anything is halted: a typo must refuse loudly,
54
+ # not drop the flag and then exit.
55
+ DRY_RUN=0
56
+ while [ $# -gt 0 ]; do
57
+ case "$1" in
58
+ --dry-run) DRY_RUN=1 ;;
59
+ -h | --help)
60
+ sed -n '2,/^$/p' "${BASH_SOURCE[0]}" | sed 's/^# \{0,1\}//'
61
+ exit 0
62
+ ;;
63
+ *)
64
+ echo "emergency-stop: unknown argument: $1" >&2
65
+ echo "Usage: emergency-stop.sh [--dry-run]" >&2
66
+ exit 2
67
+ ;;
68
+ esac
69
+ shift
70
+ done
71
+
24
72
  # Resolve agent first-name slug from SOT (config/agent.json) so the unload
25
73
  # loop targets the right launchd labels. Falls back to the basename of the
26
74
  # agent directory (stripping -ai suffix).
@@ -39,7 +87,11 @@ LAUNCH_AGENTS_DIR="$HOME/Library/LaunchAgents"
39
87
  # (deployed agents whose plists predate the rename). The loops guard with [ -f ].
40
88
  PLIST_GLOB="$LAUNCH_AGENTS_DIR/ai.maestro.${AGENT_FIRST}-*.plist $LAUNCH_AGENTS_DIR/ai.adaptic.${AGENT_FIRST}-*.plist"
41
89
 
42
- echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
90
+ if [ "$DRY_RUN" -eq 1 ]; then
91
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN INITIATED (agent=$AGENT_FIRST) — no process will be signalled" | tee -a "$LOG_FILE"
92
+ else
93
+ echo "[$TIMESTAMP] EMERGENCY STOP INITIATED (agent=$AGENT_FIRST)" | tee -a "$LOG_FILE"
94
+ fi
43
95
 
44
96
  # 1. Drop the stop flag FIRST so any in-flight work sees it on next tick.
45
97
  echo "$TIMESTAMP" > "$AGENT_DIR/.emergency-stop"
@@ -57,23 +109,72 @@ for plist in $PLIST_GLOB; do
57
109
  done
58
110
  echo "[$TIMESTAMP] Unloaded $unloaded launchd job(s)" >> "$LOG_FILE"
59
111
 
60
- # 3. Kill running Claude Code subagent processes. Filter to processes that
61
- # have AGENT_ROOT or this agent's directory in their cwd to avoid
62
- # killing unrelated claude sessions the operator may have running.
112
+ # 3. Kill running Claude Code subagent processes.
113
+ #
114
+ # THE MIRROR-IMAGE FAULT THIS FIXES: `pgrep -f claude` matches the process
115
+ # that is RUNNING THIS SCRIPT whenever an agent session invokes it — which
116
+ # is the normal way it gets invoked. The stop then killed its own caller
117
+ # mid-flight, so step 4 never ran and the halt was never logged complete:
118
+ # a script that destroys the condition it needs in order to finish, the
119
+ # same shape as resume-operations.sh refusing to lift its own stop flag.
120
+ # The comment here also claimed a cwd filter that did not exist.
121
+ #
122
+ # So: build the kill set, then subtract this process and every ancestor of
123
+ # it. Everything else matching `claude` is still terminated — the stop is
124
+ # still a kill switch, it just no longer includes the hand on the switch.
125
+ # See SCOPE OF THE KILL in the header: "everything else" really does mean
126
+ # every matching process of this user on this machine, so it is printed.
63
127
  echo "Stopping Claude Code agent processes..."
64
- # pgrep -lf is more selective than pkill -f
65
- pids=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
66
- if [ -n "$pids" ]; then
128
+
129
+ # PIDs to spare: this shell and its whole ancestor chain (the launching
130
+ # session, its shell, launchd). Walk up via ppid until PID 1.
131
+ SPARE=" $$ "
132
+ _p=$$
133
+ while [ -n "$_p" ] && [ "$_p" -gt 1 ]; do
134
+ _p=$(ps -o ppid= -p "$_p" 2>/dev/null | tr -d ' ')
135
+ [ -n "$_p" ] || break
136
+ SPARE="$SPARE$_p "
137
+ done
138
+
139
+ # claude_targets — matching PIDs (this user only) minus the spare set.
140
+ claude_targets() {
141
+ local out=""
142
+ local pid
143
+ for pid in $(pgrep -u "$(id -u)" -f "claude" 2>/dev/null || true); do
144
+ case "$SPARE" in
145
+ *" $pid "*) continue ;;
146
+ esac
147
+ out="$out$pid "
148
+ done
149
+ printf '%s' "$out"
150
+ }
151
+
152
+ pids=$(claude_targets)
153
+ if [ "$DRY_RUN" -eq 1 ]; then
154
+ echo "[DRY-RUN] sparing:$SPARE"
155
+ echo "[DRY-RUN] would terminate: ${pids:-<none>}"
156
+ echo "[$TIMESTAMP] DRY RUN — would terminate: ${pids:-<none>}" >> "$LOG_FILE"
157
+ elif [ -n "$pids" ]; then
158
+ # Name the set before signalling it — this is machine-wide for this user.
159
+ echo "Terminating (every matching process of this user, not just this agent): $pids"
67
160
  # Send SIGTERM first, give 3s, then SIGKILL stragglers.
68
161
  kill -TERM $pids 2>/dev/null || true
69
162
  sleep 3
70
- still=$(pgrep -f "claude" 2>/dev/null | tr '\n' ' ' || true)
71
- [ -n "$still" ] && kill -KILL $still 2>/dev/null || true
163
+ still=$(claude_targets)
164
+ if [ -n "$still" ]; then kill -KILL $still 2>/dev/null || true; fi
72
165
  echo "[$TIMESTAMP] Claude processes terminated ($pids)" >> "$LOG_FILE"
166
+ else
167
+ echo "[$TIMESTAMP] No Claude processes to terminate (self/ancestors spared)" >> "$LOG_FILE"
73
168
  fi
74
169
 
75
- # 4. Log completion.
76
- echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
170
+ # 4. Log completion. A dry run says so in BOTH places — the banner an operator
171
+ # reads and the log an incident review reads — because the previous version
172
+ # printed the halt banner either way.
173
+ if [ "$DRY_RUN" -eq 1 ]; then
174
+ echo "[$TIMESTAMP] EMERGENCY STOP DRY RUN — NO PROCESSES SIGNALLED (flag set, launchd unloaded)" | tee -a "$LOG_FILE"
175
+ else
176
+ echo "[$TIMESTAMP] EMERGENCY STOP COMPLETE — All operations halted" | tee -a "$LOG_FILE"
177
+ fi
77
178
  echo ""
78
179
  echo "To resume operations:"
79
180
  echo " ./scripts/resume-operations.sh"