@cohortapp/agent-sdk 2.18.14 → 2.18.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -67,6 +67,7 @@ import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
67
67
  // gets a deterministic classification (no LLM triage), no holding ack, and a
68
68
  // per-seat jitter so thirteen answers do not land in the same second.
69
69
  import { isRollCallItem, rollCallJitterMs } from "../../lib/org/inbound/broadcast.mjs";
70
+ import { stampFirstReplyPlan } from "../../lib/assurance/first-reply.mjs";
70
71
  import { isRoomSurface } from "../../lib/org/inbound/surfaces.mjs";
71
72
  import { isMembershipReason } from "../../lib/org/inbound/directedness.mjs";
72
73
  import { cohortSurfaceOf, cohortSurfaceLabel, cohortSurfaceIsDeclared } from "./deliver.mjs";
@@ -96,7 +97,7 @@ import {
96
97
  isPidAlive,
97
98
  } from "./assurance.mjs";
98
99
  import { recordPoll, recordClassification, recordSession, writeHealthDashboard } from "./health.mjs";
99
- import { acquireLock, releaseLock, updateLock, scanStaleLocks, acquireThreadLock, claimRequest, hasActiveClaim, sweepStaleItemClaims, sanitiseItemId } from "./session-lock.mjs";
100
+ import { acquireLock, releaseLock, updateLock, scanStaleLocks, acquireThreadLock, threadLockKey, claimRequest, hasActiveClaim, sweepStaleItemClaims, sanitiseItemId } from "./session-lock.mjs";
100
101
  import { markDeferred } from "./inbox-deferral.mjs";
101
102
  import { parseQueueItems, rankBacklog, resolveBacklogWeights } from "../../lib/backlog.mjs";
102
103
  import { readLatestGaps } from "../../lib/goals/gaps.mjs";
@@ -159,8 +160,10 @@ import { processOne } from "../../lib/execution/pipeline.mjs";
159
160
  // of spawning `claude --print`; not live → today's dispatch, unchanged. The
160
161
  // gate is pure (lib/session/frontdoor.mjs); the assurance sweep reopens a
161
162
  // session claim that was neither replied nor done within 20 min.
162
- import { shouldReviveFrontDoor, reviveCommand, sessionJobLabelOnDisk } from "../../lib/session/revive.mjs";
163
- import { makeFrontDoorGate, sessionLiveFromHeartbeat, readFrontDoorState } from "../../lib/session/frontdoor.mjs";
163
+ import { shouldReviveFrontDoor, reviveCommand, sessionJobLabelOnDisk, confirmRevived, isTerminalReviveFailure, detectUnrequestedRestart } from "../../lib/session/revive.mjs";
164
+ import { budgetSpentNote } from "../../lib/telemetry/collect.mjs";
165
+ import { checkLock as checkProcessLock } from "../../lib/singleton.js";
166
+ import { makeFrontDoorGate, sessionLiveFromHeartbeat, readFrontDoorState, DEFAULT_STALE_MS } from "../../lib/session/frontdoor.mjs";
164
167
  import { CADENCE_REGISTRY } from "./cadence-handlers.mjs";
165
168
  import { sweepSessionInboxForDaemon } from "../../lib/session/inbox-claims.mjs";
166
169
  import { defaultEffects, scheduleToQueue } from "../../lib/execution/effects.mjs";
@@ -293,39 +296,283 @@ const _frontDoorGate = makeFrontDoorGate({ agentRoot: AGENT_REPO_DIR });
293
296
  // we want rather than a stale count carried across it.
294
297
  /** How often the daemon asks whether the front door has gone quiet. */
295
298
  const REVIVE_CHECK_INTERVAL_MS = 60 * 1000;
296
- const _revive = { attempts: 0, lastAt: null };
297
- async function reviveFrontDoorIfShut(state, deps = {}) {
299
+ // `notLive` is the count of CONSECUTIVE not-live reads — the actor-side
300
+ // hysteresis (Ravi). Takeover is harmless and flips at one read; a kickstart
301
+ // drops a working session, so the FIRST one waits for two consecutive not-live
302
+ // reads. A live read resets it. Per-process and deliberately not persisted: a
303
+ // daemon restart is itself a change of circumstances.
304
+ // `halted` latches when an assert comes back TERMINAL (phantom pid / crash
305
+ // loop): the seat needs a person, and a blind second kickstart would bury the
306
+ // first failure's log under the loop. One attempt on that road, then hand up.
307
+ //
308
+ // `confirmedIdentity` carries the last poll's {pid, startTime} ACROSS the 60s
309
+ // poll cadence, and `kickstartThisInterval` records whether THIS daemon
310
+ // kickstarted the job since that identity was recorded. A within-window
311
+ // continuity check cannot see a crash loop whose period exceeds the settle
312
+ // window — the same identity appears twice inside it. Comparing the confirmed
313
+ // identity against the next poll's observation, with the kickstart flag,
314
+ // catches an UNREQUESTED restart at ANY period, with no window to tune.
315
+ const _revive = { attempts: 0, lastAt: null, notLive: 0, halted: false, confirmedIdentity: null, kickstartThisInterval: false };
316
+ /** TEST-ONLY: clear the per-process revive state between cases. */
317
+ export function _resetRevive() {
318
+ _revive.attempts = 0; _revive.lastAt = null; _revive.notLive = 0; _revive.halted = false;
319
+ _revive.confirmedIdentity = null; _revive.kickstartThisInterval = false;
320
+ }
321
+
322
+ /** Gap between the two settle-window reads that prove the new session did not
323
+ * come up and immediately crash. Short — the door was already dark for the
324
+ * full grace before we kickstarted, so this only has to outlast a boot wobble. */
325
+ const CONFIRM_SETTLE_MS = 5000;
326
+
327
+ /**
328
+ * One observation of the session job: its launchctl pid, whether that pid is a
329
+ * GENUINELY LIVE process (not a phantom launchctl lists but `ps` says is dead —
330
+ * a silent crash loop), the process start-time that fingerprints the instance,
331
+ * and the COMM (the binary running under the pid). A relaunch changes the
332
+ * start-time even if the pid is reused, so comparing start-times across the
333
+ * settle window is how a crash loop is told apart from one continuous process;
334
+ * the comm is the third, complementary check — a live pid running the WRONG
335
+ * binary is not the process we kickstarted.
336
+ *
337
+ * ABSOLUTE `/bin/ps`, never bare `ps`: `ps` is aliased to `pnpm start` on this
338
+ * fleet, so a bare `ps` is poison. `kill -0` (a builtin/alias-immune existence
339
+ * signal) decides liveness; `/bin/ps` reads the immutable start-time and comm.
340
+ */
341
+ async function readSessionSample(label) {
342
+ let pid = null, alive = false, startTime = null, comm = null;
343
+ try {
344
+ const { stdout } = await execFileAsync("launchctl", ["list", label]);
345
+ const m = /"PID"\s*=\s*(\d+)/.exec(String(stdout || ""));
346
+ if (m) pid = Number(m[1]);
347
+ } catch { /* no pid — the kickstart produced no running process */ }
348
+ if (Number.isInteger(pid) && pid > 0) {
349
+ try { process.kill(pid, 0); alive = true; } catch { alive = false; } // signal 0 = existence check
350
+ try {
351
+ // One /bin/ps call for both the start-time and the comm — absolute path so
352
+ // the `ps`→`pnpm start` alias cannot poison it.
353
+ const { stdout } = await execFileAsync("/bin/ps", ["-o", "lstart=,comm=", "-p", String(pid)]);
354
+ const line = String(stdout || "").trim();
355
+ // `lstart` is a fixed-width 24-char date ("Wed Sep 25 12:00:00 2026")
356
+ // followed by the comm; split on the last field boundary the date ends at.
357
+ const m = /^(.{24})\s+(.*)$/.exec(line);
358
+ if (m) { startTime = m[1].trim() || null; comm = m[2].trim() || null; }
359
+ else { startTime = line || null; } // unexpected shape — keep what we have, comm unknown
360
+ } catch { startTime = null; comm = null; } // ps gone or pid vanished — degrade; pid continuity still carries
361
+ }
362
+ return { pid, alive, startTime, comm };
363
+ }
364
+
365
+ /** The binary the session job should be running — matched by suffix against the
366
+ * observed comm. Bare "node" so any install prefix (/opt/homebrew/bin/node,
367
+ * /usr/local/bin/node) matches; a live pid running anything else (a `ps`-alias
368
+ * stray `pnpm`, a wrong process) fails the comm-check and names `wrong-comm`. */
369
+ const EXPECTED_SESSION_COMM = "node";
370
+
371
+ /**
372
+ * After a kickstart, ASSERT the door came back before recording success — a
373
+ * restart helper's exit code is not a revive (Jacob's drain-then-restart-daemon.sh
374
+ * exited 0 and left the process stopped with NO pid). A ONE-SHOT read is not
375
+ * enough: it passes a phantom pid and a come-up-then-crash loop whose local beat
376
+ * file always looks fresh. So take TWO reads across a settle window (proving the
377
+ * process is one continuous instance, not a crash loop) and pair them with the
378
+ * session's OWN beat freshness — for a session that beat IS the after-signal a
379
+ * wedge kills, because the heartbeat rides tool use, so a session that stops
380
+ * working stops beating (unlike a daemon, which a crash loop keeps re-stamping).
381
+ */
382
+ async function defaultConfirmAfter({ label, now, staleMs, sleep }) {
383
+ const _sleep = sleep || ((ms) => new Promise((r) => setTimeout(r, ms)));
384
+ const s1 = await readSessionSample(label);
385
+ await _sleep(CONFIRM_SETTLE_MS);
386
+ const s2 = await readSessionSample(label);
387
+ // serverBeatFresh for the SESSION: is the heartbeat it should now be writing
388
+ // fresh? undefined (couldn't read) does not block — continuity carries it.
389
+ let serverBeatFresh;
390
+ try {
391
+ const st = readFrontDoorState(AGENT_REPO_DIR, { now, staleMs });
392
+ if (st && st.liveness && Number.isFinite(st.liveness.ageMs)) {
393
+ serverBeatFresh = st.liveness.ageMs < (staleMs || DEFAULT_STALE_MS);
394
+ }
395
+ } catch { /* still stale — a fresh session has not beaten yet; leave undefined */ }
396
+ return { samples: [s1, s2], serverBeatFresh };
397
+ }
398
+
399
+ // A failed revive has to reach the FLEET, not just this log: a kickstart that
400
+ // returned with no live pid is a failure to escalate — the seat needs a person.
401
+ // collect.mjs reads this into `machine.reviveNote`, so the presence beat carries
402
+ // it. Best-effort (the console line already fired); cleared on a confirmed revive.
403
+ const REVIVE_NOTE_PATH = join(AGENT_REPO_DIR, "state", "telemetry", "revive-note.json");
404
+ function writeReviveNote({ reason, target, detail, at }) {
405
+ try {
406
+ mkdirSync(join(AGENT_REPO_DIR, "state", "telemetry"), { recursive: true });
407
+ const tmp = `${REVIVE_NOTE_PATH}.${process.pid}.tmp`;
408
+ writeFileSync(tmp, JSON.stringify({ reason, target: target || null, detail: detail || "", at }));
409
+ renameSync(tmp, REVIVE_NOTE_PATH);
410
+ } catch { /* the beat surface is best-effort; the log line already fired */ }
411
+ }
412
+ function clearReviveNote() { try { unlinkSync(REVIVE_NOTE_PATH); } catch { /* nothing to clear */ } }
413
+
414
+ export async function reviveFrontDoorIfShut(state, deps = {}) {
298
415
  const now = deps.now ? deps.now() : Date.now();
416
+ // TEST-ONLY seam: seed the per-process ladder state so a case can start on a
417
+ // given rung (e.g. budget already spent) without threading a dozen polls.
418
+ if (deps.seedRevive && typeof deps.seedRevive === "object") Object.assign(_revive, deps.seedRevive);
419
+ // A latched terminal failure (a prior kickstart came up and crash-looped, or
420
+ // produced a phantom pid) means the seat is a person's problem now — do not
421
+ // fire another blind kickstart on top of a loop that would bury the evidence.
422
+ if (_revive.halted) return { revive: false, reason: "halted-escalated" };
299
423
  const live = sessionLiveFromHeartbeat(state && state.heartbeat, { now });
424
+ const label = deps.label || sessionJobLabelOnDisk(readdirSync, homedir());
425
+
426
+ // ── CROSS-POLL CRASH-LOOP CATCH (unbounded, no threshold) ─────────────────
427
+ //
428
+ // The within-window continuity check inside confirmRevived cannot see a crash
429
+ // loop whose period EXCEEDS the settle window: it shows the same identity
430
+ // twice inside a few seconds and false-confirms. Between two 60s polls,
431
+ // though, the loop has restarted — so if we hold a previously CONFIRMED
432
+ // identity and this poll observes a DIFFERENT one with no kickstart issued in
433
+ // the interval, that is an unrequested restart at ANY period. HALT and note it
434
+ // rather than blindly re-confirm/re-kickstart a loop forever.
435
+ //
436
+ // Armed ONLY while the door is still in trouble (not reading live). A door
437
+ // that is healthily answering again has recovered — a later identity change is
438
+ // ordinary churn (an SDK upgrade, a `session restart`, a supervisor resume
439
+ // rotation), none of which this daemon should police as a crash loop. So a
440
+ // healthy live read DISARMS the check; the loop this hunts is precisely one
441
+ // whose relaunched instance never beats a full grace, so it stays not-live.
442
+ if (label && _revive.confirmedIdentity && state && state.sessionLive !== true) {
443
+ const readSample = deps.readSessionSample || readSessionSample;
444
+ let curr = null;
445
+ try { curr = await readSample(label); } catch { curr = null; }
446
+ const drift = detectUnrequestedRestart({
447
+ prev: _revive.confirmedIdentity,
448
+ curr,
449
+ kickstartIssuedSince: _revive.kickstartThisInterval,
450
+ });
451
+ if (drift.restarted) {
452
+ _revive.halted = true;
453
+ const detail = `crash-loop-cross-poll: pid ${_revive.confirmedIdentity.pid}(${_revive.confirmedIdentity.startTime}) → pid ${curr && curr.pid}(${curr && curr.startTime}) with no kickstart since`;
454
+ console.error(`[daemon] ${label} identity changed across the poll cadence with no kickstart issued (${detail}) — a crash loop whose period outran the settle window; halting and escalating (this seat needs a person)`);
455
+ writeReviveNote({ reason: "session-revive-failed", target: label, detail, at: new Date(now).toISOString() });
456
+ return { revive: false, reason: "crash-loop-cross-poll", confirmed: { ok: false, reason: "crash-loop-cross-poll" } };
457
+ }
458
+ } else if (state && state.sessionLive === true && _revive.confirmedIdentity) {
459
+ // Recovered and answering — disarm; a healthy session's later identity
460
+ // changes are churn, not the loop.
461
+ _revive.confirmedIdentity = null;
462
+ _revive.kickstartThisInterval = false;
463
+ }
464
+
465
+ // Track the consecutive not-live streak BEFORE the decision; a live read
466
+ // resets it so a session that beats between two quiet reads never accrues.
467
+ const notLiveNow = !(state && state.sessionLive === true);
468
+ _revive.notLive = notLiveNow ? _revive.notLive + 1 : 0;
469
+ const reviveAfterMs = Number(process.env.MAESTRO_SESSION_STALE_S || 0) * 1000 || undefined;
300
470
  const verdict = shouldReviveFrontDoor({
301
471
  frontDoor: state && state.frontDoor,
302
472
  sessionLive: state && state.sessionLive,
303
473
  silentMs: live.ageMs,
474
+ notLiveReads: _revive.notLive,
304
475
  attempts: _revive.attempts,
305
476
  lastAttemptAt: _revive.lastAt,
306
477
  now,
307
- reviveAfterMs: Number(process.env.MAESTRO_SESSION_STALE_S || 0) * 1000 || undefined,
478
+ reviveAfterMs,
308
479
  });
309
- if (!verdict.revive) return verdict;
310
- const label = sessionJobLabelOnDisk(readdirSync, homedir());
480
+ if (!verdict.revive) {
481
+ // A ladder that spent its whole budget and STOPPED is a failure that does
482
+ // not name itself — the seat goes quiet and a quiet system looks healthy.
483
+ // Turn that silence into a beat note through the same reviveNote channel.
484
+ if (verdict.reason === "budget-spent") {
485
+ const note = budgetSpentNote({ verdict: verdict.reason, target: label || null, attempts: _revive.attempts, at: new Date(now).toISOString() });
486
+ if (note) writeReviveNote(note);
487
+ }
488
+ return verdict;
489
+ }
311
490
  if (!label) return { revive: false, reason: "no-session-job" };
491
+
492
+ // Supervisor coordination via the resume loop's OWN lock, not a soft note
493
+ // (Jacob). The supervisor holds the `session` singleton for its whole life
494
+ // and writes the heartbeat as a healthy resume makes progress. We only reach
495
+ // here after the door has been dark past the FULL grace AND confirmed over
496
+ // two reads 60s apart — so a resume gap is already excluded by time, and a
497
+ // live holder here is a supervisor PARKED in its launch probe (the 3.5-day /
498
+ // 15-day case this module exists for), for which the hard kickstart is
499
+ // exactly right. Consulting the lock tells a parked supervisor (override,
500
+ // logged) apart from a dead one (a clean restart), and never blocks the
501
+ // parked case — the failure that a hard "skip if held" would reintroduce.
502
+ const supervisor = (deps.checkSessionLock || (() => checkProcessLock("session")))();
503
+ if (supervisor && supervisor.running) {
504
+ console.error(`[daemon] ${label} front door dark past grace while a supervisor (pid ${supervisor.pid}) still holds the session lock — parked in its launch probe; hard-kickstarting over it`);
505
+ }
506
+
312
507
  _revive.attempts += 1;
313
508
  _revive.lastAt = now;
509
+ // This daemon is now the one that asked for the restart, so a new identity on
510
+ // the NEXT poll is expected, not the crash loop the cross-poll check hunts.
511
+ _revive.kickstartThisInterval = true;
314
512
  const cmd = reviveCommand(label, typeof process.getuid === "function" ? process.getuid() : 0);
315
513
  console.error(`[daemon] front door shut — ${verdict.reason}; restarting ${label}`);
316
514
  try {
317
515
  await (deps.execFile || execFileAsync)(cmd.file, cmd.args);
318
- console.error(`[daemon] ${label} restarted; a fresh session clears whatever the old one was sitting on`);
319
516
  } catch (err) {
320
517
  console.error(`[daemon] could not restart ${label}: ${err && err.message ? err.message : err} — this seat needs a person`);
518
+ return { ...verdict, confirmed: { ok: false, reason: "kickstart-threw" } };
519
+ }
520
+ // The kickstart returned. Now prove it actually came back — across a settle
521
+ // window, not a one-shot read that a phantom pid or a come-up-then-crash
522
+ // would pass.
523
+ const staleMs = reviveAfterMs;
524
+ const after = deps.confirmAfter
525
+ ? await deps.confirmAfter({ label, now, staleMs })
526
+ : await defaultConfirmAfter({ label, now, staleMs });
527
+ const confirmed = confirmRevived({ ...after, staleMs, expectedComm: EXPECTED_SESSION_COMM });
528
+ if (confirmed.ok) {
529
+ console.error(`[daemon] ${label} restarted and answering (${confirmed.reason}); a fresh session clears whatever the old one was sitting on`);
530
+ clearReviveNote();
531
+ // Record the identity this revive confirmed so the NEXT poll can tell one
532
+ // continuous instance from a crash loop whose period outran the settle
533
+ // window. Reset the kickstart flag: the interval it justified is over.
534
+ const alive = Array.isArray(after && after.samples) ? after.samples.find((s) => s && Number.isInteger(s.pid) && s.pid > 0 && s.alive === true) : null;
535
+ _revive.confirmedIdentity = alive ? { pid: alive.pid, startTime: alive.startTime } : null;
536
+ _revive.kickstartThisInterval = false;
537
+ } else if (isTerminalReviveFailure(confirmed)) {
538
+ // Phantom pid or a restart boundary inside the settle window: the kickstart
539
+ // produced no living, continuous process. STOP — one attempt, then hand up
540
+ // with what we saw, rather than kickstart again into a loop that buries it.
541
+ _revive.halted = true;
542
+ const evidence = describeSamples(after && after.samples);
543
+ console.error(`[daemon] ${label} kickstart did NOT revive it (${confirmed.reason}): ${evidence} — a helper's exit code is not a revive; this seat needs a person (failure to escalate)`);
544
+ writeReviveNote({ reason: "session-revive-failed", target: label, detail: `${confirmed.reason}: ${evidence}`, at: new Date(now).toISOString() });
545
+ } else {
546
+ console.error(`[daemon] ${label} restarted; awaiting its first beat (${confirmed.reason}) — the next check confirms or re-attempts within budget`);
321
547
  }
322
- return verdict;
548
+ return { ...verdict, confirmed };
549
+ }
550
+
551
+ /** One-line evidence of what the settle-window reads actually saw, for the log
552
+ * and the beat note — so a failed escalation carries the fingerprint, not just
553
+ * a verdict. */
554
+ function describeSamples(samples) {
555
+ if (!Array.isArray(samples) || !samples.length) return "no reads";
556
+ return samples.map((s) => `pid=${s && s.pid != null ? s.pid : "none"}${s && s.alive ? "(alive)" : "(dead)"}`).join(" then ");
323
557
  }
324
558
 
325
- async function pollService(svc) {
559
+ /**
560
+ * Drain one service's inbox: poll, dedup, stamp the flurry plan, process.
561
+ *
562
+ * @param {{name:string, fn:Function}} svc
563
+ * @param {object} [deps] TEST-ONLY seams; PRODUCTION PASSES NOTHING, so the
564
+ * defaults are the real collaborators and behaviour is byte-identical. The
565
+ * same idiom as `processItem`/`sweepBacklog` above. Exported (and seamed) so
566
+ * the ONE-FIRST-REPLY-PER-FLURRY stamping has a test that fails when the call
567
+ * is removed — see `notice-emission.test.mjs`.
568
+ * @param {object} [deps.frontDoorGate] override the front-door gate
569
+ * @param {Function} [deps.processItem] override processItem
570
+ */
571
+ export async function pollService(svc, deps = {}) {
572
+ const processItemImpl = deps.processItem || processItem;
326
573
  try {
327
574
  // Front door: a live main session owns this service's inbox — leave it.
328
- const fd = _frontDoorGate.check(svc.name);
575
+ const fd = (deps.frontDoorGate || _frontDoorGate).check(svc.name);
329
576
  if (fd.changed) console.log(`[daemon] front door for ${svc.name}: ${fd.dispatch ? "daemon dispatches" : "main session owns intake"} (${fd.reason})`);
330
577
  if (!fd.dispatch) return 0;
331
578
  const result = await svc.fn();
@@ -358,8 +605,33 @@ async function pollService(svc) {
358
605
  return lock.acquired;
359
606
  });
360
607
 
608
+ // ── ONE FIRST REPLY PER FLURRY ───────────────────────────────────────
609
+ // "If there's a flurry of 4 messages in one go by others, instead of having
610
+ // a generic reply per message, it would be responding to the batch of
611
+ // messages." (owner, 2026-09-25.) The batch is computed HERE, over the whole
612
+ // poll, because this is the only place that sees more than one item at a
613
+ // time — `processItem` is called per item and structurally cannot know it is
614
+ // the third of four. Every non-leader carries `covered_by_batch`, which
615
+ // `assurance.shouldAcknowledge` turns into `ack:false` and the daemon makes
616
+ // durable as `interimForbidden`, so no later sweep in any process speaks for
617
+ // it either. The leader is told how many it stands for, so the one line it
618
+ // writes can answer all of them.
619
+ //
620
+ // WORK IS UNAFFECTED. Every item still gets its own `processItem`, its own
621
+ // obligation and its own answer; what is batched is only the FIRST thing
622
+ // said. Silence about timing is not silence about outcome.
623
+ //
624
+ // The stamping itself is `assurance/first-reply.stampFirstReplyPlan`, not
625
+ // four lines here, because four lines here had NO TEST THAT COULD SEE THEM:
626
+ // the only call site was inside this module-internal function, so deleting
627
+ // them left the whole suite green — including a test named "wired, not just
628
+ // available". `first-reply.test.mjs` now exercises the stamp and
629
+ // `notice-emission.test.mjs` drives THIS function through `pollService`'s
630
+ // injected seams, so removing the call is red in both.
631
+ stampFirstReplyPlan(newItems);
632
+
361
633
  for (const item of newItems) {
362
- await processItem(item, svc.name);
634
+ await processItemImpl(item, svc.name);
363
635
  }
364
636
 
365
637
  if (result.errors.length > 0) {
@@ -1107,7 +1379,38 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
1107
1379
  // This check runs BEFORE quick reply to prevent ALL duplicate responses.
1108
1380
  {
1109
1381
  const channel = item.channel_id || (item.raw_ref ? item.raw_ref.match(/slack:([^:]+):/)?.[1] : null) || item.channel;
1110
- const threadTs = item.thread_id || (isDm ? `dm-channel` : null);
1382
+ // THE UNTREADED CHANNEL POST USED TO TAKE NO LOCK AT ALL.
1383
+ //
1384
+ // `item.thread_id || (isDm ? "dm-channel" : null)` is null for exactly
1385
+ // one shape: a message posted to the main surface of a channel or group,
1386
+ // not in a thread, not a DM. That is the shape the owner screenshotted —
1387
+ // a flurry of messages into a room — and it was the one shape with no
1388
+ // dedup: each message cleared this block, spawned its own session, and
1389
+ // opened its own obligation, so the room got N sessions and N chances at
1390
+ // a holding line for one conversation.
1391
+ //
1392
+ // `channel-main` is the DM lock's counterpart for that surface: one
1393
+ // session at a time on a room's main feed, the rest deferred and
1394
+ // promoted at session close exactly like a DM burst.
1395
+ //
1396
+ // THIS IS ONLY SAFE BECAUSE OF THE BUNDLING FIX BESIDE IT. Deferral ends
1397
+ // in `promoteDeferred`, which collapses a group to its latest member and
1398
+ // marks the rest `.processed-bundled`. On a shared channel the deferred
1399
+ // set is several PEOPLE, not one person's flurry, so an unconditional
1400
+ // collapse here would silently bin a colleague's question — which is why
1401
+ // inbox-deferral.mjs now splits a group by sender unless the winner
1402
+ // actually carries the room's history. Widening this lock without that
1403
+ // change would trade a message storm for lost messages.
1404
+ //
1405
+ // The off-switch is real and deliberate: MAESTRO_CHANNEL_MAIN_LOCK=0
1406
+ // restores the previous behaviour on a seat without a code change, in
1407
+ // the same style as every other threshold in the assurance stack.
1408
+ const threadTs = threadLockKey({
1409
+ channel,
1410
+ threadId: item.thread_id,
1411
+ isDm,
1412
+ channelMainLock: String(process.env.MAESTRO_CHANNEL_MAIN_LOCK ?? "1") !== "0",
1413
+ });
1111
1414
  if (threadTs && channel) {
1112
1415
  const threadCheck = acquireThreadLock(channel, threadTs);
1113
1416
  if (!threadCheck.allowed) {