@cohortapp/agent-sdk 2.11.0 → 2.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -135,28 +135,28 @@ describe("composeAck", () => {
135
135
  assert.equal(a, composeAck(ITEM, CLASS_COMPLEX), "same input must give the same words");
136
136
  });
137
137
 
138
- test("names the actual ask and promises to come back either way", () => {
138
+ test("is one short human line — no filler, no promise-of-followup boilerplate", () => {
139
139
  const a = composeAck(ITEM, CLASS_COMPLEX);
140
- assert.match(a, /fix the noted issues and push the fixes to git/i);
141
- assert.match(a, /either way/i, "the promise the rest of the system exists to keep");
142
- assert.match(a, /10-20 minutes/, "an opus/critical item is honestly framed as slow");
140
+ assert.ok(a.length <= 40, `an ack is one short line, got ${a.length} chars: ${a}`);
141
+ // The templated filler communication-style.md bans.
142
+ assert.doesNotMatch(a, /either way|full answer|within a few minutes|within 10-20|I want to get this right|hear from me|come back to you/i);
143
143
  });
144
144
 
145
145
  test("degrades gracefully when the classifier gave nothing to go on", () => {
146
146
  const a = composeAck(ITEM, {});
147
- assert.ok(a.length > 40);
148
- assert.match(a, /either way/i);
147
+ assert.ok(a.length > 0);
148
+ assert.doesNotMatch(a, /\(.*\)/, "no parenthetical aside");
149
149
  });
150
150
 
151
- test("REGRESSION: an acronym-initial summary keeps its casing — no 'cEO'", () => {
152
- // Owner screenshot, verbatim: "Picking up: cEO checking in on work
153
- // progress." — topicClause downcased the first letter of whatever the
154
- // classifier wrote so it could sit mid-sentence, and "CEO…" became a typo
155
- // the requester reads as the agent's own. The summary is now carried
156
- // exactly as written.
157
- const a = composeAck(ITEM, { ...CLASS_COMPLEX, summary: "CEO checking in on work progress" });
158
- assert.match(a, /CEO checking in on work progress/);
159
- assert.doesNotMatch(a, /cEO/);
151
+ test("REGRESSION: the classifier summary NEVER reaches the human — no '(CEO …)' bleed", () => {
152
+ // Owner screenshot, verbatim: the ack carried the classifier's internal
153
+ // summary in a parenthetical aside — "(CEO asking what machine/infrastructure
154
+ // is being used)" reached a real human. The summary is now stored on the
155
+ // obligation for the board row but never shown in a message anyone reads.
156
+ const a = composeAck(ITEM, { ...CLASS_COMPLEX, summary: "CEO asking what machine/infrastructure is being used" });
157
+ assert.doesNotMatch(a, /CEO asking/i);
158
+ assert.doesNotMatch(a, /machine\/infrastructure/i);
159
+ assert.doesNotMatch(a, /\(.*\)/, "no parenthetical aside at all");
160
160
  });
161
161
 
162
162
  test("reads like a colleague, not a template — no internal framing shown to the human", () => {
@@ -223,10 +223,10 @@ describe("openAndAcknowledge", () => {
223
223
  assert.equal(openObligations().length, 1, "the debt must be visible to the sweep");
224
224
  });
225
225
 
226
- test("the stored summary keeps the classifier's casing — every later message reads from it", async () => {
227
- // `rec.summary` feeds topicAside (progress, failure, silent-success,
228
- // interrupted), the sweep's compensating ack, and board-mirror's row title.
229
- // A summary mangled at open time would resurface in all of them.
226
+ test("the stored summary keeps the classifier's casing — the board row and escalations read from it", async () => {
227
+ // `rec.summary` feeds board-mirror's row title and the needs-attention
228
+ // escalation record. A summary mangled at open time would resurface there.
229
+ // It must NOT, however, appear in any message the human reads.
230
230
  const t = fakeTransport();
231
231
  const r = await openAndAcknowledge({
232
232
  item: ITEM,
@@ -235,7 +235,7 @@ describe("openAndAcknowledge", () => {
235
235
  });
236
236
  const rec = readObligation(r.key);
237
237
  assert.equal(rec.summary, "CEO checking in on work progress");
238
- assert.match(composeProgress(rec, { now: Date.now() + 6 * 60_000 }), /\(CEO checking in on work progress\)/);
238
+ assert.doesNotMatch(composeProgress(rec, { now: Date.now() + 6 * 60_000 }), /CEO checking in on work progress/, "the summary is never leaked into a progress update");
239
239
  });
240
240
 
241
241
  test("acknowledges ONCE — a re-delivered item does not re-announce itself", async () => {
@@ -339,7 +339,7 @@ describe("settleSession", () => {
339
339
  assert.match(r.verdict, /^retrying:timeout$/);
340
340
  const msg = t.sent.at(-1).text;
341
341
  assert.match(msg, /ran past its time limit/i);
342
- assert.match(msg, /retrying it now/i);
342
+ assert.match(msg, /retrying now/i);
343
343
  assert.equal(readObligation(key).state, "open", "still owed — the retry has to be covered too");
344
344
  });
345
345
 
@@ -350,7 +350,7 @@ describe("settleSession", () => {
350
350
  assert.equal(r.willRetry, false);
351
351
  assert.match(r.verdict, /^failed:spawn_failed$/);
352
352
  const msg = t.sent.at(-1).text;
353
- assert.match(msg, /couldn't get this done/i);
353
+ assert.match(msg, /couldn't finish this/i);
354
354
  assert.match(msg, /stopped retrying/i);
355
355
  assert.match(msg, /\?$/, "a dead end must end in a question, not a shrug");
356
356
  assert.equal(readObligation(key).state, "failed");
@@ -384,7 +384,7 @@ describe("sweepObligations", () => {
384
384
  const now = Date.now() + ACK_GRACE_MS + 1000;
385
385
  const stats = await sweepObligations({ now, deps: { deliverImpl: working.impl, spokeSinceImpl: () => false } });
386
386
  assert.equal(stats.acked, 1);
387
- assert.match(working.sent[0].text, /come back to you/i);
387
+ assert.match(working.sent[0].text, /on it|looking now|digging in/i);
388
388
  assert.equal(readObligation(r.key).acknowledged, true);
389
389
  });
390
390
 
@@ -395,7 +395,7 @@ describe("sweepObligations", () => {
395
395
  const stats = await sweepObligations({ now, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
396
396
  assert.equal(stats.progressed, 1);
397
397
  const msg = t.sent.at(-1).text;
398
- assert.match(msg, /still working/i);
398
+ assert.match(msg, /still on this/i);
399
399
  assert.match(msg, /minutes in/i, "an interim update that omits how long it has been is not much of an update");
400
400
  // Six minutes in is INSIDE the 10-20 minute window the ack just promised
401
401
  // for opus work; claiming it here contradicts a message the same human read.
@@ -642,7 +642,7 @@ describe("attribution: which debt did that message actually discharge", () => {
642
642
  const stats = await sweepObligations({ now: Date.now() + ACK_GRACE_MS + 1000, deps: { deliverImpl: t.impl } });
643
643
  assert.equal(stats.closed, 0, "no session of ours had run — that message was not our answer");
644
644
  assert.equal(stats.acked, 1, "instead the missing acknowledgement is finally sent");
645
- assert.match(t.sent.at(-1).text, /looking into this|on it now|digging into this/i);
645
+ assert.match(t.sent.at(-1).text, /on it|looking now|digging in/i);
646
646
  });
647
647
  });
648
648
 
@@ -719,7 +719,7 @@ describe("a running retry is not declared dead", () => {
719
719
  now: t0 + 45 * 60_000, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false },
720
720
  });
721
721
  assert.equal(settled.willRetry, true);
722
- assert.match(t.sent.at(-1).text, /retrying it now/i);
722
+ assert.match(t.sent.at(-1).text, /retrying now/i);
723
723
 
724
724
  const stats = await sweepObligations({ now: t0 + 51 * 60_000, deps: { deliverImpl: t.impl, spokeSinceImpl: () => false } });
725
725
  assert.equal(stats.staled, 0, "the retry is minutes old, not fifty");
@@ -667,16 +667,32 @@ async function guardDynamicJobs({ event, agentRoot }, opts = {}) {
667
667
 
668
668
  const MESSAGING_CURSOR_REL = join("state", "messaging", "inbound-cursor.json");
669
669
 
670
- /** Read the org-messaging inbound cursor (seq). Fail-open → 0 (from genesis). */
670
+ /**
671
+ * Read the org-messaging inbound cursor (seq).
672
+ *
673
+ * Returns `null` when there is NO cursor on disk (or it is corrupt/unreadable) —
674
+ * NOT 0. The distinction is load-bearing, exactly as it is for the push channel
675
+ * (lib/org/push.readCursor): `cursor=0` on the events read means "replay the
676
+ * ENTIRE org ledger from genesis", so a fresh install that started at 0 would
677
+ * wake on every historical message in the workspace. `null` is the caller's cue
678
+ * to SEED from the push channel head instead (see guardMessagingInbound), so a
679
+ * (re)install starts at "now" and replays nothing. A real, persisted cursor
680
+ * (always > 0 in practice — see writeMessagingCursor) is returned as-is.
681
+ */
671
682
  function readMessagingCursor(agentRoot) {
672
683
  try {
673
684
  const p = join(agentRoot, MESSAGING_CURSOR_REL);
674
- if (!existsSync(p)) return 0;
685
+ if (!existsSync(p)) return null;
675
686
  const obj = JSON.parse(readFileSync(p, "utf8"));
676
687
  const c = Number(obj && obj.cursor);
677
- return Number.isFinite(c) ? c : 0;
688
+ // A non-finite / non-positive cursor is corrupt, not "from genesis" — treat it
689
+ // as absent (seed from head) rather than replaying all history. Aligned with
690
+ // lib/org/push.readCursor (`n > 0`): a persisted cursor is always > 0 (seeded
691
+ // from a head that is null-or->0, then only advanced forward), so 0 is never a
692
+ // real watermark — accepting it would replay the entire org ledger.
693
+ return Number.isFinite(c) && c > 0 ? c : null;
678
694
  } catch {
679
- return 0;
695
+ return null;
680
696
  }
681
697
  }
682
698
 
@@ -764,7 +780,34 @@ async function guardMessagingInbound({ event, agentRoot, log }, opts = {}) {
764
780
  opts.toInboxImpl || (await import("../../lib/channels/inbox-item.mjs")).eventToInboxItem;
765
781
  const writeInboxItem = opts.writeImpl || (await import("../poller/utils.mjs")).writeInboxItem;
766
782
 
767
- const cursor = readMessagingCursor(agentRoot);
783
+ let cursor = readMessagingCursor(agentRoot);
784
+ if (cursor == null) {
785
+ // FIRST RUN (or a corrupt cursor): no usable inbound cursor on disk. Do NOT
786
+ // start the poll lane at genesis — `events?cursor=0` replays the ENTIRE org
787
+ // ledger and wakes the fresh agent on every historical message. Seed from
788
+ // the PUSH CHANNEL HEAD instead: the SSE lane bootstraps at head and
789
+ // persists that seq to state/org/push-cursor.json, so it is the local
790
+ // "everything before this is history" watermark. Persist it as our starting
791
+ // cursor so a crash before the first advance cannot fall back to genesis.
792
+ let head = null;
793
+ try {
794
+ head = typeof opts.pushHeadImpl === "function"
795
+ ? opts.pushHeadImpl(agentRoot)
796
+ : (await import("../../lib/org/push.mjs")).readCursor(agentRoot);
797
+ } catch { head = null; }
798
+ if (head == null) {
799
+ // Neither lane has established a head yet (truly fresh install, push not
800
+ // yet connected). Replaying the backlog is the ONE thing we must not do,
801
+ // so hold this tick — the push channel writes a head shortly and the next
802
+ // tick seeds from it. Never silent: the reason says exactly why 0 routed.
803
+ if (typeof log === "function") {
804
+ log("info", "[messaging-inbound] fresh install: no inbound cursor and no push-channel head yet — holding this tick to avoid replaying the org backlog");
805
+ }
806
+ return { ok: true, decision: "inline", cadence: event.cadence, reason: "awaiting push-channel head to seed inbound cursor (fresh install)", routed: 0 };
807
+ }
808
+ cursor = Number(head);
809
+ writeMessagingCursor(agentRoot, cursor);
810
+ }
768
811
  // Pass the agent id explicitly. It is what suppresses the agent's OWN
769
812
  // outbound echoes — without it `me` is null, every echo-filter branch is
770
813
  // skipped, and the agent re-ingests its own replies as fresh inbound (a
@@ -445,6 +445,10 @@ test("messaging-inbound: pulls new directed items and routes them into the inbox
445
445
  { event: { cadence: "messaging-inbound" }, agentRoot: root, log: () => {} },
446
446
  {
447
447
  cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t", agentId: "me" } } },
448
+ // Fresh root → no inbound cursor. Seed from the push-channel head so the
449
+ // pull starts at "now" (not genesis). The stub stands in for
450
+ // lib/org/push.readCursor.
451
+ pushHeadImpl: () => 5,
448
452
  pullImpl: async () => ({ events: [ev1, ev2], nextCursor: 12 }),
449
453
  toInboxImpl: (e) => ({ id: e.message_id, service: "cohort", content: e.text }),
450
454
  writeImpl: (service, item) => written.push({ service, item }),
@@ -461,20 +465,67 @@ test("messaging-inbound: pulls new directed items and routes them into the inbox
461
465
  assert.equal(JSON.parse(readFileSync(cursorFile, "utf8")).cursor, 12);
462
466
  });
463
467
 
468
+ test("messaging-inbound: FIRST RUN seeds the inbound cursor from the push-channel head (no backlog replay)", async () => {
469
+ const root = tmpRoot();
470
+ let pulledCursor = null;
471
+ const res = await guardMessagingInbound(
472
+ { event: { cadence: "messaging-inbound" }, agentRoot: root, log: () => {} },
473
+ {
474
+ cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t" } } },
475
+ // The push channel has advanced to seq 4200. A fresh install must start
476
+ // there, NOT at 0 (which would replay the whole org ledger).
477
+ pushHeadImpl: () => 4200,
478
+ pullImpl: async ({ cursor }) => { pulledCursor = cursor; return { events: [], nextCursor: cursor }; },
479
+ toInboxImpl: () => ({}),
480
+ writeImpl: () => { throw new Error("should not write an item"); },
481
+ },
482
+ );
483
+ assert.equal(res.ok, true);
484
+ assert.equal(res.decision, "inline");
485
+ assert.equal(pulledCursor, 4200, "the pull started at the push head, not at genesis");
486
+ const cursorFile = join(root, "state", "messaging", "inbound-cursor.json");
487
+ assert.ok(existsSync(cursorFile), "the seeded head is persisted so a crash cannot fall back to genesis");
488
+ assert.equal(JSON.parse(readFileSync(cursorFile, "utf8")).cursor, 4200);
489
+ });
490
+
491
+ test("messaging-inbound: FIRST RUN with no push head yet HOLDS the tick (never replays from genesis)", async () => {
492
+ const root = tmpRoot();
493
+ const res = await guardMessagingInbound(
494
+ { event: { cadence: "messaging-inbound" }, agentRoot: root, log: () => {} },
495
+ {
496
+ cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t" } } },
497
+ pushHeadImpl: () => null, // push channel has not established a head yet
498
+ pullImpl: async () => { throw new Error("must not pull before a head exists"); },
499
+ toInboxImpl: () => ({}),
500
+ writeImpl: () => { throw new Error("must not write"); },
501
+ },
502
+ );
503
+ assert.equal(res.ok, true);
504
+ assert.equal(res.decision, "inline");
505
+ assert.equal(res.routed, 0);
506
+ assert.match(res.reason, /awaiting push-channel head/);
507
+ assert.ok(!existsSync(join(root, "state", "messaging", "inbound-cursor.json")), "no cursor seeded while we wait for a head");
508
+ });
509
+
464
510
  test("messaging-inbound: no new directed items → inline, routed 0, cursor unchanged", async () => {
465
511
  const root = tmpRoot();
512
+ // A cursor already exists (not a first run), so seeding is skipped and the pull
513
+ // starts from the persisted seq. Nothing new → the cursor must not move.
514
+ const cursorFile = join(root, "state", "messaging", "inbound-cursor.json");
515
+ mkdirSync(join(root, "state", "messaging"), { recursive: true });
516
+ writeFileSync(cursorFile, JSON.stringify({ cursor: 9, updatedAt: "2026-01-01T00:00:00Z" }));
466
517
  const res = await guardMessagingInbound(
467
518
  { event: { cadence: "messaging-inbound" }, agentRoot: root, log: () => {} },
468
519
  {
469
520
  cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t" } } },
470
- pullImpl: async ({ cursor }) => ({ events: [], nextCursor: cursor }),
521
+ pullImpl: async ({ cursor }) => { assert.equal(cursor, 9, "pull starts from the persisted cursor"); return { events: [], nextCursor: cursor }; },
471
522
  toInboxImpl: () => ({}),
472
523
  writeImpl: () => { throw new Error("should not write"); },
473
524
  },
474
525
  );
475
526
  assert.equal(res.decision, "inline");
476
527
  assert.equal(res.routed, 0);
477
- assert.ok(!existsSync(join(root, "state", "messaging", "inbound-cursor.json")), "no cursor write when nothing advanced");
528
+ assert.equal(JSON.parse(readFileSync(cursorFile, "utf8")).cursor, 9, "cursor unchanged when nothing advanced");
478
529
  });
479
530
 
480
531
  test("messaging-inbound: org not configured → inline no-op (fail-open)", async () => {
@@ -496,6 +547,7 @@ test("messaging-inbound: MESSAGING_INBOUND_ESCALATE=1 escalates when items arriv
496
547
  { event: { cadence: "messaging-inbound" }, agentRoot: tmpRoot(), log: () => {} },
497
548
  {
498
549
  cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t" } } },
550
+ pushHeadImpl: () => 1,
499
551
  pullImpl: async () => ({ events: [{ message_id: "m1", text: "@me" }], nextCursor: 1 }),
500
552
  toInboxImpl: (e) => ({ id: e.message_id }),
501
553
  writeImpl: () => {},
@@ -514,6 +566,7 @@ test("messaging-inbound: fully fail-open (pull throws → inline, never throws)"
514
566
  { event: { cadence: "messaging-inbound" }, agentRoot: tmpRoot(), log: () => {} },
515
567
  {
516
568
  cfg: { org: { cohort: { enabled: true, base: "https://x", token: "t" } } },
569
+ pushHeadImpl: () => 3,
517
570
  pullImpl: async () => { throw new Error("network down"); },
518
571
  },
519
572
  );
@@ -737,6 +790,7 @@ test("messaging-inbound (JOINT 3): the DEFAULT reader is wide — no family filt
737
790
  {
738
791
  cfg: { org: { cohort: { enabled: true, base: "https://x.test", token: "t" } } },
739
792
  agentId: ME,
793
+ pushHeadImpl: () => 1, // fresh root: seed past genesis so the ledger read runs
740
794
  fetchImpl,
741
795
  toInboxImpl: (e) => ({ id: e.id, service: "cohort", kind: e.kind, content: e.text }),
742
796
  writeImpl: (service, item) => written.push({ service, item }),
@@ -765,6 +819,7 @@ test("messaging-inbound (JOINT 3): MESSAGING_INBOUND_WIDE=0 falls back to the na
765
819
  {
766
820
  cfg: { org: { cohort: { enabled: true, base: "https://x.test", token: "t" } } },
767
821
  agentId: "member-me",
822
+ pushHeadImpl: () => 1, // fresh root: seed past genesis so the read runs
768
823
  // The narrow reader pins family=messaging,calling on its URL; assert the
769
824
  // degradation is announced rather than inferred from an empty inbox.
770
825
  fetchImpl: async () => ({ ok: true, status: 200, headers: { get: () => "application/json" }, text: async () => JSON.stringify({ events: [], nextCursor: 0 }), json: async () => ({ events: [], nextCursor: 0 }) }),
@@ -243,7 +243,8 @@ Your job: classify each incoming message and return a JSON object with exactly t
243
243
  "model": "opus" | "sonnet",
244
244
  "summary": "<one-line summary of the message>",
245
245
  "category": "action_required" | "fyi" | "ignore",
246
- "directed_at_agent": true | false
246
+ "directed_at_agent": true | false,
247
+ "answerable": true | false
247
248
  }
248
249
  ${peerContext}
249
250
  Classification rules:
@@ -271,6 +272,10 @@ CATEGORY:
271
272
  - "fyi": Informational, no action needed
272
273
  - "ignore": No value
273
274
 
275
+ ANSWERABLE (can ${agentName} answer this DIRECTLY, in one reply, right now?):
276
+ - true: A question that can be answered from what ${agentName} already knows or can see, needing NO tools, NO external lookups, NO drafting, and NO multi-step work — e.g. "what machine are you running on?", "which model are you?", "are you online?", "what's your role?", a simple factual or conversational question. These get answered immediately instead of a "I'll look into it" holding message, even from the ${principalTitle}.
277
+ - false: Anything that needs an action performed (send, schedule, file, push, draft, research, look something up), anything multi-step, or anything that is not a direct question. When unsure, use false — a wrong direct answer is worse than a brief wait.
278
+
274
279
  DIRECTED_AT_AGENT (CRITICAL — determines whether ${agentName} should respond. Default to false in channels/group chats.):
275
280
  - true: The message is a DM to ${agentName} (1:1)
276
281
  - true: The message explicitly mentions ${agentName} by name or @mention
@@ -300,43 +305,46 @@ Context:
300
305
  Examples:
301
306
 
302
307
  Input: CEO DM on Slack: "Can you send the board resolution to Jordan?"
303
- Output: {"priority":"critical","action":"respond","model":"opus","summary":"CEO requesting board resolution sent to GC Jordan","category":"action_required","directed_at_agent":true}
308
+ Output: {"priority":"critical","action":"respond","model":"opus","summary":"CEO requesting board resolution sent to GC Jordan","category":"action_required","directed_at_agent":true,"answerable":false}
309
+
310
+ Input: CEO DM on Slack: "what kind of machine are you running on?"
311
+ Output: {"priority":"critical","action":"respond","model":"opus","summary":"CEO asking what machine/infrastructure the agent runs on","category":"action_required","directed_at_agent":true,"answerable":true}
304
312
 
305
313
  Input: GitHub notification email about a merged PR
306
- Output: {"priority":"ignore","action":"ignore","model":"sonnet","summary":"GitHub PR merge notification","category":"ignore","directed_at_agent":false}
314
+ Output: {"priority":"ignore","action":"ignore","model":"sonnet","summary":"GitHub PR merge notification","category":"ignore","directed_at_agent":false,"answerable":false}
307
315
 
308
316
  Input: Leadership Slack: "We need the the regulator gap analysis updated by Thursday"
309
- Output: {"priority":"high","action":"draft","model":"opus","summary":"Leadership requesting the regulator gap analysis update by Thursday deadline","category":"action_required","directed_at_agent":true}
317
+ Output: {"priority":"high","action":"draft","model":"opus","summary":"Leadership requesting the regulator gap analysis update by Thursday deadline","category":"action_required","directed_at_agent":true,"answerable":false}
310
318
 
311
319
  Input: Team member: "Standup notes from today's sync"
312
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Team standup notes shared","category":"fyi","directed_at_agent":false}
320
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Team standup notes shared","category":"fyi","directed_at_agent":false,"answerable":false}
313
321
 
314
322
  Input: External email: "Following up on our conversation about the Singapore entity"
315
- Output: {"priority":"normal","action":"respond","model":"sonnet","summary":"External follow-up on Singapore entity discussion","category":"action_required","directed_at_agent":true}
323
+ Output: {"priority":"normal","action":"respond","model":"sonnet","summary":"External follow-up on Singapore entity discussion","category":"action_required","directed_at_agent":true,"answerable":false}
316
324
 
317
325
  Input: Channel message from Jacob: "@${agentName} can you check the engine deployment?"
318
- Output: {"priority":"normal","action":"respond","model":"sonnet","summary":"Jacob asking ${agentName} to check engine deployment","category":"action_required","directed_at_agent":true}
326
+ Output: {"priority":"normal","action":"respond","model":"sonnet","summary":"Jacob asking ${agentName} to check engine deployment","category":"action_required","directed_at_agent":true,"answerable":false}
319
327
 
320
328
  Input: Channel message from CEO: "@AnotherAgent can you review the agentic libraries in this channel?"
321
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"CEO asking another agent to review agentic libraries — not directed at ${agentName}","category":"fyi","directed_at_agent":false}
329
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"CEO asking another agent to review agentic libraries — not directed at ${agentName}","category":"fyi","directed_at_agent":false,"answerable":false}
322
330
 
323
331
  Input: Channel message from Sam to Jordan: "Jordan, did you file the compliance response yet?"
324
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Sam asking Jordan about compliance response filing","category":"fyi","directed_at_agent":false}
332
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Sam asking Jordan about compliance response filing","category":"fyi","directed_at_agent":false,"answerable":false}
325
333
 
326
334
  Input: Channel thread — Jacob: "pushed the fix" / Sam: "nice, looks good"
327
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Team exchange about code fix","category":"fyi","directed_at_agent":false}
335
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Team exchange about code fix","category":"fyi","directed_at_agent":false,"answerable":false}
328
336
 
329
337
  Input: Thread where ${agentName} previously replied — Jacob: "yeah that makes sense, I'll handle it" (responding to Sam, not ${agentName})
330
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Jacob acknowledging Sam in thread","category":"fyi","directed_at_agent":false}
338
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Jacob acknowledging Sam in thread","category":"fyi","directed_at_agent":false,"answerable":false}
331
339
 
332
340
  Input: Thread where ${agentName} previously replied — Sam: "${agentName}, can you send the updated doc?"
333
- Output: {"priority":"high","action":"respond","model":"sonnet","summary":"Sam asking ${agentName} for updated document","category":"action_required","directed_at_agent":true}
341
+ Output: {"priority":"high","action":"respond","model":"sonnet","summary":"Sam asking ${agentName} for updated document","category":"action_required","directed_at_agent":true,"answerable":false}
334
342
 
335
343
  Input: Channel message: "FYI — the Cayman entity docs are signed and filed"
336
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"FYI: Cayman entity docs signed and filed","category":"fyi","directed_at_agent":false}
344
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"FYI: Cayman entity docs signed and filed","category":"fyi","directed_at_agent":false,"answerable":false}
337
345
 
338
346
  Input: Group chat — Jordan: "Sam, can we sync on the the regulator gaps tomorrow?"
339
- Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Jordan asking Sam to sync on the regulator gaps","category":"fyi","directed_at_agent":false}
347
+ Output: {"priority":"normal","action":"archive","model":"sonnet","summary":"Jordan asking Sam to sync on the regulator gaps","category":"fyi","directed_at_agent":false,"answerable":false}
340
348
 
341
349
  Return ONLY the JSON object. No explanation, no markdown fencing, no extra text.`;
342
350
  }
@@ -407,6 +415,10 @@ function parseClassification(text) {
407
415
  summary: typeof parsed.summary === "string" ? parsed.summary : "Unclassified message",
408
416
  category: validCategories.includes(parsed.category) ? parsed.category : "fyi",
409
417
  directed_at_agent: typeof parsed.directed_at_agent === "boolean" ? parsed.directed_at_agent : false, // default false — don't respond unless we're confident the message is for this agent
418
+ // A directly-answerable question gets a quick reply instead of a session +
419
+ // holding ack (see responder.isQuickReply). Default false — when the field
420
+ // is absent or non-boolean, take the session path, never guess "answerable".
421
+ answerable: parsed.answerable === true,
410
422
  };
411
423
  }
412
424
 
@@ -781,5 +793,6 @@ function classifiedAttrs(r) {
781
793
  model: r.model,
782
794
  category: r.category,
783
795
  directed_at_agent: r.directed_at_agent,
796
+ answerable: r.answerable,
784
797
  };
785
798
  }
@@ -384,6 +384,29 @@ test("RUNG: the quick-reply responder accepts rungs 0-1 and refuses 2+", async (
384
384
  );
385
385
  });
386
386
 
387
+ test("ANSWERABLE: a directly-answerable question is quick-replied, not sessioned + acked", () => {
388
+ // "what machine are you running on?" from the CEO classifies critical/opus,
389
+ // which would otherwise force a session and a robotic holding ack. Flagged
390
+ // answerable, it is answered in the turn instead — the defect being fixed.
391
+ assert.equal(
392
+ responder.isQuickReply({ action: "respond", priority: "critical", model: "opus", answerable: true }),
393
+ true,
394
+ "an answerable question skips the session even at critical/opus",
395
+ );
396
+ // The flag only fast-paths a `respond` — a draft/research item stays on the
397
+ // session path even if the classifier also (contradictorily) flags it.
398
+ assert.equal(responder.isQuickReply({ action: "draft", priority: "high", model: "opus", answerable: true }), false);
399
+ assert.equal(responder.isQuickReply({ action: "research", priority: "normal", model: "sonnet", answerable: true }), false);
400
+ // And the rung veto still wins over answerable.
401
+ assert.equal(
402
+ responder.isQuickReply({ action: "respond", priority: "critical", model: "opus", answerable: true }, { rung: 3 }),
403
+ false,
404
+ "a rung-3 answer needs a session even when the classifier called it answerable",
405
+ );
406
+ // Without the flag, a critical/opus item still takes the session path.
407
+ assert.equal(responder.isQuickReply({ action: "respond", priority: "critical", model: "opus" }), false);
408
+ });
409
+
387
410
  // ---------------------------------------------------------------------------
388
411
  // 8) REGRESSION (live run): with no need estimate every DM routed to rung 3,
389
412
  // which vetoed the quick reply and spawned a session for a two-line message.
@@ -199,38 +199,63 @@ export function promoteDeferred(channel, agentRoot) {
199
199
  matches.push({
200
200
  file,
201
201
  timestamp: readScalar(body, "timestamp") || "",
202
+ // THE SURFACE within the channel. `channel_id` names the ROOM; a single
203
+ // room carries several independent conversations at once — a directed 1:1
204
+ // DM thread AND the actor's own doc/board/decision events that resolve to
205
+ // the same channel_id. Bundling "latest-wins across the whole channel"
206
+ // folded those unrelated surfaces into one and silently dropped
207
+ // (`.processed-bundled`) a directed DM because a newer doc-comment event
208
+ // arrived in the same room. thread_context only ever carries the SAME
209
+ // thread's history, so bundling is sound ONLY within one surface. Key on
210
+ // thread_id, else the surface's own scope_id, else "" (a bare-channel
211
+ // burst — the original same-thread case, preserved).
212
+ surface:
213
+ readScalar(body, "thread_id") ||
214
+ readScalar(body, "scope_id") ||
215
+ "",
202
216
  });
203
217
  }
204
218
  if (matches.length === 0) continue;
205
219
 
206
- // Latest-wins: lex-sort ISO timestamps, take the most recent.
207
- matches.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
208
- const latest = matches[matches.length - 1];
209
- const older = matches.slice(0, -1);
210
-
211
- // Promote the latest back to live inbox.
212
- try {
213
- const live = latest.file.replace(/\.deferred$/, "");
214
- renameSync(join(inboxDir, latest.file), join(inboxDir, live));
215
- result.promoted++;
216
- result.service = service;
217
- } catch {
218
- // If the rename failed (race with another process), leave it
219
- // as .deferred — the next session-close will retry.
220
+ // Group by surface within the channel, then bundle latest-wins WITHIN each
221
+ // group. Distinct surfaces each promote their own latest; none is dropped
222
+ // just because a newer message landed on a different surface in the room.
223
+ const groups = new Map();
224
+ for (const m of matches) {
225
+ if (!groups.has(m.surface)) groups.set(m.surface, []);
226
+ groups.get(m.surface).push(m);
220
227
  }
221
228
 
222
- // Mark older bursts as bundled — their content is already part of
223
- // the latest item's thread_context, so they don't need their own
224
- // session, but we preserve the file for audit.
225
- for (const m of older) {
229
+ for (const group of groups.values()) {
230
+ // Latest-wins: lex-sort ISO timestamps, take the most recent.
231
+ group.sort((a, b) => a.timestamp.localeCompare(b.timestamp));
232
+ const latest = group[group.length - 1];
233
+ const older = group.slice(0, -1);
234
+
235
+ // Promote the latest back to live inbox.
226
236
  try {
227
- renameSync(
228
- join(inboxDir, m.file),
229
- join(inboxDir, m.file.replace(/\.deferred$/, ".processed-bundled"))
230
- );
231
- result.bundled++;
237
+ const live = latest.file.replace(/\.deferred$/, "");
238
+ renameSync(join(inboxDir, latest.file), join(inboxDir, live));
239
+ result.promoted++;
240
+ result.service = service;
232
241
  } catch {
233
- // Best-effort; missing file is fine.
242
+ // If the rename failed (race with another process), leave it
243
+ // as .deferred — the next session-close will retry.
244
+ }
245
+
246
+ // Mark older bursts as bundled — their content is already part of
247
+ // the latest item's thread_context, so they don't need their own
248
+ // session, but we preserve the file for audit.
249
+ for (const m of older) {
250
+ try {
251
+ renameSync(
252
+ join(inboxDir, m.file),
253
+ join(inboxDir, m.file.replace(/\.deferred$/, ".processed-bundled"))
254
+ );
255
+ result.bundled++;
256
+ } catch {
257
+ // Best-effort; missing file is fine.
258
+ }
234
259
  }
235
260
  }
236
261
  }
@@ -12,11 +12,13 @@ function makeAgentRoot() {
12
12
  return root;
13
13
  }
14
14
 
15
- function writeInboxItem(root, name, { channel_id, timestamp, id = "test-id" }) {
15
+ function writeInboxItem(root, name, { channel_id, timestamp, id = "test-id", thread_id, scope_id }) {
16
16
  const body = [
17
17
  `id: "${id}"`,
18
18
  `service: "slack"`,
19
19
  `channel_id: "${channel_id}"`,
20
+ ...(thread_id ? [`thread_id: "${thread_id}"`] : []),
21
+ ...(scope_id ? [`scope_id: "${scope_id}"`] : []),
20
22
  `timestamp: "${timestamp}"`,
21
23
  `content: |`,
22
24
  ` body`,
@@ -118,6 +120,42 @@ test("promoteDeferred with multiple items: keeps latest, bundles rest", () => {
118
120
  }
119
121
  });
120
122
 
123
+ test("promoteDeferred bundles WITHIN a surface, not across the whole channel", () => {
124
+ // A directed 1:1 DM thread AND the actor's doc-comment events resolve to the
125
+ // same channel_id. Bundling latest-wins across the channel would drop the DM
126
+ // just because a newer doc event landed. Each surface must promote its own
127
+ // latest; nothing is bundled across surfaces.
128
+ const root = makeAgentRoot();
129
+ try {
130
+ // The directed DM (one message, thread T-DM).
131
+ writeInboxItem(root, "dm-1.yaml.deferred", {
132
+ channel_id: "room-1", thread_id: "T-DM", id: "dm-1",
133
+ timestamp: "2026-05-13T00:00:30Z",
134
+ });
135
+ // Two doc-comment events on the same room but a different surface (scope S-DOC),
136
+ // the later of which would have "won" under the old channel-wide bundling.
137
+ writeInboxItem(root, "doc-1.yaml.deferred", {
138
+ channel_id: "room-1", scope_id: "S-DOC", id: "doc-1",
139
+ timestamp: "2026-05-13T00:01:00Z",
140
+ });
141
+ writeInboxItem(root, "doc-2.yaml.deferred", {
142
+ channel_id: "room-1", scope_id: "S-DOC", id: "doc-2",
143
+ timestamp: "2026-05-13T00:02:00Z",
144
+ });
145
+ const r = promoteDeferred("room-1", root);
146
+ assert.equal(r.promoted, 2, "the DM and the latest doc event are BOTH promoted");
147
+ assert.equal(r.bundled, 1, "only the older doc event (same surface) is bundled");
148
+ const files = readdirSync(join(root, "state", "inbox", "slack")).sort();
149
+ assert.deepEqual(files, [
150
+ "dm-1.yaml", // the directed DM survives — NOT dropped
151
+ "doc-1.yaml.processed-bundled", // older doc event folded into doc-2
152
+ "doc-2.yaml", // latest doc event promoted
153
+ ]);
154
+ } finally {
155
+ rmSync(root, { recursive: true, force: true });
156
+ }
157
+ });
158
+
121
159
  test("promoteDeferred ignores items in other channels", () => {
122
160
  const root = makeAgentRoot();
123
161
  try {
@@ -901,6 +901,15 @@ export async function sendQuickResponse(item, classResult, routed = null) {
901
901
  via: delivered.via,
902
902
  ...(delivered.channel ? { channel: delivered.channel } : {}),
903
903
  ...(delivered.draft_path ? { draft_path: delivered.draft_path } : {}),
904
+ // PROPAGATE THE PERMANENCE. `deliver` marks a failure the send-gate refused
905
+ // (FORBIDDEN_SCOPE) or hq declared unroutable (NOT_FOUND/BAD_REQUEST) as
906
+ // `permanent` with the RPC `code`. Without carrying these up, the caller
907
+ // (agent-daemon.answerItem) could not tell a transient blip from an
908
+ // impossible send, so it fell through to a full autonomous session — which
909
+ // then posted UNRELATED work into a channel this seat was just told it may
910
+ // not write to. The daemon uses `permanent` to escalate instead of spawn.
911
+ ...(delivered.permanent ? { permanent: true } : {}),
912
+ ...(delivered.code ? { code: delivered.code } : {}),
904
913
  ...(delivered.error ? { error: delivered.error } : {}),
905
914
  };
906
915
 
@@ -1066,6 +1075,17 @@ export function isQuickReply(classResult, routed = null) {
1066
1075
  // place, or a caller that has no ladder) leaves the original rules untouched.
1067
1076
  if (!rungPermitsQuickReply(routed)) return false;
1068
1077
 
1078
+ // ── directly-answerable question: answer it, don't ack it ────────────────
1079
+ // A factual question the classifier judged answerable in one reply — no
1080
+ // tools, no drafting, no multi-step work ("what machine are you running on?")
1081
+ // — belongs on the quick path even from the CEO and even when the model or
1082
+ // priority would otherwise route it to a session. Spawning a 15-minute
1083
+ // session and posting a robotic "I'm looking into this now (CEO asking…)" to
1084
+ // a question that could be answered in one line IS the defect being fixed.
1085
+ // Gated to `respond` so a `draft`/`research` item that also gets flagged
1086
+ // answerable (a classifier contradiction) still takes the session path.
1087
+ if (classResult.answerable === true && classResult.action === "respond") return true;
1088
+
1069
1089
  // These always need full sessions — never quick reply
1070
1090
  if (classResult.action === "research") return false;
1071
1091
  if (classResult.action === "draft") return false;