@cohortapp/agent-sdk 2.11.10 → 2.11.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,58 @@
1
+ /**
2
+ * reactive-gate — a process-local signal that a REACTIVE turn (a human or peer
3
+ * agent is waiting on a reply) is actively consuming the shared Max subscription.
4
+ *
5
+ * WHY THIS EXISTS. The scarce resource on a mini is not a dispatch slot — it is
6
+ * the ONE Max subscription's inference throughput, which tracks the number of
7
+ * concurrent `claude --print` processes. When several heavy autonomous BACKLOG
8
+ * sessions run at once they saturate that subscription, and the next reactive
9
+ * reply's own claude call is starved: a quick reply measured 19s and a classify
10
+ * actually timed out at 30s while three opus backlog sessions held the line.
11
+ *
12
+ * The governor reads {@link reactiveInFlight} to YIELD the subscription: while
13
+ * any reactive turn is in flight, self-directed backlog admission is held to
14
+ * REACTIVE_INFLIGHT_BACKLOG_MAX so a reply is never starved. The dispatcher
15
+ * reads the same signal to evict already-running backlog down to that cap.
16
+ *
17
+ * A COUNTER, not a boolean, so concurrent reactive turns nest correctly and
18
+ * backlog only resumes when the LAST reply has landed. Process-local by design:
19
+ * the daemon, dispatcher, and governor share one Node process, so a module-level
20
+ * count is visible to all three with no IPC. `claude --print` children are
21
+ * separate processes and are governed through the count, not this module.
22
+ */
23
+
24
+ let _inFlight = 0;
25
+ let _peak = 0;
26
+
27
+ /** Mark the start of a reactive turn. Returns the new in-flight count. */
28
+ export function markReactiveStart() {
29
+ _inFlight += 1;
30
+ if (_inFlight > _peak) _peak = _inFlight;
31
+ return _inFlight;
32
+ }
33
+
34
+ /**
35
+ * Mark the end of a reactive turn. Clamped at zero so an unbalanced end (a
36
+ * double-finally, a test) can never drive the count negative and wedge backlog
37
+ * off permanently. Returns the new in-flight count.
38
+ */
39
+ export function markReactiveEnd() {
40
+ if (_inFlight > 0) _inFlight -= 1;
41
+ return _inFlight;
42
+ }
43
+
44
+ /** How many reactive turns are consuming the subscription right now. */
45
+ export function reactiveInFlight() {
46
+ return _inFlight;
47
+ }
48
+
49
+ /** Highest concurrent reactive count seen since the last reset (diagnostics). */
50
+ export function reactivePeak() {
51
+ return _peak;
52
+ }
53
+
54
+ /** Test seam — reset the process-local counters between cases. */
55
+ export function _resetReactiveGate() {
56
+ _inFlight = 0;
57
+ _peak = 0;
58
+ }
@@ -0,0 +1,57 @@
1
+ /**
2
+ * reactive-gate — the process-local counter that tells the governor + dispatcher
3
+ * a human/peer reply is actively consuming the subscription.
4
+ */
5
+ import { test } from "node:test";
6
+ import assert from "node:assert/strict";
7
+ import {
8
+ markReactiveStart,
9
+ markReactiveEnd,
10
+ reactiveInFlight,
11
+ reactivePeak,
12
+ _resetReactiveGate,
13
+ } from "./reactive-gate.mjs";
14
+
15
+ test("starts at zero", () => {
16
+ _resetReactiveGate();
17
+ assert.equal(reactiveInFlight(), 0);
18
+ });
19
+
20
+ test("start/end move the count by exactly one and balance to zero", () => {
21
+ _resetReactiveGate();
22
+ assert.equal(markReactiveStart(), 1);
23
+ assert.equal(reactiveInFlight(), 1);
24
+ assert.equal(markReactiveEnd(), 0);
25
+ assert.equal(reactiveInFlight(), 0);
26
+ });
27
+
28
+ test("concurrent reactive turns NEST — backlog resumes only when the last ends", () => {
29
+ _resetReactiveGate();
30
+ markReactiveStart();
31
+ markReactiveStart();
32
+ assert.equal(reactiveInFlight(), 2, "two turns in flight");
33
+ markReactiveEnd();
34
+ assert.equal(reactiveInFlight(), 1, "still one waiting — do NOT resume backlog yet");
35
+ markReactiveEnd();
36
+ assert.equal(reactiveInFlight(), 0, "now backlog may resume");
37
+ });
38
+
39
+ test("end is clamped at zero — an unbalanced end can never wedge backlog off", () => {
40
+ _resetReactiveGate();
41
+ assert.equal(markReactiveEnd(), 0, "end with nothing in flight stays at 0");
42
+ assert.equal(markReactiveEnd(), 0, "and again");
43
+ // A subsequent real turn still registers correctly (not driven negative).
44
+ assert.equal(markReactiveStart(), 1);
45
+ assert.equal(reactiveInFlight(), 1);
46
+ });
47
+
48
+ test("peak records the high-water mark since reset", () => {
49
+ _resetReactiveGate();
50
+ markReactiveStart();
51
+ markReactiveStart();
52
+ markReactiveStart();
53
+ markReactiveEnd();
54
+ markReactiveEnd();
55
+ assert.equal(reactivePeak(), 3, "peak holds even after turns end");
56
+ assert.equal(reactiveInFlight(), 1);
57
+ });
@@ -63,6 +63,7 @@ import os from "node:os";
63
63
  import { execFileSync } from "node:child_process";
64
64
  import { existsSync, readFileSync } from "node:fs";
65
65
  import { join, resolve } from "node:path";
66
+ import { reactiveInFlight } from "./reactive-gate.mjs";
66
67
 
67
68
  // ---------------------------------------------------------------------------
68
69
  // Constants
@@ -103,6 +104,19 @@ export const REACTIVE_RESERVE = num(process.env.GOV_REACTIVE_RESERVE, 2);
103
104
  * gets near-all capacity.
104
105
  */
105
106
  export const DEGRADED_BACKLOG_MAX = num(process.env.GOV_DEGRADED_BACKLOG_MAX, 1);
107
+ /**
108
+ * The most self-directed BACKLOG sessions allowed to run while a REACTIVE turn
109
+ * is in flight (a human/peer is waiting on a reply). REACTIVE_RESERVE keeps a
110
+ * dispatch slot free, but slots are not the scarce resource — the ONE Max
111
+ * subscription's throughput is, and several concurrent backlog opus sessions
112
+ * saturate it, starving the reply's own claude call (measured: a 19s quick reply
113
+ * and a 30s classify timeout under three concurrent backlog sessions). So while
114
+ * a reply generates, backlog admission is capped HARD to this (default 1): the
115
+ * subscription is left almost entirely to the human, and backlog resumes the
116
+ * instant the reply lands. The dispatcher evicts already-running backlog down to
117
+ * the same cap. 0 pauses backlog completely during a reactive turn.
118
+ */
119
+ export const REACTIVE_INFLIGHT_BACKLOG_MAX = num(process.env.GOV_REACTIVE_INFLIGHT_BACKLOG_MAX, 1);
106
120
  /**
107
121
  * A `claude` session older than this is treated as STALE and does NOT count
108
122
  * toward the concurrency ceiling — an idle/wedged lane must not hold a slot
@@ -368,6 +382,10 @@ export function defaultDeps(extra = {}) {
368
382
  liveClaude: live,
369
383
  throttleCeiling: throttle.ceiling,
370
384
  throttleReason: throttle.reason,
385
+ // How many reactive turns are consuming the subscription right now. admit()
386
+ // caps backlog HARD while this is > 0 so a waiting human's reply is never
387
+ // starved. Injectable via `extra` for tests; live from the gate otherwise.
388
+ reactiveInFlight: reactiveInFlight(),
371
389
  ...extra,
372
390
  };
373
391
  }
@@ -500,22 +518,35 @@ export function admit(req = {}, deps) {
500
518
  // can always be GENERATED (10 parallel sessions once starved the
501
519
  // quick-reply's claude call into a 60s timeout). A DEGRADED/unfunded seat
502
520
  // throttles backlog to DEGRADED_BACKLOG_MAX — a trickle in idle gaps.
521
+ // Is a reactive turn (a human/peer waiting on a reply) generating right now?
522
+ // While one is, self-directed backlog yields the SUBSCRIPTION — not merely a
523
+ // slot — so the reply's own claude call isn't starved. This is a hard cap
524
+ // ABOVE the REACTIVE_RESERVE math: reserve keeps steady-state headroom; this
525
+ // clamps to a trickle for the seconds a human is actually waiting.
526
+ const reactiveBusy = numField(d.reactiveInFlight, 0) > 0;
503
527
  let sourceCeiling;
504
528
  if (humanReply) {
505
529
  sourceCeiling = effectiveMax;
530
+ } else if (reactiveBusy) {
531
+ sourceCeiling = Math.min(effectiveMax, REACTIVE_INFLIGHT_BACKLOG_MAX);
532
+ if (mode === "degraded" && source === "backlog") {
533
+ sourceCeiling = Math.min(sourceCeiling, DEGRADED_BACKLOG_MAX);
534
+ }
506
535
  } else if (mode === "degraded" && source === "backlog") {
507
536
  sourceCeiling = Math.min(effectiveMax, DEGRADED_BACKLOG_MAX);
508
537
  } else {
509
538
  sourceCeiling = Math.max(1, effectiveMax - REACTIVE_RESERVE);
510
539
  }
511
540
  if (liveCount >= sourceCeiling) {
512
- const why = mode === "degraded" && source === "backlog"
513
- ? `budget degraded: self-directed backlog throttled to ${sourceCeiling} (${liveCount} live; inbox unaffected)`
514
- : throttleCeiling < dMax
515
- ? `at throttled ceiling (${liveCount}/${sourceCeiling}; soft-throttle active)`
516
- : source === "inbox"
517
- ? `at concurrency ceiling (${liveCount}/${sourceCeiling})`
518
- : `self-directed work yields ${REACTIVE_RESERVE} slots to the inbox (${liveCount}/${sourceCeiling})`;
541
+ const why = reactiveBusy
542
+ ? `reactive turn in flight: self-directed ${source} yields the subscription, capped to ${sourceCeiling} (${liveCount} live)`
543
+ : mode === "degraded" && source === "backlog"
544
+ ? `budget degraded: self-directed backlog throttled to ${sourceCeiling} (${liveCount} live; inbox unaffected)`
545
+ : throttleCeiling < dMax
546
+ ? `at throttled ceiling (${liveCount}/${sourceCeiling}; soft-throttle active)`
547
+ : source === "inbox"
548
+ ? `at concurrency ceiling (${liveCount}/${sourceCeiling})`
549
+ : `self-directed work yields ${REACTIVE_RESERVE} slots to the inbox (${liveCount}/${sourceCeiling})`;
519
550
  return decide(QUEUE, why);
520
551
  }
521
552
 
@@ -25,6 +25,7 @@ import {
25
25
  FREEMEM_FLOOR_FRACTION,
26
26
  etimeToMs,
27
27
  reapStaleClaude,
28
+ REACTIVE_INFLIGHT_BACKLOG_MAX,
28
29
  } from "./resource-governor.mjs";
29
30
 
30
31
  test("availableMemoryBytes counts reclaimable pages, not just free ones", () => {
@@ -128,6 +129,57 @@ test("admit: inbox may use the whole envelope; the reactive reserve is for it",
128
129
  assert.match(r.reason, /concurrency ceiling/);
129
130
  });
130
131
 
132
+ // ---------------------------------------------------------------------------
133
+ // Reactive-in-flight: backlog yields the SUBSCRIPTION (not just a slot) to a
134
+ // human/peer reply that is actively generating.
135
+ // ---------------------------------------------------------------------------
136
+
137
+ test("admit: while a reactive turn is in flight, backlog is capped to REACTIVE_INFLIGHT_BACKLOG_MAX → QUEUE", () => {
138
+ // Default cap is 1. With reactiveInFlight and even a single backlog session
139
+ // live, a second backlog spawn QUEUEs — the subscription is left to the reply.
140
+ assert.equal(REACTIVE_INFLIGHT_BACKLOG_MAX, 1);
141
+ const r = admit(
142
+ { source: "backlog" },
143
+ deps({ reactiveInFlight: 1, liveClaude: { count: 1, rssMB: 300 } }),
144
+ );
145
+ assert.equal(r.decision, DECISIONS.QUEUE);
146
+ assert.match(r.reason, /reactive turn in flight/);
147
+ });
148
+
149
+ test("admit: reactive-in-flight cap is HARDER than the steady-state reserve", () => {
150
+ // With 3 live and NO reactive turn, backlog still ADMITs (reserve leaves 5).
151
+ assert.equal(
152
+ admit({ source: "backlog" }, deps({ liveClaude: { count: 3, rssMB: 900 } })).decision,
153
+ DECISIONS.ADMIT,
154
+ );
155
+ // The SAME 3 live but a reply generating → QUEUE (cap 1, already exceeded).
156
+ assert.equal(
157
+ admit({ source: "backlog" }, deps({ reactiveInFlight: 1, liveClaude: { count: 3, rssMB: 900 } })).decision,
158
+ DECISIONS.QUEUE,
159
+ );
160
+ });
161
+
162
+ test("admit: a human reply is NEVER gated by reactive-in-flight (it IS the reactive turn)", () => {
163
+ // inbox/humanReply keeps the whole envelope even while other reactive turns run.
164
+ assert.equal(
165
+ admit({ source: "inbox" }, deps({ reactiveInFlight: 2, liveClaude: { count: 4, rssMB: 1200 } })).decision,
166
+ DECISIONS.ADMIT,
167
+ );
168
+ assert.equal(
169
+ admit({ source: "backlog", humanReply: true }, deps({ reactiveInFlight: 2, liveClaude: { count: 4, rssMB: 1200 } })).decision,
170
+ DECISIONS.ADMIT,
171
+ );
172
+ });
173
+
174
+ test("admit: one backlog session is still allowed alongside a reply (cap 1, not 0)", () => {
175
+ // A single trickle of backlog is fine — the cap is 1, so with 0 live it ADMITs.
176
+ const r = admit(
177
+ { source: "backlog" },
178
+ deps({ reactiveInFlight: 1, liveClaude: { count: 0, rssMB: 0 } }),
179
+ );
180
+ assert.equal(r.decision, DECISIONS.ADMIT);
181
+ });
182
+
131
183
  test("admit: throttle ceiling clamps effectiveMax below dynamicMax → QUEUE earlier", () => {
132
184
  // dynamicMax 8 but soft-throttle pinned the ceiling to 3; 3 live → QUEUE.
133
185
  const r = admit({ source: "backlog" }, deps({ throttleCeiling: 3, liveClaude: { count: 3, rssMB: 1000 } }));
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@cohortapp/agent-sdk",
3
- "version": "2.11.10",
3
+ "version": "2.11.12",
4
4
  "description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -51,7 +51,8 @@ import { makeInboxScanPoller } from "../poller/inbox-scan-poller.mjs";
51
51
  import { existsSync as _existsSync } from "fs";
52
52
  import { isPriorityItem } from "../poller/utils.mjs";
53
53
  import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
54
- import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions } from "./dispatcher.mjs";
54
+ import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions, yieldBacklogForReactive } from "./dispatcher.mjs";
55
+ import { markReactiveStart, markReactiveEnd } from "../../lib/reactive-gate.mjs";
55
56
  import { buildPrompt } from "./prompt-builder.mjs";
56
57
  import { sendQuickResponse, sendHoldingMessage, isQuickReply } from "./responder.mjs";
57
58
  // ANSWER ASSURANCE (scripts/daemon/assurance.mjs). The guarantee that an ask is
@@ -499,6 +500,43 @@ async function electResponderVerdict(item, deps = {}) {
499
500
  * the in-process quick-reply responder; rung ≥2 requires a spawned session.
500
501
  * @returns {Promise<{ok:boolean, path:string, reason?:string}>}
501
502
  */
503
+ // Work-verbs that mean "do a task", not "answer a question" — a DM containing one
504
+ // needs the full classify → session path (it may be multi-step work), so it is
505
+ // NEVER fast-pathed to a shallow quick reply.
506
+ const FASTPATH_WORK_RE = /\b(analy[sz]e|create|build|writ(e|ing)|draft|prepare|research|review|compile|generate|implement|fix|debug|refactor|deploy|migrat(e|ing)|audit|investigate|summari[sz]e|put together|pull together|set up|look into|go through|work on|take care of|handle|organi[sz]e|plan out|figure out)\b/;
507
+ // A DM that reads like a question / greeting / quick ask — safe to answer directly.
508
+ const FASTPATH_ASK_RE = /(\?|^(what|how|when|where|who|why|which|is |are |was |were |do |did |does |can |could |would |will |have |has |should |hey|hi |hello|thanks|thank you|yo |quick (q|question)|you (there|around|up)))/i;
509
+
510
+ /**
511
+ * Decide whether a directed DM is a "plainly simple question" that can skip the
512
+ * LLM classifier and be answered directly. Returns a default classResult (the
513
+ * quick-reply shape) when it qualifies, else null (→ full classify path).
514
+ * Deliberately CONSERVATIVE: 1:1 DM, short, question-shaped, no work-verbs.
515
+ * `deps.fastPathClassify` can override (tests); `deps.disableFastPath` disables.
516
+ */
517
+ export function fastPathClassify(item, isDm, deps = {}) {
518
+ if (deps && typeof deps.fastPathClassify === "function") return deps.fastPathClassify(item, isDm);
519
+ if (deps && deps.disableFastPath) return null;
520
+ if (process.env.DAEMON_DISABLE_FASTPATH === "1") return null;
521
+ if (!isDm) return null; // 1:1 DMs only (never group/ambient)
522
+ const content = String(item.content || item.subject || "").trim();
523
+ if (!content || content.length > 240) return null; // short only
524
+ if (item.has_attachments || item.attachments) return null;
525
+ const lc = content.toLowerCase();
526
+ if (FASTPATH_WORK_RE.test(lc)) return null; // work-request → full path
527
+ if (!FASTPATH_ASK_RE.test(content)) return null; // must look like a question/greeting
528
+ return {
529
+ priority: "high", // a directed DM is latency-sensitive → preempts backlog
530
+ action: "respond",
531
+ model: "sonnet", // fast model; a simple question does not need opus
532
+ summary: content.slice(0, 80),
533
+ category: "action_required",
534
+ directed_at_agent: true,
535
+ answerable: true, // routes to the in-process quick-reply responder
536
+ fast_path: true,
537
+ };
538
+ }
539
+
502
540
  export async function answerItem(item, service, itemId, trace_id, deps = {}, routed = null) {
503
541
  // Resolve collaborators: real imports by default (production), injected fakes
504
542
  // only when a test supplies them. Names match the imported symbols 1:1.
@@ -514,7 +552,17 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
514
552
  const _electResponderVerdict = deps.electResponderVerdict
515
553
  || ((item2) => electResponderVerdict(item2, { electResponder: deps.electResponder, orgCfg: deps.orgCfg }));
516
554
  const _claudeAvailable = deps.claudeAvailable || claudeAvailable;
517
-
555
+ // Evicts already-running backlog to hand the subscription to this reactive
556
+ // turn; injectable so a test can stub the dispatcher's eviction.
557
+ const _yieldBacklog = deps.yieldBacklogForReactive || yieldBacklogForReactive;
558
+
559
+ // A reactive turn begins HERE and is marked until the function returns. While
560
+ // it is marked the governor QUEUEs new self-directed backlog (it yields the
561
+ // subscription, not just a slot), so the classify + reply calls below are not
562
+ // starved. markReactiveEnd runs in the finally so a filtered/ignored/errored
563
+ // item releases the gate exactly once. The heavier lever — evicting backlog
564
+ // that is ALREADY running — fires only just before a real generate (below).
565
+ markReactiveStart();
518
566
  try {
519
567
  const isDm = item.is_dm === true;
520
568
  // Carry the routing decision on the item so everything downstream of here —
@@ -523,23 +571,36 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
523
571
  // session has no idea it is discharging a planned obligation.
524
572
  if (routed) item.execution = routed;
525
573
 
526
- // Classify via Haiku API
527
- const classResult = await _classify({
528
- sender: item.sender || "unknown",
529
- sender_privilege: item.sender_privilege || item.priority_signals?.from_ceo ? "ceo" : "unknown",
530
- channel: item.channel || "unknown",
531
- service,
532
- content: item.content || item.subject || "",
533
- is_reply: item.is_reply || false,
534
- thread_context: item.thread_context || null,
535
- subject: item.subject || "",
536
- is_dm: isDm,
537
- is_group: !isDm && service === "slack",
538
- agent_in_thread: item.agent_in_thread === true,
539
- kind: item.kind || "message",
540
- });
574
+ // FAST PATH: a short, directed, question-shaped DM with no work-verbs does not
575
+ // need a ~5-8s LLM classification to decide "answer this" — skip the classify
576
+ // `claude` call entirely and go straight to a quick reply. This is the single
577
+ // biggest reactive-latency cut after the fetch fix (the two sequential claude
578
+ // calls dominate processItem). CONSERVATIVE by construction: only a plain
579
+ // short question qualifies; work-requests and anything ambiguous fall through
580
+ // to the full classify → route path below, so nothing is shallow-answered.
581
+ let classResult = fastPathClassify(item, isDm, deps);
582
+ if (classResult) {
583
+ recordClassification(true);
584
+ console.log(`[daemon] fast-path DM — skipped classify (${classResult.model}): ${classResult.summary}`);
585
+ } else {
586
+ // Classify via Haiku CLI
587
+ classResult = await _classify({
588
+ sender: item.sender || "unknown",
589
+ sender_privilege: item.sender_privilege || item.priority_signals?.from_ceo ? "ceo" : "unknown",
590
+ channel: item.channel || "unknown",
591
+ service,
592
+ content: item.content || item.subject || "",
593
+ is_reply: item.is_reply || false,
594
+ thread_context: item.thread_context || null,
595
+ subject: item.subject || "",
596
+ is_dm: isDm,
597
+ is_group: !isDm && service === "slack",
598
+ agent_in_thread: item.agent_in_thread === true,
599
+ kind: item.kind || "message",
600
+ });
541
601
 
542
- recordClassification(true);
602
+ recordClassification(true);
603
+ }
543
604
 
544
605
  logEvent("classifications", {
545
606
  item_id: itemId,
@@ -729,6 +790,12 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
729
790
  // is exactly today's behaviour.
730
791
  if (_isQuickReply(classResult, routed)) {
731
792
  console.log(`[daemon] Quick reply path for ${item.sender} (${classResult.model})`);
793
+ // Free the subscription NOW: evict any already-running backlog sessions
794
+ // down to REACTIVE_INFLIGHT_BACKLOG_MAX so this reply's own claude call
795
+ // runs on a nearly-idle subscription instead of queueing behind opus
796
+ // backlog (the measured cause of 19s quick replies / 30s classify
797
+ // timeouts). No-op when backlog is already at/under the cap.
798
+ try { _yieldBacklog(); } catch { /* fail-open: never block a reply on eviction */ }
732
799
  const result = await _sendQuickResponse(item, classResult, routed);
733
800
  if (result.sent) {
734
801
  markProcessed(item, service);
@@ -920,6 +987,9 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
920
987
  // recoverInFlight() to re-deliver on restart.
921
988
  recordInFlight(item, service);
922
989
  if (obligationKeyForItem) noteSession(obligationKeyForItem, null);
990
+ // A reactive SESSION (complex reply) also competes for the subscription —
991
+ // yield already-running backlog to it, same as the quick-reply path.
992
+ try { _yieldBacklog(); } catch { /* fail-open */ }
923
993
  _dispatch(prompt, item, classResult, "inbox", {
924
994
  // Carried into the session child's env so its CLI send lanes stamp their
925
995
  // delivery receipts with the debt they discharge. This is what lets the
@@ -1058,6 +1128,12 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
1058
1128
  recordClassification(false);
1059
1129
  emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, stage: "process_item", error: err.message } });
1060
1130
  return { ok: false, path: "error", reason: err.message };
1131
+ } finally {
1132
+ // The reactive turn is over (a reply was sent, a session was spawned, or the
1133
+ // item was filtered). Release the subscription back to backlog; the next
1134
+ // sweep re-dispatches whatever was evicted or held. Balanced 1:1 with the
1135
+ // markReactiveStart above, and clamped at zero inside the gate.
1136
+ markReactiveEnd();
1061
1137
  }
1062
1138
  }
1063
1139
 
@@ -831,6 +831,21 @@ test("NO FALLTHROUGH: a TRANSIENT quick-reply failure still falls through to a f
831
831
  assert.equal(res.path, "session");
832
832
  });
833
833
 
834
+ test("REACTIVE PRIORITY: a quick reply yields the subscription (evicts backlog) BEFORE it generates", async () => {
835
+ resetState();
836
+ const item = { id: "MSG-YIELD", raw_ref: "slack:DYIELD:1", service: "slack", channel: "dm/ceo", channel_id: "DYIELD001", is_dm: true, sender: "ceo", subject: "quick q", content: "you around?" };
837
+ const order = [];
838
+ const res = await daemon.answerItem(item, "slack", "MSG-YIELD", "trace-yield", {
839
+ classify: async () => ({ priority: "high", action: "respond", model: "haiku", summary: "quick answer", category: "action_required", directed_at_agent: true }),
840
+ isQuickReply: () => true,
841
+ yieldBacklogForReactive: () => { order.push("yield"); return 2; },
842
+ sendQuickResponse: async () => { order.push("send"); return { sent: true }; },
843
+ claudeAvailable: () => true,
844
+ });
845
+ assert.equal(res.path, "quick_reply");
846
+ assert.deepEqual(order, ["yield", "send"], "backlog is yielded the subscription BEFORE the reply's claude call, not after");
847
+ });
848
+
834
849
  test("resolveSelfMemberId reads the member cuid from the res frame's `result` (own-echo keystone)", async () => {
835
850
  daemon._resetSelfMemberId();
836
851
  const pk = process.env.COHORT_API_KEY, po = process.env.COHORT_ORG_ID, pa = process.env.COHORT_AGENT_ID;
@@ -862,3 +877,29 @@ test("resolveSelfMemberId reads the member cuid from the res frame's `result` (o
862
877
  if (pa === undefined) delete process.env.COHORT_AGENT_ID; else process.env.COHORT_AGENT_ID = pa;
863
878
  }
864
879
  });
880
+
881
+ test("fastPathClassify: a short directed question skips the classifier; work-requests do not", async () => {
882
+ const fp = (content, isDm = true, extra = {}) => daemon.fastPathClassify({ content, ...extra }, isDm);
883
+ // Plain short questions → fast-path (quick reply, high priority, sonnet).
884
+ const q = fp("what OS is your machine on?");
885
+ assert.equal(q && q.fast_path, true);
886
+ assert.equal(q.answerable, true);
887
+ assert.equal(q.priority, "high");
888
+ assert.equal(q.action, "respond");
889
+ assert.ok(daemon.fastPathClassify({ content: "hey, you around?" }, true));
890
+ assert.ok(daemon.fastPathClassify({ content: "thanks!" }, true));
891
+ // Work-requests → NOT fast-pathed (fall through to full classify → session).
892
+ assert.equal(fp("can you analyze the Q3 report and send me the findings?"), null);
893
+ assert.equal(fp("please review the PR and fix the failing tests"), null);
894
+ assert.equal(fp("build a dashboard for the sales pipeline"), null);
895
+ // Non-questions, long messages, group, and non-DM → NOT fast-pathed.
896
+ assert.equal(fp("here are the notes from the meeting for your records"), null);
897
+ assert.equal(fp("x".repeat(300)), null);
898
+ assert.equal(fp("what OS?", false), null); // not a DM
899
+ assert.equal(fp(""), null);
900
+ // Attachments → full path.
901
+ assert.equal(daemon.fastPathClassify({ content: "what's this?", has_attachments: true }, true), null);
902
+ // Kill switch (deps.disableFastPath) forces the full classify path.
903
+ assert.ok(daemon.fastPathClassify({ content: "what OS?" }, true));
904
+ assert.equal(daemon.fastPathClassify({ content: "what OS?" }, true, { disableFastPath: true }), null);
905
+ });
@@ -352,6 +352,89 @@ test("M2: evicting a backlog session for a priority inbox item clears its resume
352
352
  }
353
353
  });
354
354
 
355
+ // ---------------------------------------------------------------------------
356
+ // yieldBacklogForReactive — a reply about to hit the subscription evicts
357
+ // already-running BACKLOG down to REACTIVE_INFLIGHT_BACKLOG_MAX, and NEVER
358
+ // touches an inbox/reactive session.
359
+ // ---------------------------------------------------------------------------
360
+
361
+ /** A child stub that records whether it was killed (never launches anything). */
362
+ function killTrackingProc() {
363
+ const handlers = {};
364
+ const p = {
365
+ stdout: { on: () => {} },
366
+ stderr: { on: () => {} },
367
+ on: (ev, cb) => { handlers[ev] = cb; },
368
+ kill: () => { p.killed = true; },
369
+ killed: false,
370
+ _handlers: handlers,
371
+ };
372
+ return p;
373
+ }
374
+
375
+ test("yieldBacklogForReactive: evicts backlog down to the cap and spares every inbox session", async () => {
376
+ process.env.DAEMON_MAX_CONCURRENT = "8";
377
+ const { mod, dir } = await freshDispatcher();
378
+ try {
379
+ mod.setGovernanceForTests({ governor: fakeGovernor("ADMIT"), rateGuard: allowRate, budgetGuard: noBudget });
380
+ const procs = [];
381
+ const restore = mod.setSpawnForTests(() => { const p = killTrackingProc(); procs.push(p); return p; });
382
+ try {
383
+ // 3 backlog sessions (dispatch order = age order) …
384
+ for (let i = 0; i < 3; i++) {
385
+ mod.dispatch("backlog work", { id: `BL-${i}`, title: `bl ${i}`, source_file: "q.yaml" },
386
+ { summary: `bl ${i}`, priority: "low", model: "sonnet" }, "backlog");
387
+ }
388
+ // … and 2 inbox sessions, all running concurrently.
389
+ for (let i = 0; i < 2; i++) {
390
+ mod.dispatch("inbox work", { id: `IN-${i}`, channel: `D${i}` },
391
+ { summary: `in ${i}`, priority: "normal", model: "sonnet" }, "inbox");
392
+ }
393
+ assert.equal(mod.getStatus().active_sessions, 5, "3 backlog + 2 inbox live");
394
+
395
+ // A reply is about to generate → yield the subscription. Cap is 1, so 2 of
396
+ // the 3 backlog sessions are evicted; the inbox sessions are never touched.
397
+ const evicted = mod.yieldBacklogForReactive();
398
+ assert.equal(evicted, 2, "evicts backlog down to REACTIVE_INFLIGHT_BACKLOG_MAX (1)");
399
+
400
+ const backlogProcs = procs.slice(0, 3);
401
+ const inboxProcs = procs.slice(3, 5);
402
+ assert.equal(backlogProcs.filter((p) => p.killed).length, 2, "exactly two backlog procs killed");
403
+ assert.equal(inboxProcs.filter((p) => p.killed).length, 0, "no inbox/reactive proc is ever killed");
404
+
405
+ // Idempotent: already at the cap → a second call is a no-op.
406
+ assert.equal(mod.yieldBacklogForReactive(), 0, "no further eviction once at the cap");
407
+ } finally { restore(); }
408
+ } finally {
409
+ delete process.env.DAEMON_MAX_CONCURRENT;
410
+ await cleanup(dir);
411
+ }
412
+ });
413
+
414
+ test("yieldBacklogForReactive: no-op when backlog is already at/under the cap", async () => {
415
+ process.env.DAEMON_MAX_CONCURRENT = "8";
416
+ const { mod, dir } = await freshDispatcher();
417
+ try {
418
+ mod.setGovernanceForTests({ governor: fakeGovernor("ADMIT"), rateGuard: allowRate, budgetGuard: noBudget });
419
+ const restore = mod.setSpawnForTests(() => killTrackingProc());
420
+ try {
421
+ // Only inbox sessions running — nothing self-directed to yield.
422
+ for (let i = 0; i < 3; i++) {
423
+ mod.dispatch("inbox work", { id: `IN-${i}`, channel: `D${i}` },
424
+ { summary: `in ${i}`, priority: "normal", model: "sonnet" }, "inbox");
425
+ }
426
+ assert.equal(mod.yieldBacklogForReactive(), 0, "nothing to evict");
427
+ // One backlog session — at the cap of 1 → still a no-op.
428
+ mod.dispatch("bl", { id: "BL-ONE", title: "one", source_file: "q.yaml" },
429
+ { summary: "one", priority: "low", model: "sonnet" }, "backlog");
430
+ assert.equal(mod.yieldBacklogForReactive(), 0, "a single backlog session is within the cap");
431
+ } finally { restore(); }
432
+ } finally {
433
+ delete process.env.DAEMON_MAX_CONCURRENT;
434
+ await cleanup(dir);
435
+ }
436
+ });
437
+
355
438
  // ---------------------------------------------------------------------------
356
439
  // M1 — a governor/budget/rate DEFER that force-queues an inbox item while NO
357
440
  // session is running must arm a single (debounced) re-drain timer, so the item
@@ -680,42 +680,97 @@ function countBySource(source) {
680
680
 
681
681
  /**
682
682
  * Evict the lowest-priority, longest-running backlog session to make room
683
- * for a high-priority inbox item. Returns true if a session was evicted.
683
+ * for a high-priority inbox item (or to yield the subscription to a reactive
684
+ * reply). Returns true if a session was evicted.
685
+ *
686
+ * @param {string} [why] human-readable eviction reason for the log/ledger.
687
+ */
688
+ const EVICT_PRIORITY_RANK = { low: 0, normal: 1, high: 2, critical: 3 };
689
+
690
+ /**
691
+ * SIGTERM one backlog session and retire its resume marker. Shared by the
692
+ * priority-preemption path and the reactive-yield path so both do the identical
693
+ * bookkeeping. Marks `s.evicting` so a SECOND selection in the same synchronous
694
+ * pass (before the async close handler deletes the session) cannot re-target the
695
+ * same not-yet-closed session — the bug that let one kill masquerade as many.
684
696
  */
685
- function evictForPriority() {
686
- const priorityRank = { low: 0, normal: 1, high: 2, critical: 3 };
697
+ function _evictSession(id, s, why) {
698
+ console.log(`[dispatcher] PREEMPT: Evicting backlog session ${id} (${s.classResult.summary}) for ${why}`);
699
+ logSession({ event: "evicted", sessionId: id, reason: "priority_preemption", summary: s.classResult.summary });
700
+ // M2: clear the resume-pending marker BEFORE the SIGTERM. Eviction is a
701
+ // deliberate preemption, not a crash — the item stays on disk and the
702
+ // backlog sweep re-dispatches it normally (acquiring the item-claim).
703
+ // If we left the marker, a reboot's reconcileResumePending would re-spawn
704
+ // it WITHOUT the claim while sweepBacklog independently claim+dispatched
705
+ // the same item → two concurrent runs. The non-zero SIGTERM close would
706
+ // otherwise keep the marker, so we must retire it here.
707
+ clearResumePending(id);
708
+ s.evicting = true;
709
+ s.process.kill("SIGTERM");
710
+ }
711
+
712
+ /** The single lowest-priority, oldest evictable backlog session, or null. */
713
+ function _worstBacklogSession() {
687
714
  let worst = null;
688
715
  let worstId = null;
689
-
690
716
  for (const [id, s] of activeSessions) {
691
- if (s.source !== "backlog") continue;
692
- const rank = priorityRank[s.classResult.priority] || 1;
693
- if (!worst) {
694
- worst = s; worstId = id;
695
- continue;
696
- }
697
- const worstRank = priorityRank[worst.classResult.priority] || 1;
698
- // Prefer to evict lower priority; break ties by oldest session
717
+ if (s.source !== "backlog" || s.evicting) continue;
718
+ const rank = EVICT_PRIORITY_RANK[s.classResult.priority] ?? 1;
719
+ if (!worst) { worst = s; worstId = id; continue; }
720
+ const worstRank = EVICT_PRIORITY_RANK[worst.classResult.priority] ?? 1;
721
+ // Prefer to evict lower priority; break ties by oldest session.
699
722
  if (rank < worstRank || (rank === worstRank && s.startTime < worst.startTime)) {
700
723
  worst = s; worstId = id;
701
724
  }
702
725
  }
726
+ return worst ? { id: worstId, s: worst } : null;
727
+ }
703
728
 
704
- if (worst) {
705
- console.log(`[dispatcher] PREEMPT: Evicting backlog session ${worstId} (${worst.classResult.summary}) for priority inbox item`);
706
- logSession({ event: "evicted", sessionId: worstId, reason: "priority_preemption", summary: worst.classResult.summary });
707
- // M2: clear the resume-pending marker BEFORE the SIGTERM. Eviction is a
708
- // deliberate preemption, not a crash — the item stays on disk and the
709
- // backlog sweep re-dispatches it normally (acquiring the item-claim).
710
- // If we left the marker, a reboot's reconcileResumePending would re-spawn
711
- // it WITHOUT the claim while sweepBacklog independently claim+dispatched
712
- // the same item → two concurrent runs. The non-zero SIGTERM close would
713
- // otherwise keep the marker, so we must retire it here.
714
- clearResumePending(worstId);
715
- worst.process.kill("SIGTERM");
716
- return true;
729
+ function evictForPriority(why = "priority inbox item") {
730
+ const target = _worstBacklogSession();
731
+ if (!target) return false;
732
+ _evictSession(target.id, target.s, why);
733
+ return true;
734
+ }
735
+
736
+ /**
737
+ * Yield the shared subscription to a reactive turn that is about to generate.
738
+ *
739
+ * REACTIVE_RESERVE keeps a slot free and the governor now QUEUEs NEW backlog
740
+ * while a reply is in flight — but neither of those stops backlog sessions that
741
+ * are ALREADY RUNNING from saturating the one Max subscription (the measured
742
+ * cause of 19s quick replies and 30s classify timeouts). This is the other half:
743
+ * called the instant a reactive reply is about to hit `claude`, it evicts
744
+ * already-running BACKLOG sessions down to REACTIVE_INFLIGHT_BACKLOG_MAX so the
745
+ * reply's own call runs on a nearly-idle subscription.
746
+ *
747
+ * Safe and conservative: it never touches another reactive/inbox session (only
748
+ * `source === "backlog"`); an evicted item stays on disk and the sweep re-claims
749
+ * it, so backlog loses concurrency for a few seconds, never work. No-op when
750
+ * backlog is already at/under the cap. Returns the number of sessions evicted.
751
+ */
752
+ export function yieldBacklogForReactive() {
753
+ const cap = Math.max(0, Number(governor.REACTIVE_INFLIGHT_BACKLOG_MAX ?? 1));
754
+ // Live (not-already-evicting) backlog sessions, worst-first, so we shed the
755
+ // lowest-priority / oldest work and keep the freshest `cap` running.
756
+ const live = [];
757
+ for (const [id, s] of activeSessions) {
758
+ if (s.source === "backlog" && !s.evicting) live.push({ id, s });
759
+ }
760
+ live.sort((a, b) => {
761
+ const ra = EVICT_PRIORITY_RANK[a.s.classResult.priority] ?? 1;
762
+ const rb = EVICT_PRIORITY_RANK[b.s.classResult.priority] ?? 1;
763
+ return ra - rb || a.s.startTime - b.s.startTime;
764
+ });
765
+ let evicted = 0;
766
+ for (let i = 0; i < live.length - cap; i++) {
767
+ _evictSession(live[i].id, live[i].s, "reactive reply waiting on the subscription");
768
+ evicted++;
769
+ }
770
+ if (evicted > 0) {
771
+ console.log(`[dispatcher] Yielded subscription to a reactive reply: evicted ${evicted} backlog session(s) (cap ${cap})`);
717
772
  }
718
- return false;
773
+ return evicted;
719
774
  }
720
775
 
721
776
  /**
@@ -142,7 +142,10 @@ function dm(over = {}) {
142
142
  channel_id: "D77",
143
143
  sender: "Priya Raman",
144
144
  sender_id: "U_PRIYA",
145
- content: "can you confirm the deadline?",
145
+ // A work-request ("review") so this shared fixture takes the full classify →
146
+ // route path — the fast-path (fastPathClassify) deliberately skips the
147
+ // classifier only for PLAIN short questions; that path has its own test below.
148
+ content: "can you review and confirm the deadline for the launch plan?",
146
149
  timestamp: new Date().toISOString(),
147
150
  ...over,
148
151
  };
@@ -190,6 +193,20 @@ test("LADDER: a DM is journalled intake→match→route→outcome AND still quic
190
193
  assert.equal(s.seen.dispatch.length, 0, "no session spawned for a rung-0 reply");
191
194
  });
192
195
 
196
+ test("LADDER: a plain short question DM FAST-PATHS — quick-replies WITHOUT a classifier call", async () => {
197
+ const item = dm({ id: "MSG-DM-FAST", raw_ref: "slack:D91:1", channel_id: "D91", content: "what's the deadline?" });
198
+ writeLive(item, "slack");
199
+ const s = spy(RESPOND);
200
+ await daemon.processItem(item, "slack", s.deps);
201
+
202
+ // The whole point: the ~5-8s Haiku classify call is SKIPPED for a plain question.
203
+ assert.equal(s.seen.classify, 0, "fast-path: a plain directed question skips the classifier");
204
+ // ...but the DM is still answered via the quick-reply path, still journalled.
205
+ assert.deepEqual(s.seen.quick, [{ id: "MSG-DM-FAST", rung: 0 }], "the quick reply still went out at rung 0");
206
+ assert.equal(s.seen.dispatch.length, 0, "no session spawned");
207
+ assert.ok(decisions().some((r) => r.surface === "dm" && r.disposition === "react_now"), "still journalled intake→route→outcome");
208
+ });
209
+
193
210
  // ---------------------------------------------------------------------------
194
211
  // 2) A board task is WORK: it queues, and never reaches the classifier.
195
212
  // ---------------------------------------------------------------------------