@cohortapp/agent-sdk 2.11.11 → 2.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/claude-bin.mjs +17 -0
- package/lib/claude-bin.test.mjs +22 -0
- package/lib/rate-guard.mjs +67 -1
- package/lib/rate-guard.test.mjs +71 -0
- package/lib/reactive-gate.mjs +58 -0
- package/lib/reactive-gate.test.mjs +57 -0
- package/lib/resource-governor.mjs +38 -7
- package/lib/resource-governor.test.mjs +52 -0
- package/package.json +1 -1
- package/scripts/daemon/agent-daemon.mjs +28 -2
- package/scripts/daemon/agent-daemon.test.mjs +15 -0
- package/scripts/daemon/cadence-consumer.mjs +17 -1
- package/scripts/daemon/classifier.mjs +1 -1
- package/scripts/daemon/dispatcher-governance.test.mjs +83 -0
- package/scripts/daemon/dispatcher.mjs +106 -33
- package/scripts/daemon/responder.mjs +1 -1
package/lib/claude-bin.mjs
CHANGED
|
@@ -129,9 +129,26 @@ export function daemonClaudeArgs(agentRoot, deps = {}) {
|
|
|
129
129
|
const env = deps.env || process.env;
|
|
130
130
|
if (env.DAEMON_LOAD_MCPS === "1") return [];
|
|
131
131
|
const ex = deps.existsSync || existsSync;
|
|
132
|
+
const source = deps.source || null;
|
|
132
133
|
|
|
133
134
|
const args = ["--strict-mcp-config"];
|
|
134
135
|
|
|
136
|
+
// MCP BY SOURCE — the biggest cheap per-spawn token saving. A classify or a
|
|
137
|
+
// quick-reply generates TEXT and calls NO org tools, so re-paying the ~107-tool
|
|
138
|
+
// org MCP definitions on every one of those high-frequency spawns is pure
|
|
139
|
+
// waste. Those sources get a BARE spawn (`--strict-mcp-config` with no
|
|
140
|
+
// `--mcp-config` loads zero MCP servers); only full work sessions
|
|
141
|
+
// (dispatcher/cadence — source unset) load the org toolset, so the API/tool-
|
|
142
|
+
// parity bar is honoured exactly where real work happens. Override per source
|
|
143
|
+
// with DAEMON_<SOURCE>_MCP=full|bare (e.g. DAEMON_CLASSIFIER_MCP=full).
|
|
144
|
+
const BARE_SOURCES = new Set(["classifier", "responder"]);
|
|
145
|
+
const perSource = source ? env[`DAEMON_${source.toUpperCase()}_MCP`] : null;
|
|
146
|
+
const bare = perSource === "bare" || (perSource !== "full" && BARE_SOURCES.has(source));
|
|
147
|
+
if (bare) {
|
|
148
|
+
if (env.DAEMON_BARE_MODE === "1") args.unshift("--bare");
|
|
149
|
+
return args;
|
|
150
|
+
}
|
|
151
|
+
|
|
135
152
|
const root = agentRoot || env.AGENT_ROOT || process.cwd();
|
|
136
153
|
try {
|
|
137
154
|
const cfg = join(root, ".mcp.json");
|
package/lib/claude-bin.test.mjs
CHANGED
|
@@ -107,3 +107,25 @@ test("daemonClaudeArgs never throws on a bad root (a daemon must still spawn)",
|
|
|
107
107
|
const args = daemonClaudeArgs("/agent", { existsSync: () => { throw new Error("EACCES"); }, env: {} });
|
|
108
108
|
assert.deepEqual(args, ["--strict-mcp-config"]);
|
|
109
109
|
});
|
|
110
|
+
|
|
111
|
+
// ---------------------------------------------------------------------------
|
|
112
|
+
// MCP by source — bare for tool-free spawns (classifier / responder)
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
test("daemonClaudeArgs: classifier + responder are BARE (no org MCP re-paid)", () => {
|
|
116
|
+
for (const source of ["classifier", "responder"]) {
|
|
117
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {}, source });
|
|
118
|
+
assert.ok(args.includes("--strict-mcp-config"), `${source} keeps strict-mcp-config`);
|
|
119
|
+
assert.ok(!args.includes("--mcp-config"), `${source} must NOT load the org MCP`);
|
|
120
|
+
}
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("daemonClaudeArgs: a full work session (no source) still loads the org MCP", () => {
|
|
124
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {} });
|
|
125
|
+
assert.ok(args.includes("--mcp-config"), "dispatcher/cadence session keeps full org tools");
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
test("daemonClaudeArgs: DAEMON_<SOURCE>_MCP=full overrides bare for that source", () => {
|
|
129
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: { DAEMON_CLASSIFIER_MCP: "full" }, source: "classifier" });
|
|
130
|
+
assert.ok(args.includes("--mcp-config"), "explicit override restores the org MCP for the classifier");
|
|
131
|
+
});
|
package/lib/rate-guard.mjs
CHANGED
|
@@ -243,4 +243,70 @@ export function classifyStderr(text) {
|
|
|
243
243
|
return /\b429\b|rate[\s_-]?limit|overloaded|too many requests/i.test(text);
|
|
244
244
|
}
|
|
245
245
|
|
|
246
|
-
|
|
246
|
+
// A Max/subscription USAGE or SESSION limit is NOT a transient 429 — the pool is
|
|
247
|
+
// drained until the window RESETS, so hammering it with decorrelated-jitter
|
|
248
|
+
// retries just burns the moment it re-opens. We detect it separately and hold
|
|
249
|
+
// the breaker until the reset (parsed from the message when present, else a
|
|
250
|
+
// conservative default) so the seat backs off to the reset instead of storming.
|
|
251
|
+
const USAGE_LIMIT_RE =
|
|
252
|
+
/\b(?:usage|session|weekly|5[\s-]?hour)\s+limit\b|limit reached|reached your (?:usage|session|monthly|weekly|plan)?\s*limit|you'?ve hit your [^.]*limit|approaching (?:your )?[^.]*usage limit|out of (?:usage|credits|messages)/i;
|
|
253
|
+
|
|
254
|
+
/** How long to hold the breaker when a usage limit is hit but no reset time is parseable. */
|
|
255
|
+
export const RATE_USAGE_LIMIT_HOLD_MS = num(process.env.RATE_USAGE_LIMIT_HOLD_MS, 30 * 60 * 1000);
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Best-effort parse of a reset time out of a usage-limit message —
|
|
259
|
+
* "resets 9:40pm", "resets at 9pm", "try again at 10:15am", "available again at
|
|
260
|
+
* 3pm" — into the epoch ms of its NEXT occurrence (today, or tomorrow if already
|
|
261
|
+
* past). The daemon runs in the seat's local timezone, which is the timezone the
|
|
262
|
+
* message states, so a local Date is correct. Returns null when unparseable.
|
|
263
|
+
*/
|
|
264
|
+
export function parseResetAt(text, deps) {
|
|
265
|
+
if (!text || typeof text !== "string") return null;
|
|
266
|
+
const m = /(?:reset|resets|resets at|resets in|try again at|available again at)\s+(\d{1,2})(?::(\d{2}))?\s*([ap])\.?\s?m\.?/i.exec(text);
|
|
267
|
+
if (!m) return null;
|
|
268
|
+
const nowMs = clock(deps)();
|
|
269
|
+
let hr = parseInt(m[1], 10) % 12;
|
|
270
|
+
if (/p/i.test(m[3])) hr += 12;
|
|
271
|
+
const min = m[2] ? parseInt(m[2], 10) : 0;
|
|
272
|
+
const cand = new Date(nowMs);
|
|
273
|
+
cand.setHours(hr, min, 0, 0);
|
|
274
|
+
let t = cand.getTime();
|
|
275
|
+
if (t <= nowMs) t += 24 * 60 * 60 * 1000; // already past today → next occurrence
|
|
276
|
+
return t;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Does this text look like a Max/subscription USAGE / SESSION limit (as opposed
|
|
281
|
+
* to a transient 429)? When it names a reset time, resetAt is the epoch ms to
|
|
282
|
+
* hold the breaker until.
|
|
283
|
+
*
|
|
284
|
+
* @returns {{ isLimit: boolean, resetAt: number|null }}
|
|
285
|
+
*/
|
|
286
|
+
export function classifyUsageLimit(text, deps) {
|
|
287
|
+
if (!text || typeof text !== "string") return { isLimit: false, resetAt: null };
|
|
288
|
+
if (!USAGE_LIMIT_RE.test(text)) return { isLimit: false, resetAt: null };
|
|
289
|
+
return { isLimit: true, resetAt: parseResetAt(text, deps) };
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Open the breaker until a usage window RESETS. Uses resetAt when known, else a
|
|
294
|
+
* conservative default hold. Never shortens an already-longer open window.
|
|
295
|
+
*
|
|
296
|
+
* @returns {{ openUntil:number, resetAt:number|null }}
|
|
297
|
+
*/
|
|
298
|
+
export function recordUsageLimit(provider, resetAt, deps) {
|
|
299
|
+
const now = clock(deps)();
|
|
300
|
+
const until =
|
|
301
|
+
typeof resetAt === "number" && resetAt > now ? resetAt : now + RATE_USAGE_LIMIT_HOLD_MS;
|
|
302
|
+
const prev = readState(provider, deps);
|
|
303
|
+
const next = {
|
|
304
|
+
openUntil: Math.max(prev.openUntil, until),
|
|
305
|
+
consecutive429: prev.consecutive429,
|
|
306
|
+
lastBackoffMs: prev.lastBackoffMs,
|
|
307
|
+
};
|
|
308
|
+
writeState(provider, next, deps);
|
|
309
|
+
return { openUntil: next.openUntil, resetAt: typeof resetAt === "number" && resetAt > now ? resetAt : null };
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
export const _internals = { RATE_BASE_BACKOFF_MS, RATE_MAX_BACKOFF_MS, RATE_BACKOFF_FACTOR, RATE_USAGE_LIMIT_HOLD_MS };
|
package/lib/rate-guard.test.mjs
CHANGED
|
@@ -18,11 +18,15 @@ import {
|
|
|
18
18
|
recordSuccess,
|
|
19
19
|
decorrelatedBackoff,
|
|
20
20
|
classifyStderr,
|
|
21
|
+
classifyUsageLimit,
|
|
22
|
+
parseResetAt,
|
|
23
|
+
recordUsageLimit,
|
|
21
24
|
readState,
|
|
22
25
|
sanitizeProvider,
|
|
23
26
|
RATE_BASE_BACKOFF_MS,
|
|
24
27
|
RATE_MAX_BACKOFF_MS,
|
|
25
28
|
RATE_BACKOFF_FACTOR,
|
|
29
|
+
RATE_USAGE_LIMIT_HOLD_MS,
|
|
26
30
|
} from "./rate-guard.mjs";
|
|
27
31
|
|
|
28
32
|
async function makeStateDir() {
|
|
@@ -199,3 +203,70 @@ test("readState tolerates a corrupt file (fails closed-to-empty, never throws)",
|
|
|
199
203
|
assert.equal(checkRateLimit("anthropic", { stateDir: dir }).allowed, true);
|
|
200
204
|
} finally { await rm(dir); }
|
|
201
205
|
});
|
|
206
|
+
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
// classifyUsageLimit / parseResetAt / recordUsageLimit (Max session-limit)
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
test("classifyUsageLimit detects session/usage limits, not plain 429s", () => {
|
|
212
|
+
assert.equal(classifyUsageLimit("You've hit your session limit · resets 9:40pm (Australia/Sydney)").isLimit, true);
|
|
213
|
+
assert.equal(classifyUsageLimit("usage limit reached").isLimit, true);
|
|
214
|
+
assert.equal(classifyUsageLimit("You have reached your weekly limit").isLimit, true);
|
|
215
|
+
assert.equal(classifyUsageLimit("out of usage for now").isLimit, true);
|
|
216
|
+
// A plain transient 429 is NOT a usage-window limit
|
|
217
|
+
assert.equal(classifyUsageLimit("HTTP 429 Too Many Requests").isLimit, false);
|
|
218
|
+
assert.equal(classifyUsageLimit("overloaded").isLimit, false);
|
|
219
|
+
assert.equal(classifyUsageLimit("").isLimit, false);
|
|
220
|
+
assert.equal(classifyUsageLimit(null).isLimit, false);
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
test("parseResetAt resolves a stated reset time to the next occurrence", () => {
|
|
224
|
+
// now = 2026-08-26 08:00 local; "resets 9:40pm" is later today
|
|
225
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
226
|
+
const t = parseResetAt("resets 9:40pm", { now: () => now });
|
|
227
|
+
const d = new Date(t);
|
|
228
|
+
assert.equal(d.getHours(), 21);
|
|
229
|
+
assert.equal(d.getMinutes(), 40);
|
|
230
|
+
assert.ok(t > now, "reset is in the future");
|
|
231
|
+
// A time already past today rolls to tomorrow
|
|
232
|
+
const now2 = new Date(2026, 7, 26, 22, 0, 0, 0).getTime(); // 10pm
|
|
233
|
+
const t2 = parseResetAt("try again at 9pm", { now: () => now2 });
|
|
234
|
+
assert.ok(t2 > now2 && t2 - now2 <= 24 * 60 * 60 * 1000, "rolls to next day");
|
|
235
|
+
assert.equal(parseResetAt("no time here", { now: () => now }), null);
|
|
236
|
+
});
|
|
237
|
+
|
|
238
|
+
test("recordUsageLimit holds the breaker until the parsed reset time", async () => {
|
|
239
|
+
const dir = await makeStateDir();
|
|
240
|
+
try {
|
|
241
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
242
|
+
const { resetAt } = classifyUsageLimit("session limit · resets 9:40pm", { now: () => now });
|
|
243
|
+
const rec = recordUsageLimit("anthropic", resetAt, { stateDir: dir, now: () => now });
|
|
244
|
+
assert.equal(rec.openUntil, resetAt);
|
|
245
|
+
// blocked before reset, allowed after
|
|
246
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
247
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => resetAt + 1 }).allowed, true);
|
|
248
|
+
} finally { await rm(dir); }
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
test("recordUsageLimit falls back to a default hold when no reset is known", async () => {
|
|
252
|
+
const dir = await makeStateDir();
|
|
253
|
+
try {
|
|
254
|
+
const now = 1_000_000;
|
|
255
|
+
const rec = recordUsageLimit("anthropic", null, { stateDir: dir, now: () => now });
|
|
256
|
+
assert.equal(rec.openUntil, now + RATE_USAGE_LIMIT_HOLD_MS);
|
|
257
|
+
assert.equal(rec.resetAt, null);
|
|
258
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
259
|
+
} finally { await rm(dir); }
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
test("recordUsageLimit never shortens an already-longer open window", async () => {
|
|
263
|
+
const dir = await makeStateDir();
|
|
264
|
+
try {
|
|
265
|
+
const now = 1_000_000;
|
|
266
|
+
const far = now + 10 * 60 * 60 * 1000; // 10h out
|
|
267
|
+
recordUsageLimit("anthropic", far, { stateDir: dir, now: () => now });
|
|
268
|
+
// a later, SHORTER hold must not pull the window in
|
|
269
|
+
const rec = recordUsageLimit("anthropic", now + 5 * 60 * 1000, { stateDir: dir, now: () => now });
|
|
270
|
+
assert.equal(rec.openUntil, far);
|
|
271
|
+
} finally { await rm(dir); }
|
|
272
|
+
});
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* reactive-gate — a process-local signal that a REACTIVE turn (a human or peer
|
|
3
|
+
* agent is waiting on a reply) is actively consuming the shared Max subscription.
|
|
4
|
+
*
|
|
5
|
+
* WHY THIS EXISTS. The scarce resource on a mini is not a dispatch slot — it is
|
|
6
|
+
* the ONE Max subscription's inference throughput, which tracks the number of
|
|
7
|
+
* concurrent `claude --print` processes. When several heavy autonomous BACKLOG
|
|
8
|
+
* sessions run at once they saturate that subscription, and the next reactive
|
|
9
|
+
* reply's own claude call is starved: a quick reply measured 19s and a classify
|
|
10
|
+
* actually timed out at 30s while three opus backlog sessions held the line.
|
|
11
|
+
*
|
|
12
|
+
* The governor reads {@link reactiveInFlight} to YIELD the subscription: while
|
|
13
|
+
* any reactive turn is in flight, self-directed backlog admission is held to
|
|
14
|
+
* REACTIVE_INFLIGHT_BACKLOG_MAX so a reply is never starved. The dispatcher
|
|
15
|
+
* reads the same signal to evict already-running backlog down to that cap.
|
|
16
|
+
*
|
|
17
|
+
* A COUNTER, not a boolean, so concurrent reactive turns nest correctly and
|
|
18
|
+
* backlog only resumes when the LAST reply has landed. Process-local by design:
|
|
19
|
+
* the daemon, dispatcher, and governor share one Node process, so a module-level
|
|
20
|
+
* count is visible to all three with no IPC. `claude --print` children are
|
|
21
|
+
* separate processes and are governed through the count, not this module.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
let _inFlight = 0;
|
|
25
|
+
let _peak = 0;
|
|
26
|
+
|
|
27
|
+
/** Mark the start of a reactive turn. Returns the new in-flight count. */
|
|
28
|
+
export function markReactiveStart() {
|
|
29
|
+
_inFlight += 1;
|
|
30
|
+
if (_inFlight > _peak) _peak = _inFlight;
|
|
31
|
+
return _inFlight;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Mark the end of a reactive turn. Clamped at zero so an unbalanced end (a
|
|
36
|
+
* double-finally, a test) can never drive the count negative and wedge backlog
|
|
37
|
+
* off permanently. Returns the new in-flight count.
|
|
38
|
+
*/
|
|
39
|
+
export function markReactiveEnd() {
|
|
40
|
+
if (_inFlight > 0) _inFlight -= 1;
|
|
41
|
+
return _inFlight;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** How many reactive turns are consuming the subscription right now. */
|
|
45
|
+
export function reactiveInFlight() {
|
|
46
|
+
return _inFlight;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Highest concurrent reactive count seen since the last reset (diagnostics). */
|
|
50
|
+
export function reactivePeak() {
|
|
51
|
+
return _peak;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Test seam — reset the process-local counters between cases. */
|
|
55
|
+
export function _resetReactiveGate() {
|
|
56
|
+
_inFlight = 0;
|
|
57
|
+
_peak = 0;
|
|
58
|
+
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* reactive-gate — the process-local counter that tells the governor + dispatcher
|
|
3
|
+
* a human/peer reply is actively consuming the subscription.
|
|
4
|
+
*/
|
|
5
|
+
import { test } from "node:test";
|
|
6
|
+
import assert from "node:assert/strict";
|
|
7
|
+
import {
|
|
8
|
+
markReactiveStart,
|
|
9
|
+
markReactiveEnd,
|
|
10
|
+
reactiveInFlight,
|
|
11
|
+
reactivePeak,
|
|
12
|
+
_resetReactiveGate,
|
|
13
|
+
} from "./reactive-gate.mjs";
|
|
14
|
+
|
|
15
|
+
test("starts at zero", () => {
|
|
16
|
+
_resetReactiveGate();
|
|
17
|
+
assert.equal(reactiveInFlight(), 0);
|
|
18
|
+
});
|
|
19
|
+
|
|
20
|
+
test("start/end move the count by exactly one and balance to zero", () => {
|
|
21
|
+
_resetReactiveGate();
|
|
22
|
+
assert.equal(markReactiveStart(), 1);
|
|
23
|
+
assert.equal(reactiveInFlight(), 1);
|
|
24
|
+
assert.equal(markReactiveEnd(), 0);
|
|
25
|
+
assert.equal(reactiveInFlight(), 0);
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
test("concurrent reactive turns NEST — backlog resumes only when the last ends", () => {
|
|
29
|
+
_resetReactiveGate();
|
|
30
|
+
markReactiveStart();
|
|
31
|
+
markReactiveStart();
|
|
32
|
+
assert.equal(reactiveInFlight(), 2, "two turns in flight");
|
|
33
|
+
markReactiveEnd();
|
|
34
|
+
assert.equal(reactiveInFlight(), 1, "still one waiting — do NOT resume backlog yet");
|
|
35
|
+
markReactiveEnd();
|
|
36
|
+
assert.equal(reactiveInFlight(), 0, "now backlog may resume");
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
test("end is clamped at zero — an unbalanced end can never wedge backlog off", () => {
|
|
40
|
+
_resetReactiveGate();
|
|
41
|
+
assert.equal(markReactiveEnd(), 0, "end with nothing in flight stays at 0");
|
|
42
|
+
assert.equal(markReactiveEnd(), 0, "and again");
|
|
43
|
+
// A subsequent real turn still registers correctly (not driven negative).
|
|
44
|
+
assert.equal(markReactiveStart(), 1);
|
|
45
|
+
assert.equal(reactiveInFlight(), 1);
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test("peak records the high-water mark since reset", () => {
|
|
49
|
+
_resetReactiveGate();
|
|
50
|
+
markReactiveStart();
|
|
51
|
+
markReactiveStart();
|
|
52
|
+
markReactiveStart();
|
|
53
|
+
markReactiveEnd();
|
|
54
|
+
markReactiveEnd();
|
|
55
|
+
assert.equal(reactivePeak(), 3, "peak holds even after turns end");
|
|
56
|
+
assert.equal(reactiveInFlight(), 1);
|
|
57
|
+
});
|
|
@@ -63,6 +63,7 @@ import os from "node:os";
|
|
|
63
63
|
import { execFileSync } from "node:child_process";
|
|
64
64
|
import { existsSync, readFileSync } from "node:fs";
|
|
65
65
|
import { join, resolve } from "node:path";
|
|
66
|
+
import { reactiveInFlight } from "./reactive-gate.mjs";
|
|
66
67
|
|
|
67
68
|
// ---------------------------------------------------------------------------
|
|
68
69
|
// Constants
|
|
@@ -103,6 +104,19 @@ export const REACTIVE_RESERVE = num(process.env.GOV_REACTIVE_RESERVE, 2);
|
|
|
103
104
|
* gets near-all capacity.
|
|
104
105
|
*/
|
|
105
106
|
export const DEGRADED_BACKLOG_MAX = num(process.env.GOV_DEGRADED_BACKLOG_MAX, 1);
|
|
107
|
+
/**
|
|
108
|
+
* The most self-directed BACKLOG sessions allowed to run while a REACTIVE turn
|
|
109
|
+
* is in flight (a human/peer is waiting on a reply). REACTIVE_RESERVE keeps a
|
|
110
|
+
* dispatch slot free, but slots are not the scarce resource — the ONE Max
|
|
111
|
+
* subscription's throughput is, and several concurrent backlog opus sessions
|
|
112
|
+
* saturate it, starving the reply's own claude call (measured: a 19s quick reply
|
|
113
|
+
* and a 30s classify timeout under three concurrent backlog sessions). So while
|
|
114
|
+
* a reply generates, backlog admission is capped HARD to this (default 1): the
|
|
115
|
+
* subscription is left almost entirely to the human, and backlog resumes the
|
|
116
|
+
* instant the reply lands. The dispatcher evicts already-running backlog down to
|
|
117
|
+
* the same cap. 0 pauses backlog completely during a reactive turn.
|
|
118
|
+
*/
|
|
119
|
+
export const REACTIVE_INFLIGHT_BACKLOG_MAX = num(process.env.GOV_REACTIVE_INFLIGHT_BACKLOG_MAX, 1);
|
|
106
120
|
/**
|
|
107
121
|
* A `claude` session older than this is treated as STALE and does NOT count
|
|
108
122
|
* toward the concurrency ceiling — an idle/wedged lane must not hold a slot
|
|
@@ -368,6 +382,10 @@ export function defaultDeps(extra = {}) {
|
|
|
368
382
|
liveClaude: live,
|
|
369
383
|
throttleCeiling: throttle.ceiling,
|
|
370
384
|
throttleReason: throttle.reason,
|
|
385
|
+
// How many reactive turns are consuming the subscription right now. admit()
|
|
386
|
+
// caps backlog HARD while this is > 0 so a waiting human's reply is never
|
|
387
|
+
// starved. Injectable via `extra` for tests; live from the gate otherwise.
|
|
388
|
+
reactiveInFlight: reactiveInFlight(),
|
|
371
389
|
...extra,
|
|
372
390
|
};
|
|
373
391
|
}
|
|
@@ -500,22 +518,35 @@ export function admit(req = {}, deps) {
|
|
|
500
518
|
// can always be GENERATED (10 parallel sessions once starved the
|
|
501
519
|
// quick-reply's claude call into a 60s timeout). A DEGRADED/unfunded seat
|
|
502
520
|
// throttles backlog to DEGRADED_BACKLOG_MAX — a trickle in idle gaps.
|
|
521
|
+
// Is a reactive turn (a human/peer waiting on a reply) generating right now?
|
|
522
|
+
// While one is, self-directed backlog yields the SUBSCRIPTION — not merely a
|
|
523
|
+
// slot — so the reply's own claude call isn't starved. This is a hard cap
|
|
524
|
+
// ABOVE the REACTIVE_RESERVE math: reserve keeps steady-state headroom; this
|
|
525
|
+
// clamps to a trickle for the seconds a human is actually waiting.
|
|
526
|
+
const reactiveBusy = numField(d.reactiveInFlight, 0) > 0;
|
|
503
527
|
let sourceCeiling;
|
|
504
528
|
if (humanReply) {
|
|
505
529
|
sourceCeiling = effectiveMax;
|
|
530
|
+
} else if (reactiveBusy) {
|
|
531
|
+
sourceCeiling = Math.min(effectiveMax, REACTIVE_INFLIGHT_BACKLOG_MAX);
|
|
532
|
+
if (mode === "degraded" && source === "backlog") {
|
|
533
|
+
sourceCeiling = Math.min(sourceCeiling, DEGRADED_BACKLOG_MAX);
|
|
534
|
+
}
|
|
506
535
|
} else if (mode === "degraded" && source === "backlog") {
|
|
507
536
|
sourceCeiling = Math.min(effectiveMax, DEGRADED_BACKLOG_MAX);
|
|
508
537
|
} else {
|
|
509
538
|
sourceCeiling = Math.max(1, effectiveMax - REACTIVE_RESERVE);
|
|
510
539
|
}
|
|
511
540
|
if (liveCount >= sourceCeiling) {
|
|
512
|
-
const why =
|
|
513
|
-
? `
|
|
514
|
-
:
|
|
515
|
-
? `
|
|
516
|
-
:
|
|
517
|
-
? `at
|
|
518
|
-
:
|
|
541
|
+
const why = reactiveBusy
|
|
542
|
+
? `reactive turn in flight: self-directed ${source} yields the subscription, capped to ${sourceCeiling} (${liveCount} live)`
|
|
543
|
+
: mode === "degraded" && source === "backlog"
|
|
544
|
+
? `budget degraded: self-directed backlog throttled to ${sourceCeiling} (${liveCount} live; inbox unaffected)`
|
|
545
|
+
: throttleCeiling < dMax
|
|
546
|
+
? `at throttled ceiling (${liveCount}/${sourceCeiling}; soft-throttle active)`
|
|
547
|
+
: source === "inbox"
|
|
548
|
+
? `at concurrency ceiling (${liveCount}/${sourceCeiling})`
|
|
549
|
+
: `self-directed work yields ${REACTIVE_RESERVE} slots to the inbox (${liveCount}/${sourceCeiling})`;
|
|
519
550
|
return decide(QUEUE, why);
|
|
520
551
|
}
|
|
521
552
|
|
|
@@ -25,6 +25,7 @@ import {
|
|
|
25
25
|
FREEMEM_FLOOR_FRACTION,
|
|
26
26
|
etimeToMs,
|
|
27
27
|
reapStaleClaude,
|
|
28
|
+
REACTIVE_INFLIGHT_BACKLOG_MAX,
|
|
28
29
|
} from "./resource-governor.mjs";
|
|
29
30
|
|
|
30
31
|
test("availableMemoryBytes counts reclaimable pages, not just free ones", () => {
|
|
@@ -128,6 +129,57 @@ test("admit: inbox may use the whole envelope; the reactive reserve is for it",
|
|
|
128
129
|
assert.match(r.reason, /concurrency ceiling/);
|
|
129
130
|
});
|
|
130
131
|
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// Reactive-in-flight: backlog yields the SUBSCRIPTION (not just a slot) to a
|
|
134
|
+
// human/peer reply that is actively generating.
|
|
135
|
+
// ---------------------------------------------------------------------------
|
|
136
|
+
|
|
137
|
+
test("admit: while a reactive turn is in flight, backlog is capped to REACTIVE_INFLIGHT_BACKLOG_MAX → QUEUE", () => {
|
|
138
|
+
// Default cap is 1. With reactiveInFlight and even a single backlog session
|
|
139
|
+
// live, a second backlog spawn QUEUEs — the subscription is left to the reply.
|
|
140
|
+
assert.equal(REACTIVE_INFLIGHT_BACKLOG_MAX, 1);
|
|
141
|
+
const r = admit(
|
|
142
|
+
{ source: "backlog" },
|
|
143
|
+
deps({ reactiveInFlight: 1, liveClaude: { count: 1, rssMB: 300 } }),
|
|
144
|
+
);
|
|
145
|
+
assert.equal(r.decision, DECISIONS.QUEUE);
|
|
146
|
+
assert.match(r.reason, /reactive turn in flight/);
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
test("admit: reactive-in-flight cap is HARDER than the steady-state reserve", () => {
|
|
150
|
+
// With 3 live and NO reactive turn, backlog still ADMITs (reserve leaves 5).
|
|
151
|
+
assert.equal(
|
|
152
|
+
admit({ source: "backlog" }, deps({ liveClaude: { count: 3, rssMB: 900 } })).decision,
|
|
153
|
+
DECISIONS.ADMIT,
|
|
154
|
+
);
|
|
155
|
+
// The SAME 3 live but a reply generating → QUEUE (cap 1, already exceeded).
|
|
156
|
+
assert.equal(
|
|
157
|
+
admit({ source: "backlog" }, deps({ reactiveInFlight: 1, liveClaude: { count: 3, rssMB: 900 } })).decision,
|
|
158
|
+
DECISIONS.QUEUE,
|
|
159
|
+
);
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
test("admit: a human reply is NEVER gated by reactive-in-flight (it IS the reactive turn)", () => {
|
|
163
|
+
// inbox/humanReply keeps the whole envelope even while other reactive turns run.
|
|
164
|
+
assert.equal(
|
|
165
|
+
admit({ source: "inbox" }, deps({ reactiveInFlight: 2, liveClaude: { count: 4, rssMB: 1200 } })).decision,
|
|
166
|
+
DECISIONS.ADMIT,
|
|
167
|
+
);
|
|
168
|
+
assert.equal(
|
|
169
|
+
admit({ source: "backlog", humanReply: true }, deps({ reactiveInFlight: 2, liveClaude: { count: 4, rssMB: 1200 } })).decision,
|
|
170
|
+
DECISIONS.ADMIT,
|
|
171
|
+
);
|
|
172
|
+
});
|
|
173
|
+
|
|
174
|
+
test("admit: one backlog session is still allowed alongside a reply (cap 1, not 0)", () => {
|
|
175
|
+
// A single trickle of backlog is fine — the cap is 1, so with 0 live it ADMITs.
|
|
176
|
+
const r = admit(
|
|
177
|
+
{ source: "backlog" },
|
|
178
|
+
deps({ reactiveInFlight: 1, liveClaude: { count: 0, rssMB: 0 } }),
|
|
179
|
+
);
|
|
180
|
+
assert.equal(r.decision, DECISIONS.ADMIT);
|
|
181
|
+
});
|
|
182
|
+
|
|
131
183
|
test("admit: throttle ceiling clamps effectiveMax below dynamicMax → QUEUE earlier", () => {
|
|
132
184
|
// dynamicMax 8 but soft-throttle pinned the ceiling to 3; 3 live → QUEUE.
|
|
133
185
|
const r = admit({ source: "backlog" }, deps({ throttleCeiling: 3, liveClaude: { count: 3, rssMB: 1000 } }));
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cohortapp/agent-sdk",
|
|
3
|
-
"version": "2.11.
|
|
3
|
+
"version": "2.11.13",
|
|
4
4
|
"description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -51,7 +51,8 @@ import { makeInboxScanPoller } from "../poller/inbox-scan-poller.mjs";
|
|
|
51
51
|
import { existsSync as _existsSync } from "fs";
|
|
52
52
|
import { isPriorityItem } from "../poller/utils.mjs";
|
|
53
53
|
import { classifyItem, isDirectedAtAgent } from "./classifier.mjs";
|
|
54
|
-
import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions } from "./dispatcher.mjs";
|
|
54
|
+
import { dispatch, getStatus, availableSlots, canDispatchBacklog, resetActiveSessions, yieldBacklogForReactive } from "./dispatcher.mjs";
|
|
55
|
+
import { markReactiveStart, markReactiveEnd } from "../../lib/reactive-gate.mjs";
|
|
55
56
|
import { buildPrompt } from "./prompt-builder.mjs";
|
|
56
57
|
import { sendQuickResponse, sendHoldingMessage, isQuickReply } from "./responder.mjs";
|
|
57
58
|
// ANSWER ASSURANCE (scripts/daemon/assurance.mjs). The guarantee that an ask is
|
|
@@ -551,7 +552,17 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
551
552
|
const _electResponderVerdict = deps.electResponderVerdict
|
|
552
553
|
|| ((item2) => electResponderVerdict(item2, { electResponder: deps.electResponder, orgCfg: deps.orgCfg }));
|
|
553
554
|
const _claudeAvailable = deps.claudeAvailable || claudeAvailable;
|
|
554
|
-
|
|
555
|
+
// Evicts already-running backlog to hand the subscription to this reactive
|
|
556
|
+
// turn; injectable so a test can stub the dispatcher's eviction.
|
|
557
|
+
const _yieldBacklog = deps.yieldBacklogForReactive || yieldBacklogForReactive;
|
|
558
|
+
|
|
559
|
+
// A reactive turn begins HERE and is marked until the function returns. While
|
|
560
|
+
// it is marked the governor QUEUEs new self-directed backlog (it yields the
|
|
561
|
+
// subscription, not just a slot), so the classify + reply calls below are not
|
|
562
|
+
// starved. markReactiveEnd runs in the finally so a filtered/ignored/errored
|
|
563
|
+
// item releases the gate exactly once. The heavier lever — evicting backlog
|
|
564
|
+
// that is ALREADY running — fires only just before a real generate (below).
|
|
565
|
+
markReactiveStart();
|
|
555
566
|
try {
|
|
556
567
|
const isDm = item.is_dm === true;
|
|
557
568
|
// Carry the routing decision on the item so everything downstream of here —
|
|
@@ -779,6 +790,12 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
779
790
|
// is exactly today's behaviour.
|
|
780
791
|
if (_isQuickReply(classResult, routed)) {
|
|
781
792
|
console.log(`[daemon] Quick reply path for ${item.sender} (${classResult.model})`);
|
|
793
|
+
// Free the subscription NOW: evict any already-running backlog sessions
|
|
794
|
+
// down to REACTIVE_INFLIGHT_BACKLOG_MAX so this reply's own claude call
|
|
795
|
+
// runs on a nearly-idle subscription instead of queueing behind opus
|
|
796
|
+
// backlog (the measured cause of 19s quick replies / 30s classify
|
|
797
|
+
// timeouts). No-op when backlog is already at/under the cap.
|
|
798
|
+
try { _yieldBacklog(); } catch { /* fail-open: never block a reply on eviction */ }
|
|
782
799
|
const result = await _sendQuickResponse(item, classResult, routed);
|
|
783
800
|
if (result.sent) {
|
|
784
801
|
markProcessed(item, service);
|
|
@@ -970,6 +987,9 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
970
987
|
// recoverInFlight() to re-deliver on restart.
|
|
971
988
|
recordInFlight(item, service);
|
|
972
989
|
if (obligationKeyForItem) noteSession(obligationKeyForItem, null);
|
|
990
|
+
// A reactive SESSION (complex reply) also competes for the subscription —
|
|
991
|
+
// yield already-running backlog to it, same as the quick-reply path.
|
|
992
|
+
try { _yieldBacklog(); } catch { /* fail-open */ }
|
|
973
993
|
_dispatch(prompt, item, classResult, "inbox", {
|
|
974
994
|
// Carried into the session child's env so its CLI send lanes stamp their
|
|
975
995
|
// delivery receipts with the debt they discharge. This is what lets the
|
|
@@ -1108,6 +1128,12 @@ export async function answerItem(item, service, itemId, trace_id, deps = {}, rou
|
|
|
1108
1128
|
recordClassification(false);
|
|
1109
1129
|
emitEvent({ type: EVENT_TYPES.ERROR, trace_id, attrs: { item_id: itemId, service, stage: "process_item", error: err.message } });
|
|
1110
1130
|
return { ok: false, path: "error", reason: err.message };
|
|
1131
|
+
} finally {
|
|
1132
|
+
// The reactive turn is over (a reply was sent, a session was spawned, or the
|
|
1133
|
+
// item was filtered). Release the subscription back to backlog; the next
|
|
1134
|
+
// sweep re-dispatches whatever was evicted or held. Balanced 1:1 with the
|
|
1135
|
+
// markReactiveStart above, and clamped at zero inside the gate.
|
|
1136
|
+
markReactiveEnd();
|
|
1111
1137
|
}
|
|
1112
1138
|
}
|
|
1113
1139
|
|
|
@@ -831,6 +831,21 @@ test("NO FALLTHROUGH: a TRANSIENT quick-reply failure still falls through to a f
|
|
|
831
831
|
assert.equal(res.path, "session");
|
|
832
832
|
});
|
|
833
833
|
|
|
834
|
+
test("REACTIVE PRIORITY: a quick reply yields the subscription (evicts backlog) BEFORE it generates", async () => {
|
|
835
|
+
resetState();
|
|
836
|
+
const item = { id: "MSG-YIELD", raw_ref: "slack:DYIELD:1", service: "slack", channel: "dm/ceo", channel_id: "DYIELD001", is_dm: true, sender: "ceo", subject: "quick q", content: "you around?" };
|
|
837
|
+
const order = [];
|
|
838
|
+
const res = await daemon.answerItem(item, "slack", "MSG-YIELD", "trace-yield", {
|
|
839
|
+
classify: async () => ({ priority: "high", action: "respond", model: "haiku", summary: "quick answer", category: "action_required", directed_at_agent: true }),
|
|
840
|
+
isQuickReply: () => true,
|
|
841
|
+
yieldBacklogForReactive: () => { order.push("yield"); return 2; },
|
|
842
|
+
sendQuickResponse: async () => { order.push("send"); return { sent: true }; },
|
|
843
|
+
claudeAvailable: () => true,
|
|
844
|
+
});
|
|
845
|
+
assert.equal(res.path, "quick_reply");
|
|
846
|
+
assert.deepEqual(order, ["yield", "send"], "backlog is yielded the subscription BEFORE the reply's claude call, not after");
|
|
847
|
+
});
|
|
848
|
+
|
|
834
849
|
test("resolveSelfMemberId reads the member cuid from the res frame's `result` (own-echo keystone)", async () => {
|
|
835
850
|
daemon._resetSelfMemberId();
|
|
836
851
|
const pk = process.env.COHORT_API_KEY, po = process.env.COHORT_ORG_ID, pa = process.env.COHORT_AGENT_ID;
|
|
@@ -901,7 +901,23 @@ export function startConsumer(opts = {}) {
|
|
|
901
901
|
// the shared breaker and REQUEUE the tick unchanged (decision:"deferred")
|
|
902
902
|
// instead of failTick — a 429 is an upstream gate, not a per-event failure,
|
|
903
903
|
// so it must not burn this cadence's retry budget toward the DLQ.
|
|
904
|
-
|
|
904
|
+
const cadenceOut = result.stderr_tail || result.error || result.stdout_tail || "";
|
|
905
|
+
const cadenceUl = rateGuard.classifyUsageLimit?.(cadenceOut, { agentRoot }) ?? { isLimit: false, resetAt: null };
|
|
906
|
+
if (cadenceUl.isLimit) {
|
|
907
|
+
try {
|
|
908
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, cadenceUl.resetAt, { agentRoot });
|
|
909
|
+
log({ level: "warn", stage: "subsession_usage_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
910
|
+
} catch { /* */ }
|
|
911
|
+
// Same requeue-unchanged handling as a 429: a window-usage limit is a
|
|
912
|
+
// shared, provider-side gate — not evidence THIS cadence is broken — so
|
|
913
|
+
// requeue without failTick or a circuit trip; the shared breaker (held
|
|
914
|
+
// until reset) gates re-escalation.
|
|
915
|
+
requeueTick(agentRoot, event);
|
|
916
|
+
stats.retries += 1;
|
|
917
|
+
stats.last_decision = "deferred";
|
|
918
|
+
return { ok: false, decision: "deferred" };
|
|
919
|
+
}
|
|
920
|
+
if (rateGuard.classifyStderr(cadenceOut)) {
|
|
905
921
|
try {
|
|
906
922
|
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot });
|
|
907
923
|
log({ level: "warn", stage: "subsession_rate_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil });
|
|
@@ -455,7 +455,7 @@ async function runClaudeCLI(systemPrompt, userPrompt) {
|
|
|
455
455
|
const args = [
|
|
456
456
|
"--print",
|
|
457
457
|
...sessionPermissionArgs({ source: "classifier" }),
|
|
458
|
-
...daemonClaudeArgs(),
|
|
458
|
+
...daemonClaudeArgs(undefined, { source: "classifier" }),
|
|
459
459
|
"--model", ANTHROPIC_MODEL,
|
|
460
460
|
"--append-system-prompt", systemPrompt,
|
|
461
461
|
];
|
|
@@ -352,6 +352,89 @@ test("M2: evicting a backlog session for a priority inbox item clears its resume
|
|
|
352
352
|
}
|
|
353
353
|
});
|
|
354
354
|
|
|
355
|
+
// ---------------------------------------------------------------------------
|
|
356
|
+
// yieldBacklogForReactive — a reply about to hit the subscription evicts
|
|
357
|
+
// already-running BACKLOG down to REACTIVE_INFLIGHT_BACKLOG_MAX, and NEVER
|
|
358
|
+
// touches an inbox/reactive session.
|
|
359
|
+
// ---------------------------------------------------------------------------
|
|
360
|
+
|
|
361
|
+
/** A child stub that records whether it was killed (never launches anything). */
|
|
362
|
+
function killTrackingProc() {
|
|
363
|
+
const handlers = {};
|
|
364
|
+
const p = {
|
|
365
|
+
stdout: { on: () => {} },
|
|
366
|
+
stderr: { on: () => {} },
|
|
367
|
+
on: (ev, cb) => { handlers[ev] = cb; },
|
|
368
|
+
kill: () => { p.killed = true; },
|
|
369
|
+
killed: false,
|
|
370
|
+
_handlers: handlers,
|
|
371
|
+
};
|
|
372
|
+
return p;
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
test("yieldBacklogForReactive: evicts backlog down to the cap and spares every inbox session", async () => {
|
|
376
|
+
process.env.DAEMON_MAX_CONCURRENT = "8";
|
|
377
|
+
const { mod, dir } = await freshDispatcher();
|
|
378
|
+
try {
|
|
379
|
+
mod.setGovernanceForTests({ governor: fakeGovernor("ADMIT"), rateGuard: allowRate, budgetGuard: noBudget });
|
|
380
|
+
const procs = [];
|
|
381
|
+
const restore = mod.setSpawnForTests(() => { const p = killTrackingProc(); procs.push(p); return p; });
|
|
382
|
+
try {
|
|
383
|
+
// 3 backlog sessions (dispatch order = age order) …
|
|
384
|
+
for (let i = 0; i < 3; i++) {
|
|
385
|
+
mod.dispatch("backlog work", { id: `BL-${i}`, title: `bl ${i}`, source_file: "q.yaml" },
|
|
386
|
+
{ summary: `bl ${i}`, priority: "low", model: "sonnet" }, "backlog");
|
|
387
|
+
}
|
|
388
|
+
// … and 2 inbox sessions, all running concurrently.
|
|
389
|
+
for (let i = 0; i < 2; i++) {
|
|
390
|
+
mod.dispatch("inbox work", { id: `IN-${i}`, channel: `D${i}` },
|
|
391
|
+
{ summary: `in ${i}`, priority: "normal", model: "sonnet" }, "inbox");
|
|
392
|
+
}
|
|
393
|
+
assert.equal(mod.getStatus().active_sessions, 5, "3 backlog + 2 inbox live");
|
|
394
|
+
|
|
395
|
+
// A reply is about to generate → yield the subscription. Cap is 1, so 2 of
|
|
396
|
+
// the 3 backlog sessions are evicted; the inbox sessions are never touched.
|
|
397
|
+
const evicted = mod.yieldBacklogForReactive();
|
|
398
|
+
assert.equal(evicted, 2, "evicts backlog down to REACTIVE_INFLIGHT_BACKLOG_MAX (1)");
|
|
399
|
+
|
|
400
|
+
const backlogProcs = procs.slice(0, 3);
|
|
401
|
+
const inboxProcs = procs.slice(3, 5);
|
|
402
|
+
assert.equal(backlogProcs.filter((p) => p.killed).length, 2, "exactly two backlog procs killed");
|
|
403
|
+
assert.equal(inboxProcs.filter((p) => p.killed).length, 0, "no inbox/reactive proc is ever killed");
|
|
404
|
+
|
|
405
|
+
// Idempotent: already at the cap → a second call is a no-op.
|
|
406
|
+
assert.equal(mod.yieldBacklogForReactive(), 0, "no further eviction once at the cap");
|
|
407
|
+
} finally { restore(); }
|
|
408
|
+
} finally {
|
|
409
|
+
delete process.env.DAEMON_MAX_CONCURRENT;
|
|
410
|
+
await cleanup(dir);
|
|
411
|
+
}
|
|
412
|
+
});
|
|
413
|
+
|
|
414
|
+
test("yieldBacklogForReactive: no-op when backlog is already at/under the cap", async () => {
|
|
415
|
+
process.env.DAEMON_MAX_CONCURRENT = "8";
|
|
416
|
+
const { mod, dir } = await freshDispatcher();
|
|
417
|
+
try {
|
|
418
|
+
mod.setGovernanceForTests({ governor: fakeGovernor("ADMIT"), rateGuard: allowRate, budgetGuard: noBudget });
|
|
419
|
+
const restore = mod.setSpawnForTests(() => killTrackingProc());
|
|
420
|
+
try {
|
|
421
|
+
// Only inbox sessions running — nothing self-directed to yield.
|
|
422
|
+
for (let i = 0; i < 3; i++) {
|
|
423
|
+
mod.dispatch("inbox work", { id: `IN-${i}`, channel: `D${i}` },
|
|
424
|
+
{ summary: `in ${i}`, priority: "normal", model: "sonnet" }, "inbox");
|
|
425
|
+
}
|
|
426
|
+
assert.equal(mod.yieldBacklogForReactive(), 0, "nothing to evict");
|
|
427
|
+
// One backlog session — at the cap of 1 → still a no-op.
|
|
428
|
+
mod.dispatch("bl", { id: "BL-ONE", title: "one", source_file: "q.yaml" },
|
|
429
|
+
{ summary: "one", priority: "low", model: "sonnet" }, "backlog");
|
|
430
|
+
assert.equal(mod.yieldBacklogForReactive(), 0, "a single backlog session is within the cap");
|
|
431
|
+
} finally { restore(); }
|
|
432
|
+
} finally {
|
|
433
|
+
delete process.env.DAEMON_MAX_CONCURRENT;
|
|
434
|
+
await cleanup(dir);
|
|
435
|
+
}
|
|
436
|
+
});
|
|
437
|
+
|
|
355
438
|
// ---------------------------------------------------------------------------
|
|
356
439
|
// M1 — a governor/budget/rate DEFER that force-queues an inbox item while NO
|
|
357
440
|
// session is running must arm a single (debounced) re-drain timer, so the item
|
|
@@ -622,8 +622,12 @@ function defaultSpawnResume({ marker }) {
|
|
|
622
622
|
clearResumePending(marker.sessionId);
|
|
623
623
|
logSession({ event: "resume_completed", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
624
624
|
} else {
|
|
625
|
-
// A 429 during resume must open the breaker so subsequent
|
|
626
|
-
|
|
625
|
+
// A usage limit or 429 during resume must open the breaker so subsequent
|
|
626
|
+
// spawns gate. Usage-limit → hold until reset; 429 → transient backoff.
|
|
627
|
+
const ul = rateGuard.classifyUsageLimit?.(stderr, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
628
|
+
if (ul.isLimit) {
|
|
629
|
+
try { rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
630
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
627
631
|
try { rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
628
632
|
}
|
|
629
633
|
logSession({ event: "resume_exit_nonzero", sessionId: marker.sessionId, item_id: marker.itemId, exit_code: code });
|
|
@@ -680,42 +684,97 @@ function countBySource(source) {
|
|
|
680
684
|
|
|
681
685
|
/**
|
|
682
686
|
* Evict the lowest-priority, longest-running backlog session to make room
|
|
683
|
-
* for a high-priority inbox item
|
|
687
|
+
* for a high-priority inbox item (or to yield the subscription to a reactive
|
|
688
|
+
* reply). Returns true if a session was evicted.
|
|
689
|
+
*
|
|
690
|
+
* @param {string} [why] human-readable eviction reason for the log/ledger.
|
|
691
|
+
*/
|
|
692
|
+
const EVICT_PRIORITY_RANK = { low: 0, normal: 1, high: 2, critical: 3 };
|
|
693
|
+
|
|
694
|
+
/**
|
|
695
|
+
* SIGTERM one backlog session and retire its resume marker. Shared by the
|
|
696
|
+
* priority-preemption path and the reactive-yield path so both do the identical
|
|
697
|
+
* bookkeeping. Marks `s.evicting` so a SECOND selection in the same synchronous
|
|
698
|
+
* pass (before the async close handler deletes the session) cannot re-target the
|
|
699
|
+
* same not-yet-closed session — the bug that let one kill masquerade as many.
|
|
684
700
|
*/
|
|
685
|
-
function
|
|
686
|
-
|
|
701
|
+
function _evictSession(id, s, why) {
|
|
702
|
+
console.log(`[dispatcher] PREEMPT: Evicting backlog session ${id} (${s.classResult.summary}) for ${why}`);
|
|
703
|
+
logSession({ event: "evicted", sessionId: id, reason: "priority_preemption", summary: s.classResult.summary });
|
|
704
|
+
// M2: clear the resume-pending marker BEFORE the SIGTERM. Eviction is a
|
|
705
|
+
// deliberate preemption, not a crash — the item stays on disk and the
|
|
706
|
+
// backlog sweep re-dispatches it normally (acquiring the item-claim).
|
|
707
|
+
// If we left the marker, a reboot's reconcileResumePending would re-spawn
|
|
708
|
+
// it WITHOUT the claim while sweepBacklog independently claim+dispatched
|
|
709
|
+
// the same item → two concurrent runs. The non-zero SIGTERM close would
|
|
710
|
+
// otherwise keep the marker, so we must retire it here.
|
|
711
|
+
clearResumePending(id);
|
|
712
|
+
s.evicting = true;
|
|
713
|
+
s.process.kill("SIGTERM");
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
/** The single lowest-priority, oldest evictable backlog session, or null. */
|
|
717
|
+
function _worstBacklogSession() {
|
|
687
718
|
let worst = null;
|
|
688
719
|
let worstId = null;
|
|
689
|
-
|
|
690
720
|
for (const [id, s] of activeSessions) {
|
|
691
|
-
if (s.source !== "backlog") continue;
|
|
692
|
-
const rank =
|
|
693
|
-
if (!worst) {
|
|
694
|
-
|
|
695
|
-
|
|
696
|
-
}
|
|
697
|
-
const worstRank = priorityRank[worst.classResult.priority] || 1;
|
|
698
|
-
// Prefer to evict lower priority; break ties by oldest session
|
|
721
|
+
if (s.source !== "backlog" || s.evicting) continue;
|
|
722
|
+
const rank = EVICT_PRIORITY_RANK[s.classResult.priority] ?? 1;
|
|
723
|
+
if (!worst) { worst = s; worstId = id; continue; }
|
|
724
|
+
const worstRank = EVICT_PRIORITY_RANK[worst.classResult.priority] ?? 1;
|
|
725
|
+
// Prefer to evict lower priority; break ties by oldest session.
|
|
699
726
|
if (rank < worstRank || (rank === worstRank && s.startTime < worst.startTime)) {
|
|
700
727
|
worst = s; worstId = id;
|
|
701
728
|
}
|
|
702
729
|
}
|
|
730
|
+
return worst ? { id: worstId, s: worst } : null;
|
|
731
|
+
}
|
|
703
732
|
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
713
|
-
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
|
|
733
|
+
function evictForPriority(why = "priority inbox item") {
|
|
734
|
+
const target = _worstBacklogSession();
|
|
735
|
+
if (!target) return false;
|
|
736
|
+
_evictSession(target.id, target.s, why);
|
|
737
|
+
return true;
|
|
738
|
+
}
|
|
739
|
+
|
|
740
|
+
/**
|
|
741
|
+
* Yield the shared subscription to a reactive turn that is about to generate.
|
|
742
|
+
*
|
|
743
|
+
* REACTIVE_RESERVE keeps a slot free and the governor now QUEUEs NEW backlog
|
|
744
|
+
* while a reply is in flight — but neither of those stops backlog sessions that
|
|
745
|
+
* are ALREADY RUNNING from saturating the one Max subscription (the measured
|
|
746
|
+
* cause of 19s quick replies and 30s classify timeouts). This is the other half:
|
|
747
|
+
* called the instant a reactive reply is about to hit `claude`, it evicts
|
|
748
|
+
* already-running BACKLOG sessions down to REACTIVE_INFLIGHT_BACKLOG_MAX so the
|
|
749
|
+
* reply's own call runs on a nearly-idle subscription.
|
|
750
|
+
*
|
|
751
|
+
* Safe and conservative: it never touches another reactive/inbox session (only
|
|
752
|
+
* `source === "backlog"`); an evicted item stays on disk and the sweep re-claims
|
|
753
|
+
* it, so backlog loses concurrency for a few seconds, never work. No-op when
|
|
754
|
+
* backlog is already at/under the cap. Returns the number of sessions evicted.
|
|
755
|
+
*/
|
|
756
|
+
export function yieldBacklogForReactive() {
|
|
757
|
+
const cap = Math.max(0, Number(governor.REACTIVE_INFLIGHT_BACKLOG_MAX ?? 1));
|
|
758
|
+
// Live (not-already-evicting) backlog sessions, worst-first, so we shed the
|
|
759
|
+
// lowest-priority / oldest work and keep the freshest `cap` running.
|
|
760
|
+
const live = [];
|
|
761
|
+
for (const [id, s] of activeSessions) {
|
|
762
|
+
if (s.source === "backlog" && !s.evicting) live.push({ id, s });
|
|
717
763
|
}
|
|
718
|
-
|
|
764
|
+
live.sort((a, b) => {
|
|
765
|
+
const ra = EVICT_PRIORITY_RANK[a.s.classResult.priority] ?? 1;
|
|
766
|
+
const rb = EVICT_PRIORITY_RANK[b.s.classResult.priority] ?? 1;
|
|
767
|
+
return ra - rb || a.s.startTime - b.s.startTime;
|
|
768
|
+
});
|
|
769
|
+
let evicted = 0;
|
|
770
|
+
for (let i = 0; i < live.length - cap; i++) {
|
|
771
|
+
_evictSession(live[i].id, live[i].s, "reactive reply waiting on the subscription");
|
|
772
|
+
evicted++;
|
|
773
|
+
}
|
|
774
|
+
if (evicted > 0) {
|
|
775
|
+
console.log(`[dispatcher] Yielded subscription to a reactive reply: evicted ${evicted} backlog session(s) (cap ${cap})`);
|
|
776
|
+
}
|
|
777
|
+
return evicted;
|
|
719
778
|
}
|
|
720
779
|
|
|
721
780
|
/**
|
|
@@ -1271,11 +1330,25 @@ function spawnSession(entry) {
|
|
|
1271
1330
|
if (code === 0) {
|
|
1272
1331
|
clearResumePending(sessionId);
|
|
1273
1332
|
try { rateGuard.recordSuccess(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
1274
|
-
} else
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1333
|
+
} else {
|
|
1334
|
+
// A Max/subscription USAGE limit ("you've hit your session limit · resets
|
|
1335
|
+
// 9:40pm") is checked FIRST and on BOTH streams — for `claude --print` it
|
|
1336
|
+
// lands on stdout, not stderr. It holds the breaker until the window
|
|
1337
|
+
// RESETS (not a decorrelated 429 backoff), so the seat stops storming a
|
|
1338
|
+
// drained pool. A plain 429/overload falls through to the transient path.
|
|
1339
|
+
const combined = `${stdout || ""}\n${stderr || ""}`;
|
|
1340
|
+
const ul = rateGuard.classifyUsageLimit?.(combined, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
1341
|
+
if (ul.isLimit) {
|
|
1342
|
+
try {
|
|
1343
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR });
|
|
1344
|
+
logSession({ event: "usage_limit_recorded", sessionId, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
1345
|
+
} catch { /* */ }
|
|
1346
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
1347
|
+
try {
|
|
1348
|
+
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
1349
|
+
logSession({ event: "rate_limit_recorded", sessionId, open_until: rec.openUntil, consecutive: rec.consecutive429 });
|
|
1350
|
+
} catch { /* */ }
|
|
1351
|
+
}
|
|
1279
1352
|
}
|
|
1280
1353
|
|
|
1281
1354
|
// Release item lock — MUST use same key order as acquireLock in daemon
|
|
@@ -259,7 +259,7 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
|
|
|
259
259
|
const args = [
|
|
260
260
|
"--print",
|
|
261
261
|
...sessionPermissionArgs({ source: "responder" }),
|
|
262
|
-
...daemonClaudeArgs(),
|
|
262
|
+
...daemonClaudeArgs(undefined, { source: "responder" }),
|
|
263
263
|
"--model", model,
|
|
264
264
|
"--append-system-prompt", systemPrompt,
|
|
265
265
|
// --output-format json is only valid in combination with --print (per b1
|