@cohortapp/agent-sdk 2.11.12 → 2.11.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/claude-bin.mjs +17 -0
- package/lib/claude-bin.test.mjs +22 -0
- package/lib/rate-guard.mjs +67 -1
- package/lib/rate-guard.test.mjs +71 -0
- package/package.json +1 -1
- package/scripts/daemon/cadence-consumer.mjs +17 -1
- package/scripts/daemon/classifier.mjs +1 -1
- package/scripts/daemon/dispatcher.mjs +25 -7
- package/scripts/daemon/responder.mjs +1 -1
package/lib/claude-bin.mjs
CHANGED
|
@@ -129,9 +129,26 @@ export function daemonClaudeArgs(agentRoot, deps = {}) {
|
|
|
129
129
|
const env = deps.env || process.env;
|
|
130
130
|
if (env.DAEMON_LOAD_MCPS === "1") return [];
|
|
131
131
|
const ex = deps.existsSync || existsSync;
|
|
132
|
+
const source = deps.source || null;
|
|
132
133
|
|
|
133
134
|
const args = ["--strict-mcp-config"];
|
|
134
135
|
|
|
136
|
+
// MCP BY SOURCE — the biggest cheap per-spawn token saving. A classify or a
|
|
137
|
+
// quick-reply generates TEXT and calls NO org tools, so re-paying the ~107-tool
|
|
138
|
+
// org MCP definitions on every one of those high-frequency spawns is pure
|
|
139
|
+
// waste. Those sources get a BARE spawn (`--strict-mcp-config` with no
|
|
140
|
+
// `--mcp-config` loads zero MCP servers); only full work sessions
|
|
141
|
+
// (dispatcher/cadence — source unset) load the org toolset, so the API/tool-
|
|
142
|
+
// parity bar is honoured exactly where real work happens. Override per source
|
|
143
|
+
// with DAEMON_<SOURCE>_MCP=full|bare (e.g. DAEMON_CLASSIFIER_MCP=full).
|
|
144
|
+
const BARE_SOURCES = new Set(["classifier", "responder"]);
|
|
145
|
+
const perSource = source ? env[`DAEMON_${source.toUpperCase()}_MCP`] : null;
|
|
146
|
+
const bare = perSource === "bare" || (perSource !== "full" && BARE_SOURCES.has(source));
|
|
147
|
+
if (bare) {
|
|
148
|
+
if (env.DAEMON_BARE_MODE === "1") args.unshift("--bare");
|
|
149
|
+
return args;
|
|
150
|
+
}
|
|
151
|
+
|
|
135
152
|
const root = agentRoot || env.AGENT_ROOT || process.cwd();
|
|
136
153
|
try {
|
|
137
154
|
const cfg = join(root, ".mcp.json");
|
package/lib/claude-bin.test.mjs
CHANGED
|
@@ -107,3 +107,25 @@ test("daemonClaudeArgs never throws on a bad root (a daemon must still spawn)",
|
|
|
107
107
|
const args = daemonClaudeArgs("/agent", { existsSync: () => { throw new Error("EACCES"); }, env: {} });
|
|
108
108
|
assert.deepEqual(args, ["--strict-mcp-config"]);
|
|
109
109
|
});
|
|
110
|
+
|
|
111
|
+
// ---------------------------------------------------------------------------
|
|
112
|
+
// MCP by source — bare for tool-free spawns (classifier / responder)
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
test("daemonClaudeArgs: classifier + responder are BARE (no org MCP re-paid)", () => {
|
|
116
|
+
for (const source of ["classifier", "responder"]) {
|
|
117
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {}, source });
|
|
118
|
+
assert.ok(args.includes("--strict-mcp-config"), `${source} keeps strict-mcp-config`);
|
|
119
|
+
assert.ok(!args.includes("--mcp-config"), `${source} must NOT load the org MCP`);
|
|
120
|
+
}
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("daemonClaudeArgs: a full work session (no source) still loads the org MCP", () => {
|
|
124
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {} });
|
|
125
|
+
assert.ok(args.includes("--mcp-config"), "dispatcher/cadence session keeps full org tools");
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
test("daemonClaudeArgs: DAEMON_<SOURCE>_MCP=full overrides bare for that source", () => {
|
|
129
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: { DAEMON_CLASSIFIER_MCP: "full" }, source: "classifier" });
|
|
130
|
+
assert.ok(args.includes("--mcp-config"), "explicit override restores the org MCP for the classifier");
|
|
131
|
+
});
|
package/lib/rate-guard.mjs
CHANGED
|
@@ -243,4 +243,70 @@ export function classifyStderr(text) {
|
|
|
243
243
|
return /\b429\b|rate[\s_-]?limit|overloaded|too many requests/i.test(text);
|
|
244
244
|
}
|
|
245
245
|
|
|
246
|
-
|
|
246
|
+
// A Max/subscription USAGE or SESSION limit is NOT a transient 429 — the pool is
|
|
247
|
+
// drained until the window RESETS, so hammering it with decorrelated-jitter
|
|
248
|
+
// retries just burns the moment it re-opens. We detect it separately and hold
|
|
249
|
+
// the breaker until the reset (parsed from the message when present, else a
|
|
250
|
+
// conservative default) so the seat backs off to the reset instead of storming.
|
|
251
|
+
const USAGE_LIMIT_RE =
|
|
252
|
+
/\b(?:usage|session|weekly|5[\s-]?hour)\s+limit\b|limit reached|reached your (?:usage|session|monthly|weekly|plan)?\s*limit|you'?ve hit your [^.]*limit|approaching (?:your )?[^.]*usage limit|out of (?:usage|credits|messages)/i;
|
|
253
|
+
|
|
254
|
+
/** How long to hold the breaker when a usage limit is hit but no reset time is parseable. */
|
|
255
|
+
export const RATE_USAGE_LIMIT_HOLD_MS = num(process.env.RATE_USAGE_LIMIT_HOLD_MS, 30 * 60 * 1000);
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Best-effort parse of a reset time out of a usage-limit message —
|
|
259
|
+
* "resets 9:40pm", "resets at 9pm", "try again at 10:15am", "available again at
|
|
260
|
+
* 3pm" — into the epoch ms of its NEXT occurrence (today, or tomorrow if already
|
|
261
|
+
* past). The daemon runs in the seat's local timezone, which is the timezone the
|
|
262
|
+
* message states, so a local Date is correct. Returns null when unparseable.
|
|
263
|
+
*/
|
|
264
|
+
export function parseResetAt(text, deps) {
|
|
265
|
+
if (!text || typeof text !== "string") return null;
|
|
266
|
+
const m = /(?:reset|resets|resets at|resets in|try again at|available again at)\s+(\d{1,2})(?::(\d{2}))?\s*([ap])\.?\s?m\.?/i.exec(text);
|
|
267
|
+
if (!m) return null;
|
|
268
|
+
const nowMs = clock(deps)();
|
|
269
|
+
let hr = parseInt(m[1], 10) % 12;
|
|
270
|
+
if (/p/i.test(m[3])) hr += 12;
|
|
271
|
+
const min = m[2] ? parseInt(m[2], 10) : 0;
|
|
272
|
+
const cand = new Date(nowMs);
|
|
273
|
+
cand.setHours(hr, min, 0, 0);
|
|
274
|
+
let t = cand.getTime();
|
|
275
|
+
if (t <= nowMs) t += 24 * 60 * 60 * 1000; // already past today → next occurrence
|
|
276
|
+
return t;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Does this text look like a Max/subscription USAGE / SESSION limit (as opposed
|
|
281
|
+
* to a transient 429)? When it names a reset time, resetAt is the epoch ms to
|
|
282
|
+
* hold the breaker until.
|
|
283
|
+
*
|
|
284
|
+
* @returns {{ isLimit: boolean, resetAt: number|null }}
|
|
285
|
+
*/
|
|
286
|
+
export function classifyUsageLimit(text, deps) {
|
|
287
|
+
if (!text || typeof text !== "string") return { isLimit: false, resetAt: null };
|
|
288
|
+
if (!USAGE_LIMIT_RE.test(text)) return { isLimit: false, resetAt: null };
|
|
289
|
+
return { isLimit: true, resetAt: parseResetAt(text, deps) };
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Open the breaker until a usage window RESETS. Uses resetAt when known, else a
|
|
294
|
+
* conservative default hold. Never shortens an already-longer open window.
|
|
295
|
+
*
|
|
296
|
+
* @returns {{ openUntil:number, resetAt:number|null }}
|
|
297
|
+
*/
|
|
298
|
+
export function recordUsageLimit(provider, resetAt, deps) {
|
|
299
|
+
const now = clock(deps)();
|
|
300
|
+
const until =
|
|
301
|
+
typeof resetAt === "number" && resetAt > now ? resetAt : now + RATE_USAGE_LIMIT_HOLD_MS;
|
|
302
|
+
const prev = readState(provider, deps);
|
|
303
|
+
const next = {
|
|
304
|
+
openUntil: Math.max(prev.openUntil, until),
|
|
305
|
+
consecutive429: prev.consecutive429,
|
|
306
|
+
lastBackoffMs: prev.lastBackoffMs,
|
|
307
|
+
};
|
|
308
|
+
writeState(provider, next, deps);
|
|
309
|
+
return { openUntil: next.openUntil, resetAt: typeof resetAt === "number" && resetAt > now ? resetAt : null };
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
export const _internals = { RATE_BASE_BACKOFF_MS, RATE_MAX_BACKOFF_MS, RATE_BACKOFF_FACTOR, RATE_USAGE_LIMIT_HOLD_MS };
|
package/lib/rate-guard.test.mjs
CHANGED
|
@@ -18,11 +18,15 @@ import {
|
|
|
18
18
|
recordSuccess,
|
|
19
19
|
decorrelatedBackoff,
|
|
20
20
|
classifyStderr,
|
|
21
|
+
classifyUsageLimit,
|
|
22
|
+
parseResetAt,
|
|
23
|
+
recordUsageLimit,
|
|
21
24
|
readState,
|
|
22
25
|
sanitizeProvider,
|
|
23
26
|
RATE_BASE_BACKOFF_MS,
|
|
24
27
|
RATE_MAX_BACKOFF_MS,
|
|
25
28
|
RATE_BACKOFF_FACTOR,
|
|
29
|
+
RATE_USAGE_LIMIT_HOLD_MS,
|
|
26
30
|
} from "./rate-guard.mjs";
|
|
27
31
|
|
|
28
32
|
async function makeStateDir() {
|
|
@@ -199,3 +203,70 @@ test("readState tolerates a corrupt file (fails closed-to-empty, never throws)",
|
|
|
199
203
|
assert.equal(checkRateLimit("anthropic", { stateDir: dir }).allowed, true);
|
|
200
204
|
} finally { await rm(dir); }
|
|
201
205
|
});
|
|
206
|
+
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
// classifyUsageLimit / parseResetAt / recordUsageLimit (Max session-limit)
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
test("classifyUsageLimit detects session/usage limits, not plain 429s", () => {
|
|
212
|
+
assert.equal(classifyUsageLimit("You've hit your session limit · resets 9:40pm (Australia/Sydney)").isLimit, true);
|
|
213
|
+
assert.equal(classifyUsageLimit("usage limit reached").isLimit, true);
|
|
214
|
+
assert.equal(classifyUsageLimit("You have reached your weekly limit").isLimit, true);
|
|
215
|
+
assert.equal(classifyUsageLimit("out of usage for now").isLimit, true);
|
|
216
|
+
// A plain transient 429 is NOT a usage-window limit
|
|
217
|
+
assert.equal(classifyUsageLimit("HTTP 429 Too Many Requests").isLimit, false);
|
|
218
|
+
assert.equal(classifyUsageLimit("overloaded").isLimit, false);
|
|
219
|
+
assert.equal(classifyUsageLimit("").isLimit, false);
|
|
220
|
+
assert.equal(classifyUsageLimit(null).isLimit, false);
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
test("parseResetAt resolves a stated reset time to the next occurrence", () => {
|
|
224
|
+
// now = 2026-08-26 08:00 local; "resets 9:40pm" is later today
|
|
225
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
226
|
+
const t = parseResetAt("resets 9:40pm", { now: () => now });
|
|
227
|
+
const d = new Date(t);
|
|
228
|
+
assert.equal(d.getHours(), 21);
|
|
229
|
+
assert.equal(d.getMinutes(), 40);
|
|
230
|
+
assert.ok(t > now, "reset is in the future");
|
|
231
|
+
// A time already past today rolls to tomorrow
|
|
232
|
+
const now2 = new Date(2026, 7, 26, 22, 0, 0, 0).getTime(); // 10pm
|
|
233
|
+
const t2 = parseResetAt("try again at 9pm", { now: () => now2 });
|
|
234
|
+
assert.ok(t2 > now2 && t2 - now2 <= 24 * 60 * 60 * 1000, "rolls to next day");
|
|
235
|
+
assert.equal(parseResetAt("no time here", { now: () => now }), null);
|
|
236
|
+
});
|
|
237
|
+
|
|
238
|
+
test("recordUsageLimit holds the breaker until the parsed reset time", async () => {
|
|
239
|
+
const dir = await makeStateDir();
|
|
240
|
+
try {
|
|
241
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
242
|
+
const { resetAt } = classifyUsageLimit("session limit · resets 9:40pm", { now: () => now });
|
|
243
|
+
const rec = recordUsageLimit("anthropic", resetAt, { stateDir: dir, now: () => now });
|
|
244
|
+
assert.equal(rec.openUntil, resetAt);
|
|
245
|
+
// blocked before reset, allowed after
|
|
246
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
247
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => resetAt + 1 }).allowed, true);
|
|
248
|
+
} finally { await rm(dir); }
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
test("recordUsageLimit falls back to a default hold when no reset is known", async () => {
|
|
252
|
+
const dir = await makeStateDir();
|
|
253
|
+
try {
|
|
254
|
+
const now = 1_000_000;
|
|
255
|
+
const rec = recordUsageLimit("anthropic", null, { stateDir: dir, now: () => now });
|
|
256
|
+
assert.equal(rec.openUntil, now + RATE_USAGE_LIMIT_HOLD_MS);
|
|
257
|
+
assert.equal(rec.resetAt, null);
|
|
258
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
259
|
+
} finally { await rm(dir); }
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
test("recordUsageLimit never shortens an already-longer open window", async () => {
|
|
263
|
+
const dir = await makeStateDir();
|
|
264
|
+
try {
|
|
265
|
+
const now = 1_000_000;
|
|
266
|
+
const far = now + 10 * 60 * 60 * 1000; // 10h out
|
|
267
|
+
recordUsageLimit("anthropic", far, { stateDir: dir, now: () => now });
|
|
268
|
+
// a later, SHORTER hold must not pull the window in
|
|
269
|
+
const rec = recordUsageLimit("anthropic", now + 5 * 60 * 1000, { stateDir: dir, now: () => now });
|
|
270
|
+
assert.equal(rec.openUntil, far);
|
|
271
|
+
} finally { await rm(dir); }
|
|
272
|
+
});
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cohortapp/agent-sdk",
|
|
3
|
-
"version": "2.11.
|
|
3
|
+
"version": "2.11.13",
|
|
4
4
|
"description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -901,7 +901,23 @@ export function startConsumer(opts = {}) {
|
|
|
901
901
|
// the shared breaker and REQUEUE the tick unchanged (decision:"deferred")
|
|
902
902
|
// instead of failTick — a 429 is an upstream gate, not a per-event failure,
|
|
903
903
|
// so it must not burn this cadence's retry budget toward the DLQ.
|
|
904
|
-
|
|
904
|
+
const cadenceOut = result.stderr_tail || result.error || result.stdout_tail || "";
|
|
905
|
+
const cadenceUl = rateGuard.classifyUsageLimit?.(cadenceOut, { agentRoot }) ?? { isLimit: false, resetAt: null };
|
|
906
|
+
if (cadenceUl.isLimit) {
|
|
907
|
+
try {
|
|
908
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, cadenceUl.resetAt, { agentRoot });
|
|
909
|
+
log({ level: "warn", stage: "subsession_usage_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
910
|
+
} catch { /* */ }
|
|
911
|
+
// Same requeue-unchanged handling as a 429: a window-usage limit is a
|
|
912
|
+
// shared, provider-side gate — not evidence THIS cadence is broken — so
|
|
913
|
+
// requeue without failTick or a circuit trip; the shared breaker (held
|
|
914
|
+
// until reset) gates re-escalation.
|
|
915
|
+
requeueTick(agentRoot, event);
|
|
916
|
+
stats.retries += 1;
|
|
917
|
+
stats.last_decision = "deferred";
|
|
918
|
+
return { ok: false, decision: "deferred" };
|
|
919
|
+
}
|
|
920
|
+
if (rateGuard.classifyStderr(cadenceOut)) {
|
|
905
921
|
try {
|
|
906
922
|
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot });
|
|
907
923
|
log({ level: "warn", stage: "subsession_rate_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil });
|
|
@@ -455,7 +455,7 @@ async function runClaudeCLI(systemPrompt, userPrompt) {
|
|
|
455
455
|
const args = [
|
|
456
456
|
"--print",
|
|
457
457
|
...sessionPermissionArgs({ source: "classifier" }),
|
|
458
|
-
...daemonClaudeArgs(),
|
|
458
|
+
...daemonClaudeArgs(undefined, { source: "classifier" }),
|
|
459
459
|
"--model", ANTHROPIC_MODEL,
|
|
460
460
|
"--append-system-prompt", systemPrompt,
|
|
461
461
|
];
|
|
@@ -622,8 +622,12 @@ function defaultSpawnResume({ marker }) {
|
|
|
622
622
|
clearResumePending(marker.sessionId);
|
|
623
623
|
logSession({ event: "resume_completed", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
624
624
|
} else {
|
|
625
|
-
// A 429 during resume must open the breaker so subsequent
|
|
626
|
-
|
|
625
|
+
// A usage limit or 429 during resume must open the breaker so subsequent
|
|
626
|
+
// spawns gate. Usage-limit → hold until reset; 429 → transient backoff.
|
|
627
|
+
const ul = rateGuard.classifyUsageLimit?.(stderr, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
628
|
+
if (ul.isLimit) {
|
|
629
|
+
try { rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
630
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
627
631
|
try { rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
628
632
|
}
|
|
629
633
|
logSession({ event: "resume_exit_nonzero", sessionId: marker.sessionId, item_id: marker.itemId, exit_code: code });
|
|
@@ -1326,11 +1330,25 @@ function spawnSession(entry) {
|
|
|
1326
1330
|
if (code === 0) {
|
|
1327
1331
|
clearResumePending(sessionId);
|
|
1328
1332
|
try { rateGuard.recordSuccess(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
1329
|
-
} else
|
|
1330
|
-
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1333
|
+
} else {
|
|
1334
|
+
// A Max/subscription USAGE limit ("you've hit your session limit · resets
|
|
1335
|
+
// 9:40pm") is checked FIRST and on BOTH streams — for `claude --print` it
|
|
1336
|
+
// lands on stdout, not stderr. It holds the breaker until the window
|
|
1337
|
+
// RESETS (not a decorrelated 429 backoff), so the seat stops storming a
|
|
1338
|
+
// drained pool. A plain 429/overload falls through to the transient path.
|
|
1339
|
+
const combined = `${stdout || ""}\n${stderr || ""}`;
|
|
1340
|
+
const ul = rateGuard.classifyUsageLimit?.(combined, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
1341
|
+
if (ul.isLimit) {
|
|
1342
|
+
try {
|
|
1343
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR });
|
|
1344
|
+
logSession({ event: "usage_limit_recorded", sessionId, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
1345
|
+
} catch { /* */ }
|
|
1346
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
1347
|
+
try {
|
|
1348
|
+
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
1349
|
+
logSession({ event: "rate_limit_recorded", sessionId, open_until: rec.openUntil, consecutive: rec.consecutive429 });
|
|
1350
|
+
} catch { /* */ }
|
|
1351
|
+
}
|
|
1334
1352
|
}
|
|
1335
1353
|
|
|
1336
1354
|
// Release item lock — MUST use same key order as acquireLock in daemon
|
|
@@ -259,7 +259,7 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
|
|
|
259
259
|
const args = [
|
|
260
260
|
"--print",
|
|
261
261
|
...sessionPermissionArgs({ source: "responder" }),
|
|
262
|
-
...daemonClaudeArgs(),
|
|
262
|
+
...daemonClaudeArgs(undefined, { source: "responder" }),
|
|
263
263
|
"--model", model,
|
|
264
264
|
"--append-system-prompt", systemPrompt,
|
|
265
265
|
// --output-format json is only valid in combination with --print (per b1
|