@cohortapp/agent-sdk 2.11.12 → 2.11.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/budget-guard.mjs +96 -0
- package/lib/budget-guard.test.mjs +62 -0
- package/lib/claude-bin.mjs +17 -0
- package/lib/claude-bin.test.mjs +22 -0
- package/lib/rate-guard.mjs +67 -1
- package/lib/rate-guard.test.mjs +71 -0
- package/lib/resource-governor.mjs +17 -1
- package/lib/resource-governor.test.mjs +27 -0
- package/package.json +1 -1
- package/scripts/daemon/cadence-consumer.mjs +20 -2
- package/scripts/daemon/classifier.mjs +1 -1
- package/scripts/daemon/dispatcher-governance.test.mjs +35 -1
- package/scripts/daemon/dispatcher.mjs +60 -10
- package/scripts/daemon/responder.mjs +1 -1
package/lib/budget-guard.mjs
CHANGED
|
@@ -1086,3 +1086,99 @@ function formatNotice(band, status) {
|
|
|
1086
1086
|
}
|
|
1087
1087
|
return `Seat spend at ${status.pct}% ($${usd} / $${cap} today) — heads-up only, nothing is degraded yet.${envelope}`;
|
|
1088
1088
|
}
|
|
1089
|
+
|
|
1090
|
+
// ---------------------------------------------------------------------------
|
|
1091
|
+
// Rolling usage window (subscription seats: TOKEN VOLUME, not dollars)
|
|
1092
|
+
// ---------------------------------------------------------------------------
|
|
1093
|
+
//
|
|
1094
|
+
// A Max seat has no dollar meter; what it can exhaust is a rolling token window
|
|
1095
|
+
// (≈5-hour session + weekly). The exact caps are UNPUBLISHED, so pacing is
|
|
1096
|
+
// DEFAULT-OFF: with GOV_WINDOW_5H_TOKENS / GOV_WEEK_TOKENS unset (0),
|
|
1097
|
+
// windowBand() is always "ok" and nothing paces — but the rolling volume is
|
|
1098
|
+
// still COMPUTED and surfaced so an operator can watch usage climb and calibrate
|
|
1099
|
+
// the caps. Reads the SAME cost-tracking ledger dailyStatus uses; "tokens" here
|
|
1100
|
+
// is input+output+cache summed over the window. The hard safety net remains
|
|
1101
|
+
// rate-guard's usage-limit recognition (recordUsageLimit holds the breaker until
|
|
1102
|
+
// reset); this is the proactive, opt-in layer above it.
|
|
1103
|
+
|
|
1104
|
+
const WINDOW_5H_MS = 5 * 60 * 60 * 1000;
|
|
1105
|
+
const WINDOW_WEEK_MS = 7 * 24 * 60 * 60 * 1000;
|
|
1106
|
+
export const WINDOW_5H_TOKEN_CAP = Math.max(0, parseInt(process.env.GOV_WINDOW_5H_TOKENS || "0", 10) || 0);
|
|
1107
|
+
export const WINDOW_WEEK_TOKEN_CAP = Math.max(0, parseInt(process.env.GOV_WEEK_TOKENS || "0", 10) || 0);
|
|
1108
|
+
export const WINDOW_APPROACH_FRACTION = (() => {
|
|
1109
|
+
const f = parseFloat(process.env.GOV_WINDOW_APPROACH || "0.8");
|
|
1110
|
+
return f > 0 && f < 1 ? f : 0.8;
|
|
1111
|
+
})();
|
|
1112
|
+
|
|
1113
|
+
function rowTokens(r) {
|
|
1114
|
+
return (
|
|
1115
|
+
(Number(r.input_tokens) || 0) +
|
|
1116
|
+
(Number(r.output_tokens) || 0) +
|
|
1117
|
+
(Number(r.cache_read_input_tokens) || 0) +
|
|
1118
|
+
(Number(r.cache_creation_input_tokens) || 0)
|
|
1119
|
+
);
|
|
1120
|
+
}
|
|
1121
|
+
|
|
1122
|
+
/**
|
|
1123
|
+
* Rolling token volume + session count over the last 5 hours and 7 days, from
|
|
1124
|
+
* the cost-tracking ledger. Never throws (a missing/unreadable ledger → zeros).
|
|
1125
|
+
*/
|
|
1126
|
+
export function rollingWindowUsage(deps) {
|
|
1127
|
+
const now = clock(deps)();
|
|
1128
|
+
const dir = ledgerDir(deps);
|
|
1129
|
+
let t5 = 0, s5 = 0, t7 = 0, s7 = 0;
|
|
1130
|
+
let files = [];
|
|
1131
|
+
try {
|
|
1132
|
+
files = readdirSync(dir).filter((f) => /^\d{4}-\d{2}-\d{2}\.jsonl$/.test(f)).sort().slice(-8);
|
|
1133
|
+
} catch {
|
|
1134
|
+
return { window5hTokens: 0, window5hSessions: 0, week7dTokens: 0, week7dSessions: 0 };
|
|
1135
|
+
}
|
|
1136
|
+
for (const f of files) {
|
|
1137
|
+
let body;
|
|
1138
|
+
try { body = readFileSync(join(dir, f), "utf-8"); } catch { continue; }
|
|
1139
|
+
for (const line of body.split("\n")) {
|
|
1140
|
+
if (!line.trim()) continue;
|
|
1141
|
+
let r;
|
|
1142
|
+
try { r = JSON.parse(line); } catch { continue; }
|
|
1143
|
+
const at = r && r.ts ? Date.parse(r.ts) : NaN;
|
|
1144
|
+
if (!Number.isFinite(at)) continue;
|
|
1145
|
+
const dt = now - at;
|
|
1146
|
+
if (dt < 0 || dt > WINDOW_WEEK_MS) continue;
|
|
1147
|
+
const tok = rowTokens(r);
|
|
1148
|
+
t7 += tok; s7 += 1;
|
|
1149
|
+
if (dt <= WINDOW_5H_MS) { t5 += tok; s5 += 1; }
|
|
1150
|
+
}
|
|
1151
|
+
}
|
|
1152
|
+
return { window5hTokens: t5, window5hSessions: s5, week7dTokens: t7, week7dSessions: s7 };
|
|
1153
|
+
}
|
|
1154
|
+
|
|
1155
|
+
/**
|
|
1156
|
+
* The usage-window band: "ok" | "approaching" | "at". DEFAULT-OFF — an unset
|
|
1157
|
+
* (0) cap yields "ok" for that horizon, so the fleet paces on the window only
|
|
1158
|
+
* once an operator has calibrated GOV_WINDOW_5H_TOKENS / GOV_WEEK_TOKENS. The
|
|
1159
|
+
* band is the WORSE of the 5h and weekly horizons.
|
|
1160
|
+
*/
|
|
1161
|
+
export function windowBand(deps) {
|
|
1162
|
+
const u = rollingWindowUsage(deps);
|
|
1163
|
+
// Caps come from deps first (tests / a per-seat override), else the env consts.
|
|
1164
|
+
const cap5 = deps && deps.cap5h != null ? deps.cap5h : WINDOW_5H_TOKEN_CAP;
|
|
1165
|
+
const cap7 = deps && deps.capWeek != null ? deps.capWeek : WINDOW_WEEK_TOKEN_CAP;
|
|
1166
|
+
const frac = deps && deps.approachFraction != null ? deps.approachFraction : WINDOW_APPROACH_FRACTION;
|
|
1167
|
+
const bandFor = (used, cap) => {
|
|
1168
|
+
if (!cap || cap <= 0) return "ok";
|
|
1169
|
+
if (used >= cap) return "at";
|
|
1170
|
+
if (used >= cap * frac) return "approaching";
|
|
1171
|
+
return "ok";
|
|
1172
|
+
};
|
|
1173
|
+
const rank = { ok: 0, approaching: 1, at: 2 };
|
|
1174
|
+
const b5 = bandFor(u.window5hTokens, cap5);
|
|
1175
|
+
const b7 = bandFor(u.week7dTokens, cap7);
|
|
1176
|
+
const band = rank[b5] >= rank[b7] ? b5 : b7;
|
|
1177
|
+
return {
|
|
1178
|
+
band,
|
|
1179
|
+
...u,
|
|
1180
|
+
cap5h: cap5 || null,
|
|
1181
|
+
capWeek: cap7 || null,
|
|
1182
|
+
active: cap5 > 0 || cap7 > 0,
|
|
1183
|
+
};
|
|
1184
|
+
}
|
|
@@ -18,6 +18,8 @@ import {
|
|
|
18
18
|
readNotified,
|
|
19
19
|
BANDS,
|
|
20
20
|
DEFAULT_DAILY_SPEND_CAP_USD,
|
|
21
|
+
rollingWindowUsage,
|
|
22
|
+
windowBand,
|
|
21
23
|
} from "./budget-guard.mjs";
|
|
22
24
|
|
|
23
25
|
const FIXED = Date.UTC(2026, 5, 9, 12, 0, 0); // 2026-06-09T12:00:00Z
|
|
@@ -363,3 +365,63 @@ test("dailyStatus: a subscription-auth seat STILL bands on session volume (a rea
|
|
|
363
365
|
assert.equal(sub.mode, "degraded");
|
|
364
366
|
} finally { await rm(dir); }
|
|
365
367
|
});
|
|
368
|
+
|
|
369
|
+
// ---------------------------------------------------------------------------
|
|
370
|
+
// rolling usage window (token volume) + windowBand (default-off pacing)
|
|
371
|
+
// ---------------------------------------------------------------------------
|
|
372
|
+
|
|
373
|
+
function seedWindowLedger(dir, rows) {
|
|
374
|
+
// rows: [{ minsAgo, tokens }] written to today's + yesterday's files by ts.
|
|
375
|
+
const now = FIXED;
|
|
376
|
+
const byFile = {};
|
|
377
|
+
for (const r of rows) {
|
|
378
|
+
const at = new Date(now - r.minsAgo * 60_000);
|
|
379
|
+
const day = at.toISOString().slice(0, 10);
|
|
380
|
+
(byFile[day] ||= []).push(JSON.stringify({
|
|
381
|
+
ts: at.toISOString(),
|
|
382
|
+
total_cost_usd: 0.1, model: "sonnet", cadence: "x",
|
|
383
|
+
input_tokens: r.tokens, output_tokens: 0,
|
|
384
|
+
}));
|
|
385
|
+
}
|
|
386
|
+
for (const [day, lines] of Object.entries(byFile)) {
|
|
387
|
+
writeFileSync(join(dir, `${day}.jsonl`), lines.join("\n") + "\n");
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
test("rollingWindowUsage sums tokens/sessions in the 5h and 7d windows", async () => {
|
|
392
|
+
const dir = await makeLedgerDir();
|
|
393
|
+
try {
|
|
394
|
+
seedWindowLedger(dir, [
|
|
395
|
+
{ minsAgo: 10, tokens: 1000 }, // in 5h + 7d
|
|
396
|
+
{ minsAgo: 60, tokens: 2000 }, // in 5h + 7d
|
|
397
|
+
{ minsAgo: 400, tokens: 500 }, // >5h (400m), in 7d
|
|
398
|
+
{ minsAgo: 60 * 24 * 8, tokens: 9999 }, // >7d — excluded
|
|
399
|
+
]);
|
|
400
|
+
const u = rollingWindowUsage({ ledgerDir: dir, now: clk });
|
|
401
|
+
assert.equal(u.window5hTokens, 3000);
|
|
402
|
+
assert.equal(u.window5hSessions, 2);
|
|
403
|
+
assert.equal(u.week7dTokens, 3500);
|
|
404
|
+
assert.equal(u.week7dSessions, 3);
|
|
405
|
+
} finally { await rm(dir); }
|
|
406
|
+
});
|
|
407
|
+
|
|
408
|
+
test("windowBand is DEFAULT-OFF: no caps → always 'ok' however high usage climbs", async () => {
|
|
409
|
+
const dir = await makeLedgerDir();
|
|
410
|
+
try {
|
|
411
|
+
seedWindowLedger(dir, [{ minsAgo: 5, tokens: 10_000_000 }]);
|
|
412
|
+
const b = windowBand({ ledgerDir: dir, now: clk }); // no caps
|
|
413
|
+
assert.equal(b.band, "ok");
|
|
414
|
+
assert.equal(b.active, false);
|
|
415
|
+
assert.equal(b.window5hTokens, 10_000_000);
|
|
416
|
+
} finally { await rm(dir); }
|
|
417
|
+
});
|
|
418
|
+
|
|
419
|
+
test("windowBand bands 'approaching' then 'at' once a 5h cap is calibrated", async () => {
|
|
420
|
+
const dir = await makeLedgerDir();
|
|
421
|
+
try {
|
|
422
|
+
seedWindowLedger(dir, [{ minsAgo: 5, tokens: 8500 }]);
|
|
423
|
+
assert.equal(windowBand({ ledgerDir: dir, now: clk, cap5h: 10_000 }).band, "approaching"); // 85% of 10k
|
|
424
|
+
assert.equal(windowBand({ ledgerDir: dir, now: clk, cap5h: 8000 }).band, "at"); // over 8k
|
|
425
|
+
assert.equal(windowBand({ ledgerDir: dir, now: clk, cap5h: 100_000 }).band, "ok"); // well under
|
|
426
|
+
} finally { await rm(dir); }
|
|
427
|
+
});
|
package/lib/claude-bin.mjs
CHANGED
|
@@ -129,9 +129,26 @@ export function daemonClaudeArgs(agentRoot, deps = {}) {
|
|
|
129
129
|
const env = deps.env || process.env;
|
|
130
130
|
if (env.DAEMON_LOAD_MCPS === "1") return [];
|
|
131
131
|
const ex = deps.existsSync || existsSync;
|
|
132
|
+
const source = deps.source || null;
|
|
132
133
|
|
|
133
134
|
const args = ["--strict-mcp-config"];
|
|
134
135
|
|
|
136
|
+
// MCP BY SOURCE — the biggest cheap per-spawn token saving. A classify or a
|
|
137
|
+
// quick-reply generates TEXT and calls NO org tools, so re-paying the ~107-tool
|
|
138
|
+
// org MCP definitions on every one of those high-frequency spawns is pure
|
|
139
|
+
// waste. Those sources get a BARE spawn (`--strict-mcp-config` with no
|
|
140
|
+
// `--mcp-config` loads zero MCP servers); only full work sessions
|
|
141
|
+
// (dispatcher/cadence — source unset) load the org toolset, so the API/tool-
|
|
142
|
+
// parity bar is honoured exactly where real work happens. Override per source
|
|
143
|
+
// with DAEMON_<SOURCE>_MCP=full|bare (e.g. DAEMON_CLASSIFIER_MCP=full).
|
|
144
|
+
const BARE_SOURCES = new Set(["classifier", "responder"]);
|
|
145
|
+
const perSource = source ? env[`DAEMON_${source.toUpperCase()}_MCP`] : null;
|
|
146
|
+
const bare = perSource === "bare" || (perSource !== "full" && BARE_SOURCES.has(source));
|
|
147
|
+
if (bare) {
|
|
148
|
+
if (env.DAEMON_BARE_MODE === "1") args.unshift("--bare");
|
|
149
|
+
return args;
|
|
150
|
+
}
|
|
151
|
+
|
|
135
152
|
const root = agentRoot || env.AGENT_ROOT || process.cwd();
|
|
136
153
|
try {
|
|
137
154
|
const cfg = join(root, ".mcp.json");
|
package/lib/claude-bin.test.mjs
CHANGED
|
@@ -107,3 +107,25 @@ test("daemonClaudeArgs never throws on a bad root (a daemon must still spawn)",
|
|
|
107
107
|
const args = daemonClaudeArgs("/agent", { existsSync: () => { throw new Error("EACCES"); }, env: {} });
|
|
108
108
|
assert.deepEqual(args, ["--strict-mcp-config"]);
|
|
109
109
|
});
|
|
110
|
+
|
|
111
|
+
// ---------------------------------------------------------------------------
|
|
112
|
+
// MCP by source — bare for tool-free spawns (classifier / responder)
|
|
113
|
+
// ---------------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
test("daemonClaudeArgs: classifier + responder are BARE (no org MCP re-paid)", () => {
|
|
116
|
+
for (const source of ["classifier", "responder"]) {
|
|
117
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {}, source });
|
|
118
|
+
assert.ok(args.includes("--strict-mcp-config"), `${source} keeps strict-mcp-config`);
|
|
119
|
+
assert.ok(!args.includes("--mcp-config"), `${source} must NOT load the org MCP`);
|
|
120
|
+
}
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("daemonClaudeArgs: a full work session (no source) still loads the org MCP", () => {
|
|
124
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: {} });
|
|
125
|
+
assert.ok(args.includes("--mcp-config"), "dispatcher/cadence session keeps full org tools");
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
test("daemonClaudeArgs: DAEMON_<SOURCE>_MCP=full overrides bare for that source", () => {
|
|
129
|
+
const args = daemonClaudeArgs("/agent", { existsSync: () => true, env: { DAEMON_CLASSIFIER_MCP: "full" }, source: "classifier" });
|
|
130
|
+
assert.ok(args.includes("--mcp-config"), "explicit override restores the org MCP for the classifier");
|
|
131
|
+
});
|
package/lib/rate-guard.mjs
CHANGED
|
@@ -243,4 +243,70 @@ export function classifyStderr(text) {
|
|
|
243
243
|
return /\b429\b|rate[\s_-]?limit|overloaded|too many requests/i.test(text);
|
|
244
244
|
}
|
|
245
245
|
|
|
246
|
-
|
|
246
|
+
// A Max/subscription USAGE or SESSION limit is NOT a transient 429 — the pool is
|
|
247
|
+
// drained until the window RESETS, so hammering it with decorrelated-jitter
|
|
248
|
+
// retries just burns the moment it re-opens. We detect it separately and hold
|
|
249
|
+
// the breaker until the reset (parsed from the message when present, else a
|
|
250
|
+
// conservative default) so the seat backs off to the reset instead of storming.
|
|
251
|
+
const USAGE_LIMIT_RE =
|
|
252
|
+
/\b(?:usage|session|weekly|5[\s-]?hour)\s+limit\b|limit reached|reached your (?:usage|session|monthly|weekly|plan)?\s*limit|you'?ve hit your [^.]*limit|approaching (?:your )?[^.]*usage limit|out of (?:usage|credits|messages)/i;
|
|
253
|
+
|
|
254
|
+
/** How long to hold the breaker when a usage limit is hit but no reset time is parseable. */
|
|
255
|
+
export const RATE_USAGE_LIMIT_HOLD_MS = num(process.env.RATE_USAGE_LIMIT_HOLD_MS, 30 * 60 * 1000);
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Best-effort parse of a reset time out of a usage-limit message —
|
|
259
|
+
* "resets 9:40pm", "resets at 9pm", "try again at 10:15am", "available again at
|
|
260
|
+
* 3pm" — into the epoch ms of its NEXT occurrence (today, or tomorrow if already
|
|
261
|
+
* past). The daemon runs in the seat's local timezone, which is the timezone the
|
|
262
|
+
* message states, so a local Date is correct. Returns null when unparseable.
|
|
263
|
+
*/
|
|
264
|
+
export function parseResetAt(text, deps) {
|
|
265
|
+
if (!text || typeof text !== "string") return null;
|
|
266
|
+
const m = /(?:reset|resets|resets at|resets in|try again at|available again at)\s+(\d{1,2})(?::(\d{2}))?\s*([ap])\.?\s?m\.?/i.exec(text);
|
|
267
|
+
if (!m) return null;
|
|
268
|
+
const nowMs = clock(deps)();
|
|
269
|
+
let hr = parseInt(m[1], 10) % 12;
|
|
270
|
+
if (/p/i.test(m[3])) hr += 12;
|
|
271
|
+
const min = m[2] ? parseInt(m[2], 10) : 0;
|
|
272
|
+
const cand = new Date(nowMs);
|
|
273
|
+
cand.setHours(hr, min, 0, 0);
|
|
274
|
+
let t = cand.getTime();
|
|
275
|
+
if (t <= nowMs) t += 24 * 60 * 60 * 1000; // already past today → next occurrence
|
|
276
|
+
return t;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Does this text look like a Max/subscription USAGE / SESSION limit (as opposed
|
|
281
|
+
* to a transient 429)? When it names a reset time, resetAt is the epoch ms to
|
|
282
|
+
* hold the breaker until.
|
|
283
|
+
*
|
|
284
|
+
* @returns {{ isLimit: boolean, resetAt: number|null }}
|
|
285
|
+
*/
|
|
286
|
+
export function classifyUsageLimit(text, deps) {
|
|
287
|
+
if (!text || typeof text !== "string") return { isLimit: false, resetAt: null };
|
|
288
|
+
if (!USAGE_LIMIT_RE.test(text)) return { isLimit: false, resetAt: null };
|
|
289
|
+
return { isLimit: true, resetAt: parseResetAt(text, deps) };
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Open the breaker until a usage window RESETS. Uses resetAt when known, else a
|
|
294
|
+
* conservative default hold. Never shortens an already-longer open window.
|
|
295
|
+
*
|
|
296
|
+
* @returns {{ openUntil:number, resetAt:number|null }}
|
|
297
|
+
*/
|
|
298
|
+
export function recordUsageLimit(provider, resetAt, deps) {
|
|
299
|
+
const now = clock(deps)();
|
|
300
|
+
const until =
|
|
301
|
+
typeof resetAt === "number" && resetAt > now ? resetAt : now + RATE_USAGE_LIMIT_HOLD_MS;
|
|
302
|
+
const prev = readState(provider, deps);
|
|
303
|
+
const next = {
|
|
304
|
+
openUntil: Math.max(prev.openUntil, until),
|
|
305
|
+
consecutive429: prev.consecutive429,
|
|
306
|
+
lastBackoffMs: prev.lastBackoffMs,
|
|
307
|
+
};
|
|
308
|
+
writeState(provider, next, deps);
|
|
309
|
+
return { openUntil: next.openUntil, resetAt: typeof resetAt === "number" && resetAt > now ? resetAt : null };
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
export const _internals = { RATE_BASE_BACKOFF_MS, RATE_MAX_BACKOFF_MS, RATE_BACKOFF_FACTOR, RATE_USAGE_LIMIT_HOLD_MS };
|
package/lib/rate-guard.test.mjs
CHANGED
|
@@ -18,11 +18,15 @@ import {
|
|
|
18
18
|
recordSuccess,
|
|
19
19
|
decorrelatedBackoff,
|
|
20
20
|
classifyStderr,
|
|
21
|
+
classifyUsageLimit,
|
|
22
|
+
parseResetAt,
|
|
23
|
+
recordUsageLimit,
|
|
21
24
|
readState,
|
|
22
25
|
sanitizeProvider,
|
|
23
26
|
RATE_BASE_BACKOFF_MS,
|
|
24
27
|
RATE_MAX_BACKOFF_MS,
|
|
25
28
|
RATE_BACKOFF_FACTOR,
|
|
29
|
+
RATE_USAGE_LIMIT_HOLD_MS,
|
|
26
30
|
} from "./rate-guard.mjs";
|
|
27
31
|
|
|
28
32
|
async function makeStateDir() {
|
|
@@ -199,3 +203,70 @@ test("readState tolerates a corrupt file (fails closed-to-empty, never throws)",
|
|
|
199
203
|
assert.equal(checkRateLimit("anthropic", { stateDir: dir }).allowed, true);
|
|
200
204
|
} finally { await rm(dir); }
|
|
201
205
|
});
|
|
206
|
+
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
// classifyUsageLimit / parseResetAt / recordUsageLimit (Max session-limit)
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
test("classifyUsageLimit detects session/usage limits, not plain 429s", () => {
|
|
212
|
+
assert.equal(classifyUsageLimit("You've hit your session limit · resets 9:40pm (Australia/Sydney)").isLimit, true);
|
|
213
|
+
assert.equal(classifyUsageLimit("usage limit reached").isLimit, true);
|
|
214
|
+
assert.equal(classifyUsageLimit("You have reached your weekly limit").isLimit, true);
|
|
215
|
+
assert.equal(classifyUsageLimit("out of usage for now").isLimit, true);
|
|
216
|
+
// A plain transient 429 is NOT a usage-window limit
|
|
217
|
+
assert.equal(classifyUsageLimit("HTTP 429 Too Many Requests").isLimit, false);
|
|
218
|
+
assert.equal(classifyUsageLimit("overloaded").isLimit, false);
|
|
219
|
+
assert.equal(classifyUsageLimit("").isLimit, false);
|
|
220
|
+
assert.equal(classifyUsageLimit(null).isLimit, false);
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
test("parseResetAt resolves a stated reset time to the next occurrence", () => {
|
|
224
|
+
// now = 2026-08-26 08:00 local; "resets 9:40pm" is later today
|
|
225
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
226
|
+
const t = parseResetAt("resets 9:40pm", { now: () => now });
|
|
227
|
+
const d = new Date(t);
|
|
228
|
+
assert.equal(d.getHours(), 21);
|
|
229
|
+
assert.equal(d.getMinutes(), 40);
|
|
230
|
+
assert.ok(t > now, "reset is in the future");
|
|
231
|
+
// A time already past today rolls to tomorrow
|
|
232
|
+
const now2 = new Date(2026, 7, 26, 22, 0, 0, 0).getTime(); // 10pm
|
|
233
|
+
const t2 = parseResetAt("try again at 9pm", { now: () => now2 });
|
|
234
|
+
assert.ok(t2 > now2 && t2 - now2 <= 24 * 60 * 60 * 1000, "rolls to next day");
|
|
235
|
+
assert.equal(parseResetAt("no time here", { now: () => now }), null);
|
|
236
|
+
});
|
|
237
|
+
|
|
238
|
+
test("recordUsageLimit holds the breaker until the parsed reset time", async () => {
|
|
239
|
+
const dir = await makeStateDir();
|
|
240
|
+
try {
|
|
241
|
+
const now = new Date(2026, 7, 26, 8, 0, 0, 0).getTime();
|
|
242
|
+
const { resetAt } = classifyUsageLimit("session limit · resets 9:40pm", { now: () => now });
|
|
243
|
+
const rec = recordUsageLimit("anthropic", resetAt, { stateDir: dir, now: () => now });
|
|
244
|
+
assert.equal(rec.openUntil, resetAt);
|
|
245
|
+
// blocked before reset, allowed after
|
|
246
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
247
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => resetAt + 1 }).allowed, true);
|
|
248
|
+
} finally { await rm(dir); }
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
test("recordUsageLimit falls back to a default hold when no reset is known", async () => {
|
|
252
|
+
const dir = await makeStateDir();
|
|
253
|
+
try {
|
|
254
|
+
const now = 1_000_000;
|
|
255
|
+
const rec = recordUsageLimit("anthropic", null, { stateDir: dir, now: () => now });
|
|
256
|
+
assert.equal(rec.openUntil, now + RATE_USAGE_LIMIT_HOLD_MS);
|
|
257
|
+
assert.equal(rec.resetAt, null);
|
|
258
|
+
assert.equal(checkRateLimit("anthropic", { stateDir: dir, now: () => now + 1000 }).allowed, false);
|
|
259
|
+
} finally { await rm(dir); }
|
|
260
|
+
});
|
|
261
|
+
|
|
262
|
+
test("recordUsageLimit never shortens an already-longer open window", async () => {
|
|
263
|
+
const dir = await makeStateDir();
|
|
264
|
+
try {
|
|
265
|
+
const now = 1_000_000;
|
|
266
|
+
const far = now + 10 * 60 * 60 * 1000; // 10h out
|
|
267
|
+
recordUsageLimit("anthropic", far, { stateDir: dir, now: () => now });
|
|
268
|
+
// a later, SHORTER hold must not pull the window in
|
|
269
|
+
const rec = recordUsageLimit("anthropic", now + 5 * 60 * 1000, { stateDir: dir, now: () => now });
|
|
270
|
+
assert.equal(rec.openUntil, far);
|
|
271
|
+
} finally { await rm(dir); }
|
|
272
|
+
});
|
|
@@ -504,6 +504,16 @@ export function admit(req = {}, deps) {
|
|
|
504
504
|
if (mode === "suspended" && (source === "backlog" || source === "cadence")) {
|
|
505
505
|
return decide(DEFER, "budget suspended (>=125% of the seat envelope): deferring non-inbox work");
|
|
506
506
|
}
|
|
507
|
+
// USAGE-WINDOW PACING (default-off; only active once GOV_WINDOW_* caps are
|
|
508
|
+
// calibrated — see budget-guard.windowBand). At the rolling window's limit,
|
|
509
|
+
// pause self-directed work so a burst spreads across windows instead of
|
|
510
|
+
// exhausting one; a direct human reply is never paced. "approaching" is folded
|
|
511
|
+
// into the concurrency ceiling below (throttle, don't stop). `windowBand`
|
|
512
|
+
// defaults to "ok" so an uncalibrated seat behaves exactly as before.
|
|
513
|
+
const windowBand = req.windowBand || "ok";
|
|
514
|
+
if (windowBand === "at" && !humanReply) {
|
|
515
|
+
return decide(DEFER, `usage window at limit: deferring ${source} work so it spreads across windows; direct human replies continue`);
|
|
516
|
+
}
|
|
507
517
|
// NOTE: `degraded` (>=100% / unfunded) no longer hard-stops backlog. Owner
|
|
508
518
|
// instruction (2026-08-25): THROTTLE, don't stop — the seat keeps a thin
|
|
509
519
|
// trickle of self-directed work so autonomous progress never fully halts,
|
|
@@ -524,10 +534,14 @@ export function admit(req = {}, deps) {
|
|
|
524
534
|
// ABOVE the REACTIVE_RESERVE math: reserve keeps steady-state headroom; this
|
|
525
535
|
// clamps to a trickle for the seconds a human is actually waiting.
|
|
526
536
|
const reactiveBusy = numField(d.reactiveInFlight, 0) > 0;
|
|
537
|
+
// "approaching" the usage window → throttle self-directed work toward a trickle
|
|
538
|
+
// (like a reactive turn) so the seat glides into the window rather than slamming
|
|
539
|
+
// it. Default-off: windowBand is "ok" unless GOV_WINDOW_* caps are set.
|
|
540
|
+
const windowApproaching = windowBand === "approaching";
|
|
527
541
|
let sourceCeiling;
|
|
528
542
|
if (humanReply) {
|
|
529
543
|
sourceCeiling = effectiveMax;
|
|
530
|
-
} else if (reactiveBusy) {
|
|
544
|
+
} else if (reactiveBusy || windowApproaching) {
|
|
531
545
|
sourceCeiling = Math.min(effectiveMax, REACTIVE_INFLIGHT_BACKLOG_MAX);
|
|
532
546
|
if (mode === "degraded" && source === "backlog") {
|
|
533
547
|
sourceCeiling = Math.min(sourceCeiling, DEGRADED_BACKLOG_MAX);
|
|
@@ -540,6 +554,8 @@ export function admit(req = {}, deps) {
|
|
|
540
554
|
if (liveCount >= sourceCeiling) {
|
|
541
555
|
const why = reactiveBusy
|
|
542
556
|
? `reactive turn in flight: self-directed ${source} yields the subscription, capped to ${sourceCeiling} (${liveCount} live)`
|
|
557
|
+
: windowApproaching
|
|
558
|
+
? `usage window approaching: self-directed ${source} throttled to ${sourceCeiling} to spread across windows (${liveCount} live)`
|
|
543
559
|
: mode === "degraded" && source === "backlog"
|
|
544
560
|
? `budget degraded: self-directed backlog throttled to ${sourceCeiling} (${liveCount} live; inbox unaffected)`
|
|
545
561
|
: throttleCeiling < dMax
|
|
@@ -180,6 +180,33 @@ test("admit: one backlog session is still allowed alongside a reply (cap 1, not
|
|
|
180
180
|
assert.equal(r.decision, DECISIONS.ADMIT);
|
|
181
181
|
});
|
|
182
182
|
|
|
183
|
+
// ---------------------------------------------------------------------------
|
|
184
|
+
// Usage-window pacing (default-off; driven by req.windowBand from budget-guard)
|
|
185
|
+
// ---------------------------------------------------------------------------
|
|
186
|
+
|
|
187
|
+
test("admit: windowBand 'at' DEFERS self-directed work; a human reply is never paced", () => {
|
|
188
|
+
assert.equal(admit({ source: "backlog", windowBand: "at" }, deps()).decision, DECISIONS.DEFER);
|
|
189
|
+
assert.equal(admit({ source: "cadence", windowBand: "at" }, deps()).decision, DECISIONS.DEFER);
|
|
190
|
+
// inbox / humanReply continue at the window limit
|
|
191
|
+
assert.equal(admit({ source: "inbox", windowBand: "at" }, deps()).decision, DECISIONS.ADMIT);
|
|
192
|
+
assert.equal(admit({ source: "backlog", humanReply: true, windowBand: "at" }, deps()).decision, DECISIONS.ADMIT);
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
test("admit: windowBand 'approaching' throttles backlog to a trickle (QUEUE), inbox unaffected", () => {
|
|
196
|
+
// 1 already live → backlog capped to REACTIVE_INFLIGHT_BACKLOG_MAX(1) → QUEUE
|
|
197
|
+
const r = admit({ source: "backlog", windowBand: "approaching" }, deps({ liveClaude: { count: 1, rssMB: 300 } }));
|
|
198
|
+
assert.equal(r.decision, DECISIONS.QUEUE);
|
|
199
|
+
assert.match(r.reason, /usage window approaching/);
|
|
200
|
+
// inbox still admits at the same live count
|
|
201
|
+
assert.equal(admit({ source: "inbox", windowBand: "approaching" }, deps({ liveClaude: { count: 1, rssMB: 300 } })).decision, DECISIONS.ADMIT);
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
test("admit: windowBand default (undefined/'ok') changes nothing — pacing is off by default", () => {
|
|
205
|
+
// 3 live, no window signal → backlog still ADMITs (reserve leaves 5)
|
|
206
|
+
assert.equal(admit({ source: "backlog" }, deps({ liveClaude: { count: 3, rssMB: 900 } })).decision, DECISIONS.ADMIT);
|
|
207
|
+
assert.equal(admit({ source: "backlog", windowBand: "ok" }, deps({ liveClaude: { count: 3, rssMB: 900 } })).decision, DECISIONS.ADMIT);
|
|
208
|
+
});
|
|
209
|
+
|
|
183
210
|
test("admit: throttle ceiling clamps effectiveMax below dynamicMax → QUEUE earlier", () => {
|
|
184
211
|
// dynamicMax 8 but soft-throttle pinned the ceiling to 3; 3 live → QUEUE.
|
|
185
212
|
const r = admit({ source: "backlog" }, deps({ throttleCeiling: 3, liveClaude: { count: 3, rssMB: 1000 } }));
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@cohortapp/agent-sdk",
|
|
3
|
-
"version": "2.11.
|
|
3
|
+
"version": "2.11.14",
|
|
4
4
|
"description": "Cohort Agent SDK — autonomous AI colleague runtime. Deploy senior AI colleagues on dedicated Mac minis, wired to the Cohort operating surface.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -601,8 +601,10 @@ export function startConsumer(opts = {}) {
|
|
|
601
601
|
// as a human reply, and every cadence tick arrives as `source:"cadence"`.
|
|
602
602
|
// Two gates, one predicate, or the rung silences the inbox on the second
|
|
603
603
|
// hop instead of the first.
|
|
604
|
+
let windowBand;
|
|
605
|
+
try { windowBand = budgetGuardModule.windowBand?.({ agentRoot })?.band; } catch { /* default-off */ }
|
|
604
606
|
const adm = governor.admit(
|
|
605
|
-
{ source: "cadence", mode, humanReply: cadence ? isHumanLaneCadence(cadence) : false },
|
|
607
|
+
{ source: "cadence", mode, humanReply: cadence ? isHumanLaneCadence(cadence) : false, windowBand },
|
|
606
608
|
governor.defaultDeps({ agentRoot })
|
|
607
609
|
);
|
|
608
610
|
if (adm.decision !== "ADMIT") return { admit: false, reason: adm.reason || adm.decision };
|
|
@@ -901,7 +903,23 @@ export function startConsumer(opts = {}) {
|
|
|
901
903
|
// the shared breaker and REQUEUE the tick unchanged (decision:"deferred")
|
|
902
904
|
// instead of failTick — a 429 is an upstream gate, not a per-event failure,
|
|
903
905
|
// so it must not burn this cadence's retry budget toward the DLQ.
|
|
904
|
-
|
|
906
|
+
const cadenceOut = result.stderr_tail || result.error || result.stdout_tail || "";
|
|
907
|
+
const cadenceUl = rateGuard.classifyUsageLimit?.(cadenceOut, { agentRoot }) ?? { isLimit: false, resetAt: null };
|
|
908
|
+
if (cadenceUl.isLimit) {
|
|
909
|
+
try {
|
|
910
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, cadenceUl.resetAt, { agentRoot });
|
|
911
|
+
log({ level: "warn", stage: "subsession_usage_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
912
|
+
} catch { /* */ }
|
|
913
|
+
// Same requeue-unchanged handling as a 429: a window-usage limit is a
|
|
914
|
+
// shared, provider-side gate — not evidence THIS cadence is broken — so
|
|
915
|
+
// requeue without failTick or a circuit trip; the shared breaker (held
|
|
916
|
+
// until reset) gates re-escalation.
|
|
917
|
+
requeueTick(agentRoot, event);
|
|
918
|
+
stats.retries += 1;
|
|
919
|
+
stats.last_decision = "deferred";
|
|
920
|
+
return { ok: false, decision: "deferred" };
|
|
921
|
+
}
|
|
922
|
+
if (rateGuard.classifyStderr(cadenceOut)) {
|
|
905
923
|
try {
|
|
906
924
|
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot });
|
|
907
925
|
log({ level: "warn", stage: "subsession_rate_limited", id: event.id, cadence: event.cadence, open_until: rec.openUntil });
|
|
@@ -455,7 +455,7 @@ async function runClaudeCLI(systemPrompt, userPrompt) {
|
|
|
455
455
|
const args = [
|
|
456
456
|
"--print",
|
|
457
457
|
...sessionPermissionArgs({ source: "classifier" }),
|
|
458
|
-
...daemonClaudeArgs(),
|
|
458
|
+
...daemonClaudeArgs(undefined, { source: "classifier" }),
|
|
459
459
|
"--model", ANTHROPIC_MODEL,
|
|
460
460
|
"--append-system-prompt", systemPrompt,
|
|
461
461
|
];
|
|
@@ -191,11 +191,45 @@ test("reconcile blocks a marker that is older than the freshness window", async
|
|
|
191
191
|
} finally { await cleanup(dir); }
|
|
192
192
|
});
|
|
193
193
|
|
|
194
|
+
test("resume reconcile DEFERS (never re-storms) when the backlog budget is 0 (reactive-only seat)", async () => {
|
|
195
|
+
// DAEMON_MAX_CONCURRENT=3 → backlogCap = 3 - RESERVED_INBOX_SLOTS(3) = 0, so a
|
|
196
|
+
// reactive-only seat must resume NOTHING on boot — the fresh-seat storm root.
|
|
197
|
+
process.env.DAEMON_MAX_CONCURRENT = "3";
|
|
198
|
+
const { mod, dir } = await freshDispatcher();
|
|
199
|
+
try {
|
|
200
|
+
plantMarker(dir, "s-a", { itemId: "BL-A", sourceFile: "q.yaml", startedAt: Date.now() });
|
|
201
|
+
plantMarker(dir, "s-b", { itemId: "BL-B", sourceFile: "q.yaml", startedAt: Date.now() });
|
|
202
|
+
let spawned = 0;
|
|
203
|
+
const stats = mod.reconcileResumePending({ spawnResume: () => { spawned++; } });
|
|
204
|
+
assert.equal(spawned, 0, "no resume spawns on a budget-0 seat");
|
|
205
|
+
assert.equal(stats.resumed, 0);
|
|
206
|
+
assert.equal(stats.deferred, 2, "both live markers deferred, not blocked/expired");
|
|
207
|
+
// markers are LEFT on disk (deferred, not retired) for a later reconcile
|
|
208
|
+
const left = readdirSync(join(dir, "state/sessions/resume-pending")).filter((f) => f.endsWith(".json"));
|
|
209
|
+
assert.equal(left.length, 2, "deferred markers persist");
|
|
210
|
+
} finally { delete process.env.DAEMON_MAX_CONCURRENT; await cleanup(dir); }
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
test("resume reconcile is GATED (resumes nothing) while the shared breaker is open", async () => {
|
|
214
|
+
const { mod, dir } = await freshDispatcher();
|
|
215
|
+
try {
|
|
216
|
+
mod.setGovernanceForTests({ rateGuard: { checkRateLimit: () => ({ allowed: false, retryAt: Date.now() + 60_000 }) } });
|
|
217
|
+
plantMarker(dir, "s-x", { itemId: "BL-X", sourceFile: "q.yaml", startedAt: Date.now() });
|
|
218
|
+
let spawned = 0;
|
|
219
|
+
const stats = mod.reconcileResumePending({ spawnResume: () => { spawned++; } });
|
|
220
|
+
assert.equal(spawned, 0, "an open breaker (e.g. a usage-limit hold) blocks all resume");
|
|
221
|
+
assert.equal(stats.resumed, 0);
|
|
222
|
+
assert.equal(stats.scanned, 0, "returns before scanning when the breaker is open");
|
|
223
|
+
// the marker is untouched, ready for the next reconcile after the breaker clears
|
|
224
|
+
assert.equal(readdirSync(join(dir, "state/sessions/resume-pending")).filter((f) => f.endsWith(".json")).length, 1);
|
|
225
|
+
} finally { await cleanup(dir); }
|
|
226
|
+
});
|
|
227
|
+
|
|
194
228
|
test("reconcile on an empty resume-pending dir is a no-op", async () => {
|
|
195
229
|
const { mod, dir } = await freshDispatcher();
|
|
196
230
|
try {
|
|
197
231
|
const stats = mod.reconcileResumePending({ spawnResume: () => { throw new Error("should not spawn"); } });
|
|
198
|
-
assert.deepEqual(stats, { scanned: 0, resumed: 0, blocked: 0, expired: 0 });
|
|
232
|
+
assert.deepEqual(stats, { scanned: 0, resumed: 0, blocked: 0, expired: 0, deferred: 0 });
|
|
199
233
|
} finally { await cleanup(dir); }
|
|
200
234
|
});
|
|
201
235
|
|
|
@@ -75,7 +75,11 @@ function admitFor(source, priority) {
|
|
|
75
75
|
console.warn(`[dispatcher] budget ${st.mode}: ${st.pct}% of $${st.capUSD}/day (${st.capSource}) — gating ${source} work`);
|
|
76
76
|
}
|
|
77
77
|
} catch { /* budget read best-effort */ }
|
|
78
|
-
|
|
78
|
+
// Usage-window band (default-off unless GOV_WINDOW_* caps set) → the governor
|
|
79
|
+
// paces self-directed work as the rolling window fills. Best-effort.
|
|
80
|
+
let windowBand;
|
|
81
|
+
try { windowBand = budgetGuard.windowBand?.({ agentRoot: AGENT_REPO_DIR })?.band; } catch { /* */ }
|
|
82
|
+
return governor.admit({ source, priority, mode, windowBand }, governor.defaultDeps({ agentRoot: AGENT_REPO_DIR }));
|
|
79
83
|
} catch {
|
|
80
84
|
return { decision: "ADMIT", reason: "governor-error-fail-open", snapshot: {} };
|
|
81
85
|
}
|
|
@@ -499,13 +503,32 @@ export function resetActiveSessions(opts = {}) {
|
|
|
499
503
|
*/
|
|
500
504
|
export function reconcileResumePending(opts = {}) {
|
|
501
505
|
const now = (typeof opts.now === "function" ? opts.now : Date.now)();
|
|
502
|
-
const stats = { scanned: 0, resumed: 0, blocked: 0, expired: 0 };
|
|
506
|
+
const stats = { scanned: 0, resumed: 0, blocked: 0, expired: 0, deferred: 0 };
|
|
503
507
|
let files = [];
|
|
504
508
|
try {
|
|
505
509
|
if (!existsSync(RESUME_PENDING_DIR)) return stats;
|
|
506
510
|
files = readdirSync(RESUME_PENDING_DIR).filter((f) => f.endsWith(".json") && !f.endsWith(".tmp"));
|
|
507
511
|
} catch { return stats; }
|
|
508
512
|
|
|
513
|
+
// ADMISSION GATE — resume must honour the same throttle as any backlog spawn,
|
|
514
|
+
// or a stormed seat's leftover markers bypass it and re-storm on every boot
|
|
515
|
+
// (the fresh-seat backlog-storm root). Two guards:
|
|
516
|
+
// (1) if the shared breaker is open (a usage/rate limit is holding spawns),
|
|
517
|
+
// resume NOTHING — resuming would hammer a drained pool the moment it
|
|
518
|
+
// re-opens. Markers stay for the next reconcile after the breaker clears.
|
|
519
|
+
// (2) cap resumes this pass to the BACKLOG concurrency (in-memory, race-free
|
|
520
|
+
// — unlike the governor's ps count). DAEMON_MAX_CONCURRENT=3 →
|
|
521
|
+
// RESERVED_INBOX_SLOTS(3) → budget 0 → a reactive-only seat resumes
|
|
522
|
+
// nothing on boot (the item stays re-deliverable / the marker persists).
|
|
523
|
+
try {
|
|
524
|
+
const rb = rateGuard.checkRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
525
|
+
if (rb && rb.allowed === false) {
|
|
526
|
+
logSession({ event: "resume_reconcile_gated", reason: "breaker_open", retry_at: rb.retryAt || null, markers: files.length });
|
|
527
|
+
return stats;
|
|
528
|
+
}
|
|
529
|
+
} catch { /* fail-open — a breaker read error must not strand recovery */ }
|
|
530
|
+
const resumeBudget = Math.max(0, MAX_CONCURRENT - RESERVED_INBOX_SLOTS - activeSessions.size);
|
|
531
|
+
|
|
509
532
|
for (const file of files) {
|
|
510
533
|
const path = join(RESUME_PENDING_DIR, file);
|
|
511
534
|
stats.scanned++;
|
|
@@ -554,6 +577,15 @@ export function reconcileResumePending(opts = {}) {
|
|
|
554
577
|
continue;
|
|
555
578
|
}
|
|
556
579
|
|
|
580
|
+
// Over the backlog budget for this pass → DEFER (leave the marker, do NOT
|
|
581
|
+
// bump attempts — a deferral is not a try). The next reconcile picks it up
|
|
582
|
+
// once slots free / the throttle lifts. This is what stops the re-storm.
|
|
583
|
+
if (stats.resumed >= resumeBudget) {
|
|
584
|
+
logSession({ event: "resume_deferred_throttle", sessionId: marker.sessionId, item_id: marker.itemId, budget: resumeBudget });
|
|
585
|
+
stats.deferred++;
|
|
586
|
+
continue;
|
|
587
|
+
}
|
|
588
|
+
|
|
557
589
|
// Bump the attempt count on the marker BEFORE re-dispatch so a crash mid-
|
|
558
590
|
// resume still advances toward the 3-strike cap (no infinite loop).
|
|
559
591
|
marker.recoveryAttempts = attempts + 1;
|
|
@@ -573,7 +605,7 @@ export function reconcileResumePending(opts = {}) {
|
|
|
573
605
|
}
|
|
574
606
|
}
|
|
575
607
|
if (stats.scanned > 0) {
|
|
576
|
-
console.log(`[dispatcher] resume reconcile: ${stats.resumed} resumed, ${stats.blocked} blocked, ${stats.expired} expired (of ${stats.scanned})`);
|
|
608
|
+
console.log(`[dispatcher] resume reconcile: ${stats.resumed} resumed, ${stats.deferred} deferred, ${stats.blocked} blocked, ${stats.expired} expired (of ${stats.scanned})`);
|
|
577
609
|
}
|
|
578
610
|
return stats;
|
|
579
611
|
}
|
|
@@ -622,8 +654,12 @@ function defaultSpawnResume({ marker }) {
|
|
|
622
654
|
clearResumePending(marker.sessionId);
|
|
623
655
|
logSession({ event: "resume_completed", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
624
656
|
} else {
|
|
625
|
-
// A 429 during resume must open the breaker so subsequent
|
|
626
|
-
|
|
657
|
+
// A usage limit or 429 during resume must open the breaker so subsequent
|
|
658
|
+
// spawns gate. Usage-limit → hold until reset; 429 → transient backoff.
|
|
659
|
+
const ul = rateGuard.classifyUsageLimit?.(stderr, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
660
|
+
if (ul.isLimit) {
|
|
661
|
+
try { rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
662
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
627
663
|
try { rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
628
664
|
}
|
|
629
665
|
logSession({ event: "resume_exit_nonzero", sessionId: marker.sessionId, item_id: marker.itemId, exit_code: code });
|
|
@@ -1326,11 +1362,25 @@ function spawnSession(entry) {
|
|
|
1326
1362
|
if (code === 0) {
|
|
1327
1363
|
clearResumePending(sessionId);
|
|
1328
1364
|
try { rateGuard.recordSuccess(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
1329
|
-
} else
|
|
1330
|
-
|
|
1331
|
-
|
|
1332
|
-
|
|
1333
|
-
|
|
1365
|
+
} else {
|
|
1366
|
+
// A Max/subscription USAGE limit ("you've hit your session limit · resets
|
|
1367
|
+
// 9:40pm") is checked FIRST and on BOTH streams — for `claude --print` it
|
|
1368
|
+
// lands on stdout, not stderr. It holds the breaker until the window
|
|
1369
|
+
// RESETS (not a decorrelated 429 backoff), so the seat stops storming a
|
|
1370
|
+
// drained pool. A plain 429/overload falls through to the transient path.
|
|
1371
|
+
const combined = `${stdout || ""}\n${stderr || ""}`;
|
|
1372
|
+
const ul = rateGuard.classifyUsageLimit?.(combined, { agentRoot: AGENT_REPO_DIR }) ?? { isLimit: false, resetAt: null };
|
|
1373
|
+
if (ul.isLimit) {
|
|
1374
|
+
try {
|
|
1375
|
+
const rec = rateGuard.recordUsageLimit(RATE_PROVIDER, ul.resetAt, { agentRoot: AGENT_REPO_DIR });
|
|
1376
|
+
logSession({ event: "usage_limit_recorded", sessionId, open_until: rec.openUntil, reset_at: rec.resetAt });
|
|
1377
|
+
} catch { /* */ }
|
|
1378
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
1379
|
+
try {
|
|
1380
|
+
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
1381
|
+
logSession({ event: "rate_limit_recorded", sessionId, open_until: rec.openUntil, consecutive: rec.consecutive429 });
|
|
1382
|
+
} catch { /* */ }
|
|
1383
|
+
}
|
|
1334
1384
|
}
|
|
1335
1385
|
|
|
1336
1386
|
// Release item lock — MUST use same key order as acquireLock in daemon
|
|
@@ -259,7 +259,7 @@ function runClaudeCLI(systemPrompt, userPrompt, model, opts = {}) {
|
|
|
259
259
|
const args = [
|
|
260
260
|
"--print",
|
|
261
261
|
...sessionPermissionArgs({ source: "responder" }),
|
|
262
|
-
...daemonClaudeArgs(),
|
|
262
|
+
...daemonClaudeArgs(undefined, { source: "responder" }),
|
|
263
263
|
"--model", model,
|
|
264
264
|
"--append-system-prompt", systemPrompt,
|
|
265
265
|
// --output-format json is only valid in combination with --print (per b1
|