claude-code-session-manager 0.77.0 → 0.79.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-B2ie8bbw.js → AgentLibrary-COtVRqBR.js} +1 -1
- package/dist/assets/{DataModel-BIJPYw32.js → DataModel-CSEKw_OR.js} +1 -1
- package/dist/assets/{History-CeY6dk9S.js → History-CHHovrAO.js} +1 -1
- package/dist/assets/{Hooks-BFH2ocKg.js → Hooks-BZU6C3x6.js} +1 -1
- package/dist/assets/{HostBilko-36gj9wLz.js → HostBilko-CqTUoq37.js} +1 -1
- package/dist/assets/{Library-C-hBct39.js → Library-BtxdyTLz.js} +1 -1
- package/dist/assets/{ListDetail-CNq64VWV.js → ListDetail-qZc7Zm-6.js} +1 -1
- package/dist/assets/{MarkdownEditor-Bh3qt5-1.js → MarkdownEditor-BHe_4fJR.js} +1 -1
- package/dist/assets/{McpServers-DpGN0oyz.js → McpServers-7Z98HLNo.js} +1 -1
- package/dist/assets/{Memory-D59hUjC4.js → Memory-CR72KoyP.js} +1 -1
- package/dist/assets/{Panel-DCgbaoci.js → Panel-pL6H3dpQ.js} +1 -1
- package/dist/assets/{Permissions-DAmQ0DYV.js → Permissions-CWSWjyXM.js} +1 -1
- package/dist/assets/{Plugins-Dyfgn6Is.js → Plugins-CN6lX2lt.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BiYhPO1U.js → ProvenanceBadge-BXSXwIsk.js} +1 -1
- package/dist/assets/{SaveBar-RV7B6sOh.js → SaveBar-BlB5TGpR.js} +1 -1
- package/dist/assets/{Scheduler-BPaNqx1b.js → Scheduler-DRciWUmR.js} +1 -1
- package/dist/assets/{ScopeSwitcher-P4mdLGNU.js → ScopeSwitcher-kFrXtjpr.js} +1 -1
- package/dist/assets/{Settings-BL4vf5aX.js → Settings-BXuyf4lJ.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-BRBDyi1_.js → SkillReferenceGraph-Dfacb0PE.js} +1 -1
- package/dist/assets/{Skills-BV08gDUH.js → Skills-CHqcpiyt.js} +1 -1
- package/dist/assets/{SystemPrompt-CLftSsDw.js → SystemPrompt-fxXm0BZr.js} +1 -1
- package/dist/assets/{TagLibrary-Bp8jGsd5.js → TagLibrary-DOz65ZTz.js} +1 -1
- package/dist/assets/{TiptapBody-jCpuB6E5.js → TiptapBody-D0bWx_9o.js} +1 -1
- package/dist/assets/{Toggle-D2paA1xf.js → Toggle-C9jBwGSx.js} +1 -1
- package/dist/assets/{index-BDRSqBl3.js → index-DPYa6jbM.js} +419 -419
- package/dist/assets/{settingsSchema-6IOLjZZN.js → settingsSchema-BTPw1bR3.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/src/main/__tests__/runLogRetention.test.cjs +59 -0
- package/src/main/__tests__/scheduler-never-stop.test.cjs +157 -0
- package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +81 -0
- package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +123 -0
- package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +158 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +30 -0
- package/src/main/index.cjs +8 -3
- package/src/main/ipcSchemas.cjs +2 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +45 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +90 -1
- package/src/main/lib/jobDirtFilter.cjs +54 -0
- package/src/main/lib/rateLimitDetect.cjs +35 -0
- package/src/main/lib/reaperHelpers.cjs +31 -6
- package/src/main/lib/runLogRetention.cjs +82 -4
- package/src/main/scheduler.cjs +353 -31
- package/src/preload/api.d.ts +8 -4
- package/src/preload/index.cjs +1 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -58,6 +58,8 @@ const launchFailure = require('./lib/launchFailure.cjs');
|
|
|
58
58
|
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
59
59
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
60
60
|
const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
61
|
+
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
62
|
+
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
61
63
|
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
62
64
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
63
65
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
@@ -1252,6 +1254,89 @@ function computeStallSummary(state) {
|
|
|
1252
1254
|
return { stalled, total, running, pending, byProject };
|
|
1253
1255
|
}
|
|
1254
1256
|
|
|
1257
|
+
/**
|
|
1258
|
+
* computeBlockedChains(jobs) → [{ cwd, blockedBy, blocked }]
|
|
1259
|
+
*
|
|
1260
|
+
* Pure, no IO. The gap computeStallSummary above cannot see.
|
|
1261
|
+
*
|
|
1262
|
+
* `stalled` is defined as `running === 0 && pending === 0`, which encodes an
|
|
1263
|
+
* assumption that a PENDING row is healthy in-progress work. It is not: a
|
|
1264
|
+
* pending row whose `dependsOn` chain terminates in a TERMINAL non-completed
|
|
1265
|
+
* status (`failed`/`skipped`) can never be dispatched by pickForProject, and
|
|
1266
|
+
* never will be, but it still counts toward `pending` and so reads as a
|
|
1267
|
+
* healthy queue to every monitor in the app.
|
|
1268
|
+
*
|
|
1269
|
+
* On 2026-09-05 starry-night-ships held 42 such rows behind one `failed`
|
|
1270
|
+
* job for three hours. Machine-wide `stalled` was false (42 pending),
|
|
1271
|
+
* per-project `stalled` was false (42 pending), the queue-health sweep
|
|
1272
|
+
* doesn't count `failed` at all, and the supervisor only probes `running` —
|
|
1273
|
+
* so nothing anywhere reported a problem while nothing could ever run.
|
|
1274
|
+
*
|
|
1275
|
+
* Reported per project as { blockedBy: [terminal slugs], blocked: count }.
|
|
1276
|
+
* Transitive by construction: a row blocked by a row that is itself blocked
|
|
1277
|
+
* resolves through the same walk.
|
|
1278
|
+
*/
|
|
1279
|
+
function computeBlockedChains(jobs) {
|
|
1280
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
1281
|
+
const byCwd = new Map();
|
|
1282
|
+
for (const j of rows) {
|
|
1283
|
+
const key = j.cwd || '(unknown)';
|
|
1284
|
+
if (!byCwd.has(key)) byCwd.set(key, []);
|
|
1285
|
+
byCwd.get(key).push(j);
|
|
1286
|
+
}
|
|
1287
|
+
|
|
1288
|
+
const out = [];
|
|
1289
|
+
for (const [cwd, projectJobs] of byCwd) {
|
|
1290
|
+
// Reuse the picker's OWN dep resolution so this can never disagree with
|
|
1291
|
+
// what the scheduler will actually dispatch (bare-name fallback included).
|
|
1292
|
+
const rowBySlug = new Map(projectJobs.map((j) => [j.slug, j]));
|
|
1293
|
+
const rowsByBareSlug = new Map();
|
|
1294
|
+
for (const j of projectJobs) {
|
|
1295
|
+
const bare = String(j.slug ?? '').replace(/^\d+-/, '');
|
|
1296
|
+
if (!rowsByBareSlug.has(bare)) rowsByBareSlug.set(bare, []);
|
|
1297
|
+
rowsByBareSlug.get(bare).push(j);
|
|
1298
|
+
}
|
|
1299
|
+
const rowsForDep = (slug) => {
|
|
1300
|
+
const exact = rowBySlug.get(slug);
|
|
1301
|
+
if (exact) return [exact];
|
|
1302
|
+
return rowsByBareSlug.get(String(slug ?? '').replace(/^\d+-/, '')) ?? [];
|
|
1303
|
+
};
|
|
1304
|
+
|
|
1305
|
+
// Memoised walk: does this row's dep closure hit a terminally-stuck row?
|
|
1306
|
+
const TERMINAL_STUCK = new Set(['failed', 'skipped']);
|
|
1307
|
+
const verdicts = new Map(); // slug -> Set of terminal blocker slugs
|
|
1308
|
+
const visiting = new Set();
|
|
1309
|
+
const blockersFor = (job) => {
|
|
1310
|
+
if (!job) return new Set();
|
|
1311
|
+
if (verdicts.has(job.slug)) return verdicts.get(job.slug);
|
|
1312
|
+
if (visiting.has(job.slug)) return new Set(); // dependsOn cycle — not our problem here
|
|
1313
|
+
visiting.add(job.slug);
|
|
1314
|
+
const found = new Set();
|
|
1315
|
+
for (const depSlug of job.dependsOn ?? []) {
|
|
1316
|
+
for (const dep of rowsForDep(depSlug)) {
|
|
1317
|
+
if (TERMINAL_STUCK.has(dep.status)) found.add(dep.slug);
|
|
1318
|
+
else if (dep.status !== 'completed') for (const b of blockersFor(dep)) found.add(b);
|
|
1319
|
+
}
|
|
1320
|
+
}
|
|
1321
|
+
visiting.delete(job.slug);
|
|
1322
|
+
verdicts.set(job.slug, found);
|
|
1323
|
+
return found;
|
|
1324
|
+
};
|
|
1325
|
+
|
|
1326
|
+
const blockedBy = new Set();
|
|
1327
|
+
let blocked = 0;
|
|
1328
|
+
for (const j of projectJobs) {
|
|
1329
|
+
if (j.status !== 'pending') continue;
|
|
1330
|
+
const bs = blockersFor(j);
|
|
1331
|
+
if (bs.size === 0) continue;
|
|
1332
|
+
blocked += 1;
|
|
1333
|
+
for (const b of bs) blockedBy.add(b);
|
|
1334
|
+
}
|
|
1335
|
+
if (blocked > 0) out.push({ cwd, blockedBy: [...blockedBy].sort(), blocked });
|
|
1336
|
+
}
|
|
1337
|
+
return out;
|
|
1338
|
+
}
|
|
1339
|
+
|
|
1255
1340
|
/**
|
|
1256
1341
|
* findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
1257
1342
|
*
|
|
@@ -2119,6 +2204,9 @@ let firstFailureAt = null;
|
|
|
2119
2204
|
let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
|
|
2120
2205
|
let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
|
|
2121
2206
|
let pauseClearedManuallyAt = null;
|
|
2207
|
+
// PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
|
|
2208
|
+
// See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
|
|
2209
|
+
const consecutiveRapidRateLimitsBySlug = new Map();
|
|
2122
2210
|
|
|
2123
2211
|
// ---------- timer ----------
|
|
2124
2212
|
|
|
@@ -2337,13 +2425,71 @@ async function rescheduleTimer() {
|
|
|
2337
2425
|
|
|
2338
2426
|
// ---------- pause / resume ----------
|
|
2339
2427
|
|
|
2340
|
-
|
|
2428
|
+
const MANUAL_PAUSE_COOLDOWN_MS = 300_000;
|
|
2429
|
+
// PRD 1119: after this many consecutive rate-limited dispatches of the SAME
|
|
2430
|
+
// slug that EACH also finished in under RAPID_RATE_LIMIT_WINDOW_MS, the rate
|
|
2431
|
+
// limit is not a stale/flaky auto-detection any more — it's real and
|
|
2432
|
+
// persistent for this job. Engage a hard pause the manual-clear cooldown
|
|
2433
|
+
// cannot suppress at all. This exists because the freshness check alone
|
|
2434
|
+
// (isCooldownSuppressed) is not sufficient: if the computed resumeAt is
|
|
2435
|
+
// itself wrong or stale (e.g. a failed usage-API fetch), the resume timer
|
|
2436
|
+
// can keep re-clearing the pause every ~30s, and every SUBSEQUENT dispatch
|
|
2437
|
+
// is genuinely "fresh" (it started after that re-clear) — so freshness alone
|
|
2438
|
+
// would let the spin continue indefinitely within the same 5-minute cooldown
|
|
2439
|
+
// window. The rapid-repeat count is an independent circuit breaker of last
|
|
2440
|
+
// resort for exactly that case.
|
|
2441
|
+
const CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD = 3;
|
|
2442
|
+
const RAPID_RATE_LIMIT_WINDOW_MS = 30_000;
|
|
2443
|
+
|
|
2444
|
+
/**
|
|
2445
|
+
* Pure: should setPaused()'s manual-override cooldown suppress WRITING this
|
|
2446
|
+
* pause? `force` (the rapid-repeat hard pause) always answers no — that path
|
|
2447
|
+
* exists precisely to bypass the cooldown. Otherwise, suppress only while
|
|
2448
|
+
* inside the cooldown window AND the triggering observation is stale, i.e.
|
|
2449
|
+
* it was NOT produced by a run that started after the human's manual clear.
|
|
2450
|
+
* A run that started after the clear is fresh evidence the human's fix (if
|
|
2451
|
+
* any) did not hold, and must be allowed to re-engage the pause regardless
|
|
2452
|
+
* of the cooldown — the cooldown's job is to ignore STALE auto-detections,
|
|
2453
|
+
* never to ignore new evidence.
|
|
2454
|
+
*/
|
|
2455
|
+
function isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force }) {
|
|
2456
|
+
if (force) return false;
|
|
2457
|
+
if (!clearedAt) return false;
|
|
2458
|
+
if (now - clearedAt >= MANUAL_PAUSE_COOLDOWN_MS) return false;
|
|
2459
|
+
const isFresh = typeof observedAt === 'number' && observedAt > clearedAt;
|
|
2460
|
+
return !isFresh;
|
|
2461
|
+
}
|
|
2462
|
+
|
|
2463
|
+
/**
|
|
2464
|
+
* Pure: the next consecutive-rapid-rate-limit count for a slug, given its
|
|
2465
|
+
* previous count and this run's outcome. Increments only on a rate-limited
|
|
2466
|
+
* run that ALSO ran under RAPID_RATE_LIMIT_WINDOW_MS (a genuine "dispatch,
|
|
2467
|
+
* 429, die" cycle — not a job that ran for a while before hitting the
|
|
2468
|
+
* limit). Resets to 0 on any non-rate-limited outcome. A rate-limited-but-
|
|
2469
|
+
* slow run leaves the count unchanged: still a rate limit, just not the
|
|
2470
|
+
* rapid-spin shape this cap exists to catch.
|
|
2471
|
+
*/
|
|
2472
|
+
function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
|
|
2473
|
+
if (!rateLimited) return 0;
|
|
2474
|
+
if (durationMs < RAPID_RATE_LIMIT_WINDOW_MS) return (prevCount || 0) + 1;
|
|
2475
|
+
return prevCount || 0;
|
|
2476
|
+
}
|
|
2477
|
+
|
|
2478
|
+
async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
2479
|
+
const { observedAt = null, force = false } = opts;
|
|
2341
2480
|
// Honor manual-override cooldown: if the user cleared a pause within the
|
|
2342
|
-
// last 5 minutes, suppress auto-pause re-engagement
|
|
2343
|
-
|
|
2481
|
+
// last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
|
|
2482
|
+
// backed by a fresh observation (a run that started after the clear) or is
|
|
2483
|
+
// forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
|
|
2484
|
+
if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
|
|
2344
2485
|
console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
|
|
2345
2486
|
return;
|
|
2346
2487
|
}
|
|
2488
|
+
if (force) {
|
|
2489
|
+
console.log(`[scheduler] setPaused(${reason}) forced past manual override cooldown — rapid-repeat rate-limit cap engaged`);
|
|
2490
|
+
} else if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < MANUAL_PAUSE_COOLDOWN_MS) {
|
|
2491
|
+
console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
|
|
2492
|
+
}
|
|
2347
2493
|
|
|
2348
2494
|
// For 'network' with no explicit resumeAt, auto-resume after 30 minutes.
|
|
2349
2495
|
let effectiveResumeAt = resumeAtIso;
|
|
@@ -2862,21 +3008,6 @@ async function notifyNeedsReview(job, report, {
|
|
|
2862
3008
|
}
|
|
2863
3009
|
}
|
|
2864
3010
|
|
|
2865
|
-
/** Scan the tail of a job's log for the canonical rate-limit signal. We look
|
|
2866
|
-
* at the last 16 KB — final result event always lands at the end.
|
|
2867
|
-
* Uses readTail() so no raw fd lifecycle is needed here. */
|
|
2868
|
-
function detectRateLimitInLog(logPath) {
|
|
2869
|
-
try {
|
|
2870
|
-
const text = readTail(logPath, 16384);
|
|
2871
|
-
if (!text) return false;
|
|
2872
|
-
return /"rateLimitType":"five_hour"/.test(text)
|
|
2873
|
-
|| /"api_error_status":429/.test(text)
|
|
2874
|
-
|| /You'?ve hit your limit/.test(text);
|
|
2875
|
-
} catch {
|
|
2876
|
-
return false;
|
|
2877
|
-
}
|
|
2878
|
-
}
|
|
2879
|
-
|
|
2880
3011
|
/** Scan the tail of a job's log for a network-outage signal: the structured
|
|
2881
3012
|
* `terminal_reason":"api_error"` field alongside a network-class error
|
|
2882
3013
|
* string. This is NOT a real code defect — spawning an auto-fix
|
|
@@ -3197,10 +3328,17 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
|
|
|
3197
3328
|
|
|
3198
3329
|
// ---------- execution ----------
|
|
3199
3330
|
|
|
3331
|
+
// Allocates the runId/dir pair at dispatch time WITHOUT creating the
|
|
3332
|
+
// directory — a dispatch that aborts inside spawnJob before executeJob's
|
|
3333
|
+
// openLog() call (slot-acquire miss, worktree-cap deferral, launch-gate
|
|
3334
|
+
// block, ...) must leave no trace on disk. The directory is materialised
|
|
3335
|
+
// lazily, the first time something actually needs to write into it (see
|
|
3336
|
+
// openLog's mkdirSync in executeJob below). Because tickQueue hands ONE
|
|
3337
|
+
// shared batch dir to every spawnJob in the batch, several jobs may race to
|
|
3338
|
+
// create it — `recursive: true` makes that race safe.
|
|
3200
3339
|
function pickRunDir() {
|
|
3201
3340
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
3202
3341
|
const dir = path.join(RUNS_DIR, ts);
|
|
3203
|
-
fs.mkdirSync(dir, { recursive: true });
|
|
3204
3342
|
return { runId: ts, dir };
|
|
3205
3343
|
}
|
|
3206
3344
|
|
|
@@ -3228,6 +3366,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
3228
3366
|
// fresh one via `--session-id` is the entire point of the recovery.
|
|
3229
3367
|
const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
|
|
3230
3368
|
|
|
3369
|
+
// Materialise the (possibly shared-batch) run dir lazily, right before the
|
|
3370
|
+
// first write into it — see pickRunDir's comment for why this is deferred
|
|
3371
|
+
// this far. recursive:true makes it safe if a sibling job in the same
|
|
3372
|
+
// batch dir already created it.
|
|
3373
|
+
fs.mkdirSync(runDir, { recursive: true });
|
|
3374
|
+
|
|
3231
3375
|
// Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
|
|
3232
3376
|
// error paths) before the child is created. withChildAndLog takes ownership
|
|
3233
3377
|
// of fd/safeLog/closeFd from the point it is called.
|
|
@@ -4317,6 +4461,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4317
4461
|
console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
|
|
4318
4462
|
}
|
|
4319
4463
|
|
|
4464
|
+
// Captured here (not read back off `job`, a pre-dispatch snapshot that
|
|
4465
|
+
// mutate()'s fresh-from-disk read never touches) so the rate-limited
|
|
4466
|
+
// branch below has this run's OWN start time — the freshness check
|
|
4467
|
+
// (isCooldownSuppressed) needs to know whether this specific dispatch
|
|
4468
|
+
// started after the manual clear, not whatever startedAt this row
|
|
4469
|
+
// carried from a prior run.
|
|
4470
|
+
let dispatchStartedAtMs = null;
|
|
4320
4471
|
await mutate((s) => {
|
|
4321
4472
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4322
4473
|
if (idx >= 0) {
|
|
@@ -4327,6 +4478,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4327
4478
|
delete s.jobs[idx].heldReason;
|
|
4328
4479
|
s.jobs[idx].runId = runId;
|
|
4329
4480
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
4481
|
+
dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
|
|
4330
4482
|
if (job.quietMachine === true) {
|
|
4331
4483
|
s.jobs[idx].quietMachine = true;
|
|
4332
4484
|
s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
|
|
@@ -4531,12 +4683,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4531
4683
|
// (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
|
|
4532
4684
|
// like every other best-effort git-state check in this function.
|
|
4533
4685
|
const afterGuardCwd = await uncommittedChanges(guardCwd);
|
|
4686
|
+
// stripAppOwnedChurn: the app writes session-manager-operations/ (queue.json,
|
|
4687
|
+
// history.jsonl, active-index.json, transcripts) DURING this job's own guard
|
|
4688
|
+
// window, so those land in the delta and get blamed on the job. A job can
|
|
4689
|
+
// never be responsible for them — see jobDirtFilter.cjs. Applied here, at
|
|
4690
|
+
// the single place the delta is computed, so the commit guard, the
|
|
4691
|
+
// transient-retry dirty check and leftoverPaths all agree.
|
|
4534
4692
|
const newlyDirtyAll = afterGuardCwd === null
|
|
4535
4693
|
? null
|
|
4536
|
-
: [...new Set([
|
|
4694
|
+
: stripAppOwnedChurn([...new Set([
|
|
4537
4695
|
...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
|
|
4538
4696
|
...worktreeLeftoverDirty,
|
|
4539
|
-
])];
|
|
4697
|
+
])]);
|
|
4540
4698
|
|
|
4541
4699
|
if (res.launchFailure) {
|
|
4542
4700
|
await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
|
|
@@ -4570,7 +4728,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4570
4728
|
|
|
4571
4729
|
if (res.rateLimited) {
|
|
4572
4730
|
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
4573
|
-
|
|
4731
|
+
const observedAt = dispatchStartedAtMs;
|
|
4732
|
+
const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
|
|
4733
|
+
const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
|
|
4734
|
+
consecutiveRapidRateLimitsBySlug.set(job.slug, nextCount);
|
|
4735
|
+
const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
|
|
4736
|
+
if (forceHardPause) {
|
|
4737
|
+
console.log(`[scheduler] ${job.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each — engaging hard pause`);
|
|
4738
|
+
}
|
|
4739
|
+
await setPaused('rate_limit', resetIso, { observedAt, force: forceHardPause });
|
|
4740
|
+
} else {
|
|
4741
|
+
consecutiveRapidRateLimitsBySlug.delete(job.slug);
|
|
4574
4742
|
}
|
|
4575
4743
|
|
|
4576
4744
|
// Stale queue entry: the PRD was archived (already shipped) or is gone
|
|
@@ -5366,6 +5534,109 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
5366
5534
|
tickQueue().catch((e) => console.error('[scheduler] tickQueue error', e));
|
|
5367
5535
|
}
|
|
5368
5536
|
|
|
5537
|
+
/** How long a queue may hold ready work with nothing running before the
|
|
5538
|
+
* starvation watchdog forces a tick. Deliberately longer than the poll
|
|
5539
|
+
* loop's own cadence + backoff, so this only ever fires when the normal
|
|
5540
|
+
* path has genuinely stopped driving the queue — it is a safety net, not a
|
|
5541
|
+
* second scheduler. */
|
|
5542
|
+
const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
5543
|
+
|
|
5544
|
+
/**
|
|
5545
|
+
* classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
|
|
5546
|
+
* → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
|
|
5547
|
+
*
|
|
5548
|
+
* Pure, no IO. Answers the one question the user's invariant reduces to:
|
|
5549
|
+
* "there are PRDs in a queue — is anything actually going to run them?"
|
|
5550
|
+
*
|
|
5551
|
+
* Every stall this codebase has seen was a DIFFERENT cause with the SAME
|
|
5552
|
+
* shape: ready rows, nothing running, nobody ticking. A rate-limited exit
|
|
5553
|
+
* stamped terminal `failed` (2026-09-05, 42 rows); a spin loop past the
|
|
5554
|
+
* manual-clear cooldown; a worktree merge-back that left the project on a
|
|
5555
|
+
* job branch; a job parked `needs_review` with no fix plan; app churn
|
|
5556
|
+
* counted as unfinished work. Guarding each cause individually will always
|
|
5557
|
+
* lag the next one, so this guards the SHAPE instead.
|
|
5558
|
+
*
|
|
5559
|
+
* Two outcomes, deliberately distinguished — they need opposite responses:
|
|
5560
|
+
* 'starved' — at least one pending row is dispatchable RIGHT NOW and
|
|
5561
|
+
* nothing is running. Whatever should have ticked, didn't.
|
|
5562
|
+
* Forcing a tick is safe and fixes it.
|
|
5563
|
+
* 'blocked' — every pending row is behind a terminal/parked dependency.
|
|
5564
|
+
* A tick cannot help; this needs a human (or a heal pass) to
|
|
5565
|
+
* resolve the blocker, and must be reported as such rather
|
|
5566
|
+
* than silently re-ticking forever.
|
|
5567
|
+
*
|
|
5568
|
+
* Returns null when the queue is healthy (work running, nothing pending,
|
|
5569
|
+
* paused on purpose, or simply not idle long enough yet).
|
|
5570
|
+
*/
|
|
5571
|
+
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
5572
|
+
if (paused) return null; // paused is a DECISION, not a stall
|
|
5573
|
+
if (runningCount > 0) return null; // work is flowing
|
|
5574
|
+
const rows = Array.isArray(jobs) ? jobs : [];
|
|
5575
|
+
const pending = rows.filter((j) => j && j.status === 'pending');
|
|
5576
|
+
if (pending.length === 0) return null; // nothing to run — not a stall
|
|
5577
|
+
|
|
5578
|
+
const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
|
|
5579
|
+
if (idleMs < thresholdMs) return null; // give the normal path its chance first
|
|
5580
|
+
|
|
5581
|
+
// Which pending rows could actually dispatch? Anything NOT named by a
|
|
5582
|
+
// blocked chain. computeBlockedChains already walks dependsOn with the
|
|
5583
|
+
// picker's own resolution, so the two can never disagree.
|
|
5584
|
+
const blockedChains = computeBlockedChains(rows);
|
|
5585
|
+
const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
|
|
5586
|
+
const dispatchable = pending.length - blockedTotal;
|
|
5587
|
+
|
|
5588
|
+
return {
|
|
5589
|
+
kind: dispatchable > 0 ? 'starved' : 'blocked',
|
|
5590
|
+
pending: pending.length,
|
|
5591
|
+
dispatchable,
|
|
5592
|
+
blockedChains,
|
|
5593
|
+
idleMs,
|
|
5594
|
+
};
|
|
5595
|
+
}
|
|
5596
|
+
|
|
5597
|
+
/**
|
|
5598
|
+
* The watchdog half: acts on classifyQueueStarvation. Called from the
|
|
5599
|
+
* heartbeat, which already runs on its own timer independent of the billing
|
|
5600
|
+
* poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
|
|
5601
|
+
* endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
|
|
5602
|
+
* with ready work idle indefinitely.
|
|
5603
|
+
*/
|
|
5604
|
+
async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
5605
|
+
const verdict = classifyQueueStarvation({
|
|
5606
|
+
jobs: state?.jobs,
|
|
5607
|
+
paused: state?.paused,
|
|
5608
|
+
runningCount: runningSet.size,
|
|
5609
|
+
lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
|
|
5610
|
+
now,
|
|
5611
|
+
thresholdMs,
|
|
5612
|
+
});
|
|
5613
|
+
if (!verdict) return null;
|
|
5614
|
+
|
|
5615
|
+
const mins = Math.round(verdict.idleMs / 60_000);
|
|
5616
|
+
if (verdict.kind === 'blocked') {
|
|
5617
|
+
console.warn(
|
|
5618
|
+
`[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
5619
|
+
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
5620
|
+
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
5621
|
+
);
|
|
5622
|
+
appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
5623
|
+
return verdict;
|
|
5624
|
+
}
|
|
5625
|
+
|
|
5626
|
+
console.warn(
|
|
5627
|
+
`[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
|
|
5628
|
+
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
5629
|
+
);
|
|
5630
|
+
appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
5631
|
+
// A never-populated utilization reading is itself one of the ways the
|
|
5632
|
+
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
5633
|
+
// returns early on null). Treat unknown as safe here, exactly as the
|
|
5634
|
+
// billing meter's own 429 fallback already does.
|
|
5635
|
+
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
5636
|
+
await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
|
|
5637
|
+
return verdict;
|
|
5638
|
+
}
|
|
5639
|
+
|
|
5369
5640
|
// ---------- dead-process reaper ----------
|
|
5370
5641
|
|
|
5371
5642
|
// Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
|
|
@@ -5440,11 +5711,41 @@ async function reapDeadRunningJobs() {
|
|
|
5440
5711
|
|
|
5441
5712
|
if (dead.length === 0) return;
|
|
5442
5713
|
|
|
5714
|
+
// A rate-limited death is retryable, not terminal — mirror spawnJob's own
|
|
5715
|
+
// live-process handling (PRD 1117) exactly: engage the SAME setPaused
|
|
5716
|
+
// pause here too. Skipping this would reset the row to 'pending' but
|
|
5717
|
+
// leave dispatch unpaused, so the next tick immediately re-fires it into
|
|
5718
|
+
// the same still-active rate limit — the spin loop this PRD exists to
|
|
5719
|
+
// stop. Done once, outside mutate(), before finalizing any row below.
|
|
5720
|
+
if (dead.some((d) => d.outcome === 'rate_limited')) {
|
|
5721
|
+
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
5722
|
+
const triggering = dead.find((d) => d.outcome === 'rate_limited');
|
|
5723
|
+
const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
|
|
5724
|
+
const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
|
|
5725
|
+
// Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
|
|
5726
|
+
// branch drives (see consecutiveRapidRateLimitsBySlug above) — a
|
|
5727
|
+
// process that gets rate-limited and then dies without spawnJob's own
|
|
5728
|
+
// branch ever running is reconciled HERE instead, and must feed the
|
|
5729
|
+
// same counter or a stale/wrong resumeAt could keep re-clearing this
|
|
5730
|
+
// path's "fresh" pause every reap cycle with no hard cap ever engaging.
|
|
5731
|
+
const durationMs = Number.isFinite(observedAtMs) ? Date.now() - observedAtMs : Infinity;
|
|
5732
|
+
const prevCount = consecutiveRapidRateLimitsBySlug.get(triggering.slug) || 0;
|
|
5733
|
+
const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs });
|
|
5734
|
+
consecutiveRapidRateLimitsBySlug.set(triggering.slug, nextCount);
|
|
5735
|
+
const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
|
|
5736
|
+
if (forceHardPause) {
|
|
5737
|
+
console.log(`[scheduler] ${triggering.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each (reaped) — engaging hard pause`);
|
|
5738
|
+
}
|
|
5739
|
+
await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
|
|
5740
|
+
}
|
|
5741
|
+
|
|
5443
5742
|
await mutate(async (s) => {
|
|
5444
5743
|
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
5445
5744
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
5446
5745
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
5746
|
+
const rateLimited = outcome === 'rate_limited';
|
|
5447
5747
|
const success = outcome === 'success';
|
|
5748
|
+
if (!rateLimited) consecutiveRapidRateLimitsBySlug.delete(slug);
|
|
5448
5749
|
|
|
5449
5750
|
// Best-effort in-place leftover computation: a job whose owning
|
|
5450
5751
|
// process vanished without spawnJob()'s own finally block ever
|
|
@@ -5479,13 +5780,23 @@ async function reapDeadRunningJobs() {
|
|
|
5479
5780
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
5480
5781
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
5481
5782
|
: '';
|
|
5482
|
-
const transitionReason =
|
|
5483
|
-
|
|
5484
|
-
|
|
5485
|
-
|
|
5486
|
-
|
|
5487
|
-
|
|
5488
|
-
|
|
5783
|
+
const transitionReason = rateLimited
|
|
5784
|
+
? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
|
|
5785
|
+
: (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
|
|
5786
|
+
|
|
5787
|
+
if (rateLimited) {
|
|
5788
|
+
// Retryable, never terminal (PRD 1117) — same resetJobFields path
|
|
5789
|
+
// spawnJob's own rateLimited branch uses (see ~4797's
|
|
5790
|
+
// treatAsPending), so the row comes back exactly like any other
|
|
5791
|
+
// paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
|
|
5792
|
+
resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
|
|
5793
|
+
} else {
|
|
5794
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
5795
|
+
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
5796
|
+
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
5797
|
+
s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
|
|
5798
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
5799
|
+
}
|
|
5489
5800
|
delete s.jobs[idx].runtime;
|
|
5490
5801
|
delete s.jobs[idx].guardBaseline;
|
|
5491
5802
|
delete s.jobs[idx].guardHeadBefore;
|
|
@@ -5499,7 +5810,10 @@ async function reapDeadRunningJobs() {
|
|
|
5499
5810
|
// with no exit event) wedges the lease held forever and stalls
|
|
5500
5811
|
// dispatch for every project until the app restarts.
|
|
5501
5812
|
if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
|
|
5502
|
-
if (
|
|
5813
|
+
if (rateLimited) {
|
|
5814
|
+
console.log(`[scheduler] reaped rate-limited job slug=${slug} — reset to pending, pause engaged`);
|
|
5815
|
+
appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
|
|
5816
|
+
} else if (pidless) {
|
|
5503
5817
|
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
5504
5818
|
appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
|
|
5505
5819
|
} else {
|
|
@@ -6865,6 +7179,14 @@ async function init() {
|
|
|
6865
7179
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
6866
7180
|
heartbeatInterval = setInterval(() => {
|
|
6867
7181
|
const s = readQueueSync();
|
|
7182
|
+
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
7183
|
+
// running, something must drive it. This is the only driver that does
|
|
7184
|
+
// not depend on the billing poll loop, a pause timer, or a completing
|
|
7185
|
+
// job to schedule the next tick — every one of which has failed at
|
|
7186
|
+
// least once. See classifyQueueStarvation.
|
|
7187
|
+
if (!s.unreadable) {
|
|
7188
|
+
runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
7189
|
+
}
|
|
6868
7190
|
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
6869
7191
|
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
6870
7192
|
// failed }` literal silently minted a NEW key for any other value
|
|
@@ -7465,4 +7787,4 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
7465
7787
|
});
|
|
7466
7788
|
}
|
|
7467
7789
|
|
|
7468
|
-
module.exports = { findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure };
|
|
7790
|
+
module.exports = { classifyQueueStarvation, runQueueStarvationWatchdog, QUEUE_STARVATION_MS, computeBlockedChains, stripAppOwnedChurn, findOverrunningJobs, JOB_OVERRUN_FACTOR, JOB_OVERRUN_FLOOR_MS, registerScheduleHandlers, attachWindow, init, ROOT, PRDS_DIR, healRefusalReason, writeQueue, reconcile, reconcileSourcePromptId, allocateParallelGroup, selectHistoryJobs, parsePorcelain, FINISH_PROTOCOL, IDLE_OUTPUT_KILL_MS, BASH_DEFAULT_TIMEOUT_MS, BASH_MAX_TIMEOUT_MS, remote, pickNextBatch, pickForProject, reapDeadRunningJobs, pollRecoveryClearSource, memoryLimitedBatchSize, availableForJobs, reverifyNeedsReview, isRescanCandidate, isFailedUnverifiedShaped, computeLooksDone, isPromotableOriginal, selectAutoFixTargets, applyRcaClassification, isEligibleForImmediateAutoFix, resolveRunId, isUnresolvableNeedsReview, isExhaustedAutoFix, isPlanUnqueued, fixSlugFor, healTargetForFix, buildInvestigationPrompt, isGitRepoSync, committedInWindow, computeCommittedDuringRun, classifySigtermWithCommit, isFixPlanSlug, isFixPlanBeyondDepthCap, MAX_INVESTIGATION_DEPTH, forceTickOutcome, applyPauseCleared, detectNetworkErrorInLog, detectRateLimitInLog, classifyFailureOutcome, commitGuardVerdict, leftoverFieldsFrom, applyLeftoverFields, LEFTOVER_PATHS_CAP, capDirtyPaths, buildForeignWipSection, PRE_RUN_DIRTY_PATHS_CAP, FOREIGN_WIP_DELIMITER, FOREIGN_WIP_END_DELIMITER, TRANSIENT_RETRY_CAP, buildScheduleStatePayload, partitionBootOrphans, applyOrphanOutcome, BOOT_ORPHAN_KILL_GRACE_MS, registerAdminRoutes, notifyOriginatingTab, notifyNeedsReview, isNotifiableTerminalStatus, extractResultTextFromLog, candidatePrdsDirs, candidateArchivedPrdsDirs, resolveArchivedPrdStatus, prdDirForCwd, prdPathForJob, archivedPrdPathForJob, archivedTwinExists, findPrdDir, resolveVerifyPrdPath, resolveFixPlanPath, resolveNotifyPrd, runPrdMigration, consolidateAllFlatPrds, shouldSkipInvestigationForCleanRun, archiveCompletedPrd, retireCompletedSlugs, SCHEDULER_BOOTED_AT, SCHEDULER_CODE_SHA, resetJobFields, executeJob, prdArchivedSkipResult, spawnJob, listPrdsInternal, computeStallSummary, findStaleQuarantinedJobs, QUARANTINE_ESCALATE_MS, applyClearQueueVictims, PIDLESS_SPAWN_GRACE_MS, findStrandedInvestigations, INVESTIGATION_MAX_MS, stashList, parseStashLine, pathsChangedSince, restoreSpecificStash, evaluateSharedTreeGuard, checkSharedTreeGuard, uncommittedChanges, gitHead, selectResumeRecoveryTarget, buildResumeRecoveryPreamble, buildClaudeSpawnArgs, spawnResumeRecovery, spawnInvestigation, computeLaunchHolds, handleLaunchFailure, applyLaunchFailure, setPaused, clearPause, tickQueue, runDueJobs, isCooldownSuppressed, nextRapidRateLimitCount, CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD, RAPID_RATE_LIMIT_WINDOW_MS, MANUAL_PAUSE_COOLDOWN_MS, RUNS_DIR, pickRunDir };
|
package/src/preload/api.d.ts
CHANGED
|
@@ -354,20 +354,23 @@ export interface DelegationReadinessCheck {
|
|
|
354
354
|
| 'scheduler-mcp-project-duplicate'
|
|
355
355
|
| 'dev-plugin'
|
|
356
356
|
| 'agent-personas'
|
|
357
|
-
| 'prd-write-guard'
|
|
357
|
+
| 'prd-write-guard'
|
|
358
|
+
| 'destructive-git-guard';
|
|
358
359
|
label: string;
|
|
359
360
|
ok: boolean;
|
|
360
361
|
detail: string;
|
|
361
362
|
fix: string | null;
|
|
362
363
|
/** Non-null when Session Manager can install this fix itself, one press. */
|
|
363
|
-
fixAction: 'install-prd-write-guard' | null;
|
|
364
|
+
fixAction: 'install-prd-write-guard' | 'install-destructive-git-guard' | null;
|
|
364
365
|
/** True when ok:true is a WARNING (still passing, but worth a human's attention) — today only scheduler-mcp-project-duplicate. */
|
|
365
366
|
warn?: boolean;
|
|
366
367
|
/** True when this check didn't run because a precondition (another check) already failed — reported ok:true, not a failure. */
|
|
367
368
|
skipped?: boolean;
|
|
368
369
|
}
|
|
369
370
|
|
|
370
|
-
|
|
371
|
+
/** Result of installPrdWriteGuard and installDestructiveGitGuard alike
|
|
372
|
+
* (delegationReadiness.cjs) — the two installers share one contract. */
|
|
373
|
+
export interface InstallGuardResult {
|
|
371
374
|
ok: boolean;
|
|
372
375
|
action: 'installed' | 'repaired' | 'already-installed' | 'error';
|
|
373
376
|
settingsPath?: string;
|
|
@@ -1472,7 +1475,8 @@ export interface SessionManagerAPI {
|
|
|
1472
1475
|
/** "Can this project actually delegate?" — the 4 preconditions for
|
|
1473
1476
|
* scheduler_create_prd being in an agent's tool list at all. */
|
|
1474
1477
|
delegationReadiness: (cwd: string) => Promise<DelegationReadiness>;
|
|
1475
|
-
installPrdWriteGuard: (cwd: string) => Promise<
|
|
1478
|
+
installPrdWriteGuard: (cwd: string) => Promise<InstallGuardResult>;
|
|
1479
|
+
installDestructiveGitGuard: (cwd: string) => Promise<InstallGuardResult>;
|
|
1476
1480
|
onNewSession: (handler: () => void) => () => void;
|
|
1477
1481
|
onRebootSession: (handler: () => void) => () => void;
|
|
1478
1482
|
archiveProject: (encoded: string) => Promise<{ ok: boolean; error?: string }>;
|
package/src/preload/index.cjs
CHANGED
|
@@ -24,6 +24,7 @@ contextBridge.exposeInMainWorld('api', {
|
|
|
24
24
|
seedStatus: () => ipcRenderer.invoke('app:seed-status'),
|
|
25
25
|
delegationReadiness: (cwd) => ipcRenderer.invoke('app:delegation-readiness', { cwd }),
|
|
26
26
|
installPrdWriteGuard: (cwd) => ipcRenderer.invoke('app:install-prd-write-guard', { cwd }),
|
|
27
|
+
installDestructiveGitGuard: (cwd) => ipcRenderer.invoke('app:install-destructive-git-guard', { cwd }),
|
|
27
28
|
onNewSession: (handler) => {
|
|
28
29
|
const listener = () => handler();
|
|
29
30
|
ipcRenderer.on('app:new-session', listener);
|