claude-code-session-manager 0.78.0 → 0.80.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-13pfo8uY.js → AgentLibrary-zS3jw_1e.js} +1 -1
- package/dist/assets/{DataModel-SUyQbFlg.js → DataModel-Cy_vxTpi.js} +1 -1
- package/dist/assets/{History-2GJMS703.js → History-C6JRuqfT.js} +1 -1
- package/dist/assets/{Hooks-DM2nS3RT.js → Hooks-BafPy9mB.js} +1 -1
- package/dist/assets/{HostBilko-BLeC-lpp.js → HostBilko-BZwhQOFt.js} +1 -1
- package/dist/assets/{Library-BaRkU9m0.js → Library-C8JDDliz.js} +1 -1
- package/dist/assets/{ListDetail-D5scjSKq.js → ListDetail-CqiOdwLc.js} +1 -1
- package/dist/assets/{MarkdownEditor-B1lgAo9T.js → MarkdownEditor-CyLyP67L.js} +1 -1
- package/dist/assets/{McpServers-BzyThQSM.js → McpServers-BzMv-_84.js} +1 -1
- package/dist/assets/{Memory-7UdaOTtl.js → Memory-DSBYQdJR.js} +1 -1
- package/dist/assets/{Panel-JbTMaOPq.js → Panel-CLUhkNNA.js} +1 -1
- package/dist/assets/{Permissions-UBam0bJG.js → Permissions-BfC2-HN4.js} +1 -1
- package/dist/assets/{Plugins-B3gUDkeb.js → Plugins-BKi40jT5.js} +2 -2
- package/dist/assets/{ProvenanceBadge-CeHOub7m.js → ProvenanceBadge-BzFw4KhD.js} +1 -1
- package/dist/assets/{SaveBar-BcvQEq6h.js → SaveBar-avk2p9jv.js} +1 -1
- package/dist/assets/{Scheduler-Dc5qiP24.js → Scheduler-Bf_6MdJo.js} +7 -7
- package/dist/assets/{ScopeSwitcher-BvGQmw4Y.js → ScopeSwitcher-C-RwYUVZ.js} +1 -1
- package/dist/assets/{Settings-C2dEFb-v.js → Settings-Djd8OoBA.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-CIwlosBc.js → SkillReferenceGraph-DuogY6s7.js} +1 -1
- package/dist/assets/{Skills-C0GzzVrQ.js → Skills-D_qAqxZ_.js} +1 -1
- package/dist/assets/{SystemPrompt-mtGPK8zo.js → SystemPrompt-DbHFLQV3.js} +1 -1
- package/dist/assets/{TagLibrary-DX54-mpd.js → TagLibrary-C2y91BT0.js} +1 -1
- package/dist/assets/{TiptapBody-yADC2RWE.js → TiptapBody-D9iz4xQx.js} +1 -1
- package/dist/assets/{Toggle-CRxaCYLI.js → Toggle-BGnFL2E5.js} +1 -1
- package/dist/assets/{index-D6ymGESc.js → index-_2ARyFDj.js} +4 -4
- package/dist/assets/{settingsSchema-TtMvT5Sx.js → settingsSchema-JK15eJU8.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +1 -1
- package/plugins/session-manager-dev/skills/builder/3-publish/SKILL.md +10 -0
- package/scripts/project-pages-logic/dist/logic.cjs +12 -12
- package/scripts/render-project-pages/dist/renderer.cjs +22 -22
- package/src/main/__tests__/computeDepHistorySatisfaction.test.cjs +66 -0
- package/src/main/__tests__/prdCreate.test.cjs +133 -8
- package/src/main/__tests__/prdFrontmatterDependsOn.test.cjs +136 -0
- package/src/main/__tests__/prdUpdateDependsOn.test.cjs +160 -0
- package/src/main/__tests__/queueHistory.test.cjs +33 -0
- package/src/main/__tests__/runLogRetention.test.cjs +59 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +1 -0
- package/src/main/__tests__/scheduler-autofix-outcome.test.cjs +73 -1
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +17 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +20 -0
- package/src/main/__tests__/scheduler-leftover-quarantine.test.cjs +199 -0
- package/src/main/__tests__/scheduler-mechanical-recovery.test.cjs +222 -0
- package/src/main/__tests__/scheduler-never-stop.test.cjs +157 -0
- package/src/main/__tests__/scheduler-no-orphan-run-dir.test.cjs +81 -0
- package/src/main/__tests__/scheduler-rate-limit-cooldown-freshness.test.cjs +123 -0
- package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +158 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +30 -0
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +51 -0
- package/src/main/__tests__/scheduler-resume-recovery.test.cjs +254 -0
- package/src/main/__tests__/schedulerBatchRootBlocker.test.cjs +117 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -2
- package/src/main/ipcSchemas.cjs +15 -1
- package/src/main/lib/__tests__/gitWorktree.test.cjs +129 -10
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +90 -1
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +59 -7
- package/src/main/lib/depSlugResolve.cjs +72 -0
- package/src/main/lib/epicWorktreeMerge.cjs +3 -3
- package/src/main/lib/epicWorktreeMint.cjs +17 -5
- package/src/main/lib/fixPlanSlug.cjs +62 -0
- package/src/main/lib/gitWorktree.cjs +97 -12
- package/src/main/lib/jobDirtFilter.cjs +54 -0
- package/src/main/lib/mcpToolCatalog.cjs +4 -1
- package/src/main/lib/prdCreate.cjs +84 -5
- package/src/main/lib/prdFrontmatter.cjs +56 -8
- package/src/main/lib/queueHistory.cjs +50 -5
- package/src/main/lib/rateLimitDetect.cjs +35 -0
- package/src/main/lib/reaperHelpers.cjs +31 -6
- package/src/main/lib/runLogRetention.cjs +82 -4
- package/src/main/lib/scheduleJobTransitions.cjs +12 -2
- package/src/main/lib/schedulerBatch.cjs +181 -23
- package/src/main/scheduler/prdParser.cjs +7 -0
- package/src/main/scheduler.cjs +1165 -70
- package/src/preload/api.d.ts +8 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -58,6 +58,8 @@ const launchFailure = require('./lib/launchFailure.cjs');
|
|
|
58
58
|
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
59
59
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
60
60
|
const { claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs } = require('./lib/reaperHelpers.cjs');
|
|
61
|
+
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
62
|
+
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
61
63
|
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
62
64
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
63
65
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
@@ -71,6 +73,7 @@ const { maybeEnqueueValidationPrompt } = require('./lib/epicValidationHook.cjs')
|
|
|
71
73
|
const promptSessionTranscript = require('./promptSessionTranscript.cjs');
|
|
72
74
|
const { verifyRun } = require('./runVerify.cjs');
|
|
73
75
|
const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require('./lib/terminalRunOutcome.cjs');
|
|
76
|
+
const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
|
|
74
77
|
const { landedSinceRun } = require('./lib/landedSinceRun.cjs');
|
|
75
78
|
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
76
79
|
const logs = require('./logs.cjs');
|
|
@@ -97,7 +100,7 @@ const JOB_OVERRUN_FACTOR = process.env.SM_JOB_OVERRUN_FACTOR
|
|
|
97
100
|
const JOB_OVERRUN_FLOOR_MS = process.env.SM_JOB_OVERRUN_FLOOR_MINUTES
|
|
98
101
|
? Number(process.env.SM_JOB_OVERRUN_FLOOR_MINUTES) * 60_000
|
|
99
102
|
: JOB_OVERRUN_FLOOR_MS_DEFAULT;
|
|
100
|
-
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD } = require('./lib/schedulerBatch.cjs');
|
|
103
|
+
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
|
|
101
104
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
102
105
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
103
106
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
@@ -138,6 +141,7 @@ const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
|
138
141
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
139
142
|
const queueStore = require('./lib/queueStore.cjs');
|
|
140
143
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
144
|
+
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
141
145
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
142
146
|
const { allProjectCwds } = require('../../scripts/lib/activeSessions.cjs');
|
|
143
147
|
|
|
@@ -1252,6 +1256,89 @@ function computeStallSummary(state) {
|
|
|
1252
1256
|
return { stalled, total, running, pending, byProject };
|
|
1253
1257
|
}
|
|
1254
1258
|
|
|
1259
|
+
/**
|
|
1260
|
+
* computeBlockedChains(jobs) → [{ cwd, blockedBy, blocked }]
|
|
1261
|
+
*
|
|
1262
|
+
* Pure, no IO. The gap computeStallSummary above cannot see.
|
|
1263
|
+
*
|
|
1264
|
+
* `stalled` is defined as `running === 0 && pending === 0`, which encodes an
|
|
1265
|
+
* assumption that a PENDING row is healthy in-progress work. It is not: a
|
|
1266
|
+
* pending row whose `dependsOn` chain terminates in a TERMINAL non-completed
|
|
1267
|
+
* status (`failed`/`skipped`) can never be dispatched by pickForProject, and
|
|
1268
|
+
* never will be, but it still counts toward `pending` and so reads as a
|
|
1269
|
+
* healthy queue to every monitor in the app.
|
|
1270
|
+
*
|
|
1271
|
+
* On 2026-09-05 starry-night-ships held 42 such rows behind one `failed`
|
|
1272
|
+
* job for three hours. Machine-wide `stalled` was false (42 pending),
|
|
1273
|
+
* per-project `stalled` was false (42 pending), the queue-health sweep
|
|
1274
|
+
* doesn't count `failed` at all, and the supervisor only probes `running` —
|
|
1275
|
+
* so nothing anywhere reported a problem while nothing could ever run.
|
|
1276
|
+
*
|
|
1277
|
+
* Reported per project as { blockedBy: [terminal slugs], blocked: count }.
|
|
1278
|
+
* Transitive by construction: a row blocked by a row that is itself blocked
|
|
1279
|
+
* resolves through the same walk.
|
|
1280
|
+
*/
|
|
1281
|
+
function computeBlockedChains(jobs) {
|
|
1282
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
1283
|
+
const byCwd = new Map();
|
|
1284
|
+
for (const j of rows) {
|
|
1285
|
+
const key = j.cwd || '(unknown)';
|
|
1286
|
+
if (!byCwd.has(key)) byCwd.set(key, []);
|
|
1287
|
+
byCwd.get(key).push(j);
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1290
|
+
const out = [];
|
|
1291
|
+
for (const [cwd, projectJobs] of byCwd) {
|
|
1292
|
+
// Reuse the picker's OWN dep resolution so this can never disagree with
|
|
1293
|
+
// what the scheduler will actually dispatch (bare-name fallback included).
|
|
1294
|
+
const rowBySlug = new Map(projectJobs.map((j) => [j.slug, j]));
|
|
1295
|
+
const rowsByBareSlug = new Map();
|
|
1296
|
+
for (const j of projectJobs) {
|
|
1297
|
+
const bare = String(j.slug ?? '').replace(/^\d+-/, '');
|
|
1298
|
+
if (!rowsByBareSlug.has(bare)) rowsByBareSlug.set(bare, []);
|
|
1299
|
+
rowsByBareSlug.get(bare).push(j);
|
|
1300
|
+
}
|
|
1301
|
+
const rowsForDep = (slug) => {
|
|
1302
|
+
const exact = rowBySlug.get(slug);
|
|
1303
|
+
if (exact) return [exact];
|
|
1304
|
+
return rowsByBareSlug.get(String(slug ?? '').replace(/^\d+-/, '')) ?? [];
|
|
1305
|
+
};
|
|
1306
|
+
|
|
1307
|
+
// Memoised walk: does this row's dep closure hit a terminally-stuck row?
|
|
1308
|
+
const TERMINAL_STUCK = new Set(['failed', 'skipped']);
|
|
1309
|
+
const verdicts = new Map(); // slug -> Set of terminal blocker slugs
|
|
1310
|
+
const visiting = new Set();
|
|
1311
|
+
const blockersFor = (job) => {
|
|
1312
|
+
if (!job) return new Set();
|
|
1313
|
+
if (verdicts.has(job.slug)) return verdicts.get(job.slug);
|
|
1314
|
+
if (visiting.has(job.slug)) return new Set(); // dependsOn cycle — not our problem here
|
|
1315
|
+
visiting.add(job.slug);
|
|
1316
|
+
const found = new Set();
|
|
1317
|
+
for (const depSlug of job.dependsOn ?? []) {
|
|
1318
|
+
for (const dep of rowsForDep(depSlug)) {
|
|
1319
|
+
if (TERMINAL_STUCK.has(dep.status)) found.add(dep.slug);
|
|
1320
|
+
else if (dep.status !== 'completed') for (const b of blockersFor(dep)) found.add(b);
|
|
1321
|
+
}
|
|
1322
|
+
}
|
|
1323
|
+
visiting.delete(job.slug);
|
|
1324
|
+
verdicts.set(job.slug, found);
|
|
1325
|
+
return found;
|
|
1326
|
+
};
|
|
1327
|
+
|
|
1328
|
+
const blockedBy = new Set();
|
|
1329
|
+
let blocked = 0;
|
|
1330
|
+
for (const j of projectJobs) {
|
|
1331
|
+
if (j.status !== 'pending') continue;
|
|
1332
|
+
const bs = blockersFor(j);
|
|
1333
|
+
if (bs.size === 0) continue;
|
|
1334
|
+
blocked += 1;
|
|
1335
|
+
for (const b of bs) blockedBy.add(b);
|
|
1336
|
+
}
|
|
1337
|
+
if (blocked > 0) out.push({ cwd, blockedBy: [...blockedBy].sort(), blocked });
|
|
1338
|
+
}
|
|
1339
|
+
return out;
|
|
1340
|
+
}
|
|
1341
|
+
|
|
1255
1342
|
/**
|
|
1256
1343
|
* findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
1257
1344
|
*
|
|
@@ -1987,11 +2074,24 @@ async function reconcile(state) {
|
|
|
1987
2074
|
exitCode: null,
|
|
1988
2075
|
error: null,
|
|
1989
2076
|
};
|
|
1990
|
-
//
|
|
1991
|
-
//
|
|
1992
|
-
//
|
|
1993
|
-
//
|
|
1994
|
-
|
|
2077
|
+
// Fix-plan classification (PRD 1131): a freshly-discovered PRD is a
|
|
2078
|
+
// genuine scheduler-authored fix plan only when its OWN provenance says
|
|
2079
|
+
// so — an explicit isFixPlan:true stamp (spawnInvestigation's prompt
|
|
2080
|
+
// template) or the absence of any createdVia stamp at all (legacy
|
|
2081
|
+
// fallback, matching the "no provenance = trust the name" rule the
|
|
2082
|
+
// quarantine gate below already applies) — never merely because the
|
|
2083
|
+
// slug looks like one. See lib/fixPlanSlug.cjs's header for why (PRD
|
|
2084
|
+
// 1126: a scheduler_create_prd-authored PRD whose slug happened to start
|
|
2085
|
+
// with "fix-" was wrongly stamped investigationDepth before it ever ran).
|
|
2086
|
+
// Persisted onto the queue row so every later consumer
|
|
2087
|
+
// (commitGuardVerdict, isFixPlanBeyondDepthCap, the fix-plan-completion
|
|
2088
|
+
// checks) reads this stamp instead of re-deriving it from the name.
|
|
2089
|
+
entry.isFixPlan = classifyDiscoveredFixPlan(p, slug);
|
|
2090
|
+
// Stamp investigationDepth relative to the original job it heals, so
|
|
2091
|
+
// selectAutoFixTargets/spawnInvestigation can bound the fix-of-a-fix
|
|
2092
|
+
// recursion (see MAX_INVESTIGATION_DEPTH). Non-fix-plan jobs get no
|
|
2093
|
+
// explicit field — they read as depth 1 via `?? 1`.
|
|
2094
|
+
if (entry.isFixPlan) {
|
|
1995
2095
|
const parent = healTargetForFix(slug, state.jobs);
|
|
1996
2096
|
entry.investigationDepth = parent ? (parent.investigationDepth ?? 1) + 1 : 2;
|
|
1997
2097
|
}
|
|
@@ -2003,8 +2103,9 @@ async function reconcile(state) {
|
|
|
2003
2103
|
// guard-prd-writes.cjs PreToolUse hook should have denied. Fix-plan PRDs
|
|
2004
2104
|
// are exempt: spawnInvestigation's own probe writes them directly by
|
|
2005
2105
|
// design (a trusted, scheduler-spawned internal loop, not an
|
|
2006
|
-
// agent/human authoring a PRD)
|
|
2007
|
-
//
|
|
2106
|
+
// agent/human authoring a PRD) — entry.isFixPlan (just classified above)
|
|
2107
|
+
// is the provenance-aware verdict for that exemption now, not a raw
|
|
2108
|
+
// isFixPlanSlug name check.
|
|
2008
2109
|
//
|
|
2009
2110
|
// Quarantine is loud and reversible, never a silent skip (see the
|
|
2010
2111
|
// 2026-08-01 23-PRD outage this file's header references for what a
|
|
@@ -2013,7 +2114,7 @@ async function reconcile(state) {
|
|
|
2013
2114
|
// (schedule:adopt-prd) that stamps the file via the same update-prd API
|
|
2014
2115
|
// route the MCP tool uses — reconcile()'s adopt path above promotes it
|
|
2015
2116
|
// to 'pending' on the very next pass, within one tick of being stamped.
|
|
2016
|
-
if (!p.createdVia && !
|
|
2117
|
+
if (!p.createdVia && !entry.isFixPlan) {
|
|
2017
2118
|
entry.status = 'quarantined';
|
|
2018
2119
|
// Stamped at creation (not via transitionJob, since this is a
|
|
2019
2120
|
// brand-new row minted directly at 'quarantined' rather than
|
|
@@ -2119,6 +2220,9 @@ let firstFailureAt = null;
|
|
|
2119
2220
|
let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
|
|
2120
2221
|
let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
|
|
2121
2222
|
let pauseClearedManuallyAt = null;
|
|
2223
|
+
// PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
|
|
2224
|
+
// See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
|
|
2225
|
+
const consecutiveRapidRateLimitsBySlug = new Map();
|
|
2122
2226
|
|
|
2123
2227
|
// ---------- timer ----------
|
|
2124
2228
|
|
|
@@ -2337,13 +2441,71 @@ async function rescheduleTimer() {
|
|
|
2337
2441
|
|
|
2338
2442
|
// ---------- pause / resume ----------
|
|
2339
2443
|
|
|
2340
|
-
|
|
2444
|
+
const MANUAL_PAUSE_COOLDOWN_MS = 300_000;
|
|
2445
|
+
// PRD 1119: after this many consecutive rate-limited dispatches of the SAME
|
|
2446
|
+
// slug that EACH also finished in under RAPID_RATE_LIMIT_WINDOW_MS, the rate
|
|
2447
|
+
// limit is not a stale/flaky auto-detection any more — it's real and
|
|
2448
|
+
// persistent for this job. Engage a hard pause the manual-clear cooldown
|
|
2449
|
+
// cannot suppress at all. This exists because the freshness check alone
|
|
2450
|
+
// (isCooldownSuppressed) is not sufficient: if the computed resumeAt is
|
|
2451
|
+
// itself wrong or stale (e.g. a failed usage-API fetch), the resume timer
|
|
2452
|
+
// can keep re-clearing the pause every ~30s, and every SUBSEQUENT dispatch
|
|
2453
|
+
// is genuinely "fresh" (it started after that re-clear) — so freshness alone
|
|
2454
|
+
// would let the spin continue indefinitely within the same 5-minute cooldown
|
|
2455
|
+
// window. The rapid-repeat count is an independent circuit breaker of last
|
|
2456
|
+
// resort for exactly that case.
|
|
2457
|
+
const CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD = 3;
|
|
2458
|
+
const RAPID_RATE_LIMIT_WINDOW_MS = 30_000;
|
|
2459
|
+
|
|
2460
|
+
/**
|
|
2461
|
+
* Pure: should setPaused()'s manual-override cooldown suppress WRITING this
|
|
2462
|
+
* pause? `force` (the rapid-repeat hard pause) always answers no — that path
|
|
2463
|
+
* exists precisely to bypass the cooldown. Otherwise, suppress only while
|
|
2464
|
+
* inside the cooldown window AND the triggering observation is stale, i.e.
|
|
2465
|
+
* it was NOT produced by a run that started after the human's manual clear.
|
|
2466
|
+
* A run that started after the clear is fresh evidence the human's fix (if
|
|
2467
|
+
* any) did not hold, and must be allowed to re-engage the pause regardless
|
|
2468
|
+
* of the cooldown — the cooldown's job is to ignore STALE auto-detections,
|
|
2469
|
+
* never to ignore new evidence.
|
|
2470
|
+
*/
|
|
2471
|
+
function isCooldownSuppressed({ pauseClearedManuallyAt: clearedAt, now, observedAt, force }) {
|
|
2472
|
+
if (force) return false;
|
|
2473
|
+
if (!clearedAt) return false;
|
|
2474
|
+
if (now - clearedAt >= MANUAL_PAUSE_COOLDOWN_MS) return false;
|
|
2475
|
+
const isFresh = typeof observedAt === 'number' && observedAt > clearedAt;
|
|
2476
|
+
return !isFresh;
|
|
2477
|
+
}
|
|
2478
|
+
|
|
2479
|
+
/**
|
|
2480
|
+
* Pure: the next consecutive-rapid-rate-limit count for a slug, given its
|
|
2481
|
+
* previous count and this run's outcome. Increments only on a rate-limited
|
|
2482
|
+
* run that ALSO ran under RAPID_RATE_LIMIT_WINDOW_MS (a genuine "dispatch,
|
|
2483
|
+
* 429, die" cycle — not a job that ran for a while before hitting the
|
|
2484
|
+
* limit). Resets to 0 on any non-rate-limited outcome. A rate-limited-but-
|
|
2485
|
+
* slow run leaves the count unchanged: still a rate limit, just not the
|
|
2486
|
+
* rapid-spin shape this cap exists to catch.
|
|
2487
|
+
*/
|
|
2488
|
+
function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
|
|
2489
|
+
if (!rateLimited) return 0;
|
|
2490
|
+
if (durationMs < RAPID_RATE_LIMIT_WINDOW_MS) return (prevCount || 0) + 1;
|
|
2491
|
+
return prevCount || 0;
|
|
2492
|
+
}
|
|
2493
|
+
|
|
2494
|
+
async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
2495
|
+
const { observedAt = null, force = false } = opts;
|
|
2341
2496
|
// Honor manual-override cooldown: if the user cleared a pause within the
|
|
2342
|
-
// last 5 minutes, suppress auto-pause re-engagement
|
|
2343
|
-
|
|
2497
|
+
// last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
|
|
2498
|
+
// backed by a fresh observation (a run that started after the clear) or is
|
|
2499
|
+
// forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
|
|
2500
|
+
if (isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
|
|
2344
2501
|
console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
|
|
2345
2502
|
return;
|
|
2346
2503
|
}
|
|
2504
|
+
if (force) {
|
|
2505
|
+
console.log(`[scheduler] setPaused(${reason}) forced past manual override cooldown — rapid-repeat rate-limit cap engaged`);
|
|
2506
|
+
} else if (pauseClearedManuallyAt && Date.now() - pauseClearedManuallyAt < MANUAL_PAUSE_COOLDOWN_MS) {
|
|
2507
|
+
console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
|
|
2508
|
+
}
|
|
2347
2509
|
|
|
2348
2510
|
// For 'network' with no explicit resumeAt, auto-resume after 30 minutes.
|
|
2349
2511
|
let effectiveResumeAt = resumeAtIso;
|
|
@@ -2438,6 +2600,17 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2438
2600
|
delete job.verifierVerdict;
|
|
2439
2601
|
delete job.uncommittedPaths;
|
|
2440
2602
|
delete job.resumeRecoveryAttempted;
|
|
2603
|
+
// Same one-attempt-per-episode category as resumeRecoveryAttempted above —
|
|
2604
|
+
// a re-fired row must be able to earn a fresh mechanical-recovery attempt
|
|
2605
|
+
// if it parks needs_review again (PRD 1130).
|
|
2606
|
+
delete job.mechanicalRecoveryAttempted;
|
|
2607
|
+
// Quarantine (PRD 1128) is scoped to THIS run's episode exactly like
|
|
2608
|
+
// resumeRecoveryAttempted above — a re-fired row must be able to earn a
|
|
2609
|
+
// fresh quarantine attempt if it parks needs_review again.
|
|
2610
|
+
delete job.leftoverQuarantineAttempted;
|
|
2611
|
+
delete job.quarantinedTo;
|
|
2612
|
+
delete job.quarantinedCommit;
|
|
2613
|
+
delete job.quarantinedPaths;
|
|
2441
2614
|
// Same "this run's outcome, not durable across a reset" category as the
|
|
2442
2615
|
// fields above — a stale 'archive' recoveryAction from a prior life of this
|
|
2443
2616
|
// slug must never survive a reset and silently exclude a genuinely-new
|
|
@@ -2446,6 +2619,10 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
2446
2619
|
// can otherwise linger forever when RCA is disabled or errors).
|
|
2447
2620
|
delete job.rcaFailureClass;
|
|
2448
2621
|
delete job.rcaRecoveryAction;
|
|
2622
|
+
// Same "this run's outcome, not durable across a reset" category — a
|
|
2623
|
+
// human-driven reset must genuinely start the auto-fix budget over,
|
|
2624
|
+
// including the one-time dead-fix-plan-child reopen (PRD 1129).
|
|
2625
|
+
delete job.autoFixReopened;
|
|
2449
2626
|
// Like exitCode: this run's outcome, not durable across a reset — a stale
|
|
2450
2627
|
// leak badge from a prior attempt must not linger once the job re-fires.
|
|
2451
2628
|
delete job.leakedDescendants;
|
|
@@ -2862,21 +3039,6 @@ async function notifyNeedsReview(job, report, {
|
|
|
2862
3039
|
}
|
|
2863
3040
|
}
|
|
2864
3041
|
|
|
2865
|
-
/** Scan the tail of a job's log for the canonical rate-limit signal. We look
|
|
2866
|
-
* at the last 16 KB — final result event always lands at the end.
|
|
2867
|
-
* Uses readTail() so no raw fd lifecycle is needed here. */
|
|
2868
|
-
function detectRateLimitInLog(logPath) {
|
|
2869
|
-
try {
|
|
2870
|
-
const text = readTail(logPath, 16384);
|
|
2871
|
-
if (!text) return false;
|
|
2872
|
-
return /"rateLimitType":"five_hour"/.test(text)
|
|
2873
|
-
|| /"api_error_status":429/.test(text)
|
|
2874
|
-
|| /You'?ve hit your limit/.test(text);
|
|
2875
|
-
} catch {
|
|
2876
|
-
return false;
|
|
2877
|
-
}
|
|
2878
|
-
}
|
|
2879
|
-
|
|
2880
3042
|
/** Scan the tail of a job's log for a network-outage signal: the structured
|
|
2881
3043
|
* `terminal_reason":"api_error"` field alongside a network-class error
|
|
2882
3044
|
* string. This is NOT a real code defect — spawning an auto-fix
|
|
@@ -3172,6 +3334,302 @@ As the LAST LINE of your final result text, emit exactly one of:
|
|
|
3172
3334
|
Print PASS only once the commit above has actually landed.`;
|
|
3173
3335
|
}
|
|
3174
3336
|
|
|
3337
|
+
/**
|
|
3338
|
+
* Mechanical recovery (PRD 1130). isFixPlanBeyondDepthCap (below) is the
|
|
3339
|
+
* ONLY gate on re-investigating a fix-plan job at investigationDepth >= 2 —
|
|
3340
|
+
* correct for open-ended "author another plan" recursion, but it also
|
|
3341
|
+
* strands a depth-capped job whose failure was fully mechanical (no
|
|
3342
|
+
* judgement required) with no other ladder rung, since resume-first recovery
|
|
3343
|
+
* (selectResumeRecoveryTarget above) is hard-gated on verdict
|
|
3344
|
+
* 'uncommitted_changes'. This rung is evaluated INDEPENDENTLY of
|
|
3345
|
+
* isFixPlanBeyondDepthCap — depth never disqualifies it, because unlike
|
|
3346
|
+
* auto-fix it authors no plan and spawns no model; it is pure git.
|
|
3347
|
+
*
|
|
3348
|
+
* The closed set of mechanically-resolvable verdicts starts at exactly
|
|
3349
|
+
* 'worktree_integration_failed': PRD 1125 already taught integrateBranch to
|
|
3350
|
+
* parse git's "would be overwritten by merge" stderr, verify the blocking
|
|
3351
|
+
* paths are byte-identical to the branch, discard the proven duplicates, and
|
|
3352
|
+
* retry the merge once. A job parked with this verdict has its `sm-job/
|
|
3353
|
+
* <slug>` branch preserved (integrateJobBranch never deletes the branch on
|
|
3354
|
+
* failure — see cleanupJobWorktree's `keepBranch: !integration.ok`), so a
|
|
3355
|
+
* plain re-call of integrateBranch against that same branch inherits PRD
|
|
3356
|
+
* 1125's auto-resolution for free — no re-implementation needed here.
|
|
3357
|
+
*
|
|
3358
|
+
* Bounded to exactly one attempt via job.mechanicalRecoveryAttempted,
|
|
3359
|
+
* stamped in the SAME mutate as the outcome (performMechanicalRecovery,
|
|
3360
|
+
* below) — never here — so this selector alone can be unit-tested exactly
|
|
3361
|
+
* like selectResumeRecoveryTarget/selectLeftoverQuarantineTarget.
|
|
3362
|
+
*
|
|
3363
|
+
* Kill-switch: SM_MECHANICAL_RECOVERY_DISABLE=1 restores today's behaviour
|
|
3364
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3365
|
+
*/
|
|
3366
|
+
const MECHANICALLY_RESOLVABLE_VERDICTS = new Set(['worktree_integration_failed']);
|
|
3367
|
+
|
|
3368
|
+
function selectMechanicalRecoveryTarget(job) {
|
|
3369
|
+
if (process.env.SM_MECHANICAL_RECOVERY_DISABLE === '1') return null;
|
|
3370
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3371
|
+
if (!MECHANICALLY_RESOLVABLE_VERDICTS.has(job.verifierVerdict)) return null;
|
|
3372
|
+
if (job.mechanicalRecoveryAttempted === true) return null;
|
|
3373
|
+
const cwd = job.cwd || DEFAULT_PROJECT_CWD;
|
|
3374
|
+
return { slug: job.slug, cwd, branch: jobWorktree.branchNameFor(job.slug), carriedPaths: job.carriedPaths || [] };
|
|
3375
|
+
}
|
|
3376
|
+
|
|
3377
|
+
/**
|
|
3378
|
+
* Perform an already-selected mechanical recovery (selectMechanicalRecoveryTarget
|
|
3379
|
+
* above) — a direct re-attempt of integrateBranch against the job's preserved
|
|
3380
|
+
* branch, never a fresh `claude -p` dispatch. On success the job transitions
|
|
3381
|
+
* needs_review -> completed and its verifierVerdict is cleared; the branch,
|
|
3382
|
+
* now merged, is deleted like any other successfully-integrated job branch.
|
|
3383
|
+
* On failure (including a branch that no longer exists — already deleted or
|
|
3384
|
+
* already merged) the job stays needs_review, mechanicalRecoveryAttempted is
|
|
3385
|
+
* stamped, and the retry's own failure text is appended to `error`. Either
|
|
3386
|
+
* way mechanicalRecoveryAttempted is stamped in this SAME mutate, so a crash
|
|
3387
|
+
* between the git call returning and this mutate landing simply repeats an
|
|
3388
|
+
* idempotent git operation on the next pass rather than leaving the job
|
|
3389
|
+
* re-eligible forever.
|
|
3390
|
+
*/
|
|
3391
|
+
async function performMechanicalRecovery(job, target) {
|
|
3392
|
+
const integration = await jobWorktree.integrateJobBranch({
|
|
3393
|
+
cwd: target.cwd, branch: target.branch, slug: target.slug, carriedPaths: target.carriedPaths,
|
|
3394
|
+
});
|
|
3395
|
+
if (integration.ok) {
|
|
3396
|
+
await jobWorktree.cleanupJobWorktree({ cwd: target.cwd, dir: undefined, branch: target.branch, keepBranch: false });
|
|
3397
|
+
}
|
|
3398
|
+
let becameCompleted = false;
|
|
3399
|
+
await mutate((s) => {
|
|
3400
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3401
|
+
if (!j) return;
|
|
3402
|
+
j.mechanicalRecoveryAttempted = true;
|
|
3403
|
+
if (integration.ok) {
|
|
3404
|
+
if (transitionJob(j, 'completed', {
|
|
3405
|
+
reason: `mechanical recovery: ${target.branch} re-integrated successfully`,
|
|
3406
|
+
source: 'scheduler:mechanicalRecovery',
|
|
3407
|
+
})) {
|
|
3408
|
+
delete j.verifierVerdict;
|
|
3409
|
+
j.exitCode = 0;
|
|
3410
|
+
j.error = null;
|
|
3411
|
+
becameCompleted = true;
|
|
3412
|
+
}
|
|
3413
|
+
} else {
|
|
3414
|
+
const pointer = `Mechanical recovery retry failed: ${integration.reason}`;
|
|
3415
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3416
|
+
}
|
|
3417
|
+
});
|
|
3418
|
+
if (integration.ok) {
|
|
3419
|
+
console.log(`[scheduler] mechanical-recovery: ${job.slug} → completed (branch ${target.branch} re-integrated)`);
|
|
3420
|
+
if (becameCompleted) await archiveCompletedPrd(job.slug, job.cwd);
|
|
3421
|
+
} else {
|
|
3422
|
+
console.error(`[scheduler] mechanical-recovery: ${job.slug} → retry failed: ${integration.reason}`);
|
|
3423
|
+
}
|
|
3424
|
+
}
|
|
3425
|
+
|
|
3426
|
+
/**
|
|
3427
|
+
* Leftover quarantine (PRD 1128). Resume-first recovery gets exactly one
|
|
3428
|
+
* `--resume` attempt (selectResumeRecoveryTarget above); when that attempt
|
|
3429
|
+
* ALSO parks needs_review with 'uncommitted_changes', the leftovers are
|
|
3430
|
+
* about to sit dirty in the SHARED tree forever — git then refuses any later
|
|
3431
|
+
* worktree merge for this cwd that would overwrite them, turning one parked
|
|
3432
|
+
* job into a project-wide stall (216-jupiter-sand-kazekage, 2026-09-06).
|
|
3433
|
+
* Pure/no I/O, mirroring selectResumeRecoveryTarget so the eligibility rule
|
|
3434
|
+
* is unit-testable directly.
|
|
3435
|
+
*
|
|
3436
|
+
* Bounded to exactly one attempt via job.leftoverQuarantineAttempted, stamped
|
|
3437
|
+
* synchronously by the caller in the SAME mutate as this decision (never
|
|
3438
|
+
* here) — see spawnJob's finalize and reverifyNeedsReview's periodic pass.
|
|
3439
|
+
*
|
|
3440
|
+
* Kill-switch: SM_LEFTOVER_QUARANTINE_DISABLE=1 restores today's behaviour
|
|
3441
|
+
* exactly (always returns null), mirroring SM_RESUME_RECOVERY_DISABLE.
|
|
3442
|
+
*/
|
|
3443
|
+
function selectLeftoverQuarantineTarget(job) {
|
|
3444
|
+
if (process.env.SM_LEFTOVER_QUARANTINE_DISABLE === '1') return null;
|
|
3445
|
+
if (!job || job.status !== 'needs_review') return null;
|
|
3446
|
+
if (job.verifierVerdict !== 'uncommitted_changes') return null;
|
|
3447
|
+
if (job.resumeRecoveryAttempted !== true) return null;
|
|
3448
|
+
if (job.leftoverQuarantineAttempted === true) return null;
|
|
3449
|
+
const uncommittedPaths = Array.isArray(job.uncommittedPaths)
|
|
3450
|
+
? job.uncommittedPaths.filter((p) => typeof p === 'string' && p.length > 0)
|
|
3451
|
+
: [];
|
|
3452
|
+
if (!uncommittedPaths.length) return null;
|
|
3453
|
+
// The single most important constraint: never touch a path that was
|
|
3454
|
+
// ALREADY dirty at this run's own dispatch time (preRunDirtyPaths) — that
|
|
3455
|
+
// is foreign WIP (a human's or a sibling's), not this job's own leftover.
|
|
3456
|
+
const preRunDirty = new Set(Array.isArray(job.preRunDirtyPaths) ? job.preRunDirtyPaths : []);
|
|
3457
|
+
const paths = uncommittedPaths.filter((p) => !preRunDirty.has(p));
|
|
3458
|
+
if (!paths.length) return null;
|
|
3459
|
+
return { slug: job.slug, cwd: job.cwd, paths };
|
|
3460
|
+
}
|
|
3461
|
+
|
|
3462
|
+
function execGitAt(cwd, args, { env, timeout = 20_000 } = {}) {
|
|
3463
|
+
return new Promise((resolve, reject) => {
|
|
3464
|
+
execFile(
|
|
3465
|
+
'git',
|
|
3466
|
+
['-C', cwd, ...args],
|
|
3467
|
+
{ timeout, windowsHide: true, encoding: 'utf8', env: env ? { ...process.env, ...env } : process.env },
|
|
3468
|
+
(err, stdout, stderr) => {
|
|
3469
|
+
if (err) {
|
|
3470
|
+
err.stderrText = stderr;
|
|
3471
|
+
reject(err);
|
|
3472
|
+
return;
|
|
3473
|
+
}
|
|
3474
|
+
resolve(stdout || '');
|
|
3475
|
+
},
|
|
3476
|
+
);
|
|
3477
|
+
});
|
|
3478
|
+
}
|
|
3479
|
+
|
|
3480
|
+
async function pathExistsInTree(cwd, treeish, p) {
|
|
3481
|
+
try {
|
|
3482
|
+
await execGitAt(cwd, ['cat-file', '-e', `${treeish}:${p}`]);
|
|
3483
|
+
return true;
|
|
3484
|
+
} catch {
|
|
3485
|
+
return false;
|
|
3486
|
+
}
|
|
3487
|
+
}
|
|
3488
|
+
|
|
3489
|
+
/**
|
|
3490
|
+
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
3491
|
+
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
3492
|
+
* unavailable) via a THROWAWAY `GIT_INDEX_FILE` — never touches the live
|
|
3493
|
+
* index, never moves the checked-out branch — then restores those paths to
|
|
3494
|
+
* match that baseline commit's tree, so the shared working tree returns to
|
|
3495
|
+
* its pre-run state. This is deliberately NOT `git stash` (the destructive-
|
|
3496
|
+
* git guard blocks stash on a shared tree, and a stash nobody restores
|
|
3497
|
+
* strands the work invisibly — see standards.md).
|
|
3498
|
+
*
|
|
3499
|
+
* Never throws: any git failure, or a non-git cwd, aborts the WHOLE attempt
|
|
3500
|
+
* with the tree untouched (no partial restore) — restore only ever runs
|
|
3501
|
+
* after the salvage ref/commit has safely landed, so a failure there leaves
|
|
3502
|
+
* the data recoverable from the ref even though the tree stayed dirty.
|
|
3503
|
+
* A path no longer dirty on disk (already committed, or reverted since) is
|
|
3504
|
+
* skipped, never force-restored.
|
|
3505
|
+
*/
|
|
3506
|
+
async function quarantineLeftovers({ cwd, slug, paths, headBefore }) {
|
|
3507
|
+
if (!cwd || !slug || !Array.isArray(paths) || paths.length === 0) {
|
|
3508
|
+
return { ok: false, reason: 'no cwd/slug/paths given' };
|
|
3509
|
+
}
|
|
3510
|
+
let baseline = headBefore || null;
|
|
3511
|
+
try {
|
|
3512
|
+
if (!baseline) {
|
|
3513
|
+
baseline = (await execGitAt(cwd, ['rev-parse', 'HEAD'])).trim();
|
|
3514
|
+
}
|
|
3515
|
+
if (!baseline) return { ok: false, reason: 'could not resolve a baseline commit (non-git cwd?)' };
|
|
3516
|
+
|
|
3517
|
+
const dirtyNowRaw = await execGitAt(cwd, ['status', '--porcelain', '--', ...paths]);
|
|
3518
|
+
const dirtyNow = new Set(parsePorcelain(dirtyNowRaw));
|
|
3519
|
+
const toQuarantine = paths.filter((p) => dirtyNow.has(p));
|
|
3520
|
+
const skippedPaths = paths.filter((p) => !dirtyNow.has(p));
|
|
3521
|
+
if (!toQuarantine.length) {
|
|
3522
|
+
return { ok: true, ref: null, commit: null, quarantinedPaths: [], skippedPaths };
|
|
3523
|
+
}
|
|
3524
|
+
|
|
3525
|
+
const tmpIndex = path.join(os.tmpdir(), `sm-salvage-index-${slug}-${process.pid}-${Date.now()}`);
|
|
3526
|
+
const env = { GIT_INDEX_FILE: tmpIndex };
|
|
3527
|
+
let treeSha;
|
|
3528
|
+
let commitSha;
|
|
3529
|
+
try {
|
|
3530
|
+
await execGitAt(cwd, ['read-tree', baseline], { env });
|
|
3531
|
+
for (const p of toQuarantine) {
|
|
3532
|
+
if (fs.existsSync(path.join(cwd, p))) {
|
|
3533
|
+
await execGitAt(cwd, ['add', '--', p], { env });
|
|
3534
|
+
} else {
|
|
3535
|
+
await execGitAt(cwd, ['rm', '--cached', '--ignore-unmatch', '--', p], { env });
|
|
3536
|
+
}
|
|
3537
|
+
}
|
|
3538
|
+
treeSha = (await execGitAt(cwd, ['write-tree'], { env })).trim();
|
|
3539
|
+
commitSha = (await execGitAt(cwd, ['commit-tree', treeSha, '-p', baseline, '-m', `salvage: leftover changes from ${slug}`], { env })).trim();
|
|
3540
|
+
} catch (e) {
|
|
3541
|
+
return { ok: false, reason: `git command failed while building the salvage commit: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3542
|
+
} finally {
|
|
3543
|
+
await fsp.rm(tmpIndex, { force: true }).catch(() => {});
|
|
3544
|
+
}
|
|
3545
|
+
|
|
3546
|
+
const ref = `sm-salvage/${slug}`;
|
|
3547
|
+
try {
|
|
3548
|
+
await execGitAt(cwd, ['update-ref', `refs/heads/${ref}`, commitSha]);
|
|
3549
|
+
} catch (e) {
|
|
3550
|
+
return { ok: false, reason: `git command failed updating ${ref}: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3551
|
+
}
|
|
3552
|
+
|
|
3553
|
+
// The salvage commit is safely landed at this point — a failure from here
|
|
3554
|
+
// on is reported with the ref/commit still attached so nothing looks lost
|
|
3555
|
+
// even if the tree itself couldn't be fully restored.
|
|
3556
|
+
try {
|
|
3557
|
+
const inBaseline = [];
|
|
3558
|
+
const notInBaseline = [];
|
|
3559
|
+
for (const p of toQuarantine) {
|
|
3560
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3561
|
+
if (await pathExistsInTree(cwd, baseline, p)) inBaseline.push(p); else notInBaseline.push(p);
|
|
3562
|
+
}
|
|
3563
|
+
if (inBaseline.length) {
|
|
3564
|
+
await execGitAt(cwd, ['checkout', baseline, '--', ...inBaseline]);
|
|
3565
|
+
}
|
|
3566
|
+
if (notInBaseline.length) {
|
|
3567
|
+
await execGitAt(cwd, ['reset', '--', ...notInBaseline]).catch(() => {});
|
|
3568
|
+
for (const p of notInBaseline) {
|
|
3569
|
+
// eslint-disable-next-line no-await-in-loop
|
|
3570
|
+
await fsp.rm(path.join(cwd, p), { force: true });
|
|
3571
|
+
}
|
|
3572
|
+
}
|
|
3573
|
+
} catch (e) {
|
|
3574
|
+
return {
|
|
3575
|
+
ok: false,
|
|
3576
|
+
ref,
|
|
3577
|
+
commit: commitSha,
|
|
3578
|
+
reason: `salvage commit landed at ${ref} (${commitSha}) but restoring the working tree failed: ${(e && (e.stderrText || e.message)) || e}`,
|
|
3579
|
+
};
|
|
3580
|
+
}
|
|
3581
|
+
|
|
3582
|
+
return { ok: true, ref, commit: commitSha, quarantinedPaths: toQuarantine, skippedPaths };
|
|
3583
|
+
} catch (e) {
|
|
3584
|
+
return { ok: false, reason: `git command failed: ${(e && (e.stderrText || e.message)) || e}` };
|
|
3585
|
+
}
|
|
3586
|
+
}
|
|
3587
|
+
|
|
3588
|
+
/**
|
|
3589
|
+
* Perform an already-selected quarantine (job.leftoverQuarantineAttempted
|
|
3590
|
+
* must already be true, stamped by the caller) and persist the outcome onto
|
|
3591
|
+
* the job row: `quarantinedTo`/`quarantinedCommit`/`quarantinedPaths` on
|
|
3592
|
+
* success, plus a one-line pointer appended to `error` naming the ref so a
|
|
3593
|
+
* human can recover with a single named command
|
|
3594
|
+
* (`git show sm-salvage/<slug>`). The belt-and-braces `salvagePatch` (when
|
|
3595
|
+
* present) is referenced alongside it, never removed. On failure, only a
|
|
3596
|
+
* diagnostic is appended — the job row's dirt-describing fields are left as
|
|
3597
|
+
* they were, since the tree itself was left untouched (or, for a
|
|
3598
|
+
* restore-only failure, the salvage ref is still named in the note).
|
|
3599
|
+
*
|
|
3600
|
+
* `headBefore`, when the caller has it fresh (spawnJob's own finalize still
|
|
3601
|
+
* has the local `guardHeadBefore` in scope for the run that just parked —
|
|
3602
|
+
* the same value is deleted off the job ROW earlier in that same finalize),
|
|
3603
|
+
* is used as the salvage ref's baseline commit; otherwise (the periodic
|
|
3604
|
+
* reverifyNeedsReview pass, re-discovering an already-parked row) this falls
|
|
3605
|
+
* back to the current HEAD inside quarantineLeftovers itself.
|
|
3606
|
+
*/
|
|
3607
|
+
async function performLeftoverQuarantine(job, paths, headBefore = null) {
|
|
3608
|
+
const result = await quarantineLeftovers({
|
|
3609
|
+
cwd: job.cwd || DEFAULT_PROJECT_CWD,
|
|
3610
|
+
slug: job.slug,
|
|
3611
|
+
paths,
|
|
3612
|
+
headBefore: headBefore || job.guardHeadBefore || null,
|
|
3613
|
+
});
|
|
3614
|
+
await mutate((s) => {
|
|
3615
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
3616
|
+
if (!j) return;
|
|
3617
|
+
if (result.ok && Array.isArray(result.quarantinedPaths) && result.quarantinedPaths.length) {
|
|
3618
|
+
j.quarantinedTo = result.ref;
|
|
3619
|
+
j.quarantinedCommit = result.commit;
|
|
3620
|
+
j.quarantinedPaths = capDirtyPaths(result.quarantinedPaths);
|
|
3621
|
+
const salvageNote = j.salvagePatch ? `; salvage patch also at ${j.salvagePatch}` : '';
|
|
3622
|
+
const pointer = `Leftovers quarantined to ${result.ref} (commit ${result.commit}) — recover via \`git show ${result.ref}\`${salvageNote}`;
|
|
3623
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3624
|
+
console.log(`[scheduler] ${job.slug}: quarantined ${result.quarantinedPaths.length} leftover path(s) to ${result.ref} (${result.commit})`);
|
|
3625
|
+
} else if (!result.ok) {
|
|
3626
|
+
const pointer = `Leftover quarantine failed: ${result.reason}`;
|
|
3627
|
+
j.error = j.error ? `${j.error}\n${pointer}` : pointer;
|
|
3628
|
+
console.error(`[scheduler] ${job.slug}: leftover quarantine failed: ${result.reason}`);
|
|
3629
|
+
}
|
|
3630
|
+
});
|
|
3631
|
+
}
|
|
3632
|
+
|
|
3175
3633
|
/**
|
|
3176
3634
|
* Pure argv builder for a `claude -p` child spawn, shared so the
|
|
3177
3635
|
* resume-vs-fresh-session choice is made in exactly one place. `resume`
|
|
@@ -3197,10 +3655,17 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
|
|
|
3197
3655
|
|
|
3198
3656
|
// ---------- execution ----------
|
|
3199
3657
|
|
|
3658
|
+
// Allocates the runId/dir pair at dispatch time WITHOUT creating the
|
|
3659
|
+
// directory — a dispatch that aborts inside spawnJob before executeJob's
|
|
3660
|
+
// openLog() call (slot-acquire miss, worktree-cap deferral, launch-gate
|
|
3661
|
+
// block, ...) must leave no trace on disk. The directory is materialised
|
|
3662
|
+
// lazily, the first time something actually needs to write into it (see
|
|
3663
|
+
// openLog's mkdirSync in executeJob below). Because tickQueue hands ONE
|
|
3664
|
+
// shared batch dir to every spawnJob in the batch, several jobs may race to
|
|
3665
|
+
// create it — `recursive: true` makes that race safe.
|
|
3200
3666
|
function pickRunDir() {
|
|
3201
3667
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
3202
3668
|
const dir = path.join(RUNS_DIR, ts);
|
|
3203
|
-
fs.mkdirSync(dir, { recursive: true });
|
|
3204
3669
|
return { runId: ts, dir };
|
|
3205
3670
|
}
|
|
3206
3671
|
|
|
@@ -3228,6 +3693,12 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
3228
3693
|
// fresh one via `--session-id` is the entire point of the recovery.
|
|
3229
3694
|
const sessionId = resumeTarget ? resumeTarget.sessionId : randomUUID();
|
|
3230
3695
|
|
|
3696
|
+
// Materialise the (possibly shared-batch) run dir lazily, right before the
|
|
3697
|
+
// first write into it — see pickRunDir's comment for why this is deferred
|
|
3698
|
+
// this far. recursive:true makes it safe if a sibling job in the same
|
|
3699
|
+
// batch dir already created it.
|
|
3700
|
+
fs.mkdirSync(runDir, { recursive: true });
|
|
3701
|
+
|
|
3231
3702
|
// Phase 1: open log fd so we can emit pre-spawn diagnostics (early-exit
|
|
3232
3703
|
// error paths) before the child is created. withChildAndLog takes ownership
|
|
3233
3704
|
// of fd/safeLog/closeFd from the point it is called.
|
|
@@ -3638,13 +4109,9 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
3638
4109
|
// project (keyed by cwd) so jobs in different repos run concurrently up to
|
|
3639
4110
|
// the cap; within one project, sequential-group semantics are preserved.
|
|
3640
4111
|
|
|
3641
|
-
|
|
3642
|
-
|
|
3643
|
-
|
|
3644
|
-
*/
|
|
3645
|
-
function isFixPlanSlug(slug) {
|
|
3646
|
-
return /^\d+-fix-/.test(slug);
|
|
3647
|
-
}
|
|
4112
|
+
// isFixPlanSlug/classifyDiscoveredFixPlan/resolveIsFixPlan now live in
|
|
4113
|
+
// lib/fixPlanSlug.cjs (PRD 1131) — see that module's header for why slug
|
|
4114
|
+
// shape alone is no longer sufficient to classify a fix plan.
|
|
3648
4115
|
|
|
3649
4116
|
/**
|
|
3650
4117
|
* The fix-plan slug spawnInvestigation authors for a given failed job —
|
|
@@ -3683,7 +4150,25 @@ function healTargetForFix(fixSlug, jobs) {
|
|
|
3683
4150
|
* unit-tested (no spawn, no fs). Inputs are the already-resolved values that
|
|
3684
4151
|
* spawnInvestigation computes.
|
|
3685
4152
|
*/
|
|
3686
|
-
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group }) {
|
|
4153
|
+
function buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild = null }) {
|
|
4154
|
+
const deadFixChildNote = deadChild ? `
|
|
4155
|
+
|
|
4156
|
+
# This is a REOPENED investigation — your own prior fix plan died
|
|
4157
|
+
You already investigated this job once and produced a fix-plan PRD, \`${deadChild.slug}\`, which was
|
|
4158
|
+
supposed to heal it. That fix-plan job itself reached a terminal, non-completed status
|
|
4159
|
+
(\`${deadChild.status}\`) without ever fixing the original failure — so the parent job you are now
|
|
4160
|
+
investigating is stuck again with nothing left to retry it automatically. This is the ONE reopen
|
|
4161
|
+
this parent gets; do not cold-read the log and re-derive the plan that already failed.
|
|
4162
|
+
|
|
4163
|
+
Dead fix-plan child's own outcome:
|
|
4164
|
+
- Slug: ${deadChild.slug}
|
|
4165
|
+
- Status: ${deadChild.status}
|
|
4166
|
+
- Verifier verdict: ${deadChild.verifierVerdict ?? '(none recorded)'}
|
|
4167
|
+
- Error: ${deadChild.error ?? '(none recorded)'}
|
|
4168
|
+
|
|
4169
|
+
Read why THAT job died (its own run log, if any, under the runs directory) before writing a new
|
|
4170
|
+
fix-plan PRD, and make sure your new plan is genuinely different from — not a repeat of — whatever
|
|
4171
|
+
that dead child attempted.` : '';
|
|
3687
4172
|
const abandonedBackgroundTaskNote = failedJob.verifierVerdict === 'abandoned_background_task' ? `
|
|
3688
4173
|
|
|
3689
4174
|
# Known failure class: abandoned background task
|
|
@@ -3704,7 +4189,7 @@ The fix-plan PRD you write for this MUST instruct its executor to, in order:
|
|
|
3704
4189
|
1. Check for a salvage patch (named \`<slug>.uncommitted.patch\` in the run directory${failedJob.salvagePatch ? `, e.g. \`${failedJob.salvagePatch}\`` : ''}) and, if found, apply it to the working tree BEFORE inspecting \`git status\`/\`git diff\` in ${cwd} for uncommitted changes matching the original PRD's acceptance criteria.
|
|
3705
4190
|
2. If the work is present (via the applied patch or already in the tree) and satisfies the acceptance criteria, run the project's verify commands and COMMIT it — do not re-implement or re-plan the PRD from scratch.
|
|
3706
4191
|
3. Only fall back to re-implementing whatever acceptance criteria are genuinely missing after applying any salvage patch, not the whole PRD.` : '';
|
|
3707
|
-
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${abandonedBackgroundTaskNote}
|
|
4192
|
+
return `You are investigating a failed scheduled job in the session-manager queue. Your ONLY job is to write a fix-plan PRD file. Do NOT attempt the fix yourself.${deadFixChildNote}${abandonedBackgroundTaskNote}
|
|
3708
4193
|
|
|
3709
4194
|
# Failed job
|
|
3710
4195
|
- Slug: ${failedJob.slug}
|
|
@@ -3749,8 +4234,13 @@ ${logTail}
|
|
|
3749
4234
|
cwd: ${cwd}
|
|
3750
4235
|
parallelGroup: ${group}
|
|
3751
4236
|
estimateMinutes: <your time estimate>
|
|
4237
|
+
isFixPlan: true
|
|
3752
4238
|
---
|
|
3753
4239
|
\`\`\`
|
|
4240
|
+
\`isFixPlan: true\` is REQUIRED — it is the scheduler's provenance signal that this PRD is a
|
|
4241
|
+
genuine auto-authored fix plan (not a human/agent PRD whose slug merely happens to start with
|
|
4242
|
+
"fix-"); omitting it means this fix plan will not get its depth-cap/zero-edit-commit-guard
|
|
4243
|
+
exemptions.
|
|
3754
4244
|
\`cwd\` must be the git repo root where the fix will actually land. If the failed job's cwd is
|
|
3755
4245
|
not that repo (e.g. a scratch dir like \`/tmp\`), set \`cwd:\` to the correct repo root instead —
|
|
3756
4246
|
the scheduler's commit guard and post-run verifier read git state from this path, and a
|
|
@@ -3824,7 +4314,7 @@ function readRunOutcomeSidecars(runDir, slug) {
|
|
|
3824
4314
|
*/
|
|
3825
4315
|
const INVESTIGATION_LAUNCH_KEY = 'investigation';
|
|
3826
4316
|
|
|
3827
|
-
async function spawnInvestigation(failedJob, runDir) {
|
|
4317
|
+
async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {}) {
|
|
3828
4318
|
// The probe launches with the same CLI as the job it diagnoses. While
|
|
3829
4319
|
// that CLI cannot launch at all (launch circuit breaker, issue #11 list
|
|
3830
4320
|
// B1: probes e4f82da2/d374e6bf died on the same HTTP 400 as the runs
|
|
@@ -3853,7 +4343,14 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3853
4343
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is resume-recovery eligible`);
|
|
3854
4344
|
return { deferred: false };
|
|
3855
4345
|
}
|
|
3856
|
-
|
|
4346
|
+
// Mechanical recovery (PRD 1130): same first-refusal treatment — a job
|
|
4347
|
+
// eligible for a pure-git retry must never also get a cold-read fix-plan
|
|
4348
|
+
// PRD authored in the same pass.
|
|
4349
|
+
if (selectMechanicalRecoveryTarget(failedJob)) {
|
|
4350
|
+
console.log(`[scheduler] skip investigation: ${failedJob.slug} is mechanical-recovery eligible`);
|
|
4351
|
+
return { deferred: false };
|
|
4352
|
+
}
|
|
4353
|
+
if (isFixPlanBeyondDepthCap(failedJob.slug, failedJob.investigationDepth, failedJob.isFixPlan)) {
|
|
3857
4354
|
console.log(`[scheduler] skip investigation: ${failedJob.slug} is a fix plan at/beyond depth cap (depth=${failedJob.investigationDepth ?? 'none'})`);
|
|
3858
4355
|
return { deferred: false };
|
|
3859
4356
|
}
|
|
@@ -3910,7 +4407,13 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3910
4407
|
|
|
3911
4408
|
const logTail = readTail(failedLogPath, 16 * 1024) || '(failed to read log)';
|
|
3912
4409
|
|
|
3913
|
-
|
|
4410
|
+
// A dead-fix-plan reopen (PRD 1129) targets the SAME fixPath its dead
|
|
4411
|
+
// child was originally authored at, by construction (fixSlugFor is a pure
|
|
4412
|
+
// function of the parent) — the file existing is not staleness here, it's
|
|
4413
|
+
// the whole reason a reopen was offered. Skip the guard in that one case
|
|
4414
|
+
// so the second investigation can overwrite the dead plan; every other
|
|
4415
|
+
// caller keeps the original protection against clobbering a live sibling.
|
|
4416
|
+
if (fs.existsSync(fixPath) && !deadChild) {
|
|
3914
4417
|
console.log(`[scheduler] skip investigation: fix plan already exists at ${fixPath}`);
|
|
3915
4418
|
releaseSlot();
|
|
3916
4419
|
return { deferred: false };
|
|
@@ -3936,7 +4439,7 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
3936
4439
|
console.warn(`[scheduler] investigation cwd is not a git repo (${cwd}); falling back to ${DEFAULT_PROJECT_CWD}`);
|
|
3937
4440
|
cwd = DEFAULT_PROJECT_CWD;
|
|
3938
4441
|
}
|
|
3939
|
-
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group });
|
|
4442
|
+
const prompt = buildInvestigationPrompt({ failedJob, cwd, failedLogPath, originalBody, logTail, fixPath, group, deadChild });
|
|
3940
4443
|
|
|
3941
4444
|
// Phase 1: open log fd for pre-spawn diagnostics.
|
|
3942
4445
|
const { fd, safeLog, closeFd } = openLog(investigationLogPath);
|
|
@@ -4065,6 +4568,22 @@ async function spawnInvestigation(failedJob, runDir) {
|
|
|
4065
4568
|
mutate((s) => {
|
|
4066
4569
|
const j = s.jobs.find((x) => x.slug === failedJob.slug);
|
|
4067
4570
|
if (j) j.autoFixOutcome = 'plan';
|
|
4571
|
+
// Dead-fix-plan reopen (PRD 1129): fixSlugFor is a pure function of
|
|
4572
|
+
// the parent, so the freshly-authored plan landed at the SAME slug
|
|
4573
|
+
// as the dead child — reconcile() sees an already-known slug and
|
|
4574
|
+
// will never re-mint a pending row for it. Explicitly reset the
|
|
4575
|
+
// dead child's own row here so the overwritten plan actually gets
|
|
4576
|
+
// a chance to run, rather than sitting inert behind a permanently
|
|
4577
|
+
// terminal queue row. force:true because 'skipped' (a valid dead
|
|
4578
|
+
// status here) is otherwise reset-refused by design.
|
|
4579
|
+
if (deadChild) {
|
|
4580
|
+
const child = s.jobs.find((x) => x.slug === deadChild.slug);
|
|
4581
|
+
if (child) {
|
|
4582
|
+
resetJobFields(child, 'reset by dead-fix-plan reopen: parent investigation authored a new plan', {
|
|
4583
|
+
force: true, source: 'spawnInvestigation:dead-fix-plan-reopen',
|
|
4584
|
+
});
|
|
4585
|
+
}
|
|
4586
|
+
}
|
|
4068
4587
|
}).catch(() => {});
|
|
4069
4588
|
} else {
|
|
4070
4589
|
console.log(`[scheduler] investigation finished WITHOUT producing fix plan (slug=${failedJob.slug}, code=${exitCode})`);
|
|
@@ -4147,6 +4666,58 @@ async function computeLaunchHolds(state, { now = Date.now(), claudeVersion } = {
|
|
|
4147
4666
|
return held;
|
|
4148
4667
|
}
|
|
4149
4668
|
|
|
4669
|
+
/**
|
|
4670
|
+
* computeDepHistorySatisfaction(state) → Map<cwd, Set<string>|symbol>
|
|
4671
|
+
*
|
|
4672
|
+
* PRD 1122's once-per-tick dependsOn history/archive lookup: for every
|
|
4673
|
+
* distinct project cwd with jobs this tick, builds the set of dep slugs that
|
|
4674
|
+
* have no live queue row but are nonetheless known-satisfied — a completed
|
|
4675
|
+
* record in that project's own `state/history.jsonl` shard
|
|
4676
|
+
* (queueHistory.completedSlugsForCwd, scoped per-project so a same-named PRD
|
|
4677
|
+
* in an unrelated project can never satisfy a dep here), or a `.md` file
|
|
4678
|
+
* under any of that project's `prds-archived/` dirs (listArchivedPrdDirs —
|
|
4679
|
+
* covers both the retired flat layout and every Epic's own sibling archive).
|
|
4680
|
+
* findBlockingDep (schedulerBatch.cjs) treats a dep slug as blocking
|
|
4681
|
+
* whenever it has no live row AND is absent from this set, so a typo or a
|
|
4682
|
+
* double-prefixed slug (the exact 2026-09-06 starry-night-ships incident)
|
|
4683
|
+
* HOLDS its dependent instead of silently dispatching it.
|
|
4684
|
+
*
|
|
4685
|
+
* Fails OPEN per project, never queue-wide: a history-shard or archive-scan
|
|
4686
|
+
* read error for one cwd degrades that cwd's value to
|
|
4687
|
+
* `DEP_HISTORY_FAIL_OPEN` (findBlockingDep then treats every rowless dep in
|
|
4688
|
+
* that project as satisfied, exactly today's pre-1122 behaviour) with a
|
|
4689
|
+
* logged warning — it never throws out of this function and never blocks
|
|
4690
|
+
* every OTHER project's dispatch for one project's bad fs state.
|
|
4691
|
+
*
|
|
4692
|
+
* Computed ONCE here, before pickNextBatch runs, and threaded down as pure
|
|
4693
|
+
* data (quietOpts.satisfiedSlugsByCwd) — schedulerBatch.cjs itself does no
|
|
4694
|
+
* I/O, so this is the only fs read this gate costs per tick, not one per job
|
|
4695
|
+
* per dep.
|
|
4696
|
+
*/
|
|
4697
|
+
async function computeDepHistorySatisfaction(state) {
|
|
4698
|
+
const byCwd = new Map();
|
|
4699
|
+
const cwds = new Set((state?.jobs || []).map((j) => j.cwd || DEFAULT_PROJECT_CWD));
|
|
4700
|
+
for (const cwd of cwds) {
|
|
4701
|
+
const satisfied = new Set();
|
|
4702
|
+
try {
|
|
4703
|
+
for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
|
|
4704
|
+
for (const dir of listArchivedPrdDirs(cwd)) {
|
|
4705
|
+
let entries;
|
|
4706
|
+
try { entries = await fsp.readdir(dir); } catch { continue; }
|
|
4707
|
+
for (const name of entries) {
|
|
4708
|
+
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
4709
|
+
}
|
|
4710
|
+
}
|
|
4711
|
+
} catch (e) {
|
|
4712
|
+
console.warn(`[scheduler] depHistorySatisfaction: history/archive lookup failed for ${cwd} (${e?.message}) — falling back to fail-open dep resolution for this project this tick`);
|
|
4713
|
+
byCwd.set(cwd, DEP_HISTORY_FAIL_OPEN);
|
|
4714
|
+
continue;
|
|
4715
|
+
}
|
|
4716
|
+
byCwd.set(cwd, satisfied);
|
|
4717
|
+
}
|
|
4718
|
+
return byCwd;
|
|
4719
|
+
}
|
|
4720
|
+
|
|
4150
4721
|
/**
|
|
4151
4722
|
* A run that never got a turn (res.launchFailure — see executeJob's onExit)
|
|
4152
4723
|
* is routed here instead of the failed/investigation path (issue #11 lists
|
|
@@ -4317,6 +4888,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4317
4888
|
console.log(`[scheduler] ${job.slug}: dispatching as launch probe for '${launchKey}'${launchEnv ? ` with mitigation ${JSON.stringify(launchEnv)}` : ''}`);
|
|
4318
4889
|
}
|
|
4319
4890
|
|
|
4891
|
+
// Captured here (not read back off `job`, a pre-dispatch snapshot that
|
|
4892
|
+
// mutate()'s fresh-from-disk read never touches) so the rate-limited
|
|
4893
|
+
// branch below has this run's OWN start time — the freshness check
|
|
4894
|
+
// (isCooldownSuppressed) needs to know whether this specific dispatch
|
|
4895
|
+
// started after the manual clear, not whatever startedAt this row
|
|
4896
|
+
// carried from a prior run.
|
|
4897
|
+
let dispatchStartedAtMs = null;
|
|
4320
4898
|
await mutate((s) => {
|
|
4321
4899
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
4322
4900
|
if (idx >= 0) {
|
|
@@ -4327,6 +4905,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4327
4905
|
delete s.jobs[idx].heldReason;
|
|
4328
4906
|
s.jobs[idx].runId = runId;
|
|
4329
4907
|
s.jobs[idx].startedAt = new Date().toISOString();
|
|
4908
|
+
dispatchStartedAtMs = Date.parse(s.jobs[idx].startedAt);
|
|
4330
4909
|
if (job.quietMachine === true) {
|
|
4331
4910
|
s.jobs[idx].quietMachine = true;
|
|
4332
4911
|
s.jobs[idx].quietLeaseDegraded = job.quietLeaseDegraded === true;
|
|
@@ -4434,6 +5013,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4434
5013
|
let res;
|
|
4435
5014
|
let worktreeLeftoverDirty = [];
|
|
4436
5015
|
let worktreeIntegrationFailure = null;
|
|
5016
|
+
// Set only when integrateJobBranch's stderr-parsing auto-resolve fired
|
|
5017
|
+
// (PRD 1125) — surfaced on the job row so the Queue UI can say the merge
|
|
5018
|
+
// self-healed rather than silently looking like an ordinary merge.
|
|
5019
|
+
let mergeAutoResolved = null;
|
|
5020
|
+
let mergeAutoResolvedPaths = null;
|
|
4437
5021
|
// A job's uncommitted-work patch, whichever isolation mode produced it —
|
|
4438
5022
|
// set by EITHER branch below, never both (worktree.ok picks exactly one
|
|
4439
5023
|
// shape for the whole run). Named generically (not "worktree...") because
|
|
@@ -4481,6 +5065,11 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4481
5065
|
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
4482
5066
|
} else if (integration.integrated) {
|
|
4483
5067
|
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
5068
|
+
if (integration.autoResolved) {
|
|
5069
|
+
mergeAutoResolved = integration.autoResolved;
|
|
5070
|
+
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
5071
|
+
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
5072
|
+
}
|
|
4484
5073
|
}
|
|
4485
5074
|
await jobWorktree.cleanupJobWorktree({
|
|
4486
5075
|
cwd: guardCwd,
|
|
@@ -4531,12 +5120,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4531
5120
|
// (non-git cwd / git errored) — NEVER treated as "left nothing", exactly
|
|
4532
5121
|
// like every other best-effort git-state check in this function.
|
|
4533
5122
|
const afterGuardCwd = await uncommittedChanges(guardCwd);
|
|
5123
|
+
// stripAppOwnedChurn: the app writes session-manager-operations/ (queue.json,
|
|
5124
|
+
// history.jsonl, active-index.json, transcripts) DURING this job's own guard
|
|
5125
|
+
// window, so those land in the delta and get blamed on the job. A job can
|
|
5126
|
+
// never be responsible for them — see jobDirtFilter.cjs. Applied here, at
|
|
5127
|
+
// the single place the delta is computed, so the commit guard, the
|
|
5128
|
+
// transient-retry dirty check and leftoverPaths all agree.
|
|
4534
5129
|
const newlyDirtyAll = afterGuardCwd === null
|
|
4535
5130
|
? null
|
|
4536
|
-
: [...new Set([
|
|
5131
|
+
: stripAppOwnedChurn([...new Set([
|
|
4537
5132
|
...afterGuardCwd.filter((p) => !new Set(guardBaseline || []).has(p)),
|
|
4538
5133
|
...worktreeLeftoverDirty,
|
|
4539
|
-
])];
|
|
5134
|
+
])]);
|
|
4540
5135
|
|
|
4541
5136
|
if (res.launchFailure) {
|
|
4542
5137
|
await handleLaunchFailure({ job, res, runId, runDir, launchKey, launchEnv, claudeVersion: claudeVersionNow });
|
|
@@ -4570,7 +5165,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4570
5165
|
|
|
4571
5166
|
if (res.rateLimited) {
|
|
4572
5167
|
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
4573
|
-
|
|
5168
|
+
const observedAt = dispatchStartedAtMs;
|
|
5169
|
+
const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
|
|
5170
|
+
const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs: res.durationMs });
|
|
5171
|
+
consecutiveRapidRateLimitsBySlug.set(job.slug, nextCount);
|
|
5172
|
+
const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
|
|
5173
|
+
if (forceHardPause) {
|
|
5174
|
+
console.log(`[scheduler] ${job.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each — engaging hard pause`);
|
|
5175
|
+
}
|
|
5176
|
+
await setPaused('rate_limit', resetIso, { observedAt, force: forceHardPause });
|
|
5177
|
+
} else {
|
|
5178
|
+
consecutiveRapidRateLimitsBySlug.delete(job.slug);
|
|
4574
5179
|
}
|
|
4575
5180
|
|
|
4576
5181
|
// Stale queue entry: the PRD was archived (already shipped) or is gone
|
|
@@ -4700,7 +5305,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4700
5305
|
ranInWorktree: worktree.ok,
|
|
4701
5306
|
jobSelfCommitted,
|
|
4702
5307
|
legitimateNoOp: guardIsLegitimateNoOp,
|
|
4703
|
-
isFixPlanJob:
|
|
5308
|
+
isFixPlanJob: resolveIsFixPlan(job.slug, job.isFixPlan),
|
|
4704
5309
|
verifyResult,
|
|
4705
5310
|
salvagePatch,
|
|
4706
5311
|
});
|
|
@@ -4777,9 +5382,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4777
5382
|
let failedJobSnapshot = null;
|
|
4778
5383
|
let needsInvestigationNow = false;
|
|
4779
5384
|
let investigationJobSnapshot = null;
|
|
5385
|
+
let investigationDeadChildSnapshot = null;
|
|
4780
5386
|
let needsReviewRcaSnapshot = null;
|
|
4781
5387
|
let resumeRecoveryJob = null;
|
|
4782
5388
|
let resumeRecoveryTarget = null;
|
|
5389
|
+
let quarantineJob = null;
|
|
5390
|
+
let quarantinePaths = null;
|
|
5391
|
+
let mechanicalRecoveryJob = null;
|
|
5392
|
+
let mechanicalRecoveryTarget = null;
|
|
4783
5393
|
let terminalNotifySnapshot = null;
|
|
4784
5394
|
const newlyCompletedPrds = [];
|
|
4785
5395
|
await mutate((s) => {
|
|
@@ -4874,6 +5484,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4874
5484
|
} else {
|
|
4875
5485
|
delete s.jobs[i2].uncommittedPaths;
|
|
4876
5486
|
}
|
|
5487
|
+
// Worktree merge self-healed (PRD 1125) — every blocking path was
|
|
5488
|
+
// proven byte-identical to the branch, so the duplicate was
|
|
5489
|
+
// discarded and the merge retried once, successfully. Surfaced so
|
|
5490
|
+
// the Queue UI shows a self-heal instead of an ordinary merge.
|
|
5491
|
+
if (mergeAutoResolved) {
|
|
5492
|
+
s.jobs[i2].mergeAutoResolved = mergeAutoResolved;
|
|
5493
|
+
s.jobs[i2].mergeAutoResolvedPaths = capDirtyPaths(mergeAutoResolvedPaths);
|
|
5494
|
+
} else {
|
|
5495
|
+
delete s.jobs[i2].mergeAutoResolved;
|
|
5496
|
+
delete s.jobs[i2].mergeAutoResolvedPaths;
|
|
5497
|
+
}
|
|
4877
5498
|
// Non-blocking notes (e.g. a recovered missing-dependency probe, or a
|
|
4878
5499
|
// pattern hit demoted because a materially-checkable verdict outranked
|
|
4879
5500
|
// it) — surfaced even on completed jobs so the signal isn't lost.
|
|
@@ -4924,6 +5545,18 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4924
5545
|
// takes the treatAsPending branch above and never reaches here).
|
|
4925
5546
|
needsReviewRcaSnapshot = { ...s.jobs[i2] };
|
|
4926
5547
|
|
|
5548
|
+
// Mechanical recovery (PRD 1130): evaluated FIRST, ahead of both
|
|
5549
|
+
// resume-first recovery and auto-fix — a job parked with a
|
|
5550
|
+
// mechanically-resolvable verdict (see
|
|
5551
|
+
// selectMechanicalRecoveryTarget) needs no model, no plan, and no
|
|
5552
|
+
// depth-cap check, so it must never fall through to either.
|
|
5553
|
+
// Snapshot only (no I/O inside mutate()); the actual git retry
|
|
5554
|
+
// happens outside mutate(), below.
|
|
5555
|
+
const mTarget = selectMechanicalRecoveryTarget(s.jobs[i2]);
|
|
5556
|
+
if (mTarget) {
|
|
5557
|
+
mechanicalRecoveryJob = { ...s.jobs[i2] };
|
|
5558
|
+
mechanicalRecoveryTarget = mTarget;
|
|
5559
|
+
} else {
|
|
4927
5560
|
// Resume-first recovery (PRD 1111): evaluated BEFORE the auto-fix
|
|
4928
5561
|
// eligibility check below — a job whose verdict is
|
|
4929
5562
|
// 'uncommitted_changes' with a live sessionId gets one bounded
|
|
@@ -4937,6 +5570,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4937
5570
|
resumeRecoveryJob = { ...s.jobs[i2] };
|
|
4938
5571
|
resumeRecoveryTarget = target;
|
|
4939
5572
|
} else {
|
|
5573
|
+
// Leftover quarantine (PRD 1128): resume recovery is spent
|
|
5574
|
+
// (resumeRecoveryAttempted already true) and this run STILL parked
|
|
5575
|
+
// needs_review with uncommitted_changes — the leftovers are about
|
|
5576
|
+
// to sit dirty in the shared tree forever, poisoning every later
|
|
5577
|
+
// worktree merge for this cwd. Stamp the one-attempt marker HERE,
|
|
5578
|
+
// synchronously in the same mutate as this decision (mirrors
|
|
5579
|
+
// resumeRecoveryAttempted's own stamp-before-acting rule above),
|
|
5580
|
+
// so a concurrent reverifyNeedsReview pass can never double-fire
|
|
5581
|
+
// this. The actual git work is async and runs outside mutate(),
|
|
5582
|
+
// below (performLeftoverQuarantine).
|
|
5583
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(s.jobs[i2]);
|
|
5584
|
+
if (quarantineTarget) {
|
|
5585
|
+
s.jobs[i2].leftoverQuarantineAttempted = true;
|
|
5586
|
+
quarantineJob = { ...s.jobs[i2] };
|
|
5587
|
+
quarantinePaths = quarantineTarget.paths;
|
|
5588
|
+
}
|
|
4940
5589
|
// Same-tick auto-fix (feedback 2026-07-12): rather than waiting up to
|
|
4941
5590
|
// 10 min for reverifyNeedsReview()'s periodic pass, check right here
|
|
4942
5591
|
// whether this job qualifies for auto-fix (same eligibility rule
|
|
@@ -4951,8 +5600,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4951
5600
|
isEligibleForImmediateAutoFix(s.jobs[i2], s.jobs, fixSlugExists)
|
|
4952
5601
|
) {
|
|
4953
5602
|
const isRetryAttempt = s.jobs[i2].autoFixAttempted === true;
|
|
5603
|
+
const isDeadFixPlanReopen = isFixPlanDead(s.jobs[i2], s.jobs);
|
|
5604
|
+
if (isDeadFixPlanReopen) {
|
|
5605
|
+
investigationDeadChildSnapshot = s.jobs.find((x) => x.slug === fixSlugFor(s.jobs[i2])) || null;
|
|
5606
|
+
}
|
|
4954
5607
|
s.jobs[i2].autoFixAttempted = true;
|
|
4955
5608
|
if (!s.jobs[i2].runId) s.jobs[i2].runId = runId;
|
|
5609
|
+
if (isDeadFixPlanReopen) s.jobs[i2].autoFixReopened = true;
|
|
4956
5610
|
if (isRetryAttempt) {
|
|
4957
5611
|
s.jobs[i2].autoFixRetries = (s.jobs[i2].autoFixRetries ?? 0) + 1;
|
|
4958
5612
|
delete s.jobs[i2].autoFixOutcome;
|
|
@@ -4961,13 +5615,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
4961
5615
|
investigationJobSnapshot = { ...s.jobs[i2] };
|
|
4962
5616
|
}
|
|
4963
5617
|
}
|
|
5618
|
+
}
|
|
4964
5619
|
}
|
|
4965
5620
|
// Auto-promote: when a fix-* PRD completes successfully, the original
|
|
4966
5621
|
// failed PRD's work is logically done. Flip its status to 'completed'
|
|
4967
5622
|
// so the cross-group failure gate in pickNextBatch releases. Without
|
|
4968
5623
|
// this, the queue stalls indefinitely behind a stale failure even
|
|
4969
5624
|
// though the auto-recovery did its job.
|
|
4970
|
-
if (effectiveStatus === 'completed' &&
|
|
5625
|
+
if (effectiveStatus === 'completed' && resolveIsFixPlan(job.slug, job.isFixPlan)) {
|
|
4971
5626
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
4972
5627
|
if (orig) {
|
|
4973
5628
|
const priorStatus = orig.status;
|
|
@@ -5039,6 +5694,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5039
5694
|
});
|
|
5040
5695
|
}
|
|
5041
5696
|
|
|
5697
|
+
if (mechanicalRecoveryJob && mechanicalRecoveryTarget) {
|
|
5698
|
+
console.log(`[scheduler] needs_review ${job.slug} → mechanical-recovery (re-integrating ${mechanicalRecoveryTarget.branch})`);
|
|
5699
|
+
performMechanicalRecovery(mechanicalRecoveryJob, mechanicalRecoveryTarget).catch((e) => {
|
|
5700
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
5701
|
+
});
|
|
5702
|
+
}
|
|
5703
|
+
|
|
5042
5704
|
if (resumeRecoveryJob && resumeRecoveryTarget) {
|
|
5043
5705
|
console.log(`[scheduler] needs_review ${job.slug} → resume-recovery (session ${resumeRecoveryTarget.sessionId}, ${resumeRecoveryTarget.dirtyPaths.length} dirty path(s))`);
|
|
5044
5706
|
spawnResumeRecovery(resumeRecoveryJob, resumeRecoveryTarget).catch((e) => {
|
|
@@ -5046,6 +5708,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5046
5708
|
});
|
|
5047
5709
|
}
|
|
5048
5710
|
|
|
5711
|
+
if (quarantineJob && quarantinePaths) {
|
|
5712
|
+
console.log(`[scheduler] needs_review ${job.slug} → quarantining ${quarantinePaths.length} leftover path(s) (resume recovery already spent)`);
|
|
5713
|
+
performLeftoverQuarantine(quarantineJob, quarantinePaths, guardHeadBefore).catch((e) => {
|
|
5714
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
5715
|
+
});
|
|
5716
|
+
}
|
|
5717
|
+
|
|
5049
5718
|
if (actuallyFailed && failedJobSnapshot) {
|
|
5050
5719
|
// Transient-failure detector. A 143/137 exit is ALWAYS a signal kill — the
|
|
5051
5720
|
// agent never self-exits with those — so the only question is WHO killed it.
|
|
@@ -5118,7 +5787,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5118
5787
|
}
|
|
5119
5788
|
} else if (needsInvestigationNow && investigationJobSnapshot) {
|
|
5120
5789
|
console.log(`[scheduler] needs_review ${job.slug} → immediate auto-fix investigation (not waiting for periodic reverify)`);
|
|
5121
|
-
spawnInvestigation(investigationJobSnapshot, runDir).catch((e) => {
|
|
5790
|
+
spawnInvestigation(investigationJobSnapshot, runDir, { deadChild: investigationDeadChildSnapshot }).catch((e) => {
|
|
5122
5791
|
console.error('[scheduler] spawnInvestigation error', job.slug, e);
|
|
5123
5792
|
});
|
|
5124
5793
|
}
|
|
@@ -5188,11 +5857,13 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
5188
5857
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
5189
5858
|
const freeSlots = sessionSlots.available();
|
|
5190
5859
|
const heldSlugs = await computeLaunchHolds(state);
|
|
5860
|
+
const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
|
|
5191
5861
|
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
5192
5862
|
leaseHeld: quietMachineLease.isHeld(),
|
|
5193
5863
|
machineInUse: sessionSlots.inUse(),
|
|
5194
5864
|
now: Date.now(),
|
|
5195
5865
|
heldSlugs,
|
|
5866
|
+
satisfiedSlugsByCwd,
|
|
5196
5867
|
});
|
|
5197
5868
|
if (batch.length === 0 && freeSlots === 0) {
|
|
5198
5869
|
const snap = sessionSlots.snapshot();
|
|
@@ -5366,6 +6037,109 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
5366
6037
|
tickQueue().catch((e) => console.error('[scheduler] tickQueue error', e));
|
|
5367
6038
|
}
|
|
5368
6039
|
|
|
6040
|
+
/** How long a queue may hold ready work with nothing running before the
|
|
6041
|
+
* starvation watchdog forces a tick. Deliberately longer than the poll
|
|
6042
|
+
* loop's own cadence + backoff, so this only ever fires when the normal
|
|
6043
|
+
* path has genuinely stopped driving the queue — it is a safety net, not a
|
|
6044
|
+
* second scheduler. */
|
|
6045
|
+
const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
6046
|
+
|
|
6047
|
+
/**
|
|
6048
|
+
* classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
|
|
6049
|
+
* → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
|
|
6050
|
+
*
|
|
6051
|
+
* Pure, no IO. Answers the one question the user's invariant reduces to:
|
|
6052
|
+
* "there are PRDs in a queue — is anything actually going to run them?"
|
|
6053
|
+
*
|
|
6054
|
+
* Every stall this codebase has seen was a DIFFERENT cause with the SAME
|
|
6055
|
+
* shape: ready rows, nothing running, nobody ticking. A rate-limited exit
|
|
6056
|
+
* stamped terminal `failed` (2026-09-05, 42 rows); a spin loop past the
|
|
6057
|
+
* manual-clear cooldown; a worktree merge-back that left the project on a
|
|
6058
|
+
* job branch; a job parked `needs_review` with no fix plan; app churn
|
|
6059
|
+
* counted as unfinished work. Guarding each cause individually will always
|
|
6060
|
+
* lag the next one, so this guards the SHAPE instead.
|
|
6061
|
+
*
|
|
6062
|
+
* Two outcomes, deliberately distinguished — they need opposite responses:
|
|
6063
|
+
* 'starved' — at least one pending row is dispatchable RIGHT NOW and
|
|
6064
|
+
* nothing is running. Whatever should have ticked, didn't.
|
|
6065
|
+
* Forcing a tick is safe and fixes it.
|
|
6066
|
+
* 'blocked' — every pending row is behind a terminal/parked dependency.
|
|
6067
|
+
* A tick cannot help; this needs a human (or a heal pass) to
|
|
6068
|
+
* resolve the blocker, and must be reported as such rather
|
|
6069
|
+
* than silently re-ticking forever.
|
|
6070
|
+
*
|
|
6071
|
+
* Returns null when the queue is healthy (work running, nothing pending,
|
|
6072
|
+
* paused on purpose, or simply not idle long enough yet).
|
|
6073
|
+
*/
|
|
6074
|
+
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6075
|
+
if (paused) return null; // paused is a DECISION, not a stall
|
|
6076
|
+
if (runningCount > 0) return null; // work is flowing
|
|
6077
|
+
const rows = Array.isArray(jobs) ? jobs : [];
|
|
6078
|
+
const pending = rows.filter((j) => j && j.status === 'pending');
|
|
6079
|
+
if (pending.length === 0) return null; // nothing to run — not a stall
|
|
6080
|
+
|
|
6081
|
+
const idleMs = Number.isFinite(lastRunAtMs) ? now - lastRunAtMs : Infinity;
|
|
6082
|
+
if (idleMs < thresholdMs) return null; // give the normal path its chance first
|
|
6083
|
+
|
|
6084
|
+
// Which pending rows could actually dispatch? Anything NOT named by a
|
|
6085
|
+
// blocked chain. computeBlockedChains already walks dependsOn with the
|
|
6086
|
+
// picker's own resolution, so the two can never disagree.
|
|
6087
|
+
const blockedChains = computeBlockedChains(rows);
|
|
6088
|
+
const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
|
|
6089
|
+
const dispatchable = pending.length - blockedTotal;
|
|
6090
|
+
|
|
6091
|
+
return {
|
|
6092
|
+
kind: dispatchable > 0 ? 'starved' : 'blocked',
|
|
6093
|
+
pending: pending.length,
|
|
6094
|
+
dispatchable,
|
|
6095
|
+
blockedChains,
|
|
6096
|
+
idleMs,
|
|
6097
|
+
};
|
|
6098
|
+
}
|
|
6099
|
+
|
|
6100
|
+
/**
|
|
6101
|
+
* The watchdog half: acts on classifyQueueStarvation. Called from the
|
|
6102
|
+
* heartbeat, which already runs on its own timer independent of the billing
|
|
6103
|
+
* poll loop — so a wedged or never-succeeding poll (the /api/oauth/usage
|
|
6104
|
+
* endpoint was itself 429ing all of 2026-09-05) can no longer leave a queue
|
|
6105
|
+
* with ready work idle indefinitely.
|
|
6106
|
+
*/
|
|
6107
|
+
async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6108
|
+
const verdict = classifyQueueStarvation({
|
|
6109
|
+
jobs: state?.jobs,
|
|
6110
|
+
paused: state?.paused,
|
|
6111
|
+
runningCount: runningSet.size,
|
|
6112
|
+
lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
|
|
6113
|
+
now,
|
|
6114
|
+
thresholdMs,
|
|
6115
|
+
});
|
|
6116
|
+
if (!verdict) return null;
|
|
6117
|
+
|
|
6118
|
+
const mins = Math.round(verdict.idleMs / 60_000);
|
|
6119
|
+
if (verdict.kind === 'blocked') {
|
|
6120
|
+
console.warn(
|
|
6121
|
+
`[scheduler] QUEUE BLOCKED: ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
6122
|
+
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
6123
|
+
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
6124
|
+
);
|
|
6125
|
+
appendAuditEvent('queue_blocked_stall', { pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
6126
|
+
return verdict;
|
|
6127
|
+
}
|
|
6128
|
+
|
|
6129
|
+
console.warn(
|
|
6130
|
+
`[scheduler] QUEUE STARVED: ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
|
|
6131
|
+
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
6132
|
+
);
|
|
6133
|
+
appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
6134
|
+
// A never-populated utilization reading is itself one of the ways the
|
|
6135
|
+
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
6136
|
+
// returns early on null). Treat unknown as safe here, exactly as the
|
|
6137
|
+
// billing meter's own 429 fallback already does.
|
|
6138
|
+
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
6139
|
+
await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
|
|
6140
|
+
return verdict;
|
|
6141
|
+
}
|
|
6142
|
+
|
|
5369
6143
|
// ---------- dead-process reaper ----------
|
|
5370
6144
|
|
|
5371
6145
|
// Queue-health sweep cadence: hangs off reapDeadRunningJobs's own cycle
|
|
@@ -5440,11 +6214,41 @@ async function reapDeadRunningJobs() {
|
|
|
5440
6214
|
|
|
5441
6215
|
if (dead.length === 0) return;
|
|
5442
6216
|
|
|
6217
|
+
// A rate-limited death is retryable, not terminal — mirror spawnJob's own
|
|
6218
|
+
// live-process handling (PRD 1117) exactly: engage the SAME setPaused
|
|
6219
|
+
// pause here too. Skipping this would reset the row to 'pending' but
|
|
6220
|
+
// leave dispatch unpaused, so the next tick immediately re-fires it into
|
|
6221
|
+
// the same still-active rate limit — the spin loop this PRD exists to
|
|
6222
|
+
// stop. Done once, outside mutate(), before finalizing any row below.
|
|
6223
|
+
if (dead.some((d) => d.outcome === 'rate_limited')) {
|
|
6224
|
+
const resetIso = await refreshNextReset().catch(() => cachedNextReset);
|
|
6225
|
+
const triggering = dead.find((d) => d.outcome === 'rate_limited');
|
|
6226
|
+
const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
|
|
6227
|
+
const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
|
|
6228
|
+
// Same rapid-repeat circuit breaker spawnJob's own res.rateLimited
|
|
6229
|
+
// branch drives (see consecutiveRapidRateLimitsBySlug above) — a
|
|
6230
|
+
// process that gets rate-limited and then dies without spawnJob's own
|
|
6231
|
+
// branch ever running is reconciled HERE instead, and must feed the
|
|
6232
|
+
// same counter or a stale/wrong resumeAt could keep re-clearing this
|
|
6233
|
+
// path's "fresh" pause every reap cycle with no hard cap ever engaging.
|
|
6234
|
+
const durationMs = Number.isFinite(observedAtMs) ? Date.now() - observedAtMs : Infinity;
|
|
6235
|
+
const prevCount = consecutiveRapidRateLimitsBySlug.get(triggering.slug) || 0;
|
|
6236
|
+
const nextCount = nextRapidRateLimitCount(prevCount, { rateLimited: true, durationMs });
|
|
6237
|
+
consecutiveRapidRateLimitsBySlug.set(triggering.slug, nextCount);
|
|
6238
|
+
const forceHardPause = nextCount >= CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD;
|
|
6239
|
+
if (forceHardPause) {
|
|
6240
|
+
console.log(`[scheduler] ${triggering.slug}: ${nextCount} consecutive rate-limited dispatches under ${RAPID_RATE_LIMIT_WINDOW_MS / 1000}s each (reaped) — engaging hard pause`);
|
|
6241
|
+
}
|
|
6242
|
+
await setPaused('rate_limit', resetIso, { observedAt: observedAtMs, force: forceHardPause });
|
|
6243
|
+
}
|
|
6244
|
+
|
|
5443
6245
|
await mutate(async (s) => {
|
|
5444
6246
|
for (const { slug, pid, outcome, gateOutcome, pidless, reason } of dead) {
|
|
5445
6247
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
5446
6248
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
6249
|
+
const rateLimited = outcome === 'rate_limited';
|
|
5447
6250
|
const success = outcome === 'success';
|
|
6251
|
+
if (!rateLimited) consecutiveRapidRateLimitsBySlug.delete(slug);
|
|
5448
6252
|
|
|
5449
6253
|
// Best-effort in-place leftover computation: a job whose owning
|
|
5450
6254
|
// process vanished without spawnJob()'s own finally block ever
|
|
@@ -5479,13 +6283,23 @@ async function reapDeadRunningJobs() {
|
|
|
5479
6283
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
5480
6284
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
5481
6285
|
: '';
|
|
5482
|
-
const transitionReason =
|
|
5483
|
-
|
|
5484
|
-
|
|
5485
|
-
|
|
5486
|
-
|
|
5487
|
-
|
|
5488
|
-
|
|
6286
|
+
const transitionReason = rateLimited
|
|
6287
|
+
? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
|
|
6288
|
+
: (pidless ? reason : `reaped: process gone (outcome=${outcome})`) + leftoverSuffix;
|
|
6289
|
+
|
|
6290
|
+
if (rateLimited) {
|
|
6291
|
+
// Retryable, never terminal (PRD 1117) — same resetJobFields path
|
|
6292
|
+
// spawnJob's own rateLimited branch uses (see ~4797's
|
|
6293
|
+
// treatAsPending), so the row comes back exactly like any other
|
|
6294
|
+
// paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
|
|
6295
|
+
resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
|
|
6296
|
+
} else {
|
|
6297
|
+
transitionJob(s.jobs[idx], success ? 'completed' : 'failed', { reason: transitionReason, source: 'reapDeadRunningJobs' });
|
|
6298
|
+
s.jobs[idx].exitCode = success ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
6299
|
+
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
6300
|
+
s.jobs[idx].error = success ? null : `${transitionReason} (outcome=${outcome})`;
|
|
6301
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
6302
|
+
}
|
|
5489
6303
|
delete s.jobs[idx].runtime;
|
|
5490
6304
|
delete s.jobs[idx].guardBaseline;
|
|
5491
6305
|
delete s.jobs[idx].guardHeadBefore;
|
|
@@ -5499,7 +6313,10 @@ async function reapDeadRunningJobs() {
|
|
|
5499
6313
|
// with no exit event) wedges the lease held forever and stalls
|
|
5500
6314
|
// dispatch for every project until the app restarts.
|
|
5501
6315
|
if (s.jobs[idx].quietMachine === true) quietMachineLease.release(slug);
|
|
5502
|
-
if (
|
|
6316
|
+
if (rateLimited) {
|
|
6317
|
+
console.log(`[scheduler] reaped rate-limited job slug=${slug} — reset to pending, pause engaged`);
|
|
6318
|
+
appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
|
|
6319
|
+
} else if (pidless) {
|
|
5503
6320
|
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
5504
6321
|
appendAuditEvent('job_reaped_pidless', { slug, cwd: s.jobs[idx].cwd ?? null, outcome, graceMs: PIDLESS_SPAWN_GRACE_MS });
|
|
5505
6322
|
} else {
|
|
@@ -5711,10 +6528,16 @@ const MAX_INVESTIGATION_DEPTH = 1;
|
|
|
5711
6528
|
* no recorded investigationDepth (a job already in the queue before this
|
|
5712
6529
|
* depth tracking shipped) is treated as excluded too, preserving the
|
|
5713
6530
|
* pre-existing blanket-exclusion behavior for legacy jobs — no retroactive
|
|
5714
|
-
* migration. Non-fix-plan
|
|
6531
|
+
* migration. Non-fix-plan jobs are never capped here.
|
|
6532
|
+
*
|
|
6533
|
+
* `isFixPlan` (PRD 1131) is the job's own persisted classification stamp
|
|
6534
|
+
* (see lib/fixPlanSlug.cjs's resolveIsFixPlan) — an explicit true/false wins
|
|
6535
|
+
* over the slug; only a row with the field entirely absent (persisted
|
|
6536
|
+
* before this change shipped) falls back to the legacy slug-only heuristic.
|
|
6537
|
+
* Exported for tests.
|
|
5715
6538
|
*/
|
|
5716
|
-
function isFixPlanBeyondDepthCap(slug, investigationDepth) {
|
|
5717
|
-
if (!
|
|
6539
|
+
function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
|
|
6540
|
+
if (!resolveIsFixPlan(slug, isFixPlan)) return false;
|
|
5718
6541
|
if (investigationDepth == null) return true;
|
|
5719
6542
|
return investigationDepth >= MAX_INVESTIGATION_DEPTH + 1;
|
|
5720
6543
|
}
|
|
@@ -5767,12 +6590,18 @@ function isUnresolvableNeedsReview(job, { hasRunDir }) {
|
|
|
5767
6590
|
* ('no-plan', 'error', and unstamped/undefined) — mirrors the retry
|
|
5768
6591
|
* eligibility rule in selectAutoFixTargets so a job can never be retry-
|
|
5769
6592
|
* eligible there and simultaneously un-annotatable here.
|
|
6593
|
+
*
|
|
6594
|
+
* A parent stamped `autoFixReopened: true` (its dead fix-plan child earned
|
|
6595
|
+
* it exactly one further attempt — see isFixPlanDead) is a separate
|
|
6596
|
+
* exhaustion path: it is spent as soon as that second investigation
|
|
6597
|
+
* concludes with ANY outcome, including another 'plan' — a reopened parent
|
|
6598
|
+
* never gets a third attempt, so unlike the fresh case a 'plan' outcome does
|
|
6599
|
+
* not exempt it here.
|
|
5770
6600
|
*/
|
|
5771
6601
|
function isExhaustedAutoFix(job) {
|
|
5772
|
-
|
|
5773
|
-
|
|
5774
|
-
|
|
5775
|
-
&& (job.autoFixRetries ?? 0) >= 1;
|
|
6602
|
+
if (!job || job.status !== 'needs_review' || job.autoFixAttempted !== true) return false;
|
|
6603
|
+
if (job.autoFixReopened === true) return job.autoFixOutcome != null;
|
|
6604
|
+
return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
|
|
5776
6605
|
}
|
|
5777
6606
|
|
|
5778
6607
|
/**
|
|
@@ -5788,6 +6617,31 @@ function isPlanUnqueued(job, queuedSlugs) {
|
|
|
5788
6617
|
return !queuedSlugs.has(fixSlugFor(job));
|
|
5789
6618
|
}
|
|
5790
6619
|
|
|
6620
|
+
// Terminal-and-not-completed statuses a fix-plan child can die in — see
|
|
6621
|
+
// isFixPlanDead.
|
|
6622
|
+
const DEAD_FIX_CHILD_STATUSES = new Set(['needs_review', 'failed', 'quarantined', 'skipped']);
|
|
6623
|
+
|
|
6624
|
+
/**
|
|
6625
|
+
* Pure predicate: a parent stuck at outcome 'plan' whose own fix-plan child
|
|
6626
|
+
* (fixSlugFor(job)) has ITSELF died — reached a terminal non-completed
|
|
6627
|
+
* status — with nothing left in the ladder that will ever revisit either
|
|
6628
|
+
* row again (selectAutoFixTargets skips a 'plan' outcome outright, and
|
|
6629
|
+
* isPlanUnqueued only fires when the child never reached the queue at all,
|
|
6630
|
+
* which isn't true once a dead child row exists). `job.autoFixReopened`
|
|
6631
|
+
* gates this to exactly once per parent — once stamped, this always returns
|
|
6632
|
+
* false so the parent can never be reopened a second time. Exported for
|
|
6633
|
+
* tests.
|
|
6634
|
+
*/
|
|
6635
|
+
function isFixPlanDead(job, jobsInProject) {
|
|
6636
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
6637
|
+
if (job.autoFixOutcome !== 'plan') return false;
|
|
6638
|
+
if (job.autoFixReopened === true) return false;
|
|
6639
|
+
const fixSlug = fixSlugFor(job);
|
|
6640
|
+
const child = (jobsInProject || []).find((j) => j.slug === fixSlug);
|
|
6641
|
+
if (!child) return false;
|
|
6642
|
+
return DEAD_FIX_CHILD_STATUSES.has(child.status);
|
|
6643
|
+
}
|
|
6644
|
+
|
|
5791
6645
|
/**
|
|
5792
6646
|
* Pure predicate: is this job eligible for the boot re-verify self-heal? Only
|
|
5793
6647
|
* needs_review jobs with a run log (own or backfilled via resolveRunId) AND a
|
|
@@ -5933,16 +6787,30 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
5933
6787
|
// bounded `--resume` attempt must never also become a fix-plan target
|
|
5934
6788
|
// in the same pass — see spawnInvestigation's own identical guard.
|
|
5935
6789
|
if (selectResumeRecoveryTarget(job)) return false;
|
|
6790
|
+
// Mechanical recovery (PRD 1130): a job eligible for a pure-git retry
|
|
6791
|
+
// must never also become a fix-plan target — it needs no plan and no
|
|
6792
|
+
// model. Defensive: today's single mechanically-resolvable verdict
|
|
6793
|
+
// (worktree_integration_failed) is already excluded below via the depth
|
|
6794
|
+
// cap, but this must hold even if that stops being true.
|
|
6795
|
+
if (selectMechanicalRecoveryTarget(job)) return false;
|
|
5936
6796
|
const runId = job.runId || resolveJobRunId(job);
|
|
5937
6797
|
if (!runId) return false;
|
|
5938
|
-
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth)) return false;
|
|
6798
|
+
if (isFixPlanBeyondDepthCap(job.slug, job.investigationDepth, job.isFixPlan)) return false;
|
|
6799
|
+
// A dead fix-plan child (PRD 1129) earns its parent exactly one further
|
|
6800
|
+
// attempt, bypassing the normal 'plan' exclusion and the fix-slug/queue
|
|
6801
|
+
// membership checks below — those checks exist to stop a FRESH
|
|
6802
|
+
// investigation from clobbering a live sibling, but here the sibling is
|
|
6803
|
+
// dead and reusing its slug is the whole point of the reopen.
|
|
6804
|
+
const dead = isFixPlanDead(job, jobs);
|
|
5939
6805
|
if (job.autoFixAttempted) {
|
|
5940
|
-
const retryEligible =
|
|
6806
|
+
const retryEligible = dead
|
|
6807
|
+
|| job.autoFixOutcome === 'no-plan'
|
|
5941
6808
|
|| job.autoFixOutcome === 'error'
|
|
5942
6809
|
|| job.autoFixOutcome == null;
|
|
5943
6810
|
if (!retryEligible) return false;
|
|
5944
|
-
if ((job.autoFixRetries ?? 0) >= 1) return false;
|
|
6811
|
+
if (!dead && (job.autoFixRetries ?? 0) >= 1) return false;
|
|
5945
6812
|
}
|
|
6813
|
+
if (dead) return true;
|
|
5946
6814
|
const fixSlug = fixSlugFor(job);
|
|
5947
6815
|
if (fixSlugExists(fixSlug)) return false;
|
|
5948
6816
|
if (slugsInQueue.has(fixSlug)) return false;
|
|
@@ -6121,7 +6989,7 @@ async function reverifyNeedsReview() {
|
|
|
6121
6989
|
const promotedPrds = [];
|
|
6122
6990
|
await mutate((s) => {
|
|
6123
6991
|
for (const job of s.jobs) {
|
|
6124
|
-
if (job.status !== 'completed' || !
|
|
6992
|
+
if (job.status !== 'completed' || !resolveIsFixPlan(job.slug, job.isFixPlan)) continue;
|
|
6125
6993
|
const orig = healTargetForFix(job.slug, s.jobs);
|
|
6126
6994
|
if (!orig) continue;
|
|
6127
6995
|
const priorStatus = orig.status;
|
|
@@ -6205,6 +7073,23 @@ async function reverifyNeedsReview() {
|
|
|
6205
7073
|
? await readQueue()
|
|
6206
7074
|
: afterHealForAnnotate;
|
|
6207
7075
|
|
|
7076
|
+
// Mechanical recovery (PRD 1130): evaluated first, ahead of both
|
|
7077
|
+
// resume-first recovery and auto-fix below — catches a job whose
|
|
7078
|
+
// mechanically-resolvable verdict this periodic pass finds still eligible
|
|
7079
|
+
// (e.g. one already parked before this rung shipped, or one the same-tick
|
|
7080
|
+
// check in spawnJob missed because the app restarted in between). Depth
|
|
7081
|
+
// never disqualifies it, so it runs regardless of investigationDepth.
|
|
7082
|
+
{
|
|
7083
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7084
|
+
const target = selectMechanicalRecoveryTarget(job);
|
|
7085
|
+
if (!target) continue;
|
|
7086
|
+
console.log(`[scheduler] mechanical-recovery: needs_review ${job.slug} → re-integrating ${target.branch}`);
|
|
7087
|
+
performMechanicalRecovery(job, target).catch((e) => {
|
|
7088
|
+
console.error('[scheduler] performMechanicalRecovery error', job.slug, e);
|
|
7089
|
+
});
|
|
7090
|
+
}
|
|
7091
|
+
}
|
|
7092
|
+
|
|
6208
7093
|
// Resume-first recovery (PRD 1111): before any fix-plan investigation is
|
|
6209
7094
|
// authored below, offer the bounded one-attempt `--resume` dispatch to any
|
|
6210
7095
|
// needs_review job this periodic pass finds still eligible — e.g. one the
|
|
@@ -6223,6 +7108,26 @@ async function reverifyNeedsReview() {
|
|
|
6223
7108
|
}
|
|
6224
7109
|
}
|
|
6225
7110
|
|
|
7111
|
+
// Leftover quarantine (PRD 1128), periodic pass: catches a job parked
|
|
7112
|
+
// needs_review with resume recovery already spent BEFORE this feature
|
|
7113
|
+
// shipped, or one the same-tick check in spawnJob missed because the app
|
|
7114
|
+
// restarted in between. Stamps the one-attempt marker in its own mutate
|
|
7115
|
+
// BEFORE the async git work starts (same race-closing rule as the resume
|
|
7116
|
+
// loop above and spawnJob's own dispatch stamp).
|
|
7117
|
+
{
|
|
7118
|
+
for (const job of queueForResumeAndAutofix.jobs) {
|
|
7119
|
+
const quarantineTarget = selectLeftoverQuarantineTarget(job);
|
|
7120
|
+
if (!quarantineTarget) continue;
|
|
7121
|
+
console.log(`[scheduler] leftover-quarantine: needs_review ${job.slug} → quarantining ${quarantineTarget.paths.length} leftover path(s)`);
|
|
7122
|
+
mutate((s) => {
|
|
7123
|
+
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
7124
|
+
if (j) j.leftoverQuarantineAttempted = true;
|
|
7125
|
+
}).then(() => performLeftoverQuarantine(job, quarantineTarget.paths)).catch((e) => {
|
|
7126
|
+
console.error('[scheduler] performLeftoverQuarantine error', job.slug, e);
|
|
7127
|
+
});
|
|
7128
|
+
}
|
|
7129
|
+
}
|
|
7130
|
+
|
|
6226
7131
|
// Auto-fix: spawn a fix-plan investigation for each job still in
|
|
6227
7132
|
// needs_review after the heal pass (kill-switch: SM_AUTOFIX_DISABLE=1).
|
|
6228
7133
|
// spawnInvestigation early-returns once investigationsInFlight reaches
|
|
@@ -6236,23 +7141,30 @@ async function reverifyNeedsReview() {
|
|
|
6236
7141
|
const runId = job.runId || resolveRunId(job);
|
|
6237
7142
|
const runDir = path.join(RUNS_DIR, runId);
|
|
6238
7143
|
const isRetryAttempt = job.autoFixAttempted === true;
|
|
7144
|
+
const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
|
|
7145
|
+
const deadChild = isDeadFixPlanReopen
|
|
7146
|
+
? queueForResumeAndAutofix.jobs.find((j) => j.slug === fixSlugFor(job))
|
|
7147
|
+
: null;
|
|
6239
7148
|
// Persist the attempt BEFORE spawning — a crash mid-investigation still
|
|
6240
7149
|
// counts it (mirrors orphanRetries). Safe even when the slot is busy: the
|
|
6241
7150
|
// investigation is queued and drained as slots free, so it is genuinely
|
|
6242
|
-
// attempted rather than silently dropped.
|
|
7151
|
+
// attempted rather than silently dropped. autoFixReopened is stamped in
|
|
7152
|
+
// this SAME mutate so a crash between selection and dispatch can never
|
|
7153
|
+
// leave the parent re-eligible for a second reopen (PRD 1129).
|
|
6243
7154
|
await mutate((s) => {
|
|
6244
7155
|
const j = s.jobs.find((x) => x.slug === job.slug);
|
|
6245
7156
|
if (j) {
|
|
6246
7157
|
j.autoFixAttempted = true;
|
|
6247
7158
|
if (!j.runId && runId) j.runId = runId;
|
|
7159
|
+
if (isDeadFixPlanReopen) j.autoFixReopened = true;
|
|
6248
7160
|
if (isRetryAttempt) {
|
|
6249
7161
|
j.autoFixRetries = (j.autoFixRetries ?? 0) + 1;
|
|
6250
7162
|
delete j.autoFixOutcome;
|
|
6251
7163
|
}
|
|
6252
7164
|
}
|
|
6253
7165
|
});
|
|
6254
|
-
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'})`);
|
|
6255
|
-
spawnInvestigation(job, runDir).catch((e) => {
|
|
7166
|
+
console.log(`[scheduler] auto-fix: needs_review ${job.slug} → authoring fix-plan (${isRetryAttempt ? 'retry' : '1/1'}${isDeadFixPlanReopen ? ', dead fix-plan child reopen' : ''})`);
|
|
7167
|
+
spawnInvestigation(job, runDir, { deadChild }).catch((e) => {
|
|
6256
7168
|
console.error('[scheduler] auto-fix spawnInvestigation error', job.slug, e);
|
|
6257
7169
|
});
|
|
6258
7170
|
}
|
|
@@ -6865,6 +7777,14 @@ async function init() {
|
|
|
6865
7777
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
6866
7778
|
heartbeatInterval = setInterval(() => {
|
|
6867
7779
|
const s = readQueueSync();
|
|
7780
|
+
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
7781
|
+
// running, something must drive it. This is the only driver that does
|
|
7782
|
+
// not depend on the billing poll loop, a pause timer, or a completing
|
|
7783
|
+
// job to schedule the next tick — every one of which has failed at
|
|
7784
|
+
// least once. See classifyQueueStarvation.
|
|
7785
|
+
if (!s.unreadable) {
|
|
7786
|
+
runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
7787
|
+
}
|
|
6868
7788
|
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
6869
7789
|
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
6870
7790
|
// failed }` literal silently minted a NEW key for any other value
|
|
@@ -7335,6 +8255,36 @@ const remote = {
|
|
|
7335
8255
|
return { ok: false, error: `job status is "${job.status}" — only a not-yet-running PRD (status "pending"/"quarantined", or no queue row yet) may be edited` };
|
|
7336
8256
|
}
|
|
7337
8257
|
|
|
8258
|
+
// Write-time FK check for a patched dependsOn (PRD 1124), reusing the
|
|
8259
|
+
// SAME resolution rule scheduler_create_prd's prdCreate.cjs applies (exact
|
|
8260
|
+
// slug, else bare-name after stripping one leading `NN-`) so update and
|
|
8261
|
+
// create can never disagree about what a dependsOn entry resolves to. An
|
|
8262
|
+
// explicit empty array CLEARS the dependency and skips validation — there
|
|
8263
|
+
// is nothing to resolve. A listPrds() read failure is skipped-with-a-
|
|
8264
|
+
// warning, matching createPrd's tolerance for an I/O hiccup.
|
|
8265
|
+
if (frontmatter && Array.isArray(frontmatter.dependsOn) && frontmatter.dependsOn.length) {
|
|
8266
|
+
let listing;
|
|
8267
|
+
try {
|
|
8268
|
+
listing = await this.listPrds({ cwd, limit: Number.MAX_SAFE_INTEGER });
|
|
8269
|
+
} catch (e) {
|
|
8270
|
+
console.warn(`[scheduler] updatePrd: dependsOn validation skipped (listPrds failed): ${e?.message ?? e}`);
|
|
8271
|
+
listing = null;
|
|
8272
|
+
}
|
|
8273
|
+
if (listing) {
|
|
8274
|
+
const candidateSlugs = (listing.prds ?? []).map((p) => p.slug);
|
|
8275
|
+
for (const dep of frontmatter.dependsOn) {
|
|
8276
|
+
if (resolveDepSlug(dep, candidateSlugs).length > 0) continue;
|
|
8277
|
+
const near = findNearMatches(dep, candidateSlugs);
|
|
8278
|
+
const suggestion = near.length ? ` Closest existing slug(s): ${near.join(', ')}.` : '';
|
|
8279
|
+
return {
|
|
8280
|
+
ok: false,
|
|
8281
|
+
error: `dependsOn entry "${dep}" does not resolve to any existing PRD in this project.${suggestion} ` +
|
|
8282
|
+
'Pass the bare name (preferred) or the exact NN-prefixed slug of an existing PRD.',
|
|
8283
|
+
};
|
|
8284
|
+
}
|
|
8285
|
+
}
|
|
8286
|
+
}
|
|
8287
|
+
|
|
7338
8288
|
let dir = null;
|
|
7339
8289
|
let filePath = null;
|
|
7340
8290
|
if (cwd) {
|
|
@@ -7465,4 +8415,149 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
7465
8415
|
});
|
|
7466
8416
|
}
|
|
7467
8417
|
|
|
7468
|
-
module.exports = {
|
|
8418
|
+
module.exports = {
|
|
8419
|
+
classifyQueueStarvation,
|
|
8420
|
+
runQueueStarvationWatchdog,
|
|
8421
|
+
QUEUE_STARVATION_MS,
|
|
8422
|
+
computeBlockedChains,
|
|
8423
|
+
stripAppOwnedChurn,
|
|
8424
|
+
findOverrunningJobs,
|
|
8425
|
+
JOB_OVERRUN_FACTOR,
|
|
8426
|
+
JOB_OVERRUN_FLOOR_MS,
|
|
8427
|
+
registerScheduleHandlers,
|
|
8428
|
+
attachWindow,
|
|
8429
|
+
init,
|
|
8430
|
+
ROOT,
|
|
8431
|
+
PRDS_DIR,
|
|
8432
|
+
healRefusalReason,
|
|
8433
|
+
writeQueue,
|
|
8434
|
+
reconcile,
|
|
8435
|
+
reconcileSourcePromptId,
|
|
8436
|
+
allocateParallelGroup,
|
|
8437
|
+
selectHistoryJobs,
|
|
8438
|
+
parsePorcelain,
|
|
8439
|
+
FINISH_PROTOCOL,
|
|
8440
|
+
IDLE_OUTPUT_KILL_MS,
|
|
8441
|
+
BASH_DEFAULT_TIMEOUT_MS,
|
|
8442
|
+
BASH_MAX_TIMEOUT_MS,
|
|
8443
|
+
remote,
|
|
8444
|
+
pickNextBatch,
|
|
8445
|
+
pickForProject,
|
|
8446
|
+
reapDeadRunningJobs,
|
|
8447
|
+
pollRecoveryClearSource,
|
|
8448
|
+
memoryLimitedBatchSize,
|
|
8449
|
+
availableForJobs,
|
|
8450
|
+
reverifyNeedsReview,
|
|
8451
|
+
isRescanCandidate,
|
|
8452
|
+
isFailedUnverifiedShaped,
|
|
8453
|
+
computeLooksDone,
|
|
8454
|
+
isPromotableOriginal,
|
|
8455
|
+
selectAutoFixTargets,
|
|
8456
|
+
applyRcaClassification,
|
|
8457
|
+
isEligibleForImmediateAutoFix,
|
|
8458
|
+
resolveRunId,
|
|
8459
|
+
isUnresolvableNeedsReview,
|
|
8460
|
+
isExhaustedAutoFix,
|
|
8461
|
+
isPlanUnqueued,
|
|
8462
|
+
isFixPlanDead,
|
|
8463
|
+
fixSlugFor,
|
|
8464
|
+
healTargetForFix,
|
|
8465
|
+
buildInvestigationPrompt,
|
|
8466
|
+
isGitRepoSync,
|
|
8467
|
+
committedInWindow,
|
|
8468
|
+
computeCommittedDuringRun,
|
|
8469
|
+
classifySigtermWithCommit,
|
|
8470
|
+
isFixPlanSlug,
|
|
8471
|
+
classifyDiscoveredFixPlan,
|
|
8472
|
+
resolveIsFixPlan,
|
|
8473
|
+
isFixPlanBeyondDepthCap,
|
|
8474
|
+
MAX_INVESTIGATION_DEPTH,
|
|
8475
|
+
forceTickOutcome,
|
|
8476
|
+
applyPauseCleared,
|
|
8477
|
+
detectNetworkErrorInLog,
|
|
8478
|
+
detectRateLimitInLog,
|
|
8479
|
+
classifyFailureOutcome,
|
|
8480
|
+
commitGuardVerdict,
|
|
8481
|
+
leftoverFieldsFrom,
|
|
8482
|
+
applyLeftoverFields,
|
|
8483
|
+
LEFTOVER_PATHS_CAP,
|
|
8484
|
+
capDirtyPaths,
|
|
8485
|
+
buildForeignWipSection,
|
|
8486
|
+
PRE_RUN_DIRTY_PATHS_CAP,
|
|
8487
|
+
FOREIGN_WIP_DELIMITER,
|
|
8488
|
+
FOREIGN_WIP_END_DELIMITER,
|
|
8489
|
+
TRANSIENT_RETRY_CAP,
|
|
8490
|
+
buildScheduleStatePayload,
|
|
8491
|
+
partitionBootOrphans,
|
|
8492
|
+
applyOrphanOutcome,
|
|
8493
|
+
BOOT_ORPHAN_KILL_GRACE_MS,
|
|
8494
|
+
registerAdminRoutes,
|
|
8495
|
+
notifyOriginatingTab,
|
|
8496
|
+
notifyNeedsReview,
|
|
8497
|
+
isNotifiableTerminalStatus,
|
|
8498
|
+
extractResultTextFromLog,
|
|
8499
|
+
candidatePrdsDirs,
|
|
8500
|
+
candidateArchivedPrdsDirs,
|
|
8501
|
+
resolveArchivedPrdStatus,
|
|
8502
|
+
prdDirForCwd,
|
|
8503
|
+
prdPathForJob,
|
|
8504
|
+
archivedPrdPathForJob,
|
|
8505
|
+
archivedTwinExists,
|
|
8506
|
+
findPrdDir,
|
|
8507
|
+
resolveVerifyPrdPath,
|
|
8508
|
+
resolveFixPlanPath,
|
|
8509
|
+
resolveNotifyPrd,
|
|
8510
|
+
runPrdMigration,
|
|
8511
|
+
consolidateAllFlatPrds,
|
|
8512
|
+
shouldSkipInvestigationForCleanRun,
|
|
8513
|
+
archiveCompletedPrd,
|
|
8514
|
+
retireCompletedSlugs,
|
|
8515
|
+
SCHEDULER_BOOTED_AT,
|
|
8516
|
+
SCHEDULER_CODE_SHA,
|
|
8517
|
+
resetJobFields,
|
|
8518
|
+
executeJob,
|
|
8519
|
+
prdArchivedSkipResult,
|
|
8520
|
+
spawnJob,
|
|
8521
|
+
listPrdsInternal,
|
|
8522
|
+
computeStallSummary,
|
|
8523
|
+
findStaleQuarantinedJobs,
|
|
8524
|
+
QUARANTINE_ESCALATE_MS,
|
|
8525
|
+
applyClearQueueVictims,
|
|
8526
|
+
PIDLESS_SPAWN_GRACE_MS,
|
|
8527
|
+
findStrandedInvestigations,
|
|
8528
|
+
INVESTIGATION_MAX_MS,
|
|
8529
|
+
stashList,
|
|
8530
|
+
parseStashLine,
|
|
8531
|
+
pathsChangedSince,
|
|
8532
|
+
restoreSpecificStash,
|
|
8533
|
+
evaluateSharedTreeGuard,
|
|
8534
|
+
checkSharedTreeGuard,
|
|
8535
|
+
uncommittedChanges,
|
|
8536
|
+
gitHead,
|
|
8537
|
+
selectResumeRecoveryTarget,
|
|
8538
|
+
buildResumeRecoveryPreamble,
|
|
8539
|
+
buildClaudeSpawnArgs,
|
|
8540
|
+
spawnResumeRecovery,
|
|
8541
|
+
selectMechanicalRecoveryTarget,
|
|
8542
|
+
performMechanicalRecovery,
|
|
8543
|
+
MECHANICALLY_RESOLVABLE_VERDICTS,
|
|
8544
|
+
selectLeftoverQuarantineTarget,
|
|
8545
|
+
quarantineLeftovers,
|
|
8546
|
+
performLeftoverQuarantine,
|
|
8547
|
+
spawnInvestigation,
|
|
8548
|
+
computeLaunchHolds,
|
|
8549
|
+
computeDepHistorySatisfaction,
|
|
8550
|
+
handleLaunchFailure,
|
|
8551
|
+
applyLaunchFailure,
|
|
8552
|
+
setPaused,
|
|
8553
|
+
clearPause,
|
|
8554
|
+
tickQueue,
|
|
8555
|
+
runDueJobs,
|
|
8556
|
+
isCooldownSuppressed,
|
|
8557
|
+
nextRapidRateLimitCount,
|
|
8558
|
+
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|
|
8559
|
+
RAPID_RATE_LIMIT_WINDOW_MS,
|
|
8560
|
+
MANUAL_PAUSE_COOLDOWN_MS,
|
|
8561
|
+
RUNS_DIR,
|
|
8562
|
+
pickRunDir,
|
|
8563
|
+
};
|