claude-code-session-manager 0.85.0 → 0.87.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/AgentLibrary-DyLWzZDf.js +3 -0
- package/dist/assets/{DataModel-DRH-Ty20.js → DataModel--mISIJ6h.js} +1 -1
- package/dist/assets/{History-CfRhT1Im.js → History-C2ahUXTg.js} +2 -2
- package/dist/assets/{Hooks-wEmh_U6c.js → Hooks-BiC6oyR2.js} +3 -3
- package/dist/assets/{HostBilko-D_t7Rbi7.js → HostBilko-BPleEOld.js} +1 -1
- package/dist/assets/{Library-CpArQ-OJ.js → Library-Dc8Qst1R.js} +1 -1
- package/dist/assets/{ListDetail-pjaKYs84.js → ListDetail-DIXh-OLX.js} +1 -1
- package/dist/assets/MarkdownEditor-C90bkLXK.js +1 -0
- package/dist/assets/{McpServers-ftqaV3kn.js → McpServers-DqcbLOLZ.js} +2 -2
- package/dist/assets/{Memory-ChMWkNd0.js → Memory-CW62MXlh.js} +4 -4
- package/dist/assets/{Panel-D9Kr40Ai.js → Panel-Bw1FhRuF.js} +1 -1
- package/dist/assets/Permissions-BcUC-5y8.js +3 -0
- package/dist/assets/{Plugins-BtChISho.js → Plugins-BnKx9flD.js} +2 -2
- package/dist/assets/{ProvenanceBadge-DBA5EcYy.js → ProvenanceBadge-Bw5vNVPT.js} +1 -1
- package/dist/assets/{SaveBar-I0_dWNTX.js → SaveBar-CWr0O_w-.js} +1 -1
- package/dist/assets/Scheduler-DYdLuUqq.js +14 -0
- package/dist/assets/{ScopeSwitcher-5GTEveb2.js → ScopeSwitcher-CrBLbg8s.js} +1 -1
- package/dist/assets/Settings-DluB-vN1.js +3 -0
- package/dist/assets/{SkillReferenceGraph-DNBFGrYE.js → SkillReferenceGraph-CHLSseay.js} +1 -1
- package/dist/assets/{Skills-DJB6-bBM.js → Skills-gNdo_HNK.js} +2 -2
- package/dist/assets/{SystemPrompt-BiDDrJUA.js → SystemPrompt-Cru05-Ia.js} +1 -1
- package/dist/assets/{TagLibrary-_Wrevtop.js → TagLibrary-DNHY0xou.js} +1 -1
- package/dist/assets/{TiptapBody-OWWXdLRy.js → TiptapBody-I4lmbCgP.js} +1 -1
- package/dist/assets/{Toggle-B122N0HL.js → Toggle-bWMHjmRh.js} +1 -1
- package/dist/assets/{index-CDo9xBR9.css → index-DV3PorRY.css} +1 -1
- package/dist/assets/{index-DhvuQL4C.js → index-fc_JjdxL.js} +724 -724
- package/dist/assets/settingsSchema-BfhtZnGD.js +3 -0
- package/dist/index.html +2 -2
- package/package.json +15 -14
- package/plugins/CLAUDE.md +61 -0
- package/plugins/session-manager-dev/.claude-plugin/plugin.json +1 -1
- package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +1 -1
- package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +1 -1
- package/scripts/scheduler-mcp-server.cjs +7 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +100 -9
- package/src/main/__tests__/broadcastCoalescer.test.cjs +18 -0
- package/src/main/__tests__/epicMint.test.cjs +2 -2
- package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
- package/src/main/__tests__/health-starve-escalation.test.cjs +94 -0
- package/src/main/__tests__/loadGateDetailTick.test.cjs +31 -0
- package/src/main/__tests__/machineProfile.test.cjs +19 -1
- package/src/main/__tests__/needsReviewLedger.test.cjs +162 -0
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +3 -3
- package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +15 -1
- package/src/main/__tests__/prdCreateDisposition.test.cjs +201 -0
- package/src/main/__tests__/prdFrontmatterDisposition.test.cjs +125 -0
- package/src/main/__tests__/prdLocations.test.cjs +100 -2
- package/src/main/__tests__/prdLocationsArchived.test.cjs +43 -1
- package/src/main/__tests__/prdSetDisposition.test.cjs +222 -0
- package/src/main/__tests__/pty-session-open-telemetry.test.cjs +96 -0
- package/src/main/__tests__/queue-health-verdict.test.cjs +170 -0
- package/src/main/__tests__/queue-starvation-per-project.test.cjs +147 -0
- package/src/main/__tests__/queueHistory.test.cjs +63 -0
- package/src/main/__tests__/reconcileTiming.test.cjs +135 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +101 -1
- package/src/main/__tests__/scheduler-boot-orphans.test.cjs +2 -2
- package/src/main/__tests__/scheduler-broadcast-reconcile.test.cjs +121 -0
- package/src/main/__tests__/scheduler-cross-project-batch.test.cjs +43 -0
- package/src/main/__tests__/scheduler-failed-autoreset.test.cjs +121 -0
- package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +344 -0
- package/src/main/__tests__/scheduler-job-budget.test.cjs +172 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +93 -3
- package/src/main/__tests__/scheduler-needs-review-autoresolve.test.cjs +189 -0
- package/src/main/__tests__/scheduler-no-dead-end-status.test.cjs +152 -0
- package/src/main/__tests__/scheduler-porcelain-rename.test.cjs +164 -0
- package/src/main/__tests__/scheduler-quarantine-autoresolve.test.cjs +165 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +160 -0
- package/src/main/__tests__/scheduler-reaper-helpers-basics.test.cjs +87 -0
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +88 -0
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +154 -0
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +14 -0
- package/src/main/__tests__/telemetryClient.test.cjs +75 -5
- package/src/main/__tests__/telemetrySettings.test.cjs +32 -0
- package/src/main/chatRunner.cjs +8 -5
- package/src/main/health.cjs +76 -3
- package/src/main/historyAggregator.cjs +5 -0
- package/src/main/index.cjs +95 -44
- package/src/main/ipcSchemas.cjs +47 -0
- package/src/main/lib/__tests__/active-sessions.test.cjs +251 -0
- package/src/main/lib/__tests__/bootSelfHeal.test.cjs +107 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +322 -43
- package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +239 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +89 -0
- package/src/main/lib/__tests__/guardShims.test.cjs +151 -0
- package/src/main/lib/__tests__/loadGate.test.cjs +103 -2
- package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +5 -5
- package/src/main/lib/__tests__/prdDisposition.test.cjs +224 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +179 -1
- package/src/main/lib/__tests__/telemetryBoot.test.cjs +11 -0
- package/src/main/lib/__tests__/usageCircuit.test.cjs +224 -0
- package/src/main/lib/__tests__/watchdog-helpers.test.cjs +312 -0
- package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +193 -0
- package/{scripts → src/main}/lib/activeSessions.cjs +50 -4
- package/src/main/lib/agentModelResolve.cjs +65 -27
- package/src/main/lib/bootSelfHeal.cjs +88 -0
- package/src/main/lib/delegationReadiness.cjs +290 -225
- package/src/main/lib/effectiveModelInfo.cjs +333 -0
- package/src/main/lib/ephemeralCwd.cjs +1 -1
- package/src/main/lib/epicMint.cjs +3 -3
- package/src/main/lib/gitWorktree.cjs +42 -12
- package/src/main/lib/guardShims.cjs +156 -0
- package/src/main/lib/jobDirtFilter.cjs +7 -2
- package/src/main/lib/launchFailure.cjs +2 -1
- package/src/main/lib/loadGate.cjs +23 -1
- package/src/main/lib/machineProfile.cjs +15 -0
- package/src/main/lib/mcpToolCatalog.cjs +4 -1
- package/src/main/lib/needsReviewLedger.cjs +205 -0
- package/src/main/lib/opsErrorLog.cjs +1 -1
- package/src/main/lib/opsOwnership.cjs +1 -1
- package/src/main/lib/prdCreate.cjs +56 -1
- package/src/main/lib/prdDisposition.cjs +199 -0
- package/src/main/lib/prdFrontmatter.cjs +8 -2
- package/src/main/lib/prdLocations.cjs +167 -45
- package/src/main/lib/projectHomeAdminRoutes.cjs +4 -4
- package/src/main/lib/projectPageSummarySchema.cjs +1 -1
- package/src/main/lib/projectRootResolve.cjs +1 -1
- package/src/main/lib/queueHistory.cjs +19 -1
- package/src/main/lib/queueStore.cjs +6 -1
- package/src/main/lib/reaperHelpers.cjs +181 -15
- package/src/main/lib/scheduleJobSchema.cjs +8 -0
- package/src/main/lib/scheduleJobTransitions.cjs +33 -0
- package/src/main/lib/schedulerBatch.cjs +12 -1
- package/src/main/lib/schedulerConfig.cjs +37 -0
- package/src/main/lib/telemetryBoot.cjs +11 -8
- package/src/main/lib/telemetryClient.cjs +44 -2
- package/src/main/lib/telemetrySettings.cjs +20 -3
- package/src/main/lib/usageCircuit.cjs +159 -0
- package/{scripts → src/main}/lib/watchdogHelpers.cjs +1 -1
- package/src/main/pty.cjs +9 -0
- package/src/main/scheduler/prdParser.cjs +13 -0
- package/src/main/scheduler.cjs +1869 -193
- package/src/main/templates/PRD_AUTHORING.md +50 -0
- package/src/main/templates/project-pages-catalog.json +1 -1
- package/src/main/usage.cjs +21 -3
- package/src/preload/api.d.ts +92 -1
- package/src/preload/index.cjs +10 -0
- package/web/README.md +41 -0
- package/{scripts/render-project-pages.cjs → web/project-pages/render.cjs} +4 -4
- package/{scripts/render-project-pages → web/project-pages/renderer}/dist/renderer.cjs +1 -1
- package/{scripts/validate-project-pages-summary.cjs → web/project-pages/validate-summary.cjs} +5 -5
- package/dist/assets/AgentLibrary-Bkv-HcP1.js +0 -3
- package/dist/assets/MarkdownEditor-Xc141kjj.js +0 -1
- package/dist/assets/Permissions-DKoNVgzj.js +0 -3
- package/dist/assets/Scheduler-CbES7MC8.js +0 -14
- package/dist/assets/Settings-BX3FElXk.js +0 -3
- package/dist/assets/settingsSchema-sGoCTd7J.js +0 -3
- /package/{scripts/project-pages-logic → web/project-pages/logic}/dist/logic.cjs +0 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -60,7 +60,9 @@ const { readTail } = require('./lib/fileTail.cjs');
|
|
|
60
60
|
const {
|
|
61
61
|
claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
|
|
62
62
|
findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
|
|
63
|
+
readSpawnedPidFromLog, readLogMtimeMs,
|
|
63
64
|
} = require('./lib/reaperHelpers.cjs');
|
|
65
|
+
const { resolveProjectRoot } = require('./lib/opsOwnership.cjs');
|
|
64
66
|
const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
|
|
65
67
|
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
66
68
|
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
@@ -89,22 +91,42 @@ const {
|
|
|
89
91
|
USAGE_REFRESH_INTERVAL_MS,
|
|
90
92
|
MAX_JOB_DURATION_MS,
|
|
91
93
|
BROADCAST_COALESCE_MS,
|
|
94
|
+
RECONCILE_SLOW_PASS_MS,
|
|
92
95
|
QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
|
|
93
96
|
JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
|
|
94
97
|
JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
|
|
98
|
+
JOB_BUDGET_FACTOR: JOB_BUDGET_FACTOR_DEFAULT,
|
|
99
|
+
JOB_BUDGET_FLOOR_MS: JOB_BUDGET_FLOOR_MS_DEFAULT,
|
|
100
|
+
JOB_BUDGET_CEILING_MS: JOB_BUDGET_CEILING_MS_DEFAULT,
|
|
95
101
|
PIDLESS_SPAWN_GRACE_MS,
|
|
96
102
|
INVESTIGATION_MAX_MS,
|
|
97
103
|
STARVATION_ESCALATE_MS,
|
|
104
|
+
STARVE_ESCALATION_MS,
|
|
98
105
|
} = require('./lib/schedulerConfig.cjs');
|
|
99
106
|
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
100
107
|
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
101
108
|
: QUARANTINE_ESCALATE_MS_DEFAULT;
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
:
|
|
109
|
+
// Shared by every SM_*-env-overridable numeric constant below (bare factors
|
|
110
|
+
// use unitMs=1; minute-denominated knobs use unitMs=60_000) — one parse rule
|
|
111
|
+
// instead of one hand-copied ternary per constant.
|
|
112
|
+
function numEnvOverride(envVar, unitMs, fallback) {
|
|
113
|
+
const raw = process.env[envVar];
|
|
114
|
+
return raw ? Number(raw) * unitMs : fallback;
|
|
115
|
+
}
|
|
116
|
+
const JOB_OVERRUN_FACTOR = numEnvOverride('SM_JOB_OVERRUN_FACTOR', 1, JOB_OVERRUN_FACTOR_DEFAULT);
|
|
117
|
+
const JOB_OVERRUN_FLOOR_MS = numEnvOverride('SM_JOB_OVERRUN_FLOOR_MINUTES', 60_000, JOB_OVERRUN_FLOOR_MS_DEFAULT);
|
|
118
|
+
// Same three numbers as JOB_OVERRUN_FACTOR/JOB_OVERRUN_FLOOR_MS today (3x,
|
|
119
|
+
// 45min) is coincidental, not structural — this triad ACTS (kills) where
|
|
120
|
+
// JOB_OVERRUN_* only ever escalates (see JOB_OVERRUN_FACTOR's own header);
|
|
121
|
+
// tune them independently, don't re-couple on a future pass just because the
|
|
122
|
+
// defaults happen to match right now.
|
|
123
|
+
const JOB_BUDGET_FACTOR = numEnvOverride('SM_JOB_BUDGET_FACTOR', 1, JOB_BUDGET_FACTOR_DEFAULT);
|
|
124
|
+
const JOB_BUDGET_FLOOR_MS = numEnvOverride('SM_JOB_BUDGET_FLOOR_MINUTES', 60_000, JOB_BUDGET_FLOOR_MS_DEFAULT);
|
|
125
|
+
const JOB_BUDGET_CEILING_MS = numEnvOverride('SM_JOB_BUDGET_CEILING_MINUTES', 60_000, JOB_BUDGET_CEILING_MS_DEFAULT);
|
|
126
|
+
// A running job past this fraction of its own budget gets a durable
|
|
127
|
+
// `budgetWarning` stamp on its row (see the budget watchdog below) so the
|
|
128
|
+
// renderer can warn BEFORE the kill, not only after.
|
|
129
|
+
const BUDGET_WARNING_FRACTION = 0.75;
|
|
108
130
|
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
|
|
109
131
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
110
132
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
@@ -150,8 +172,9 @@ const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
|
150
172
|
const queueStore = require('./lib/queueStore.cjs');
|
|
151
173
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
152
174
|
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
175
|
+
const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
|
|
153
176
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
154
|
-
const { allProjectCwds } = require('
|
|
177
|
+
const { allProjectCwds } = require('./lib/activeSessions.cjs');
|
|
155
178
|
|
|
156
179
|
// Captured once at module load so every run's meta sidecar can record how
|
|
157
180
|
// stale the running process is relative to on-disk source (incident: PRD
|
|
@@ -310,22 +333,95 @@ only post-AC work. If a review finding can't be fixed within scope, commit what
|
|
|
310
333
|
you have, describe the finding in the commit body, and note the follow-up in your
|
|
311
334
|
final result.`;
|
|
312
335
|
|
|
313
|
-
//
|
|
314
|
-
//
|
|
315
|
-
//
|
|
316
|
-
//
|
|
317
|
-
|
|
336
|
+
// Unquote a single git porcelain v1 path token. Defined once in
|
|
337
|
+
// gitWorktree.cjs (which this file already requires — the reverse would be
|
|
338
|
+
// circular, since gitWorktree.cjs's own salvageDirtyDelta needs the exact
|
|
339
|
+
// same unquoting) and reused here rather than re-implemented, so the two
|
|
340
|
+
// porcelain consumers in this codebase can never drift apart.
|
|
341
|
+
const { unquotePorcelainPath } = gitWorktree;
|
|
342
|
+
|
|
343
|
+
// Split a rename/copy porcelain path field ("old -> new") into its two real
|
|
344
|
+
// paths. Each side is independently quoted per unquotePorcelainPath's rule —
|
|
345
|
+
// only the side that needs escaping is wrapped in quotes, the literal " -> "
|
|
346
|
+
// arrow between them never is. Returns null when no " -> " separator is
|
|
347
|
+
// found (a malformed/unexpected line) so the caller can fall back to treating
|
|
348
|
+
// the whole field as one opaque path rather than guessing.
|
|
349
|
+
function splitRenamePorcelainField(field) {
|
|
350
|
+
const arrow = ' -> ';
|
|
351
|
+
let head;
|
|
352
|
+
let rest;
|
|
353
|
+
if (field[0] === '"') {
|
|
354
|
+
let end = -1;
|
|
355
|
+
for (let i = 1; i < field.length; i += 1) {
|
|
356
|
+
if (field[i] === '\\') { i += 1; continue; }
|
|
357
|
+
if (field[i] === '"') { end = i; break; }
|
|
358
|
+
}
|
|
359
|
+
if (end === -1) return null;
|
|
360
|
+
head = field.slice(0, end + 1);
|
|
361
|
+
rest = field.slice(end + 1);
|
|
362
|
+
} else {
|
|
363
|
+
const idx = field.indexOf(arrow);
|
|
364
|
+
if (idx === -1) return null;
|
|
365
|
+
// An unquoted path containing a literal " -> " substring (git only
|
|
366
|
+
// quotes for a quote/backslash/control-byte/non-ASCII byte — a plain
|
|
367
|
+
// ASCII arrow inside a filename is never quoted) makes the true
|
|
368
|
+
// old/new boundary genuinely ambiguous from this text alone: the first
|
|
369
|
+
// occurrence could be the real separator, or it could be sitting
|
|
370
|
+
// inside the old path with the real separator later in the field.
|
|
371
|
+
// Guessing wrong silently corrupts oldPath/path for downstream
|
|
372
|
+
// fs.existsSync/Set-membership checks, which is worse than the
|
|
373
|
+
// existing "malformed line" fallback below — so more than one
|
|
374
|
+
// occurrence falls back to treating the whole field as one opaque
|
|
375
|
+
// path, same as any other line this function can't confidently parse.
|
|
376
|
+
if (field.indexOf(arrow, idx + arrow.length) !== -1) return null;
|
|
377
|
+
head = field.slice(0, idx);
|
|
378
|
+
rest = field.slice(idx);
|
|
379
|
+
}
|
|
380
|
+
if (!rest.startsWith(arrow)) return null;
|
|
381
|
+
return { oldPath: unquotePorcelainPath(head), path: unquotePorcelainPath(rest.slice(arrow.length)) };
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
// Parse `git status --porcelain` output into `{ code, path }` entries (plus
|
|
385
|
+
// `oldPath` for a rename/copy). Pure + exported for unit testing. Each
|
|
386
|
+
// porcelain line is "XY<space>PATH"; a staged rename/copy line is
|
|
387
|
+
// "XY<space>OLD -> NEW" instead — X (index status) is 'R' or 'C' — and NEW is
|
|
388
|
+
// the path git will report in any later `git status` call, so callers that
|
|
389
|
+
// key off `.path` (dirtyAfter membership, pathsCommittedDuringRun membership,
|
|
390
|
+
// fs.existsSync) must compare against NEW, never the fused "OLD -> NEW"
|
|
391
|
+
// string. `oldPath` is retained on the entry for callers that need the
|
|
392
|
+
// original path too. `code` is the raw 2-char status (e.g. '??' for
|
|
393
|
+
// untracked) — callers that need to distinguish "untracked" from
|
|
394
|
+
// "tracked-but-modified" (the shared-tree guard's revert-vs-now-ignored
|
|
395
|
+
// split) read it off the entry instead of re-deriving it later.
|
|
396
|
+
function parsePorcelainEntries(stdout) {
|
|
318
397
|
return String(stdout || '')
|
|
319
398
|
.split('\n')
|
|
320
399
|
.filter((l) => l.length > 0)
|
|
321
|
-
.map((l) =>
|
|
322
|
-
|
|
400
|
+
.map((l) => {
|
|
401
|
+
const code = l.slice(0, 2);
|
|
402
|
+
const field = l.slice(3);
|
|
403
|
+
if (code.includes('R') || code.includes('C')) {
|
|
404
|
+
const split = splitRenamePorcelainField(field);
|
|
405
|
+
if (split) return { code, path: split.path, oldPath: split.oldPath };
|
|
406
|
+
}
|
|
407
|
+
return { code, path: unquotePorcelainPath(field) };
|
|
408
|
+
})
|
|
409
|
+
.filter((e) => e.path);
|
|
323
410
|
}
|
|
324
411
|
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
|
|
328
|
-
|
|
412
|
+
// Parse `git status --porcelain` output into a list of changed paths. Pure +
|
|
413
|
+
// exported for unit testing.
|
|
414
|
+
function parsePorcelain(stdout) {
|
|
415
|
+
return parsePorcelainEntries(stdout).map((e) => e.path);
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
// Same as uncommittedChanges but keeps each path's porcelain status code —
|
|
419
|
+
// the shared-tree guard's baseline needs this to tell "was untracked" apart
|
|
420
|
+
// from "was tracked-and-modified" (see evaluateSharedTreeGuard). Returns null
|
|
421
|
+
// when the guard does not apply (cwd is not a git work tree, git is missing,
|
|
422
|
+
// or the call errors); never throws — a guard failure must not fail an
|
|
423
|
+
// otherwise-successful job.
|
|
424
|
+
function uncommittedChangesWithStatus(cwd) {
|
|
329
425
|
return new Promise((resolve) => {
|
|
330
426
|
if (!cwd) { resolve(null); return; }
|
|
331
427
|
execFile(
|
|
@@ -333,13 +429,24 @@ function uncommittedChanges(cwd) {
|
|
|
333
429
|
['-C', cwd, 'status', '--porcelain'],
|
|
334
430
|
{ timeout: 10_000, windowsHide: true },
|
|
335
431
|
(err, stdout) => {
|
|
336
|
-
if (err) { resolve(null); return; }
|
|
337
|
-
resolve(
|
|
432
|
+
if (err) { resolve(null); return; }
|
|
433
|
+
resolve(parsePorcelainEntries(stdout));
|
|
338
434
|
},
|
|
339
435
|
);
|
|
340
436
|
});
|
|
341
437
|
}
|
|
342
438
|
|
|
439
|
+
// Return the list of uncommitted paths in cwd, or null under the same
|
|
440
|
+
// conditions as uncommittedChangesWithStatus (never throws). Kept as a thin
|
|
441
|
+
// path-only projection of that call rather than its own execFile, so a future
|
|
442
|
+
// fix to the git invocation (timeout, error handling) can't land in one and
|
|
443
|
+
// silently miss the other.
|
|
444
|
+
function uncommittedChanges(cwd) {
|
|
445
|
+
return uncommittedChangesWithStatus(cwd).then((entries) => (
|
|
446
|
+
entries === null ? null : entries.map((e) => e.path)
|
|
447
|
+
));
|
|
448
|
+
}
|
|
449
|
+
|
|
343
450
|
// Return the current HEAD commit sha in cwd, or null on any error. Used by the
|
|
344
451
|
// commit-guard to detect whether the job self-committed during its run (HEAD
|
|
345
452
|
// moved) — in which case leftover working-tree dirt is presumptively from a
|
|
@@ -428,17 +535,52 @@ function restoreSpecificStash(cwd, ref) {
|
|
|
428
535
|
// - reverted: a path that was dirty in the baseline, is clean now, and was
|
|
429
536
|
// not touched by any commit landed during the run — the job reset/
|
|
430
537
|
// checked-out over pre-existing uncommitted work without stashing it.
|
|
431
|
-
//
|
|
432
|
-
//
|
|
433
|
-
|
|
538
|
+
//
|
|
539
|
+
// A THIRD outcome is not a revert at all: an untracked path can drop out of
|
|
540
|
+
// `git status` because the run committed a `.gitignore` change that now
|
|
541
|
+
// matches it — the file is untouched on disk, just no longer visible to git
|
|
542
|
+
// (Incident: 2026-09-12, PRD 1181 added a bare `logs/` ignore pattern, eleven
|
|
543
|
+
// untracked `session-manager-operations/logs/*` paths vanished from status,
|
|
544
|
+
// and an otherwise-perfect run was parked in needs_review for a human who had
|
|
545
|
+
// nothing to decide). `dirtyBefore` entries therefore carry each path's
|
|
546
|
+
// porcelain status code (`{ code, path }`, from parsePorcelainEntries) so this
|
|
547
|
+
// function can tell "was untracked" apart from "was tracked-and-modified":
|
|
548
|
+
// - a TRACKED path (any code other than '??') leaving the dirty set always
|
|
549
|
+
// means its content was restored to HEAD — still `reverted`, even though
|
|
550
|
+
// the file still exists on disk, because for a tracked file "exists" is
|
|
551
|
+
// not the question; "matches what the human left uncommitted" is.
|
|
552
|
+
// - an UNTRACKED path ('??') leaving the dirty set is `reverted` only if it
|
|
553
|
+
// no longer exists on disk; if it still exists, it merely became ignored
|
|
554
|
+
// and is reported separately as `nowIgnored`.
|
|
555
|
+
// Plain path strings are still accepted in `dirtyBefore` for callers that
|
|
556
|
+
// have no status code (e.g. the stash-detection pass, which always passes an
|
|
557
|
+
// empty array) — an entry with no `code` is treated as tracked, matching the
|
|
558
|
+
// old behavior exactly.
|
|
559
|
+
//
|
|
560
|
+
// Pure/no I/O — status-code parsing and on-disk existence checks both happen
|
|
561
|
+
// at the call site (checkSharedTreeGuard); this function never stats the
|
|
562
|
+
// filesystem. Exported for unit testing.
|
|
563
|
+
function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun, existsAfter }) {
|
|
434
564
|
const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
|
|
435
565
|
const newStashes = (stashAfter || [])
|
|
436
566
|
.map(parseStashLine)
|
|
437
567
|
.filter((e) => e && !beforeHashes.has(e.hash));
|
|
438
568
|
const dirtyAfterSet = new Set(dirtyAfter || []);
|
|
439
569
|
const committedSet = new Set(pathsCommittedDuringRun || []);
|
|
440
|
-
const
|
|
441
|
-
|
|
570
|
+
const existsSet = new Set(existsAfter || []);
|
|
571
|
+
const reverted = [];
|
|
572
|
+
const nowIgnored = [];
|
|
573
|
+
for (const entry of dirtyBefore || []) {
|
|
574
|
+
const p = typeof entry === 'string' ? entry : entry.path;
|
|
575
|
+
const code = typeof entry === 'string' ? undefined : entry.code;
|
|
576
|
+
if (dirtyAfterSet.has(p) || committedSet.has(p)) continue;
|
|
577
|
+
if (code === '??' && existsSet.has(p)) {
|
|
578
|
+
nowIgnored.push(p);
|
|
579
|
+
} else {
|
|
580
|
+
reverted.push(p);
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
return { newStashes, reverted, nowIgnored };
|
|
442
584
|
}
|
|
443
585
|
|
|
444
586
|
// Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
|
|
@@ -488,18 +630,40 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
|
|
|
488
630
|
// restored stash is not ALSO reported as an unexplained revert (it was
|
|
489
631
|
// explained — by the stash this guard just restored).
|
|
490
632
|
const dirtyAfter = await module.exports.uncommittedChanges(cwd);
|
|
491
|
-
|
|
633
|
+
// Existence check for the "now ignored, not reverted" split (2026-09-12
|
|
634
|
+
// incident) — only untracked baseline entries need it; a tracked path
|
|
635
|
+
// leaving the dirty set is always a revert regardless of disk state (see
|
|
636
|
+
// evaluateSharedTreeGuard). Scoped to entries carrying a status code —
|
|
637
|
+
// plain path strings (no code) fall back to the old always-reverted path.
|
|
638
|
+
const untrackedBaselinePaths = (dirtyBaseline || [])
|
|
639
|
+
.filter((e) => e && typeof e === 'object' && e.code === '??')
|
|
640
|
+
.map((e) => e.path);
|
|
641
|
+
// A large untracked baseline (the 2026-09-12 incident's shared tree had
|
|
642
|
+
// ~240 such paths) makes this a lot of stat calls — fs.promises.access
|
|
643
|
+
// run concurrently instead of fs.existsSync run synchronously one at a
|
|
644
|
+
// time keeps this off the event loop instead of blocking every other
|
|
645
|
+
// in-flight scheduler/IPC task for the duration.
|
|
646
|
+
const existsChecks = await Promise.all(
|
|
647
|
+
untrackedBaselinePaths.map((p) => fsp.access(path.join(cwd, p)).then(() => true, () => false)),
|
|
648
|
+
);
|
|
649
|
+
const existsAfter = untrackedBaselinePaths.filter((_, i) => existsChecks[i]);
|
|
650
|
+
const { reverted, nowIgnored } = module.exports.evaluateSharedTreeGuard({
|
|
492
651
|
stashBefore: stashBaseline,
|
|
493
652
|
stashAfter,
|
|
494
653
|
dirtyBefore: dirtyBaseline,
|
|
495
654
|
dirtyAfter,
|
|
496
655
|
pathsCommittedDuringRun,
|
|
656
|
+
existsAfter,
|
|
497
657
|
});
|
|
498
658
|
if (reverted.length) {
|
|
499
659
|
result.reverted = reverted;
|
|
500
660
|
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
501
661
|
}
|
|
502
|
-
|
|
662
|
+
if (nowIgnored.length) {
|
|
663
|
+
result.nowIgnored = nowIgnored;
|
|
664
|
+
console.log(`[scheduler] ${slug}: shared-tree guard: ${nowIgnored.length} path(s) no longer shown by git status but still present on disk — likely a new ignore rule, not a revert (${nowIgnored.slice(0, 3).join(', ')})`);
|
|
665
|
+
}
|
|
666
|
+
return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted || result.nowIgnored) ? result : null;
|
|
503
667
|
} catch (e) {
|
|
504
668
|
console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
|
|
505
669
|
return null;
|
|
@@ -1151,8 +1315,9 @@ function ensureDirs() {
|
|
|
1151
1315
|
* reconcile()-level call is what makes "anything written to the retired flat
|
|
1152
1316
|
* prds/ dir is swept into prds-archived/ without being executed" actually
|
|
1153
1317
|
* true regardless of which of reconcile's several callers (tickQueue's poll,
|
|
1154
|
-
* job completion, the schedule:
|
|
1155
|
-
* rescheduleTimer) triggers the pass
|
|
1318
|
+
* job completion, the schedule:rescan/schedule:adopt-prd IPC handlers,
|
|
1319
|
+
* broadcast()'s coalescer, rescheduleTimer) triggers the pass — schedule:state
|
|
1320
|
+
* no longer reconciles on read. A PRD dropped in the flat dir has no
|
|
1156
1321
|
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
1157
1322
|
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
1158
1323
|
*/
|
|
@@ -1216,7 +1381,22 @@ async function runPrdMigration() {
|
|
|
1216
1381
|
// on every pass, but stays here so a fresh boot's very first log line
|
|
1217
1382
|
// still reports the initial sweep — see consolidateAllFlatPrds's own
|
|
1218
1383
|
// comment for why reconcile() is the load-bearing call site.)
|
|
1219
|
-
|
|
1384
|
+
//
|
|
1385
|
+
// Deferred off scheduler.init()'s synchronous critical path: allProjectCwds()
|
|
1386
|
+
// is a synchronous ~270ms directory scan, and awaiting it inline here
|
|
1387
|
+
// competed with the renderer's first IPC round trips (schedule.state,
|
|
1388
|
+
// billing.fetch, teams.list) for the event loop during boot. Dropping the
|
|
1389
|
+
// await doesn't weaken the consolidation guarantee — consolidateAllFlatPrds
|
|
1390
|
+
// also runs at the top of every reconcile() (see its own comment above),
|
|
1391
|
+
// and a flat PRD can only ever execute via tickQueue, which always
|
|
1392
|
+
// reconciles first, so nothing dropped in the flat dir can run before a
|
|
1393
|
+
// reconcile() pass sweeps it regardless of whether this boot-time pass has
|
|
1394
|
+
// finished yet.
|
|
1395
|
+
setImmediate(() => {
|
|
1396
|
+
consolidateAllFlatPrds(allProjectCwds()).catch((e) => {
|
|
1397
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'deferred flat-PRD consolidation failed', meta: { error: e?.message } });
|
|
1398
|
+
});
|
|
1399
|
+
});
|
|
1220
1400
|
|
|
1221
1401
|
// Rollout migration for the PRD-authoring-lockdown feature: stamp every
|
|
1222
1402
|
// pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
|
|
@@ -1484,14 +1664,15 @@ function computeBlockedChains(jobs) {
|
|
|
1484
1664
|
* findStaleQuarantinedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
1485
1665
|
*
|
|
1486
1666
|
* Pure, no IO. A 'quarantined' row (no createdVia provenance) can otherwise
|
|
1487
|
-
* sit forever with nothing looking at it — quarantine
|
|
1488
|
-
* human adopting or archiving it
|
|
1489
|
-
*
|
|
1490
|
-
*
|
|
1491
|
-
*
|
|
1492
|
-
*
|
|
1493
|
-
*
|
|
1494
|
-
*
|
|
1667
|
+
* sit forever with nothing looking at it — quarantine used to clear only via
|
|
1668
|
+
* a human adopting or archiving it; autoResolveQuarantine below now gives it
|
|
1669
|
+
* a bounded automatic exit too. This function stays the escalation/warn half
|
|
1670
|
+
* of that gate: any quarantined row whose recorded quarantine timestamp
|
|
1671
|
+
* (statusHistory's `to === 'quarantined'` entry — stamped at creation, or
|
|
1672
|
+
* backfilled from the PRD file's mtime by reconcile() for rows quarantined
|
|
1673
|
+
* before that stamp existed) is older than `thresholdMs` is reported so the
|
|
1674
|
+
* caller can warn-log and surface it distinctly. A row with no recoverable
|
|
1675
|
+
* timestamp is skipped rather than guessed at.
|
|
1495
1676
|
*/
|
|
1496
1677
|
function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
|
|
1497
1678
|
const stale = [];
|
|
@@ -1507,6 +1688,106 @@ function findStaleQuarantinedJobs(jobs, now, thresholdMs) {
|
|
|
1507
1688
|
return stale;
|
|
1508
1689
|
}
|
|
1509
1690
|
|
|
1691
|
+
// Bounded automatic exit for a quarantined row (this PRD): up to
|
|
1692
|
+
// QUARANTINE_RESOLVE_CAP auto-resolve attempts, each gated on having sat
|
|
1693
|
+
// `quarantined` for QUARANTINE_ESCALATE_MS, before autoResolveQuarantine
|
|
1694
|
+
// below settles the row to 'skipped' rather than leaving it as a dead end
|
|
1695
|
+
// only a human `scheduler_reset_job`/adopt action could ever clear. A single
|
|
1696
|
+
// attempt is enough in practice — the outcome is terminal — but the counter
|
|
1697
|
+
// still guards against two overlapping ticks both trying to resolve the
|
|
1698
|
+
// same row.
|
|
1699
|
+
const QUARANTINE_RESOLVE_CAP = 1;
|
|
1700
|
+
|
|
1701
|
+
/**
|
|
1702
|
+
* Kill-switch gate for the quarantine auto-resolve pass below
|
|
1703
|
+
* (SM_QUARANTINE_AUTORESOLVE_DISABLE=1), same shape as
|
|
1704
|
+
* failedAutoResetDisabled/needsReviewAutoResolveDisabled.
|
|
1705
|
+
*/
|
|
1706
|
+
function quarantineAutoResolveDisabled() {
|
|
1707
|
+
return process.env.SM_QUARANTINE_AUTORESOLVE_DISABLE === '1';
|
|
1708
|
+
}
|
|
1709
|
+
|
|
1710
|
+
/**
|
|
1711
|
+
* selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) →
|
|
1712
|
+
* [{ slug, cwd, ageMs }]
|
|
1713
|
+
*
|
|
1714
|
+
* Pure selector — no IO. Same age computation as findStaleQuarantinedJobs
|
|
1715
|
+
* above, bounded additionally by quarantineResolveAttempts so a row already
|
|
1716
|
+
* auto-resolved (or mid-resolve on a race) is never re-selected. Deliberately
|
|
1717
|
+
* does NOT check createdVia here — that requires a disk read of the PRD
|
|
1718
|
+
* file, and doing it at selection time would let this pass act on a
|
|
1719
|
+
* snapshot that's gone stale by the time the mutate() pass actually runs.
|
|
1720
|
+
* autoResolveQuarantine below re-reads createdVia fresh, immediately before
|
|
1721
|
+
* transitioning, inside the same mutate() callback that applies this
|
|
1722
|
+
* selector's targets — see that function's own header for why.
|
|
1723
|
+
*/
|
|
1724
|
+
function selectQuarantineAutoResolveTargets(jobs, now, thresholdMs) {
|
|
1725
|
+
const targets = [];
|
|
1726
|
+
for (const j of jobs ?? []) {
|
|
1727
|
+
if (j.status !== 'quarantined') continue;
|
|
1728
|
+
if ((j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue;
|
|
1729
|
+
const entry = (j.statusHistory || []).find((h) => h.to === 'quarantined');
|
|
1730
|
+
if (!entry) continue;
|
|
1731
|
+
const since = Date.parse(entry.at);
|
|
1732
|
+
if (Number.isNaN(since)) continue;
|
|
1733
|
+
const ageMs = now - since;
|
|
1734
|
+
if (ageMs < thresholdMs) continue;
|
|
1735
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs });
|
|
1736
|
+
}
|
|
1737
|
+
return targets;
|
|
1738
|
+
}
|
|
1739
|
+
|
|
1740
|
+
/**
|
|
1741
|
+
* autoResolveQuarantine(job, ageMs) → Promise<'skipped'|null>
|
|
1742
|
+
*
|
|
1743
|
+
* Applies the bounded automatic exit to a single quarantined row (mutates in
|
|
1744
|
+
* place; calls transitionJob + appendAuditEvent) — extracted so it's
|
|
1745
|
+
* unit-testable without going through mutate()/queue.json IO, same shape as
|
|
1746
|
+
* applyNeedsReviewAutoResolve above.
|
|
1747
|
+
*
|
|
1748
|
+
* Re-validates status + the attempts cap itself (race guard, mirrors the
|
|
1749
|
+
* other auto-resolve loops in the 10-minute interval body), THEN re-reads the
|
|
1750
|
+
* PRD file's createdVia frontmatter fresh from disk before doing anything
|
|
1751
|
+
* else. That ordering is load-bearing: reconcile()'s adopt path (the only
|
|
1752
|
+
* OTHER route off 'quarantined') promotes a row to 'pending' the instant it
|
|
1753
|
+
* observes a createdVia stamp, on its own independent pass — if this
|
|
1754
|
+
* function trusted a snapshot taken before its own turn to run, it could
|
|
1755
|
+
* transition a row to 'skipped' the same tick reconcile() already adopted it
|
|
1756
|
+
* to 'pending', silently discarding a PRD a human just fixed. Checking here,
|
|
1757
|
+
* immediately before the transition, inside the caller's mutate() callback,
|
|
1758
|
+
* closes that window.
|
|
1759
|
+
*
|
|
1760
|
+
* A PRD file that cannot be found or parsed at all is treated as still
|
|
1761
|
+
* lacking provenance — there is no proof it has one, and stalling forever on
|
|
1762
|
+
* an unreadable file would defeat the point of a bounded exit (same
|
|
1763
|
+
* can't-prove-it/don't-guess-but-don't-stall posture as the rest of this
|
|
1764
|
+
* file's stale-row detectors).
|
|
1765
|
+
*/
|
|
1766
|
+
async function autoResolveQuarantine(job, ageMs) {
|
|
1767
|
+
if (!job || job.status !== 'quarantined') return null;
|
|
1768
|
+
if ((job.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) return null;
|
|
1769
|
+
|
|
1770
|
+
let createdVia = null;
|
|
1771
|
+
try {
|
|
1772
|
+
const resolvedDir = await findPrdDir(job.slug);
|
|
1773
|
+
const prdPath = resolvedDir ? path.join(resolvedDir, `${job.slug}.md`) : prdPathForJob(job);
|
|
1774
|
+
const parsed = await parsePrd(prdPath);
|
|
1775
|
+
createdVia = parsed.createdVia ?? null;
|
|
1776
|
+
} catch { /* unreadable/gone — no provenance found, so it stays "lacking" */ }
|
|
1777
|
+
if (createdVia) return null; // reconcile()'s own adopt path owns this row now
|
|
1778
|
+
|
|
1779
|
+
const attempt = (job.quarantineResolveAttempts ?? 0) + 1;
|
|
1780
|
+
job.quarantineResolveAttempts = attempt;
|
|
1781
|
+
job.error = `quarantined without createdVia provenance past the ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h `
|
|
1782
|
+
+ 'escalation window — auto-resolved to skipped';
|
|
1783
|
+
transitionJob(job, 'skipped', {
|
|
1784
|
+
reason: 'quarantined without createdVia provenance past escalation window',
|
|
1785
|
+
source: 'autoResolveQuarantine',
|
|
1786
|
+
});
|
|
1787
|
+
appendAuditEvent('quarantine_auto_resolved', { slug: job.slug, cwd: job.cwd ?? null, ageMs: ageMs ?? null, attempt });
|
|
1788
|
+
return 'skipped';
|
|
1789
|
+
}
|
|
1790
|
+
|
|
1510
1791
|
/**
|
|
1511
1792
|
* findOverrunningJobs(jobs, now, { factor, floorMs }) → [{ slug, cwd, estimateMinutes, ranMs, ratio }]
|
|
1512
1793
|
*
|
|
@@ -1553,6 +1834,99 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
|
|
|
1553
1834
|
return out;
|
|
1554
1835
|
}
|
|
1555
1836
|
|
|
1837
|
+
/**
|
|
1838
|
+
* computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs }) → number
|
|
1839
|
+
*
|
|
1840
|
+
* Pure. `budgetMs = clamp(estimateMinutes * factor, floorMs, ceilingMs)` — see
|
|
1841
|
+
* JOB_BUDGET_FACTOR's header comment (schedulerConfig.cjs) for the measured
|
|
1842
|
+
* p50/p90/max this is calibrated against. A missing/zero/non-finite estimate
|
|
1843
|
+
* is treated as 0, which the floor clamp then dominates — "jobs with a
|
|
1844
|
+
* missing estimate get the floor" falls straight out of the clamp, no
|
|
1845
|
+
* special-casing needed.
|
|
1846
|
+
*/
|
|
1847
|
+
function computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs } = {}) {
|
|
1848
|
+
const f = typeof factor === 'number' && factor > 0 ? factor : JOB_BUDGET_FACTOR;
|
|
1849
|
+
const floor = typeof floorMs === 'number' && floorMs >= 0 ? floorMs : JOB_BUDGET_FLOOR_MS;
|
|
1850
|
+
const ceiling = typeof ceilingMs === 'number' && ceilingMs > 0 ? ceilingMs : JOB_BUDGET_CEILING_MS;
|
|
1851
|
+
const est = Number(estimateMinutes);
|
|
1852
|
+
const minutes = Number.isFinite(est) && est > 0 ? est : 0;
|
|
1853
|
+
return Math.min(Math.max(minutes * f * 60_000, floor), ceiling);
|
|
1854
|
+
}
|
|
1855
|
+
|
|
1856
|
+
/**
|
|
1857
|
+
* classifyBudgetKill(res, landedCommitEvidence) → { status, reason, landedCommit } | null
|
|
1858
|
+
*
|
|
1859
|
+
* Pure. `res` is executeJob's resolved outcome — only fires when
|
|
1860
|
+
* `res.killedByWatchdog === 'budget'` (stamped by the budget watchdog inside
|
|
1861
|
+
* executeJob, never inferred from exit code/duration alone, so it can never
|
|
1862
|
+
* collide with an ordinary idle-tail/deadman/external kill). ALWAYS routes to
|
|
1863
|
+
* needs_review — never 'failed' (spawnJob's ordinary non-zero-exit default)
|
|
1864
|
+
* and never silently 'completed' (executeJob's onExit excludes
|
|
1865
|
+
* killedByWatchdog === 'budget' from the result=success → exit 0 mapping
|
|
1866
|
+
* idle-tail/deadman get) — so a budget kill is a visible, actionable park,
|
|
1867
|
+
* never a retry (classifyFailureOutcome/selectAutoFixTargets only ever see
|
|
1868
|
+
* 'failed'/ordinary needs_review rows, not this one — see
|
|
1869
|
+
* selectAutoFixTargets' own budget_exceeded exclusion) and never a discard:
|
|
1870
|
+
* `landedCommitEvidence`, when the caller resolved one via the SAME
|
|
1871
|
+
* commit-guard evidence check the plain sigterm/exit paths already use, is
|
|
1872
|
+
* threaded onto the row as `landedCommit` so a job that HAD already
|
|
1873
|
+
* committed before overrunning is still adjudicated on its git evidence.
|
|
1874
|
+
*/
|
|
1875
|
+
function classifyBudgetKill(res, landedCommitEvidence) {
|
|
1876
|
+
if (!res || res.killedByWatchdog !== 'budget') return null;
|
|
1877
|
+
return {
|
|
1878
|
+
status: 'needs_review',
|
|
1879
|
+
reason: res.budgetKillReason || `wall-clock budget exceeded (exit ${res.exitCode})`,
|
|
1880
|
+
landedCommit: landedCommitEvidence || null,
|
|
1881
|
+
};
|
|
1882
|
+
}
|
|
1883
|
+
|
|
1884
|
+
/**
|
|
1885
|
+
* isJobBudgetExempt(job) → boolean
|
|
1886
|
+
*
|
|
1887
|
+
* Pure. `quietMachine: true` PRDs (their whole point is running alone,
|
|
1888
|
+
* un-contended, for a timing-sensitive measurement) and any PRD with an
|
|
1889
|
+
* explicit `budgetExempt: true` opt-out have no wall-clock kill ceiling.
|
|
1890
|
+
*/
|
|
1891
|
+
function isJobBudgetExempt(job) {
|
|
1892
|
+
return job?.quietMachine === true || job?.budgetExempt === true;
|
|
1893
|
+
}
|
|
1894
|
+
|
|
1895
|
+
/** Pure predicate the budget watchdog's shouldFire calls — single source of
|
|
1896
|
+
* truth for "has this job run past its own budget" so it's unit-testable
|
|
1897
|
+
* without spinning up real timers. */
|
|
1898
|
+
function shouldKillForBudget(elapsedMs, budgetMs) {
|
|
1899
|
+
return elapsedMs >= budgetMs;
|
|
1900
|
+
}
|
|
1901
|
+
|
|
1902
|
+
/**
|
|
1903
|
+
* resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes })
|
|
1904
|
+
* → { killedByWatchdog: 'budget'|null, budgetKillReason: string|null }
|
|
1905
|
+
*
|
|
1906
|
+
* Pure. `ctx.killedByWatchdog` is stamped by the budget watchdog's action()
|
|
1907
|
+
* the instant its periodic shouldFire() observes elapsedMs >= jobBudgetMs —
|
|
1908
|
+
* but that setInterval tick and the child's real 'exit' event both run on
|
|
1909
|
+
* the SAME single-threaded event loop, so it's possible to observe the
|
|
1910
|
+
* budget threshold crossed and call ctx.killTree() in the same window the
|
|
1911
|
+
* agent happens to exit cleanly (exit 0) or fails on its own for an
|
|
1912
|
+
* unrelated reason — ctx.killTree() against an already-exited pid is a
|
|
1913
|
+
* silent no-op (ESRCH, caught), but the flag would still read 'budget'
|
|
1914
|
+
* unless gated here. Only trusted when the exit SHAPE actually looks like a
|
|
1915
|
+
* signal kill (killedBySignal — mirrors the exact same check onExit already
|
|
1916
|
+
* uses for its own mappedToSuccess exclusion), so a clean exit=0 or an
|
|
1917
|
+
* ordinary unrelated non-zero failure racing the watchdog's tick is never
|
|
1918
|
+
* misclassified as a budget kill downstream.
|
|
1919
|
+
*/
|
|
1920
|
+
function resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes }) {
|
|
1921
|
+
if (killedByWatchdog !== 'budget' || !killedBySignal) {
|
|
1922
|
+
return { killedByWatchdog: killedByWatchdog === 'budget' ? null : (killedByWatchdog ?? null), budgetKillReason: null };
|
|
1923
|
+
}
|
|
1924
|
+
return {
|
|
1925
|
+
killedByWatchdog: 'budget',
|
|
1926
|
+
budgetKillReason: `wall-clock budget exceeded: ran ${Math.round(durationMs / 60_000)}m against a ${Math.round(jobBudgetMs / 60_000)}m budget (estimateMinutes=${estimateMinutes ?? 0})`,
|
|
1927
|
+
};
|
|
1928
|
+
}
|
|
1929
|
+
|
|
1556
1930
|
/**
|
|
1557
1931
|
* findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
1558
1932
|
* → [{ slug, cwd, ageMs, restoreStatus }]
|
|
@@ -1772,7 +2146,7 @@ async function listPrdFiles() {
|
|
|
1772
2146
|
ensureDirs();
|
|
1773
2147
|
const dirs = candidatePrdsDirs();
|
|
1774
2148
|
const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
|
|
1775
|
-
return perDir.flat().sort();
|
|
2149
|
+
return { files: perDir.flat().sort(), dirCount: dirs.length };
|
|
1776
2150
|
}
|
|
1777
2151
|
|
|
1778
2152
|
/**
|
|
@@ -1916,16 +2290,31 @@ async function reconcile(state) {
|
|
|
1916
2290
|
if (state && state.unreadable) {
|
|
1917
2291
|
throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
|
|
1918
2292
|
}
|
|
2293
|
+
// Per-phase timing (PRD: reconcile evidence trail) — plain Date.now() diffs,
|
|
2294
|
+
// matching the ad-hoc elapsedMs idiom already used in health.cjs/
|
|
2295
|
+
// definitionOfDone.cjs. Only logged when the total exceeds
|
|
2296
|
+
// RECONCILE_SLOW_PASS_MS (see the warn emission at the bottom of this
|
|
2297
|
+
// function); a normal-speed pass logs nothing.
|
|
2298
|
+
const reconcileStartMs = Date.now();
|
|
2299
|
+
const phaseMs = {};
|
|
2300
|
+
|
|
1919
2301
|
// Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
|
|
1920
|
-
// has several callers besides tickQueue's ~60s poll (broadcast
|
|
1921
|
-
// rescheduleTimer,
|
|
2302
|
+
// has several callers besides tickQueue's ~60s poll (broadcast()'s
|
|
2303
|
+
// coalescer, rescheduleTimer, schedule:rescan, schedule:adopt-prd) — this
|
|
1922
2304
|
// lives here, not in any one caller, so the "a hand-written PRD in the flat
|
|
1923
2305
|
// dir is swept before it can become a job" guarantee holds regardless of
|
|
1924
2306
|
// which caller triggers this reconcile pass. A freshly hand-written file
|
|
1925
2307
|
// has no queue row yet, so it is never "live" and gets archived here
|
|
1926
2308
|
// instead of ever reaching the onDisk scan below.
|
|
2309
|
+
let phaseStartMs = Date.now();
|
|
1927
2310
|
await consolidateAllFlatPrds(allProjectCwds());
|
|
1928
|
-
|
|
2311
|
+
phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
|
|
2312
|
+
|
|
2313
|
+
phaseStartMs = Date.now();
|
|
2314
|
+
const { files, dirCount } = await listPrdFiles();
|
|
2315
|
+
phaseMs.prdDirResolve = Date.now() - phaseStartMs;
|
|
2316
|
+
|
|
2317
|
+
phaseStartMs = Date.now();
|
|
1929
2318
|
const onDisk = new Map();
|
|
1930
2319
|
for (const f of files) {
|
|
1931
2320
|
try {
|
|
@@ -1937,6 +2326,7 @@ async function reconcile(state) {
|
|
|
1937
2326
|
console.warn('[scheduler] failed to parse', f, e?.message);
|
|
1938
2327
|
}
|
|
1939
2328
|
}
|
|
2329
|
+
phaseMs.parseLoop = Date.now() - phaseStartMs;
|
|
1940
2330
|
|
|
1941
2331
|
const next = [];
|
|
1942
2332
|
const seen = new Set();
|
|
@@ -2002,7 +2392,9 @@ async function reconcile(state) {
|
|
|
2002
2392
|
// membership, so moving the file between Epic dirs must re-point the row.
|
|
2003
2393
|
epicId: p.epicId ?? job.epicId ?? null,
|
|
2004
2394
|
dependsOn: p.dependsOn,
|
|
2395
|
+
disposition: p.disposition ?? null,
|
|
2005
2396
|
quietMachine: p.quietMachine === true,
|
|
2397
|
+
budgetExempt: p.budgetExempt === true,
|
|
2006
2398
|
originSessionId: job.originSessionId
|
|
2007
2399
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
2008
2400
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
@@ -2057,9 +2449,11 @@ async function reconcile(state) {
|
|
|
2057
2449
|
// ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
|
|
2058
2450
|
// see the repair pass below, right after historyBySlug is available.
|
|
2059
2451
|
const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
|
|
2452
|
+
phaseStartMs = Date.now();
|
|
2060
2453
|
const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
|
|
2061
2454
|
? await queueHistory.historyTerminalBySlug()
|
|
2062
2455
|
: new Map();
|
|
2456
|
+
phaseMs.historyLookup = Date.now() - phaseStartMs;
|
|
2063
2457
|
|
|
2064
2458
|
// Backfill: any terminal job dropped above whose slug isn't already in
|
|
2065
2459
|
// history.jsonl gets written now, before its row is gone for good. This is
|
|
@@ -2115,7 +2509,9 @@ async function reconcile(state) {
|
|
|
2115
2509
|
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
2116
2510
|
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
2117
2511
|
dependsOn: p.dependsOn,
|
|
2512
|
+
disposition: p.disposition ?? null,
|
|
2118
2513
|
quietMachine: p.quietMachine === true,
|
|
2514
|
+
budgetExempt: p.budgetExempt === true,
|
|
2119
2515
|
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
2120
2516
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2121
2517
|
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
@@ -2237,7 +2633,9 @@ async function reconcile(state) {
|
|
|
2237
2633
|
sourceTabId: p.sourceTabId,
|
|
2238
2634
|
epicId: p.epicId ?? null,
|
|
2239
2635
|
dependsOn: p.dependsOn,
|
|
2636
|
+
disposition: p.disposition ?? null,
|
|
2240
2637
|
quietMachine: p.quietMachine === true,
|
|
2638
|
+
budgetExempt: p.budgetExempt === true,
|
|
2241
2639
|
originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
2242
2640
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2243
2641
|
agentType: p.agentType ?? null,
|
|
@@ -2334,7 +2732,8 @@ async function reconcile(state) {
|
|
|
2334
2732
|
// small. Append BEFORE dropping so a crash between the two can't lose a
|
|
2335
2733
|
// record — appendHistory dedupes by slug+runId, so a replay of the same
|
|
2336
2734
|
// batch on next boot is a safe no-op.
|
|
2337
|
-
|
|
2735
|
+
phaseStartMs = Date.now();
|
|
2736
|
+
const nowMs = phaseStartMs;
|
|
2338
2737
|
const { hot, toArchive } = queueHistory.partitionJobs(sorted, nowMs);
|
|
2339
2738
|
if (toArchive.length > 0) {
|
|
2340
2739
|
await queueHistory.appendHistory(toArchive);
|
|
@@ -2370,6 +2769,17 @@ async function reconcile(state) {
|
|
|
2370
2769
|
} catch (e) {
|
|
2371
2770
|
console.warn('[scheduler] autoArchiveCompleted failed', e?.message);
|
|
2372
2771
|
}
|
|
2772
|
+
phaseMs.queueWrite = Date.now() - phaseStartMs;
|
|
2773
|
+
|
|
2774
|
+
const totalMs = Date.now() - reconcileStartMs;
|
|
2775
|
+
if (totalMs > RECONCILE_SLOW_PASS_MS) {
|
|
2776
|
+
logs.writeLine({
|
|
2777
|
+
level: 'warn',
|
|
2778
|
+
scope: 'scheduler',
|
|
2779
|
+
message: `reconcile() pass took ${totalMs}ms (threshold ${RECONCILE_SLOW_PASS_MS}ms)`,
|
|
2780
|
+
meta: { totalMs, phaseMs, prdFileCount: files.length, resolvedDirCount: dirCount },
|
|
2781
|
+
});
|
|
2782
|
+
}
|
|
2373
2783
|
|
|
2374
2784
|
return state;
|
|
2375
2785
|
}
|
|
@@ -2559,20 +2969,36 @@ function applyPauseCleared(wasPaused, token) {
|
|
|
2559
2969
|
return token;
|
|
2560
2970
|
}
|
|
2561
2971
|
|
|
2972
|
+
/**
|
|
2973
|
+
* Human-readable explanation for a `reason: 'load-deferred'` tick, surfaced
|
|
2974
|
+
* to the renderer via lastTick.detail. Names the gate, the measured ratio,
|
|
2975
|
+
* the threshold and how long the stretch has been held — the box could sit
|
|
2976
|
+
* gated for 80+ minutes with nothing in the UI naming why (PRD: load gate
|
|
2977
|
+
* hysteresis). Pure so it's unit-testable without driving tickQueue's full
|
|
2978
|
+
* fs/worktree machinery.
|
|
2979
|
+
*/
|
|
2980
|
+
function formatLoadGateDetail(load) {
|
|
2981
|
+
const heldMinutes = Math.round(load.gatedSinceMs / 60_000);
|
|
2982
|
+
return `CPU load gate: loadavg1 ${load.loadavg1} / ${load.cores} cores = ${load.ratio} > threshold ${load.threshold}, held for ${heldMinutes}m`;
|
|
2983
|
+
}
|
|
2984
|
+
|
|
2562
2985
|
function attachWindow(w) { mainWindow = w; }
|
|
2563
2986
|
|
|
2564
2987
|
/**
|
|
2565
2988
|
* Build the snapshot payload consumed by both the `schedule:state` IPC
|
|
2566
|
-
* handler and the `schedule:state` broadcast event.
|
|
2567
|
-
* `paths` map (renderer uses it for "open folder" actions); broadcast omits
|
|
2568
|
-
* it because subscribers don't need to re-derive paths on every tick.
|
|
2989
|
+
* handler and the `schedule:state` broadcast event.
|
|
2569
2990
|
*/
|
|
2570
|
-
function buildScheduleStatePayload(state
|
|
2991
|
+
function buildScheduleStatePayload(state) {
|
|
2571
2992
|
const payload = {
|
|
2572
2993
|
config: state.config,
|
|
2573
2994
|
jobs: state.jobs,
|
|
2574
2995
|
scheduledFor: state.scheduledFor,
|
|
2575
2996
|
lastRunAt: state.lastRunAt,
|
|
2997
|
+
// Distinct from lastRunAt (only stamped when a batch actually launches):
|
|
2998
|
+
// stamped every time tickQueue reaches the picker at all. See
|
|
2999
|
+
// classifyQueueHealth/classifyQueueStarvation's header comments for why
|
|
3000
|
+
// the two must never merge.
|
|
3001
|
+
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
2576
3002
|
nextReset: getNextResetCached(),
|
|
2577
3003
|
paused: state.paused,
|
|
2578
3004
|
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
@@ -2603,9 +3029,6 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
2603
3029
|
};
|
|
2604
3030
|
})(),
|
|
2605
3031
|
};
|
|
2606
|
-
if (withPaths) {
|
|
2607
|
-
payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
|
|
2608
|
-
}
|
|
2609
3032
|
return payload;
|
|
2610
3033
|
}
|
|
2611
3034
|
|
|
@@ -2616,20 +3039,33 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
2616
3039
|
// per mutation. Callers where latency matters (pause/resume, job
|
|
2617
3040
|
// start/finish/reap/reset) pass `{ flush: true }` to bypass the window and
|
|
2618
3041
|
// send immediately.
|
|
3042
|
+
// getPayload is the coalescer's ONLY entry point back into queue state, so
|
|
3043
|
+
// routing the reconcile+write pair through it (rather than broadcast() doing
|
|
3044
|
+
// its own bare readQueue/reconcile/writeQueue) is what makes a burst of
|
|
3045
|
+
// broadcast() calls cost exactly one reconcile + one write per coalesce
|
|
3046
|
+
// window. It goes through mutate() (not a bare read/write pair) so it
|
|
3047
|
+
// serializes against every other concurrent mutation and inherits mutate's
|
|
3048
|
+
// pre-fn `state.unreadable` bail — never write a state derived from a failed
|
|
3049
|
+
// read. `module.exports.reconcile` (not the bare local binding) is the seam
|
|
3050
|
+
// tests spy on, matching this file's existing testable-seam convention (see
|
|
3051
|
+
// module.exports.stashList/evaluateSharedTreeGuard/committedInWindow above).
|
|
2619
3052
|
const broadcastCoalescer = createBroadcastCoalescer({
|
|
2620
3053
|
delayMs: BROADCAST_COALESCE_MS,
|
|
2621
3054
|
send: (payload) => {
|
|
2622
3055
|
if (!mainWindow || mainWindow.isDestroyed()) return;
|
|
2623
3056
|
sendIfAlive(mainWindow, 'schedule:state', payload);
|
|
2624
3057
|
},
|
|
2625
|
-
getPayload:
|
|
3058
|
+
getPayload: () => mutate(async (state) => {
|
|
3059
|
+
await module.exports.reconcile(state);
|
|
3060
|
+
return buildScheduleStatePayload(state);
|
|
3061
|
+
}),
|
|
2626
3062
|
});
|
|
2627
3063
|
|
|
3064
|
+
// Reconcile is unconditional — even with no window attached (a
|
|
3065
|
+
// scheduler-passive or headless instance), discovery must still run so
|
|
3066
|
+
// on-disk PRDs get onboarded. Only the actual IPC push is window-gated,
|
|
3067
|
+
// inside the coalescer's own `send`.
|
|
2628
3068
|
async function broadcast(opts = {}) {
|
|
2629
|
-
if (!mainWindow || mainWindow.isDestroyed()) return;
|
|
2630
|
-
const state = await readQueue();
|
|
2631
|
-
await reconcile(state);
|
|
2632
|
-
await writeQueue(state);
|
|
2633
3069
|
if (opts.flush) {
|
|
2634
3070
|
await broadcastCoalescer.flush();
|
|
2635
3071
|
} else {
|
|
@@ -2929,7 +3365,7 @@ const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
|
|
|
2929
3365
|
* process may still be writing to it, so reading now risks misclassifying a
|
|
2930
3366
|
* job that is about to emit result:success as no_result and double-running it.
|
|
2931
3367
|
* Ported from reconcileQueueOffline's cross-tick escalation (see
|
|
2932
|
-
*
|
|
3368
|
+
* src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
|
|
2933
3369
|
* this process stays up to revisit it, rather than a separate short-lived
|
|
2934
3370
|
* watchdog process needing another tick.
|
|
2935
3371
|
*/
|
|
@@ -2949,17 +3385,26 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
|
|
|
2949
3385
|
}
|
|
2950
3386
|
|
|
2951
3387
|
/**
|
|
2952
|
-
* applyOrphanOutcome(job, outcome, killNote?) → void
|
|
3388
|
+
* applyOrphanOutcome(job, outcome, killNote?, confirmedLandedCommit?) → void
|
|
2953
3389
|
*
|
|
2954
3390
|
* Mutates `job` in place to finalize a boot-orphaned 'running' job given its
|
|
2955
3391
|
* classified run outcome: success/failed finalize terminally; no_result/unknown
|
|
2956
3392
|
* re-queues to pending bounded by ORPHAN_REQUEUE_CAP. The status-mutation
|
|
2957
3393
|
* semantics (and the cap-exhaustion boundary) match the now-deleted
|
|
2958
|
-
* reconcileQueueOffline (
|
|
3394
|
+
* reconcileQueueOffline (src/main/lib/watchdogHelpers.cjs) verbatim; killNote
|
|
2959
3395
|
* plumbing differs slightly (see call sites) since this path always knows
|
|
2960
3396
|
* pid liveness up front rather than re-checking per tick.
|
|
3397
|
+
*
|
|
3398
|
+
* `confirmedLandedCommit` is the same evidence-before-failure gate
|
|
3399
|
+
* reapDeadRunningJobs applies (see resolveLandedCommitEvidence): a job that
|
|
3400
|
+
* dies while the app itself is offline is classified 'failed' from its log
|
|
3401
|
+
* tail alone, exactly like the pre-fix reap path was — so without this, an
|
|
3402
|
+
* orphaned job that actually landed a real commit is reachable via boot
|
|
3403
|
+
* reconciliation even though the live reap path is now guarded. Callers
|
|
3404
|
+
* must resolve this (a git spawn) BEFORE calling mutate(), never inside it —
|
|
3405
|
+
* pass null to skip the gate (e.g. when the outcome isn't 'failed').
|
|
2961
3406
|
*/
|
|
2962
|
-
function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
3407
|
+
function applyOrphanOutcome(job, outcome, killNote = '', confirmedLandedCommit = null) {
|
|
2963
3408
|
const now = new Date().toISOString();
|
|
2964
3409
|
if (outcome === 'success') {
|
|
2965
3410
|
transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
|
|
@@ -2968,9 +3413,16 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
2968
3413
|
job.finishedAt = now;
|
|
2969
3414
|
delete job.runtime;
|
|
2970
3415
|
} else if (outcome === 'failed') {
|
|
2971
|
-
|
|
2972
|
-
|
|
2973
|
-
|
|
3416
|
+
if (confirmedLandedCommit) {
|
|
3417
|
+
transitionJob(job, 'completed', { reason: `orphaned: app restarted while running${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
|
|
3418
|
+
job.exitCode = 0;
|
|
3419
|
+
job.error = null;
|
|
3420
|
+
job.landedCommit = confirmedLandedCommit;
|
|
3421
|
+
} else {
|
|
3422
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
|
|
3423
|
+
job.exitCode = job.exitCode ?? 1;
|
|
3424
|
+
job.error = `orphaned: app restarted while running${killNote}`;
|
|
3425
|
+
}
|
|
2974
3426
|
job.finishedAt = now;
|
|
2975
3427
|
delete job.runtime;
|
|
2976
3428
|
} else {
|
|
@@ -2978,6 +3430,13 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
2978
3430
|
if (tries < ORPHAN_REQUEUE_CAP) {
|
|
2979
3431
|
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
|
|
2980
3432
|
job.orphanRetries = tries + 1;
|
|
3433
|
+
} else if (confirmedLandedCommit) {
|
|
3434
|
+
transitionJob(job, 'completed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
|
|
3435
|
+
job.exitCode = 0;
|
|
3436
|
+
job.error = null;
|
|
3437
|
+
job.landedCommit = confirmedLandedCommit;
|
|
3438
|
+
job.finishedAt = now;
|
|
3439
|
+
delete job.runtime;
|
|
2981
3440
|
} else {
|
|
2982
3441
|
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
|
|
2983
3442
|
job.exitCode = job.exitCode ?? 1;
|
|
@@ -3865,6 +4324,60 @@ async function pathExistsInTree(cwd, treeish, p) {
|
|
|
3865
4324
|
}
|
|
3866
4325
|
}
|
|
3867
4326
|
|
|
4327
|
+
/**
|
|
4328
|
+
* resolveLandedCommitEvidence(cwd, sha, sinceIso) → Promise<boolean>
|
|
4329
|
+
*
|
|
4330
|
+
* Bounded, non-fatal proof that `sha` is a real, resolvable commit in the
|
|
4331
|
+
* repo at `cwd`, committed no earlier than `sinceIso` — `git cat-file -e
|
|
4332
|
+
* <sha>^{commit}` plus a `git log -1 --format=%cI` timestamp check, both via
|
|
4333
|
+
* execGitAt's existing spawn+timeout bound (never shell:true, never an
|
|
4334
|
+
* unbounded execSync). This is the evidence gate reapDeadRunningJobs (PRD:
|
|
4335
|
+
* reaper must consult completion evidence) adds ahead of stamping a reaped
|
|
4336
|
+
* row 'failed': a landedCommit field being non-empty is not proof by itself
|
|
4337
|
+
* (job 1192 had one and still got reaped 'failed') — only a git-verified
|
|
4338
|
+
* resolution is.
|
|
4339
|
+
*
|
|
4340
|
+
* The timestamp bound matters because `landedCommit` deliberately survives
|
|
4341
|
+
* resetJobFields (see the comment there) so a re-fired run can consult it as
|
|
4342
|
+
* priorLandedCommit — which means a STALE landedCommit from an earlier
|
|
4343
|
+
* dispatch of the same slug can still be sitting on the row when a LATER
|
|
4344
|
+
* dispatch dies for real. Without `sinceIso`, that stale-but-real sha would
|
|
4345
|
+
* satisfy `cat-file -e` and wrongly promote a genuine failure to
|
|
4346
|
+
* 'completed'. Passing the current dispatch's `row.startedAt` as `sinceIso`
|
|
4347
|
+
* closes that: only a commit landed during THIS run counts as evidence.
|
|
4348
|
+
*
|
|
4349
|
+
* `cwd` is normalized through opsOwnership's resolveProjectRoot first (the
|
|
4350
|
+
* same "never trust a raw agent cwd" reasoning delegationReadiness.cjs
|
|
4351
|
+
* already relies on) so a row reaped while its cwd is an ephemeral worktree
|
|
4352
|
+
* checkout resolves the commit against the real project root instead.
|
|
4353
|
+
*
|
|
4354
|
+
* Never throws: a missing sha, a resolveProjectRoot failure (ephemeral cwd,
|
|
4355
|
+
* thrown error), a spawn failure, a timeout, or a cwd that no longer exists
|
|
4356
|
+
* on disk all resolve to `false` — the caller's safe default is 'failed',
|
|
4357
|
+
* exactly like today, whenever this can't positively prove landing.
|
|
4358
|
+
*/
|
|
4359
|
+
async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
|
|
4360
|
+
if (!sha || typeof sha !== 'string') return false;
|
|
4361
|
+
try {
|
|
4362
|
+
const root = resolveProjectRoot(cwd);
|
|
4363
|
+
await execGitAt(root, ['cat-file', '-e', `${sha}^{commit}`], { timeout: 10_000 });
|
|
4364
|
+
if (sinceIso) {
|
|
4365
|
+
const since = new Date(sinceIso).getTime();
|
|
4366
|
+
if (Number.isFinite(since)) {
|
|
4367
|
+
const committedIso = (await execGitAt(root, ['log', '-1', '--format=%cI', sha], { timeout: 10_000 })).trim();
|
|
4368
|
+
const committedAt = new Date(committedIso).getTime();
|
|
4369
|
+
// A commit dated before this dispatch even started can only be a
|
|
4370
|
+
// stale sha surviving from an earlier life of the row — never
|
|
4371
|
+
// evidence that THIS dispatch landed anything.
|
|
4372
|
+
if (Number.isFinite(committedAt) && committedAt < since) return false;
|
|
4373
|
+
}
|
|
4374
|
+
}
|
|
4375
|
+
return true;
|
|
4376
|
+
} catch {
|
|
4377
|
+
return false;
|
|
4378
|
+
}
|
|
4379
|
+
}
|
|
4380
|
+
|
|
3868
4381
|
/**
|
|
3869
4382
|
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
3870
4383
|
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
@@ -4407,6 +4920,56 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4407
4920
|
},
|
|
4408
4921
|
};
|
|
4409
4922
|
|
|
4923
|
+
// Wall-clock budget watchdog: unlike idleTailWatchdog above (which only
|
|
4924
|
+
// fires when the log mtime STALLS), this fires on total elapsed wall-
|
|
4925
|
+
// clock time regardless of whether the job keeps writing output — the
|
|
4926
|
+
// gap a chatty-but-runaway executor slips through (see
|
|
4927
|
+
// computeJobBudgetMs's header for the measured p50/p90/max this budget
|
|
4928
|
+
// is calibrated against). `quietMachine` jobs and any PRD with an
|
|
4929
|
+
// explicit `budgetExempt: true` opt out entirely — logged once here so
|
|
4930
|
+
// an unbounded job is never silently unbounded.
|
|
4931
|
+
const jobBudgetMs = computeJobBudgetMs(job.estimateMinutes);
|
|
4932
|
+
const budgetExempt = isJobBudgetExempt(job);
|
|
4933
|
+
if (budgetExempt) {
|
|
4934
|
+
safeLog(`[scheduler] wall-clock budget watchdog EXEMPT for ${job.slug} ` +
|
|
4935
|
+
`(${job.quietMachine === true ? 'quietMachine' : 'budgetExempt'}) — no wall-clock kill ceiling this run\n`);
|
|
4936
|
+
}
|
|
4937
|
+
let budgetWarningStamped = false;
|
|
4938
|
+
const budgetWatchdog = {
|
|
4939
|
+
label: 'budget',
|
|
4940
|
+
intervalMs: IDLE_CHECK_INTERVAL_MS,
|
|
4941
|
+
shouldFire(ctx) {
|
|
4942
|
+
if (budgetExempt) return false;
|
|
4943
|
+
const elapsedMs = Date.now() - ctx.startedAt;
|
|
4944
|
+
if (!budgetWarningStamped && elapsedMs >= jobBudgetMs * BUDGET_WARNING_FRACTION) {
|
|
4945
|
+
budgetWarningStamped = true;
|
|
4946
|
+
// Fire-and-forget (side effect inside a sync predicate, same pattern
|
|
4947
|
+
// resultTailWatchdog's shouldFire already uses for agentResultSubtype)
|
|
4948
|
+
// — exposes the warning on the row well before the kill fires, so
|
|
4949
|
+
// the renderer can show it without waiting for the next tick.
|
|
4950
|
+
mutate((state) => {
|
|
4951
|
+
const j = state.jobs.find((x) => x.slug === job.slug);
|
|
4952
|
+
if (!j) return;
|
|
4953
|
+
j.budgetWarning = { budgetMs: jobBudgetMs, elapsedMs, at: new Date().toISOString() };
|
|
4954
|
+
}).catch((e) => console.warn('[scheduler] budget-warning stamp failed', job.slug, e?.message));
|
|
4955
|
+
}
|
|
4956
|
+
return shouldKillForBudget(elapsedMs, jobBudgetMs);
|
|
4957
|
+
},
|
|
4958
|
+
action(ctx) {
|
|
4959
|
+
const elapsedMs = Date.now() - ctx.startedAt;
|
|
4960
|
+
ctx.safeLog(`\n[scheduler] wall-clock budget watchdog: ran ${Math.round(elapsedMs / 60_000)}m ` +
|
|
4961
|
+
`(> ${Math.round(jobBudgetMs / 60_000)}m budget, estimateMinutes=${job.estimateMinutes ?? 0}) — SIGTERM process group\n`);
|
|
4962
|
+
ctx.killedByWatchdog = 'budget';
|
|
4963
|
+
ctx.killTree('SIGTERM');
|
|
4964
|
+
const budgetKillTimer = setTimeout(() => {
|
|
4965
|
+
ctx.safeLog(`\n[scheduler] budget watchdog: still alive ${Math.round(POST_RESULT_KILL_MS/1000)}s after SIGTERM — SIGKILL\n`);
|
|
4966
|
+
ctx.killTree('SIGKILL');
|
|
4967
|
+
}, POST_RESULT_KILL_MS);
|
|
4968
|
+
if (budgetKillTimer.unref) budgetKillTimer.unref();
|
|
4969
|
+
ctx.addTimer(budgetKillTimer);
|
|
4970
|
+
},
|
|
4971
|
+
};
|
|
4972
|
+
|
|
4410
4973
|
// ---------- spawn ----------
|
|
4411
4974
|
|
|
4412
4975
|
const { child } = withChildAndLog({
|
|
@@ -4437,8 +5000,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4437
5000
|
detached: true,
|
|
4438
5001
|
},
|
|
4439
5002
|
},
|
|
4440
|
-
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
|
|
4441
|
-
onExit({ exitCode, signal, killedByWatchdog
|
|
5003
|
+
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog, budgetWatchdog],
|
|
5004
|
+
onExit({ exitCode, signal, killedByWatchdog, error, spawnFailed, leakedDescendants, safeLog: sl }) {
|
|
4442
5005
|
const durationMs = Date.now() - startedAt;
|
|
4443
5006
|
const leaked = leakedDescendants ?? [];
|
|
4444
5007
|
if (leaked.length > 0) {
|
|
@@ -4468,7 +5031,11 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4468
5031
|
// and 137 (128+SIGKILL) in case the process exited via signal-as-code.
|
|
4469
5032
|
let effectiveCode = exitCode;
|
|
4470
5033
|
const killedBySignal = signal === 'SIGTERM' || signal === 'SIGKILL' || exitCode === 143 || exitCode === 137 || exitCode === null;
|
|
4471
|
-
|
|
5034
|
+
// A budget kill must NEVER be laundered into a clean exit=0, even when
|
|
5035
|
+
// the agent had already emitted result=success before it fired — the
|
|
5036
|
+
// AC requires it always park needs_review, never silently 'completed'.
|
|
5037
|
+
// idle-tail/deadman/result-tail kills keep the existing success-mapping.
|
|
5038
|
+
const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal && killedByWatchdog !== 'budget';
|
|
4472
5039
|
if (mappedToSuccess) {
|
|
4473
5040
|
effectiveCode = 0;
|
|
4474
5041
|
sl(`\n[scheduler] mapping exit code=${exitCode} signal=${signal} → 0 ` +
|
|
@@ -4491,6 +5058,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4491
5058
|
sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
|
|
4492
5059
|
`${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
|
|
4493
5060
|
}
|
|
5061
|
+
// Formatted once, here, off the FINAL durationMs (more accurate than
|
|
5062
|
+
// the watchdog action's own snapshot at kill time) — matches the
|
|
5063
|
+
// reason string format the AC requires verbatim. See
|
|
5064
|
+
// resolveBudgetKillOutcome's own header for why this is gated on
|
|
5065
|
+
// killedBySignal, not on killedByWatchdog alone.
|
|
5066
|
+
const budgetKillOutcome = resolveBudgetKillOutcome({
|
|
5067
|
+
killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes: job.estimateMinutes,
|
|
5068
|
+
});
|
|
5069
|
+
const { budgetKillReason } = budgetKillOutcome;
|
|
5070
|
+
const effectiveKilledByWatchdog = budgetKillOutcome.killedByWatchdog;
|
|
4494
5071
|
// Sync write: child 'exit' handler must flush meta before resolve()
|
|
4495
5072
|
// so the spawnJob mutate() that follows sees the persisted exit code.
|
|
4496
5073
|
config.writeJsonSync(metaPath, {
|
|
@@ -4501,10 +5078,14 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4501
5078
|
launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
|
|
4502
5079
|
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
4503
5080
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
5081
|
+
killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
4504
5082
|
schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
4505
5083
|
originSessionId, contextDigestApplied,
|
|
4506
5084
|
});
|
|
4507
|
-
resolve({
|
|
5085
|
+
resolve({
|
|
5086
|
+
exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats,
|
|
5087
|
+
leakedDescendants: leaked, sessionId, killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
5088
|
+
});
|
|
4508
5089
|
},
|
|
4509
5090
|
});
|
|
4510
5091
|
|
|
@@ -4512,8 +5093,27 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4512
5093
|
safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
|
|
4513
5094
|
// Make this job the OOM killer's preferred victim over Electron.
|
|
4514
5095
|
biasJobOomScore(child.pid);
|
|
4515
|
-
//
|
|
4516
|
-
|
|
5096
|
+
// Persist runtime.pid with one retry — still fire-and-forget (must
|
|
5097
|
+
// never block the spawn), but a final failure is now loud instead of
|
|
5098
|
+
// silently swallowed. A silent failure here is exactly what let the
|
|
5099
|
+
// pidless-grace reaper terminalize a live, working job (runtime.pid
|
|
5100
|
+
// never landed, so selectReapableJobs had no way to tell "never
|
|
5101
|
+
// spawned" from "spawned but unrecorded").
|
|
5102
|
+
if (onPid) {
|
|
5103
|
+
(async () => {
|
|
5104
|
+
try {
|
|
5105
|
+
await onPid(child.pid, sessionId, cwd);
|
|
5106
|
+
} catch (firstErr) {
|
|
5107
|
+
try {
|
|
5108
|
+
await onPid(child.pid, sessionId, cwd);
|
|
5109
|
+
} catch (finalErr) {
|
|
5110
|
+
const message = finalErr?.message ?? String(finalErr);
|
|
5111
|
+
console.error(`[scheduler] FAILED to persist runtime.pid for ${job.slug} pid=${child.pid}: ${message}`);
|
|
5112
|
+
appendAuditEvent('job_pid_persist_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
|
|
5113
|
+
}
|
|
5114
|
+
}
|
|
5115
|
+
})();
|
|
5116
|
+
}
|
|
4517
5117
|
}
|
|
4518
5118
|
});
|
|
4519
5119
|
}
|
|
@@ -5435,8 +6035,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5435
6035
|
|
|
5436
6036
|
// Commit-guard baseline: snapshot the working tree BEFORE the run so the
|
|
5437
6037
|
// post-run check flags only paths THIS job left dirty, not pre-existing WIP.
|
|
6038
|
+
// Captured once, with status codes, so the shared-tree guard below can
|
|
6039
|
+
// tell "was untracked" apart from "was tracked-and-modified" (2026-09-12
|
|
6040
|
+
// incident) without a second `git status` call; every other consumer of
|
|
6041
|
+
// `guardBaseline` still gets the plain path-string array it always did.
|
|
5438
6042
|
const guardCwd = job.cwd || defaultCwd;
|
|
5439
|
-
const
|
|
6043
|
+
const guardBaselineEntries = await uncommittedChangesWithStatus(guardCwd);
|
|
6044
|
+
const guardBaseline = guardBaselineEntries ? guardBaselineEntries.map((e) => e.path) : guardBaselineEntries;
|
|
5440
6045
|
const guardHeadBefore = await gitHead(guardCwd);
|
|
5441
6046
|
// Shared-tree stash guard baseline (incident 2026-09-01): captured
|
|
5442
6047
|
// unconditionally, before worktree isolation is even attempted, so an
|
|
@@ -5488,6 +6093,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5488
6093
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
5489
6094
|
} else {
|
|
5490
6095
|
console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
|
|
6096
|
+
// A job losing worktree isolation must never be a silent downgrade
|
|
6097
|
+
// discoverable only by reading queue.json afterwards — every genuine
|
|
6098
|
+
// fallback (never the deliberate SM_JOB_WORKTREE_DISABLE opt-out) is
|
|
6099
|
+
// logged at warn in the durable ops error log, with the job slug, cwd,
|
|
6100
|
+
// and specific reason attached.
|
|
6101
|
+
if (!jobWorktree.isWorktreeDisabled()) {
|
|
6102
|
+
try {
|
|
6103
|
+
appendError({
|
|
6104
|
+
cwd: job.cwd || defaultCwd,
|
|
6105
|
+
scope: 'scheduler',
|
|
6106
|
+
level: 'warn',
|
|
6107
|
+
message: `${job.slug}: worktree isolation fell back to the SHARED working tree — ${worktree.reason}`,
|
|
6108
|
+
meta: { slug: job.slug, cwd: job.cwd || defaultCwd, reason: worktree.reason },
|
|
6109
|
+
});
|
|
6110
|
+
} catch { /* durable logging must never break dispatch */ }
|
|
6111
|
+
}
|
|
5491
6112
|
}
|
|
5492
6113
|
// dispatchPhase stamp folded into a single unconditional mutate covering
|
|
5493
6114
|
// both branches above — the degraded-isolation fallback flag (skipped
|
|
@@ -5882,7 +6503,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5882
6503
|
sharedTreeGuard = await module.exports.checkSharedTreeGuard({
|
|
5883
6504
|
cwd: guardCwd,
|
|
5884
6505
|
stashBaseline,
|
|
5885
|
-
dirtyBaseline:
|
|
6506
|
+
dirtyBaseline: guardBaselineEntries,
|
|
5886
6507
|
headBefore: guardHeadBefore,
|
|
5887
6508
|
slug: job.slug,
|
|
5888
6509
|
});
|
|
@@ -5909,6 +6530,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5909
6530
|
// narrowly to exit 143 — never applied to other non-zero exit codes or to
|
|
5910
6531
|
// rateLimited (already handled separately, above).
|
|
5911
6532
|
let sigtermCommitFound = false;
|
|
6533
|
+
// Verified SHA twin of sigtermCommitFound's boolean — only the budget-kill
|
|
6534
|
+
// path (below) threads this onto the row's landedCommit; classifySigtermWithCommit's
|
|
6535
|
+
// own needs_review branch is unchanged and keeps using the boolean alone.
|
|
6536
|
+
let sigtermLandedCommitEvidence = null;
|
|
5912
6537
|
if (res.exitCode === 143 && !res.rateLimited) {
|
|
5913
6538
|
const guardHeadAtSigterm = await gitHead(guardCwd);
|
|
5914
6539
|
sigtermCommitFound = await computeCommittedDuringRun(
|
|
@@ -5918,6 +6543,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5918
6543
|
job.startedAt,
|
|
5919
6544
|
new Date().toISOString(),
|
|
5920
6545
|
);
|
|
6546
|
+
if (guardHeadBefore && guardHeadAtSigterm && guardHeadAtSigterm !== guardHeadBefore) {
|
|
6547
|
+
const verified = await resolveLandedCommitEvidence(guardCwd, guardHeadAtSigterm, job.startedAt);
|
|
6548
|
+
if (verified) sigtermLandedCommitEvidence = guardHeadAtSigterm;
|
|
6549
|
+
}
|
|
5921
6550
|
}
|
|
5922
6551
|
|
|
5923
6552
|
// BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
|
|
@@ -5939,6 +6568,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5939
6568
|
}
|
|
5940
6569
|
}
|
|
5941
6570
|
|
|
6571
|
+
// Evidence gate for a PLAIN non-zero exit — not SIGTERM-with-commit
|
|
6572
|
+
// (classifySigtermWithCommit above already routes exit 143 to
|
|
6573
|
+
// needs_review when a commit landed) and not rate-limited (handled
|
|
6574
|
+
// separately). A process that dies non-zero for any OTHER reason
|
|
6575
|
+
// (crash during post-commit cleanup, an overrun watchdog's SIGKILL) is
|
|
6576
|
+
// observed directly by THIS exit handler — it never reaches
|
|
6577
|
+
// reapDeadRunningJobs' own git-verified landedCommit evidence gate, so
|
|
6578
|
+
// without this check the exact bug that gate exists to prevent (job
|
|
6579
|
+
// 1192: a landedCommit non-empty is not proof by itself, but discarding
|
|
6580
|
+
// proof of real landed work with no evidence check at all is worse)
|
|
6581
|
+
// recurs here, one call site over. Computed outside mutate() (I/O) like
|
|
6582
|
+
// every other pre-finalize git check above.
|
|
6583
|
+
let plainExitLandedCommitEvidence = null;
|
|
6584
|
+
if (res.exitCode !== 0 && res.exitCode !== 143 && !res.rateLimited) {
|
|
6585
|
+
const headAtPlainExit = await gitHead(guardCwd);
|
|
6586
|
+
if (guardHeadBefore && headAtPlainExit && headAtPlainExit !== guardHeadBefore) {
|
|
6587
|
+
const verified = await resolveLandedCommitEvidence(guardCwd, headAtPlainExit, job.startedAt);
|
|
6588
|
+
if (verified) plainExitLandedCommitEvidence = headAtPlainExit;
|
|
6589
|
+
}
|
|
6590
|
+
}
|
|
6591
|
+
|
|
5942
6592
|
let actuallyFailed = false;
|
|
5943
6593
|
let failedJobSnapshot = null;
|
|
5944
6594
|
let needsInvestigationNow = false;
|
|
@@ -6009,7 +6659,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6009
6659
|
// Determine effective status, applying the verifier verdict for exit=0 runs.
|
|
6010
6660
|
let effectiveStatus;
|
|
6011
6661
|
let sigtermOverrideReason = null;
|
|
6012
|
-
|
|
6662
|
+
// Wall-clock budget kill — checked FIRST and unconditionally wins:
|
|
6663
|
+
// never 'failed', never silently 'completed', and (via
|
|
6664
|
+
// sigtermLandedCommitEvidence/plainExitLandedCommitEvidence, whichever
|
|
6665
|
+
// this exit code populated) still adjudicated on git evidence rather
|
|
6666
|
+
// than discarded. See classifyBudgetKill's own header.
|
|
6667
|
+
const budgetKill = classifyBudgetKill(res, sigtermLandedCommitEvidence || plainExitLandedCommitEvidence);
|
|
6668
|
+
const sigtermOverride = (!budgetKill && res.exitCode !== 0)
|
|
6013
6669
|
? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
|
|
6014
6670
|
: null;
|
|
6015
6671
|
// Validated against the LIVE row's own foreign-WIP manifest — never
|
|
@@ -6017,7 +6673,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6017
6673
|
// launder a real regression into a block (PRD: give the executor a
|
|
6018
6674
|
// first-class verdict for "the gate failed on a sibling's in-flight
|
|
6019
6675
|
// file", but VALIDATE the claim rather than trust it).
|
|
6020
|
-
const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
|
|
6676
|
+
const foreignWipValidation = (!budgetKill && !sigtermOverride && foreignWipClaimedPaths !== null)
|
|
6021
6677
|
? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
|
|
6022
6678
|
: null;
|
|
6023
6679
|
// Consecutive-block streak: cleared by default on every outcome and
|
|
@@ -6026,7 +6682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6026
6682
|
// between two blocks always resets "in a row" back to zero.
|
|
6027
6683
|
const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
|
|
6028
6684
|
delete s.jobs[i2].foreignWipBlockCount;
|
|
6029
|
-
if (
|
|
6685
|
+
if (budgetKill) {
|
|
6686
|
+
effectiveStatus = budgetKill.status;
|
|
6687
|
+
sigtermOverrideReason = budgetKill.reason;
|
|
6688
|
+
s.jobs[i2].verifierVerdict = 'budget_exceeded';
|
|
6689
|
+
if (budgetKill.landedCommit) jobLandedCommitThisRun = budgetKill.landedCommit;
|
|
6690
|
+
} else if (sigtermOverride) {
|
|
6030
6691
|
effectiveStatus = sigtermOverride.status;
|
|
6031
6692
|
sigtermOverrideReason = sigtermOverride.reason;
|
|
6032
6693
|
} else if (foreignWipValidation && foreignWipValidation.ok) {
|
|
@@ -6061,7 +6722,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6061
6722
|
effectiveStatus = 'failed';
|
|
6062
6723
|
sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
|
|
6063
6724
|
} else if (res.exitCode !== 0) {
|
|
6064
|
-
|
|
6725
|
+
if (plainExitLandedCommitEvidence) {
|
|
6726
|
+
// Same conservative posture as the SIGTERM+commit case above:
|
|
6727
|
+
// a landed commit doesn't prove every AC line passed, so this
|
|
6728
|
+
// still routes to needs_review for a human/reverify pass,
|
|
6729
|
+
// never silently to completed.
|
|
6730
|
+
effectiveStatus = 'needs_review';
|
|
6731
|
+
sigtermOverrideReason = `exited ${res.exitCode} after landing a git-verified commit — verify AC before treating as done`;
|
|
6732
|
+
jobLandedCommitThisRun = plainExitLandedCommitEvidence;
|
|
6733
|
+
} else {
|
|
6734
|
+
effectiveStatus = 'failed';
|
|
6735
|
+
}
|
|
6065
6736
|
} else if (
|
|
6066
6737
|
!verifyResult
|
|
6067
6738
|
|| COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
|
|
@@ -6090,15 +6761,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6090
6761
|
const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
|
|
6091
6762
|
? verifyResult.reason
|
|
6092
6763
|
: (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
|
|
6093
|
-
|
|
6094
|
-
s
|
|
6095
|
-
|
|
6096
|
-
s
|
|
6097
|
-
|
|
6098
|
-
|
|
6099
|
-
|
|
6100
|
-
|
|
6101
|
-
}
|
|
6764
|
+
// error/verifierVerdict are stamped BEFORE transitionJob() below —
|
|
6765
|
+
// needsReviewLedger's buildNeedsReviewEntryLine reads job.
|
|
6766
|
+
// verifierVerdict/heldReason/error synchronously off `job` the
|
|
6767
|
+
// instant transitionJob() runs (it's called inside transitionJob,
|
|
6768
|
+
// not deferred), so setting these after that call fed the durable
|
|
6769
|
+
// needs_review ledger stale/leftover values from before this run,
|
|
6770
|
+
// defeating its whole `byReason` rollup for the two escalation
|
|
6771
|
+
// paths that land here.
|
|
6102
6772
|
s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
|
|
6103
6773
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
6104
6774
|
// A failed job (non-zero exit) never consults verifyResult above,
|
|
@@ -6116,18 +6786,28 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6116
6786
|
s.jobs[i2].landedCommit = jobLandedCommitThisRun;
|
|
6117
6787
|
}
|
|
6118
6788
|
// Persist the verifier's verdict string so the renderer can show it.
|
|
6119
|
-
// 'blocked_by_foreign_wip_streak'
|
|
6120
|
-
// exit-code path, never from verifyResult (which
|
|
6121
|
-
// non-zero exit) — never clobber
|
|
6789
|
+
// 'blocked_by_foreign_wip_streak'/'budget_exceeded' are set above
|
|
6790
|
+
// from the sigterm/exit-code path, never from verifyResult (which
|
|
6791
|
+
// stays null on a non-zero exit) — never clobber either here.
|
|
6122
6792
|
if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
|
|
6123
6793
|
s.jobs[i2].verifierVerdict = verifyResult.verdict;
|
|
6124
|
-
} else if (s.jobs[i2].verifierVerdict
|
|
6794
|
+
} else if (!['blocked_by_foreign_wip_streak', 'budget_exceeded'].includes(s.jobs[i2].verifierVerdict)) {
|
|
6125
6795
|
delete s.jobs[i2].verifierVerdict;
|
|
6126
6796
|
}
|
|
6797
|
+
transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
|
|
6798
|
+
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
6799
|
+
s.jobs[i2].exitCode = res.exitCode;
|
|
6800
|
+
s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
|
|
6801
|
+
if (salvagePatch) {
|
|
6802
|
+
s.jobs[i2].salvagePatch = salvagePatch;
|
|
6803
|
+
} else {
|
|
6804
|
+
delete s.jobs[i2].salvagePatch;
|
|
6805
|
+
}
|
|
6127
6806
|
// Closed-set outcome taxonomy (issue #11 list A2) so a queue row
|
|
6128
6807
|
// says WHY it ended without anyone opening the transcript.
|
|
6129
6808
|
s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
|
|
6130
6809
|
effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
|
|
6810
|
+
budgetKill: !!budgetKill,
|
|
6131
6811
|
});
|
|
6132
6812
|
delete s.jobs[i2].launchFailure;
|
|
6133
6813
|
delete s.jobs[i2].heldReason;
|
|
@@ -6714,7 +7394,7 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
6714
7394
|
}
|
|
6715
7395
|
return recordTick(
|
|
6716
7396
|
{ fired: false, reason: 'load-deferred', deferredCount: gatedBatch.length, ratio: load.ratio, threshold: load.threshold },
|
|
6717
|
-
{ detail:
|
|
7397
|
+
{ detail: formatLoadGateDetail(load), holds },
|
|
6718
7398
|
);
|
|
6719
7399
|
}
|
|
6720
7400
|
|
|
@@ -6860,43 +7540,220 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
|
|
|
6860
7540
|
}
|
|
6861
7541
|
|
|
6862
7542
|
/**
|
|
6863
|
-
*
|
|
6864
|
-
*
|
|
6865
|
-
*
|
|
6866
|
-
*
|
|
6867
|
-
*
|
|
7543
|
+
* classifyQueueStarvationByProject({ jobs, paused, runningSet, lastRunAtMs, now, thresholdMs })
|
|
7544
|
+
* → [{ cwd, kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }]
|
|
7545
|
+
*
|
|
7546
|
+
* Per-project driver around classifyQueueStarvation's pure single-project
|
|
7547
|
+
* core. `runningCount > 0` inside that core used to be fed the MACHINE-WIDE
|
|
7548
|
+
* `runningSet.size`, which meant one long-lived job in ANY project disarmed
|
|
7549
|
+
* the watchdog for EVERY other project on the box — observed live
|
|
7550
|
+
* 2026-09-12: a job in starry-night-ships ran 80+ minutes while two other
|
|
7551
|
+
* projects sat starved/blocked for hours, and the watchdog never fired once
|
|
7552
|
+
* because "work is flowing" was true somewhere else. Partitioning by cwd
|
|
7553
|
+
* (the same grouping computeBlockedChains already does) fixes DETECTION only
|
|
7554
|
+
* — the idle clock (`lastRunAtMs`) stays machine-wide, since
|
|
7555
|
+
* `lastDispatchAttemptAt` is machine-level state, and only one tick is ever
|
|
7556
|
+
* forced per watchdog pass regardless of how many cwds are starved.
|
|
7557
|
+
*
|
|
7558
|
+
* Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
|
|
7559
|
+
* project has a verdict.
|
|
7560
|
+
*/
|
|
7561
|
+
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
7562
|
+
if (paused) return [];
|
|
7563
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7564
|
+
const byCwd = new Map();
|
|
7565
|
+
for (const j of rows) {
|
|
7566
|
+
const key = j.cwd || '(unknown)';
|
|
7567
|
+
if (!byCwd.has(key)) byCwd.set(key, []);
|
|
7568
|
+
byCwd.get(key).push(j);
|
|
7569
|
+
}
|
|
7570
|
+
|
|
7571
|
+
const verdicts = [];
|
|
7572
|
+
for (const [cwd, projectJobs] of byCwd) {
|
|
7573
|
+
// Same source of truth tickQueue itself uses for "is anything running":
|
|
7574
|
+
// the in-process runningSet OR a row already stamped status:'running'.
|
|
7575
|
+
const projRunningCount = projectJobs.filter(
|
|
7576
|
+
(j) => j.status === 'running' || runningSlugs?.has?.(j.slug),
|
|
7577
|
+
).length;
|
|
7578
|
+
const verdict = classifyQueueStarvation({
|
|
7579
|
+
jobs: projectJobs,
|
|
7580
|
+
paused: false,
|
|
7581
|
+
runningCount: projRunningCount,
|
|
7582
|
+
lastRunAtMs,
|
|
7583
|
+
now,
|
|
7584
|
+
thresholdMs,
|
|
7585
|
+
});
|
|
7586
|
+
if (verdict) verdicts.push({ cwd, ...verdict });
|
|
7587
|
+
}
|
|
7588
|
+
return verdicts;
|
|
7589
|
+
}
|
|
7590
|
+
|
|
7591
|
+
/**
|
|
7592
|
+
* classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
|
|
7593
|
+
* totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
|
|
7594
|
+
* → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
|
|
7595
|
+
*
|
|
7596
|
+
* Single source of truth for the Scheduler page's queue-health header: the
|
|
7597
|
+
* one thing a human staring at a stale-looking queue needs is "which of the
|
|
7598
|
+
* genuinely different causes is this" (all slots busy? every pending row
|
|
7599
|
+
* blocked on a dependency? the dispatch driver itself never ticked?) — this
|
|
7600
|
+
* function names that cause instead of leaving the renderer to re-derive it.
|
|
7601
|
+
*
|
|
7602
|
+
* Reuses classifyQueueStarvation for the blocked/stalled read so the header
|
|
7603
|
+
* can never disagree with runQueueStarvationWatchdog's own decision to force
|
|
7604
|
+
* a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
|
|
7605
|
+
* the same computeBlockedChains walk under the hood. Called here with
|
|
7606
|
+
* `thresholdMs: 0` first (a live header must say "blocked" the instant every
|
|
7607
|
+
* pending row is dependency-stuck, not wait out the watchdog's own 10-minute
|
|
7608
|
+
* grace period) — the returned `idleMs` is then compared against the REAL
|
|
7609
|
+
* `thresholdMs` to decide 'stalled' vs the healthy 'running' default, which
|
|
7610
|
+
* is exactly the comparison classifyQueueStarvation would make internally.
|
|
7611
|
+
*
|
|
7612
|
+
* `pending`/`dispatchable`/`blockedChains`/`needsReviewCount` are always
|
|
7613
|
+
* populated (via computeBlockedChains — the exact primitive
|
|
7614
|
+
* classifyQueueStarvation itself calls) regardless of kind, so a 'saturated'
|
|
7615
|
+
* or 'running' header can still say how much of the backlog is dependency-
|
|
7616
|
+
* blocked, not just the kinds where that's the headline cause.
|
|
7617
|
+
*
|
|
7618
|
+
* Kinds, in the priority order they're checked (paused is a decision, not a
|
|
7619
|
+
* stall; an open launch breaker explains an otherwise-inexplicable
|
|
7620
|
+
* non-dispatch before slot/dependency causes are even considered):
|
|
7621
|
+
* 'paused' — the scheduler itself is paused.
|
|
7622
|
+
* 'launch-blocked' — a pending row's persona has an active circuit-breaker
|
|
7623
|
+
* entry (lib/launchFailure.cjs).
|
|
7624
|
+
* 'idle' — nothing pending in this scope.
|
|
7625
|
+
* 'saturated' — pending work exists but every session slot is in use.
|
|
7626
|
+
* 'blocked' — nothing running, slots free, every pending row's
|
|
7627
|
+
* dependsOn chain terminates in a non-completed row.
|
|
7628
|
+
* 'stalled' — nothing running, slots free, at least one row is
|
|
7629
|
+
* dispatchable right now, and the dispatch driver has
|
|
7630
|
+
* been idle >= thresholdMs (agrees with the watchdog).
|
|
7631
|
+
* 'running' — the healthy default: work is flowing, or the driver
|
|
7632
|
+
* hasn't been idle long enough to call a stall yet.
|
|
7633
|
+
*
|
|
7634
|
+
* Pure, no IO. `cwd` scopes jobs/pending/blocked/needsReview to one project
|
|
7635
|
+
* (the Scheduler nav row is PROJECT-face — see CLAUDE.md); `freeSlots` /
|
|
7636
|
+
* `totalSlots` / `launchBlocks` stay machine-wide inputs by design, same as
|
|
7637
|
+
* WindowStrip's existing scopeCwd split.
|
|
7638
|
+
*/
|
|
7639
|
+
function classifyQueueHealth({
|
|
7640
|
+
jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
|
|
7641
|
+
lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
|
|
7642
|
+
} = {}) {
|
|
7643
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7644
|
+
const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
|
|
7645
|
+
const pendingRows = projectJobs.filter((j) => j.status === 'pending');
|
|
7646
|
+
const runningRows = projectJobs.filter((j) => j.status === 'running' || runningSlugs?.has?.(j.slug));
|
|
7647
|
+
const needsReviewCount = projectJobs.filter((j) => j.status === 'needs_review').length;
|
|
7648
|
+
|
|
7649
|
+
// computeBlockedChains is the SAME primitive classifyQueueStarvation calls
|
|
7650
|
+
// internally, computed once here so EVERY kind (not just 'blocked'/
|
|
7651
|
+
// 'stalled') carries real dispatchable-vs-blocked counts instead of a null.
|
|
7652
|
+
const blockedChains = computeBlockedChains(projectJobs);
|
|
7653
|
+
const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
|
|
7654
|
+
const dispatchable = Math.max(0, pendingRows.length - blockedTotal);
|
|
7655
|
+
const base = {
|
|
7656
|
+
cwd, pending: pendingRows.length, dispatchable, blockedChains, needsReviewCount, runningCount: runningRows.length,
|
|
7657
|
+
};
|
|
7658
|
+
|
|
7659
|
+
if (paused) return { ...base, kind: 'paused', reason: paused.reason ?? null };
|
|
7660
|
+
|
|
7661
|
+
// launch-blocked: only a persona a PENDING row in this scope actually uses
|
|
7662
|
+
// — a breaker open for a persona nothing here needs is not this scope's
|
|
7663
|
+
// problem (matches WindowStrip's own unconditional-banner-per-block read).
|
|
7664
|
+
const neededAgentTypes = new Set(pendingRows.map((j) => launchFailure.launchBlockKeyFor(j)));
|
|
7665
|
+
for (const [key, block] of Object.entries(launchBlocks ?? {})) {
|
|
7666
|
+
if (block && neededAgentTypes.has(key)) {
|
|
7667
|
+
return { ...base, kind: 'launch-blocked', agentType: key, block };
|
|
7668
|
+
}
|
|
7669
|
+
}
|
|
7670
|
+
|
|
7671
|
+
if (pendingRows.length === 0) return { ...base, kind: 'idle' };
|
|
7672
|
+
|
|
7673
|
+
// classifyQueueStarvation only ever classifies while nothing is running
|
|
7674
|
+
// (its own runningCount > 0 guard) — that boundary is also exactly where
|
|
7675
|
+
// slot saturation, not dependency shape, is the honest cause.
|
|
7676
|
+
if (runningRows.length > 0) {
|
|
7677
|
+
if (Number.isFinite(freeSlots) && freeSlots <= 0) {
|
|
7678
|
+
return { ...base, kind: 'saturated', totalSlots: totalSlots ?? null };
|
|
7679
|
+
}
|
|
7680
|
+
return { ...base, kind: 'running' };
|
|
7681
|
+
}
|
|
7682
|
+
|
|
7683
|
+
// Nothing running: hand the SAME rows + idle clock to classifyQueueStarvation
|
|
7684
|
+
// (thresholdMs: 0 — a live header must say "blocked" the instant every
|
|
7685
|
+
// pending row is dependency-stuck, not wait out the watchdog's own grace
|
|
7686
|
+
// period) purely for its idleMs reading; its own dispatchable/blockedChains
|
|
7687
|
+
// are mathematically identical to `base`'s (same computeBlockedChains walk
|
|
7688
|
+
// over the same rows), so `base` already carries them.
|
|
7689
|
+
const immediate = classifyQueueStarvation({
|
|
7690
|
+
jobs: projectJobs, paused: false, runningCount: 0,
|
|
7691
|
+
lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
|
|
7692
|
+
});
|
|
7693
|
+
// pending.length is already > 0 above, so `immediate` can only be null when
|
|
7694
|
+
// lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
|
|
7695
|
+
// to computing idleMs the same way rather than asserting a kind we can't
|
|
7696
|
+
// back up with a real number.
|
|
7697
|
+
const idleMs = immediate ? immediate.idleMs
|
|
7698
|
+
: (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
|
|
7699
|
+
if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
|
|
7700
|
+
const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
|
|
7701
|
+
return { ...base, kind, idleMs };
|
|
7702
|
+
}
|
|
7703
|
+
|
|
7704
|
+
/**
|
|
7705
|
+
* The watchdog half: acts on classifyQueueStarvationByProject. Called from
|
|
7706
|
+
* the heartbeat, which already runs on its own timer independent of the
|
|
7707
|
+
* billing poll loop — so a wedged or never-succeeding poll (the
|
|
7708
|
+
* /api/oauth/usage endpoint was itself 429ing all of 2026-09-05) can no
|
|
7709
|
+
* longer leave a queue with ready work idle indefinitely.
|
|
7710
|
+
*
|
|
7711
|
+
* Logs and audits one event PER starved/blocked cwd (each carrying that
|
|
7712
|
+
* cwd), but still forces at most one machine-wide tickQueue() per pass —
|
|
7713
|
+
* the tick itself is machine-wide (it drives whatever the picker finds
|
|
7714
|
+
* across every project), only the DETECTION is per-project.
|
|
6868
7715
|
*/
|
|
6869
7716
|
async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
6870
7717
|
// lastDispatchAttemptAt, not lastRunAt: the latter only advances when a
|
|
6871
7718
|
// batch actually launches, so a poll that keeps succeeding while dispatch
|
|
6872
7719
|
// itself never gets invoked would otherwise mask a stall behind a fresh-
|
|
6873
7720
|
// looking timestamp that was never actually tracking dispatch liveness.
|
|
6874
|
-
const
|
|
7721
|
+
const verdicts = classifyQueueStarvationByProject({
|
|
6875
7722
|
jobs: state?.jobs,
|
|
6876
7723
|
paused: state?.paused,
|
|
6877
|
-
|
|
7724
|
+
runningSet,
|
|
6878
7725
|
lastRunAtMs: Date.parse(state?.lastDispatchAttemptAt ?? ''),
|
|
6879
7726
|
now,
|
|
6880
7727
|
thresholdMs,
|
|
6881
7728
|
});
|
|
6882
|
-
if (
|
|
7729
|
+
if (verdicts.length === 0) return null;
|
|
7730
|
+
|
|
7731
|
+
let anyStarved = false;
|
|
7732
|
+
let primary = null;
|
|
7733
|
+
for (const verdict of verdicts) {
|
|
7734
|
+
const mins = Math.round(verdict.idleMs / 60_000);
|
|
7735
|
+
if (verdict.kind === 'blocked') {
|
|
7736
|
+
console.warn(
|
|
7737
|
+
`[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
7738
|
+
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
7739
|
+
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7740
|
+
);
|
|
7741
|
+
appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
7742
|
+
if (!primary) primary = verdict;
|
|
7743
|
+
continue;
|
|
7744
|
+
}
|
|
6883
7745
|
|
|
6884
|
-
const mins = Math.round(verdict.idleMs / 60_000);
|
|
6885
|
-
if (verdict.kind === 'blocked') {
|
|
6886
7746
|
console.warn(
|
|
6887
|
-
`[scheduler] QUEUE
|
|
6888
|
-
+ `
|
|
6889
|
-
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7747
|
+
`[scheduler] QUEUE STARVED (${verdict.cwd}): ${verdict.dispatchable} dispatchable job(s) of ${verdict.pending} pending, 0 running, `
|
|
7748
|
+
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
6890
7749
|
);
|
|
6891
|
-
appendAuditEvent('
|
|
6892
|
-
|
|
7750
|
+
appendAuditEvent('queue_starvation_forced_tick', { cwd: verdict.cwd, pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
7751
|
+
anyStarved = true;
|
|
7752
|
+
primary = verdict;
|
|
6893
7753
|
}
|
|
6894
7754
|
|
|
6895
|
-
|
|
6896
|
-
|
|
6897
|
-
+ `idle ${mins}m (>= ${Math.round(thresholdMs / 60_000)}m) — forcing a tick`,
|
|
6898
|
-
);
|
|
6899
|
-
appendAuditEvent('queue_starvation_forced_tick', { pending: verdict.pending, dispatchable: verdict.dispatchable, idleMs: verdict.idleMs });
|
|
7755
|
+
if (!anyStarved) return primary;
|
|
7756
|
+
|
|
6900
7757
|
// A never-populated utilization reading is itself one of the ways the
|
|
6901
7758
|
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
6902
7759
|
// returns early on null). Treat unknown as safe here, exactly as the
|
|
@@ -6911,8 +7768,76 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
|
|
|
6911
7768
|
// actually ticked. The watchdog is the last line of defence against a
|
|
6912
7769
|
// wedged dispatcher, so it must be able to un-wedge this too.
|
|
6913
7770
|
cancelToken.cancelled = false;
|
|
7771
|
+
// A forced tick is machine-wide by construction (the picker considers
|
|
7772
|
+
// every project's rows) — one call here services every starved cwd found
|
|
7773
|
+
// this pass, not one call per cwd.
|
|
6914
7774
|
await tickQueue({ bypassLoadGate: false }).catch((e) => console.error('[scheduler] starvation tick error', e));
|
|
6915
|
-
return
|
|
7775
|
+
return primary;
|
|
7776
|
+
}
|
|
7777
|
+
|
|
7778
|
+
// One-shot latch, keyed per cwd, for the starve-escalation consequence below —
|
|
7779
|
+
// never escalate the same starve stretch twice. Cleared the moment that cwd
|
|
7780
|
+
// stops appearing in findStarvedProjects at all (dispatched, or the machine
|
|
7781
|
+
// went idle/paused), mirroring the heartbeat's stallSince/stallToasted pair.
|
|
7782
|
+
const starveEscalated = new Set();
|
|
7783
|
+
|
|
7784
|
+
/**
|
|
7785
|
+
* selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs)
|
|
7786
|
+
* → { toEscalate: [...sp], toClear: [cwd, ...] }
|
|
7787
|
+
*
|
|
7788
|
+
* Pure. `starvedProjects` is this sweep's findStarvedProjects() output (the
|
|
7789
|
+
* per-cwd STARVED verdict — cwd, pendingCount, oldestPendingSlug, ageMs);
|
|
7790
|
+
* `escalatedCwds` is the Set already latched from a prior sweep.
|
|
7791
|
+
*
|
|
7792
|
+
* toEscalate: rows crossing thresholdMs for the FIRST time this stretch —
|
|
7793
|
+
* i.e. old enough AND not already latched.
|
|
7794
|
+
* toClear: previously-latched cwds no longer reported as starved at all this
|
|
7795
|
+
* sweep, so a LATER starve on that project escalates again instead of being
|
|
7796
|
+
* silently suppressed forever by a stale latch.
|
|
7797
|
+
*/
|
|
7798
|
+
function selectStarveEscalations(starvedProjects, escalatedCwds, thresholdMs = STARVE_ESCALATION_MS) {
|
|
7799
|
+
const stillStarved = new Set(starvedProjects.map((sp) => sp.cwd));
|
|
7800
|
+
const toClear = [...escalatedCwds].filter((cwd) => !stillStarved.has(cwd));
|
|
7801
|
+
const toEscalate = starvedProjects.filter((sp) => sp.ageMs >= thresholdMs && !escalatedCwds.has(sp.cwd));
|
|
7802
|
+
return { toEscalate, toClear };
|
|
7803
|
+
}
|
|
7804
|
+
|
|
7805
|
+
/**
|
|
7806
|
+
* runStarveEscalationSweep(starvedProjects) — acts on selectStarveEscalations'
|
|
7807
|
+
* verdict: audits a DISTINCT 'project_starve_escalated' event (once per starve
|
|
7808
|
+
* stretch, per cwd) and pushes the same toast-channel error the heartbeat's
|
|
7809
|
+
* stall detector already uses ('schedule:stall' → renderer toast.error), so a
|
|
7810
|
+
* starve that has gone on long enough to matter is visible without grepping
|
|
7811
|
+
* the audit log. The hold reason is read from `lastTick` (recordTick's own
|
|
7812
|
+
* last-computed outcome) — never re-evaluated here, so this can never
|
|
7813
|
+
* disagree with what actually happened on the last tick.
|
|
7814
|
+
*
|
|
7815
|
+
* Escalation only: never mutates a job, never dispatches, never bypasses a
|
|
7816
|
+
* gate. Exported for direct unit testing (attach a fake window via
|
|
7817
|
+
* attachWindow() first to assert the toast send).
|
|
7818
|
+
*/
|
|
7819
|
+
function runStarveEscalationSweep(starvedProjects) {
|
|
7820
|
+
const { toEscalate, toClear } = selectStarveEscalations(starvedProjects, starveEscalated, STARVE_ESCALATION_MS);
|
|
7821
|
+
for (const cwd of toClear) starveEscalated.delete(cwd);
|
|
7822
|
+
for (const sp of toEscalate) {
|
|
7823
|
+
starveEscalated.add(sp.cwd);
|
|
7824
|
+
const holdReason = lastTick?.reason ?? 'unknown';
|
|
7825
|
+
const mins = Math.round(sp.ageMs / 60_000);
|
|
7826
|
+
console.error(
|
|
7827
|
+
`[scheduler] PROJECT STARVE ESCALATED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
7828
|
+
+ `waiting=${mins}m (>= ${Math.round(STARVE_ESCALATION_MS / 60_000)}m escalation threshold), hold reason=${holdReason} — `
|
|
7829
|
+
+ 'a bounded escalation only; nothing was auto-reset, cancelled, or dispatched',
|
|
7830
|
+
);
|
|
7831
|
+
appendAuditEvent('project_starve_escalated', {
|
|
7832
|
+
cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs, holdReason,
|
|
7833
|
+
});
|
|
7834
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
7835
|
+
message: `Project starved: ${sp.cwd} has ${sp.pendingCount} pending PRD(s), oldest waiting ~${mins}m `
|
|
7836
|
+
+ `(hold reason: ${holdReason}) — check the Scheduler tab.`,
|
|
7837
|
+
total: sp.pendingCount,
|
|
7838
|
+
byProject: { [sp.cwd]: { starved: sp.pendingCount } },
|
|
7839
|
+
});
|
|
7840
|
+
}
|
|
6916
7841
|
}
|
|
6917
7842
|
|
|
6918
7843
|
// ---------- dead-process reaper ----------
|
|
@@ -7007,12 +7932,19 @@ async function reapDeadRunningJobs() {
|
|
|
7007
7932
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
7008
7933
|
// queue.json is the source of truth for which jobs are actually running.
|
|
7009
7934
|
const state = await readQueue();
|
|
7935
|
+
// Shared by the log-evidence injections below and the reapable-processing
|
|
7936
|
+
// loop further down — same `j.runId` → run log path formula either way.
|
|
7937
|
+
const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
|
|
7010
7938
|
const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
|
|
7011
7939
|
pidAlive: claudePidAlive,
|
|
7012
7940
|
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
7013
7941
|
findLiveProcess: (j) => findLiveProcessForJob(j, {
|
|
7014
7942
|
worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
|
|
7943
|
+
runCwd: j.runtime?.cwd || j.cwd,
|
|
7015
7944
|
}),
|
|
7945
|
+
getLogPid: (j) => readSpawnedPidFromLog(logPathForJob(j)),
|
|
7946
|
+
getLogMtimeMs: (j) => readLogMtimeMs(logPathForJob(j)),
|
|
7947
|
+
logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
|
|
7016
7948
|
});
|
|
7017
7949
|
for (const w of warnings) {
|
|
7018
7950
|
console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
|
|
@@ -7037,11 +7969,9 @@ async function reapDeadRunningJobs() {
|
|
|
7037
7969
|
}
|
|
7038
7970
|
|
|
7039
7971
|
const dead = [];
|
|
7040
|
-
for (const { slug, pid, pidless, reason } of reapable) {
|
|
7972
|
+
for (const { slug, pid, pidless, reason, failureOverride } of reapable) {
|
|
7041
7973
|
const j = state.jobs.find((x) => x.slug === slug);
|
|
7042
|
-
const logPath = j
|
|
7043
|
-
? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
|
|
7044
|
-
: null;
|
|
7974
|
+
const logPath = logPathForJob(j);
|
|
7045
7975
|
// Absent/empty run dir → classifyRunOutcome finds no result event →
|
|
7046
7976
|
// 'no_result' → non-success below → filed as failed, never completed.
|
|
7047
7977
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
@@ -7060,7 +7990,7 @@ async function reapDeadRunningJobs() {
|
|
|
7060
7990
|
// re-derived so a phantom link never survives the reap.
|
|
7061
7991
|
const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
|
|
7062
7992
|
const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
|
|
7063
|
-
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
|
|
7993
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact, failureOverride });
|
|
7064
7994
|
}
|
|
7065
7995
|
|
|
7066
7996
|
queueHealthSweepCycle += 1;
|
|
@@ -7158,8 +8088,44 @@ async function reapDeadRunningJobs() {
|
|
|
7158
8088
|
integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
|
|
7159
8089
|
}
|
|
7160
8090
|
|
|
8091
|
+
// Evidence-before-failure guard for a row about to be stamped 'failed'
|
|
8092
|
+
// (this PRD — job 1192 shipped a real 3-file commit and was still
|
|
8093
|
+
// reaped 'failed' because this check did not exist): a `landedCommit`
|
|
8094
|
+
// already recorded on the row is only ever stamped from an actual HEAD
|
|
8095
|
+
// advance or a proven branch-integration (jobLandedCommitThisRun / the
|
|
8096
|
+
// dead-pid integration proof above / the dispatch-time sidecar
|
|
8097
|
+
// backfill) — never speculative — but it can still be STALE by the time
|
|
8098
|
+
// this row is reaped (the branch it named could have been force-pushed
|
|
8099
|
+
// over, or the row could be carrying a sidecar-backfilled sha from a
|
|
8100
|
+
// run that was later discarded). git-resolving it here is what turns
|
|
8101
|
+
// "the field is non-empty" into "this sha is a real commit in this
|
|
8102
|
+
// repo right now". Computed OUTSIDE mutate() for the same reason
|
|
8103
|
+
// integrationResults is above: git spawn work must never run inside
|
|
8104
|
+
// mutate()'s single global serialization chain.
|
|
8105
|
+
//
|
|
8106
|
+
// Scoped to exactly the rows that would otherwise fall through to
|
|
8107
|
+
// 'failed' below: a 'success' outcome is already resolved by
|
|
8108
|
+
// integrationResults above (never reaches 'failed'), a rate-limited
|
|
8109
|
+
// death is retryable and never terminal, and a pidless row that already
|
|
8110
|
+
// carries a `failureOverride` (PRD 1173) is already diverted to
|
|
8111
|
+
// needs_review — this gate must never re-litigate either of those.
|
|
8112
|
+
const landedCommitEvidence = new Map();
|
|
8113
|
+
// Each row's evidence check is an independent read-only `git cat-file`/
|
|
8114
|
+
// `git log` pair with no shared mutable state between iterations, so
|
|
8115
|
+
// this runs the whole dead-job batch concurrently rather than one
|
|
8116
|
+
// dispatch's git-spawn latency at a time.
|
|
8117
|
+
await Promise.all(dead.map(async (d) => {
|
|
8118
|
+
if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
|
|
8119
|
+
if (d.pidless && d.failureOverride) return;
|
|
8120
|
+
const row = state.jobs.find((x) => x.slug === d.slug);
|
|
8121
|
+
if (!row?.landedCommit) return;
|
|
8122
|
+
const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
8123
|
+
const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
|
|
8124
|
+
if (resolved) landedCommitEvidence.set(d.slug, row.landedCommit);
|
|
8125
|
+
}));
|
|
8126
|
+
|
|
7161
8127
|
await mutate(async (s) => {
|
|
7162
|
-
for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
|
|
8128
|
+
for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact, failureOverride } of dead) {
|
|
7163
8129
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
7164
8130
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
7165
8131
|
const rateLimited = outcome === 'rate_limited';
|
|
@@ -7218,6 +8184,20 @@ async function reapDeadRunningJobs() {
|
|
|
7218
8184
|
notLandedInfo = ir.notLandedInfo;
|
|
7219
8185
|
}
|
|
7220
8186
|
}
|
|
8187
|
+
// Evidence-before-failure guard for the pidless-reap path (PRD 1173):
|
|
8188
|
+
// a pidless reap about to stamp 'failed' purely because runtime.pid
|
|
8189
|
+
// was never recorded must first check whether this row already
|
|
8190
|
+
// carries a landedCommit from an earlier dispatch of the same slug
|
|
8191
|
+
// (landedCommit survives a reset — see the comment near
|
|
8192
|
+
// resetJobFields). Diverted to needs_review, never silently
|
|
8193
|
+
// 'completed' — see resolvePidlessFailureOverride's header in
|
|
8194
|
+
// reaperHelpers.cjs for why needs_review is the correct destination.
|
|
8195
|
+
// Scoped strictly to the pidless branch: dead-pid and rate-limited
|
|
8196
|
+
// rows are untouched, and a pidless row that already resolved to
|
|
8197
|
+
// effectiveSuccess/notLandedInfo above is left alone too.
|
|
8198
|
+
if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
|
|
8199
|
+
notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
|
|
8200
|
+
}
|
|
7221
8201
|
|
|
7222
8202
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
7223
8203
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
@@ -7225,9 +8205,20 @@ async function reapDeadRunningJobs() {
|
|
|
7225
8205
|
const baseReason = notLandedInfo
|
|
7226
8206
|
? `reaped: ${notLandedInfo.reason}`
|
|
7227
8207
|
: (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
|
|
8208
|
+
// Evidence gate (this PRD): a row that would otherwise fall through
|
|
8209
|
+
// to 'failed' below, but whose landedCommit was proven to resolve
|
|
8210
|
+
// via git cat-file BEFORE this mutate() ran (see landedCommitEvidence
|
|
8211
|
+
// above), gets promoted to 'completed' instead — the row already
|
|
8212
|
+
// shipped real work, so a bookkeeping gap (no runtime.pid recorded)
|
|
8213
|
+
// must never override git-verified evidence with a false failure.
|
|
8214
|
+
const confirmedLandedCommit = (!effectiveSuccess && !notLandedInfo && !rateLimited)
|
|
8215
|
+
? (landedCommitEvidence.get(slug) || null)
|
|
8216
|
+
: null;
|
|
7228
8217
|
const transitionReason = rateLimited
|
|
7229
8218
|
? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
|
|
7230
|
-
:
|
|
8219
|
+
: confirmedLandedCommit
|
|
8220
|
+
? `${baseReason}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence${leftoverSuffix}`
|
|
8221
|
+
: baseReason + leftoverSuffix;
|
|
7231
8222
|
|
|
7232
8223
|
if (rateLimited) {
|
|
7233
8224
|
// Retryable, never terminal (PRD 1117) — same resetJobFields path
|
|
@@ -7236,17 +8227,27 @@ async function reapDeadRunningJobs() {
|
|
|
7236
8227
|
// paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
|
|
7237
8228
|
resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
|
|
7238
8229
|
} else {
|
|
7239
|
-
const
|
|
7240
|
-
|
|
7241
|
-
|
|
7242
|
-
|
|
7243
|
-
|
|
7244
|
-
|
|
8230
|
+
const landed = effectiveSuccess || Boolean(confirmedLandedCommit);
|
|
8231
|
+
const targetStatus = effectiveSuccess
|
|
8232
|
+
? 'completed'
|
|
8233
|
+
: (notLandedInfo ? 'needs_review' : (confirmedLandedCommit ? 'completed' : 'failed'));
|
|
8234
|
+
const source = confirmedLandedCommit ? 'reapDeadRunningJobs:landed' : 'reapDeadRunningJobs';
|
|
8235
|
+
// error/verifierVerdict are stamped BEFORE transitionJob() below —
|
|
8236
|
+
// see the identical ordering fix (and its rationale) in spawnJob's
|
|
8237
|
+
// finalize path: transitionJob's needs_review ledger entry reads
|
|
8238
|
+
// these fields off `job` synchronously the instant it runs, so
|
|
8239
|
+
// setting them after fed the ledger a stale/leftover reason.
|
|
8240
|
+
s.jobs[idx].error = landed ? null : `${transitionReason} (outcome=${outcome})`;
|
|
7245
8241
|
if (notLandedInfo) {
|
|
7246
8242
|
s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
|
|
7247
8243
|
} else {
|
|
7248
8244
|
delete s.jobs[idx].verifierVerdict;
|
|
7249
8245
|
}
|
|
8246
|
+
transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source });
|
|
8247
|
+
s.jobs[idx].exitCode = landed ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
8248
|
+
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
8249
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
8250
|
+
if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
|
|
7250
8251
|
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
7251
8252
|
}
|
|
7252
8253
|
// A pidless spawn that never wrote its own '<slug>.log' into the
|
|
@@ -7282,7 +8283,14 @@ async function reapDeadRunningJobs() {
|
|
|
7282
8283
|
appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
|
|
7283
8284
|
} else if (pidless) {
|
|
7284
8285
|
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
7285
|
-
appendAuditEvent('job_reaped_pidless', {
|
|
8286
|
+
appendAuditEvent('job_reaped_pidless', {
|
|
8287
|
+
slug,
|
|
8288
|
+
cwd: s.jobs[idx].cwd ?? null,
|
|
8289
|
+
outcome,
|
|
8290
|
+
graceMs: PIDLESS_SPAWN_GRACE_MS,
|
|
8291
|
+
landedCommit: s.jobs[idx].landedCommit ?? null,
|
|
8292
|
+
verifierVerdict: s.jobs[idx].verifierVerdict ?? null,
|
|
8293
|
+
});
|
|
7286
8294
|
} else {
|
|
7287
8295
|
console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
|
|
7288
8296
|
}
|
|
@@ -7477,6 +8485,12 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
7477
8485
|
const seen = new Set(hot.map((j) => `${j.slug}|${j.runId ?? ''}`));
|
|
7478
8486
|
const archived = (Array.isArray(historyEntries) ? historyEntries : []).filter((j) => {
|
|
7479
8487
|
if (!j) return false;
|
|
8488
|
+
// needs_review_entry/needs_review_resolution lines (needsReviewLedger.cjs)
|
|
8489
|
+
// share history.jsonl with terminal job rows but carry no `status` — the
|
|
8490
|
+
// History view (SchedulerHistoryView.tsx) renders ScheduleJob rows, so a
|
|
8491
|
+
// ledger line slipping through here would show up as a statusless,
|
|
8492
|
+
// meaningless row in that table.
|
|
8493
|
+
if (j.kind && j.kind !== 'terminal') return false;
|
|
7480
8494
|
const key = `${j.slug}|${j.runId ?? ''}`;
|
|
7481
8495
|
if (seen.has(key)) return false;
|
|
7482
8496
|
seen.add(key);
|
|
@@ -7602,6 +8616,69 @@ function isExhaustedAutoFix(job) {
|
|
|
7602
8616
|
return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
|
|
7603
8617
|
}
|
|
7604
8618
|
|
|
8619
|
+
/**
|
|
8620
|
+
* Verdicts from the POST-RUN GUARDS (commit-guard / shared-tree guard) that a
|
|
8621
|
+
* later, independently-checkable commit/looksDone signal can meaningfully
|
|
8622
|
+
* confirm or refute. Auto-fix investigations only ever launch for FAILING
|
|
8623
|
+
* runs (see isExhaustedAutoFix above and selectAutoFixTargets) — a job parked
|
|
8624
|
+
* by one of these GUARD verdicts exits 0 and never has autoFixAttempted set,
|
|
8625
|
+
* so it is invisible to isExhaustedAutoFix and can sit in needs_review
|
|
8626
|
+
* forever with nothing to spend and nothing to exhaust (PRD 1181, 2026-09-12:
|
|
8627
|
+
* exit 0, commit aff5607 landed, parked on a shared-tree verdict, cleared
|
|
8628
|
+
* only by a human).
|
|
8629
|
+
*
|
|
8630
|
+
* 'worktree_integration_failed' is deliberately EXCLUDED — its damage IS a
|
|
8631
|
+
* commit: one stranded on an unmerged `sm-job/<slug>` branch. A `landedCommit`
|
|
8632
|
+
* existing is not evidence against that verdict, it is a restatement of it,
|
|
8633
|
+
* so admitting it here would auto-complete a row whose work never actually
|
|
8634
|
+
* reached the target branch. It already has its own dedicated, git-native
|
|
8635
|
+
* resolution path (selectMechanicalRecoveryTarget / performMechanicalRecovery
|
|
8636
|
+
* — a real re-attempted merge) and must never be pulled into this ladder.
|
|
8637
|
+
*
|
|
8638
|
+
* 'pidless_reap_with_landed_commit' (PRD 1173, resolvePidlessFailureOverride
|
|
8639
|
+
* in reaperHelpers.cjs) is included: it parks on the exact same shape (exit
|
|
8640
|
+
* never observed / no autoFixAttempted, real landedCommit evidence) as
|
|
8641
|
+
* 'silent_no_op' and 'shared_tree_reverted', and this ladder never trusts
|
|
8642
|
+
* landedCommit alone anyway — applyNeedsReviewAutoResolve only resolves once
|
|
8643
|
+
* job.looksDone independently reconfirms via a fresh commits-since-this-run
|
|
8644
|
+
* scan, which is exactly the "does the commit correspond to THIS dispatch"
|
|
8645
|
+
* re-verification resolvePidlessFailureOverride's own header says the
|
|
8646
|
+
* pidless-reap path itself cannot do. Omitting it here reproduces the same
|
|
8647
|
+
* "nothing to spend, nothing to exhaust" needs_review stall this PRD exists
|
|
8648
|
+
* to fix, just for a third verdict.
|
|
8649
|
+
*/
|
|
8650
|
+
const GUARD_VERDICT_EVIDENCE_ELIGIBLE = new Set(['silent_no_op', 'shared_tree_reverted', 'pidless_reap_with_landed_commit']);
|
|
8651
|
+
|
|
8652
|
+
/**
|
|
8653
|
+
* Pure predicate, no I/O: a needs_review row parked directly by one of the
|
|
8654
|
+
* GUARD_VERDICT_EVIDENCE_ELIGIBLE verdicts, that never went through an
|
|
8655
|
+
* auto-fix investigation at all (autoFixAttempted is not true) — the
|
|
8656
|
+
* structural gap this PRD closes, distinct from isExhaustedAutoFix's "went
|
|
8657
|
+
* through auto-fix and spent it" case. A job that DID get an auto-fix
|
|
8658
|
+
* investigation is left to isExhaustedAutoFix's own ladder rather than this
|
|
8659
|
+
* one, even if its verifierVerdict happens to also be in the eligible set.
|
|
8660
|
+
* Exported for tests.
|
|
8661
|
+
*/
|
|
8662
|
+
function isGuardParkedWithoutAutoFix(job) {
|
|
8663
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
8664
|
+
if (job.autoFixAttempted === true) return false;
|
|
8665
|
+
return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
|
|
8666
|
+
}
|
|
8667
|
+
|
|
8668
|
+
/**
|
|
8669
|
+
* Pure predicate, no I/O: is this needs_review row eligible for the bounded
|
|
8670
|
+
* auto-resolve ladder at all — either because its auto-fix path is genuinely
|
|
8671
|
+
* spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
|
|
8672
|
+
* that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
|
|
8673
|
+
* Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
|
|
8674
|
+
* duplicated one — the ladder itself doesn't care which door a row came
|
|
8675
|
+
* through, only whether it now carries completion evidence (job.looksDone).
|
|
8676
|
+
* Exported for tests.
|
|
8677
|
+
*/
|
|
8678
|
+
function isEligibleForNeedsReviewAutoResolve(job) {
|
|
8679
|
+
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
|
|
8680
|
+
}
|
|
8681
|
+
|
|
7605
8682
|
/**
|
|
7606
8683
|
* Pure predicate: an investigation produced a fix plan (autoFixOutcome ===
|
|
7607
8684
|
* 'plan') but its fix-plan slug is not present among `queuedSlugs` — the
|
|
@@ -7749,17 +8826,26 @@ function isRescanCandidate(job) {
|
|
|
7749
8826
|
* selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
|
|
7750
8827
|
* again be narrower than the work reverifyNeedsReview performs.
|
|
7751
8828
|
*
|
|
7752
|
-
*
|
|
7753
|
-
*
|
|
7754
|
-
*
|
|
7755
|
-
*
|
|
7756
|
-
*
|
|
7757
|
-
*
|
|
8829
|
+
* Widened again (this PRD): a `needs_review` row parked directly by a GUARD
|
|
8830
|
+
* verdict with no auto-fix history (isGuardParkedWithoutAutoFix) is not an
|
|
8831
|
+
* isRescanCandidate either — RESCANNABLE_VERDICTS covers transcript-verifier
|
|
8832
|
+
* verdicts, not commit-guard/shared-tree-guard verdicts — but
|
|
8833
|
+
* reverifyNeedsReview's looksDone-annotation pass now runs for it too (see
|
|
8834
|
+
* that function). Same rule as always: never let this guard be narrower than
|
|
8835
|
+
* the work reverifyNeedsReview actually performs.
|
|
8836
|
+
*
|
|
8837
|
+
* Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
|
|
8838
|
+
* isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
|
|
8839
|
+
* called with an injected fixSlugExists that always returns false — cheap
|
|
8840
|
+
* and deliberately over-inclusive (a false positive here just means one
|
|
8841
|
+
* extra periodic pass, never a missed one) so this guard never pays
|
|
8842
|
+
* selectAutoFixTargets's production fs.existsSync scan per tick.
|
|
8843
|
+
* resolveRunId's IO only fires for rows missing job.runId, same as
|
|
7758
8844
|
* isRescanCandidate already incurs above.
|
|
7759
8845
|
*/
|
|
7760
8846
|
function shouldRunPeriodicReverify(jobs) {
|
|
7761
8847
|
if (!Array.isArray(jobs)) return false;
|
|
7762
|
-
if (jobs.some((j) => isRescanCandidate(j))) return true;
|
|
8848
|
+
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
|
|
7763
8849
|
if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
|
|
7764
8850
|
return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
|
|
7765
8851
|
}
|
|
@@ -7781,24 +8867,90 @@ function stuckFailedEscalationDisabled() {
|
|
|
7781
8867
|
return process.env.SM_STUCK_FAILED_ESCALATE_DISABLE === '1';
|
|
7782
8868
|
}
|
|
7783
8869
|
|
|
8870
|
+
// Bounded automatic failed -> pending recovery (PRD 1151): a failed row gets
|
|
8871
|
+
// up to FAILED_AUTORESET_CAP auto-reset attempts, each gated on having sat
|
|
8872
|
+
// `failed` for FAILED_AUTORESET_MS, before the stuck-failed escalation below
|
|
8873
|
+
// is allowed to page a human. Same env-override shape as
|
|
8874
|
+
// STUCK_FAILED_ESCALATE_MS/QUARANTINE_ESCALATE_MS above.
|
|
8875
|
+
const FAILED_AUTORESET_CAP = 3;
|
|
8876
|
+
const FAILED_AUTORESET_MS = process.env.SM_FAILED_AUTORESET_MINUTES
|
|
8877
|
+
? Number(process.env.SM_FAILED_AUTORESET_MINUTES) * 60_000
|
|
8878
|
+
: 10 * 60_000;
|
|
8879
|
+
|
|
8880
|
+
/**
|
|
8881
|
+
* Kill-switch gate for the failed-autoreset pass below
|
|
8882
|
+
* (SM_FAILED_AUTORESET_DISABLE=1), same shape as stuckFailedEscalationDisabled
|
|
8883
|
+
* above.
|
|
8884
|
+
*/
|
|
8885
|
+
function failedAutoResetDisabled() {
|
|
8886
|
+
return process.env.SM_FAILED_AUTORESET_DISABLE === '1';
|
|
8887
|
+
}
|
|
8888
|
+
|
|
8889
|
+
/**
|
|
8890
|
+
* selectFailedAutoResetTargets(jobs, now, thresholdMs) →
|
|
8891
|
+
* [{ slug, cwd, ageMs, attempts }]
|
|
8892
|
+
*
|
|
8893
|
+
* Pure selector — no IO, no `require` inside the function. Selects `failed`
|
|
8894
|
+
* rows whose newest statusHistory entry with `to === 'failed'` is older than
|
|
8895
|
+
* `thresholdMs` and whose failedAutoResetAttempts counter hasn't yet spent
|
|
8896
|
+
* FAILED_AUTORESET_CAP attempts. "Newest" (not first) matters because a row
|
|
8897
|
+
* can have failed more than once across its lifetime (an earlier auto-reset
|
|
8898
|
+
* attempt that itself failed again) — only the most recent failed-since
|
|
8899
|
+
* timestamp should gate the next attempt.
|
|
8900
|
+
*
|
|
8901
|
+
* Excludes a row whose newest failed-entry came from spawnJob:fail-dirty —
|
|
8902
|
+
* that source means a transient failure left genuinely uncommitted work in
|
|
8903
|
+
* the job's worktree and the system already decided once, deliberately, not
|
|
8904
|
+
* to auto-requeue it (see that call site's own comment: "could discard
|
|
8905
|
+
* uncommitted work left by the failed run"). This bounded auto-reset is a
|
|
8906
|
+
* different, slower mechanism and must not quietly override that decision
|
|
8907
|
+
* 10 minutes later — a human should look at a dirty worktree before it gets
|
|
8908
|
+
* re-driven.
|
|
8909
|
+
*/
|
|
8910
|
+
function selectFailedAutoResetTargets(jobs, now, thresholdMs) {
|
|
8911
|
+
const targets = [];
|
|
8912
|
+
for (const j of jobs ?? []) {
|
|
8913
|
+
if (j.status !== 'failed') continue;
|
|
8914
|
+
const attempts = j.failedAutoResetAttempts ?? 0;
|
|
8915
|
+
if (attempts >= FAILED_AUTORESET_CAP) continue;
|
|
8916
|
+
const history = j.statusHistory || [];
|
|
8917
|
+
let entry = null;
|
|
8918
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
8919
|
+
if (history[i].to === 'failed') { entry = history[i]; break; }
|
|
8920
|
+
}
|
|
8921
|
+
if (!entry) continue;
|
|
8922
|
+
if (entry.source === 'spawnJob:fail-dirty') continue;
|
|
8923
|
+
const since = Date.parse(entry.at);
|
|
8924
|
+
if (Number.isNaN(since)) continue;
|
|
8925
|
+
const ageMs = now - since;
|
|
8926
|
+
if (ageMs < thresholdMs) continue;
|
|
8927
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts });
|
|
8928
|
+
}
|
|
8929
|
+
return targets;
|
|
8930
|
+
}
|
|
8931
|
+
|
|
7784
8932
|
/**
|
|
7785
8933
|
* findStuckFailedJobs(jobs, now, thresholdMs) → [{ slug, cwd, ageMs }]
|
|
7786
8934
|
*
|
|
7787
8935
|
* Pure predicate (isRescanCandidate's own resolveRunId/classifyRunOutcome log
|
|
7788
8936
|
* read is the only IO, gated per-job exactly like shouldRunPeriodicReverify
|
|
7789
|
-
* above). `failed`
|
|
7790
|
-
* path — selectResumeRecoveryTarget/selectAutoFixTargets both
|
|
7791
|
-
* needs_review, reapDeadRunningJobs only ever writes running →
|
|
7792
|
-
* reconcile-repair's to-pending is for structurally invalid rows.
|
|
7793
|
-
*
|
|
7794
|
-
*
|
|
7795
|
-
*
|
|
7796
|
-
*
|
|
7797
|
-
*
|
|
7798
|
-
*
|
|
7799
|
-
*
|
|
7800
|
-
*
|
|
7801
|
-
*
|
|
8937
|
+
* above). `failed` used to be a fully terminal state for every automated
|
|
8938
|
+
* recovery path — selectResumeRecoveryTarget/selectAutoFixTargets both
|
|
8939
|
+
* require needs_review, reapDeadRunningJobs only ever writes running →
|
|
8940
|
+
* failed, and reconcile-repair's to-pending is for structurally invalid rows.
|
|
8941
|
+
* That is no longer true: selectFailedAutoResetTargets above now drives a
|
|
8942
|
+
* bounded failed → pending auto-reset (LEGAL_TRANSITIONS already allowed the
|
|
8943
|
+
* edge). This escalation now only fires once that auto-reset budget is
|
|
8944
|
+
* genuinely spent (see the interval body's filter on failedAutoResetAttempts)
|
|
8945
|
+
* — a rescan candidate (isRescanCandidate) that has sat failed longer than
|
|
8946
|
+
* `thresholdMs` AND exhausted its auto-reset attempts can therefore go
|
|
8947
|
+
* silently stuck forever — job 4056-outcome-stats sat `failed` for five days
|
|
8948
|
+
* with no operator signal (reported 2026-09-10, social-signals-trader) even
|
|
8949
|
+
* though the periodic reverify pass (once shouldRunPeriodicReverify's guard
|
|
8950
|
+
* was fixed) WAS firing on it — reverifyNeedsReview's failed branch can
|
|
8951
|
+
* annotate looksDone but can never resolve a failed row itself (see its own
|
|
8952
|
+
* header). This is the visibility half that guard fix was missing: escalate
|
|
8953
|
+
* once per exhausted row, never requeue from here.
|
|
7802
8954
|
*
|
|
7803
8955
|
* `stuckFailedNotified` gates this to exactly once per row — once the caller
|
|
7804
8956
|
* stamps it, this always excludes that row so a human is never re-paged on
|
|
@@ -7812,7 +8964,18 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
|
|
|
7812
8964
|
if (j.status !== 'failed') continue;
|
|
7813
8965
|
if (j.stuckFailedNotified === true) continue;
|
|
7814
8966
|
if (!isRescanCandidate(j)) continue;
|
|
7815
|
-
|
|
8967
|
+
// Newest (not first) to === 'failed' entry — same rationale as
|
|
8968
|
+
// selectFailedAutoResetTargets above: a row can have failed more than
|
|
8969
|
+
// once across its lifetime (an earlier auto-reset attempt that itself
|
|
8970
|
+
// failed again), and only the CURRENT failure episode's age should gate
|
|
8971
|
+
// escalation. Using the first/oldest entry would report a stale age
|
|
8972
|
+
// (and become instantly escalation-eligible) for a row that failed
|
|
8973
|
+
// months ago, recovered, and has only just failed again.
|
|
8974
|
+
const history = j.statusHistory || [];
|
|
8975
|
+
let entry = null;
|
|
8976
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
8977
|
+
if (history[i].to === 'failed') { entry = history[i]; break; }
|
|
8978
|
+
}
|
|
7816
8979
|
if (!entry) continue;
|
|
7817
8980
|
const since = Date.parse(entry.at);
|
|
7818
8981
|
if (Number.isNaN(since)) continue;
|
|
@@ -7822,6 +8985,138 @@ function findStuckFailedJobs(jobs, now, thresholdMs) {
|
|
|
7822
8985
|
return stuck;
|
|
7823
8986
|
}
|
|
7824
8987
|
|
|
8988
|
+
// Bounded automatic terminal decision for an EXHAUSTED needs_review row
|
|
8989
|
+
// (isExhaustedAutoFix === true — auto-fix attempted, no plan produced,
|
|
8990
|
+
// retries spent): up to NEEDS_REVIEW_RESOLVE_CAP requeue attempts (a
|
|
8991
|
+
// needs_review -> pending -> ... -> needs_review round trip counts as one
|
|
8992
|
+
// spent attempt), each gated on having sat exhausted-needs_review for
|
|
8993
|
+
// NEEDS_REVIEW_RESOLVE_MS, before the row is auto-skipped so a `dependsOn`
|
|
8994
|
+
// chain behind it always drains without an operator. Same env-override
|
|
8995
|
+
// shape as FAILED_AUTORESET_MS above.
|
|
8996
|
+
const NEEDS_REVIEW_RESOLVE_CAP = 2;
|
|
8997
|
+
const NEEDS_REVIEW_RESOLVE_MS = process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES
|
|
8998
|
+
? Number(process.env.SM_NEEDS_REVIEW_RESOLVE_MINUTES) * 60_000
|
|
8999
|
+
: 30 * 60_000;
|
|
9000
|
+
|
|
9001
|
+
/**
|
|
9002
|
+
* Kill-switch gate for the needs_review auto-resolve pass below
|
|
9003
|
+
* (SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1), same shape as
|
|
9004
|
+
* failedAutoResetDisabled/stuckFailedEscalationDisabled above.
|
|
9005
|
+
*/
|
|
9006
|
+
function needsReviewAutoResolveDisabled() {
|
|
9007
|
+
return process.env.SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE === '1';
|
|
9008
|
+
}
|
|
9009
|
+
|
|
9010
|
+
/**
|
|
9011
|
+
* selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
|
|
9012
|
+
* [{ slug, cwd, ageMs, attempts }]
|
|
9013
|
+
*
|
|
9014
|
+
* Pure selector — no IO. Selects `needs_review` rows eligible for the
|
|
9015
|
+
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
|
|
9016
|
+
* auto-fix genuinely spent, or parked by a GUARD verdict that never entered
|
|
9017
|
+
* auto-fix at all), whose newest statusHistory entry with `to ===
|
|
9018
|
+
* 'needs_review'` is older than `thresholdMs`, and whose
|
|
9019
|
+
* exhaustedResolveAttempts counter has not yet spent its cap.
|
|
9020
|
+
*
|
|
9021
|
+
* The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
|
|
9022
|
+
* CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
|
|
9023
|
+
* the pass that observes attempts === CAP is exactly the one that must fire
|
|
9024
|
+
* the terminal skip (see the interval body's branch below) — excluding that
|
|
9025
|
+
* row here would mean the cap-exhausted row is never selected again and the
|
|
9026
|
+
* dependsOn chain behind it never drains, defeating this PRD's own purpose.
|
|
9027
|
+
*/
|
|
9028
|
+
function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
9029
|
+
const targets = [];
|
|
9030
|
+
for (const j of jobs ?? []) {
|
|
9031
|
+
if (j.status !== 'needs_review') continue;
|
|
9032
|
+
if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
|
|
9033
|
+
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
|
|
9034
|
+
const history = j.statusHistory || [];
|
|
9035
|
+
let entry = null;
|
|
9036
|
+
for (let i = history.length - 1; i >= 0; i--) {
|
|
9037
|
+
if (history[i].to === 'needs_review') { entry = history[i]; break; }
|
|
9038
|
+
}
|
|
9039
|
+
if (!entry) continue;
|
|
9040
|
+
const since = Date.parse(entry.at);
|
|
9041
|
+
if (Number.isNaN(since)) continue;
|
|
9042
|
+
const ageMs = now - since;
|
|
9043
|
+
if (ageMs < thresholdMs) continue;
|
|
9044
|
+
targets.push({ slug: j.slug, cwd: j.cwd ?? null, ageMs, attempts: j.exhaustedResolveAttempts ?? 0 });
|
|
9045
|
+
}
|
|
9046
|
+
return targets;
|
|
9047
|
+
}
|
|
9048
|
+
|
|
9049
|
+
/**
|
|
9050
|
+
* Applies the needs_review auto-resolve decision to a single job (mutates in
|
|
9051
|
+
* place; calls transitionJob + appendAuditEvent). Extracted from the
|
|
9052
|
+
* interval body so the three branches are unit-testable without going
|
|
9053
|
+
* through mutate()/queue.json IO. Order of decision:
|
|
9054
|
+
* 1. job.looksDone (the annotation reverifyNeedsReview writes when a
|
|
9055
|
+
* later commit touches the PRD's declared paths) -> 'completed'.
|
|
9056
|
+
* 2. otherwise, one more bounded requeue -> 'pending', incrementing
|
|
9057
|
+
* exhaustedResolveAttempts.
|
|
9058
|
+
* 3. once NEEDS_REVIEW_RESOLVE_CAP requeue attempts are spent -> 'skipped',
|
|
9059
|
+
* with job.error naming the exhausted path so the Queue UI still shows
|
|
9060
|
+
* why, and a marker (needsReviewAutoResolvedSkip) that findBlockingDep
|
|
9061
|
+
* reads to stop treating this SPECIFIC skip as a permanent dependsOn
|
|
9062
|
+
* block — unlike a generic "PRD source vanished" skip, this row was
|
|
9063
|
+
* given every bounded chance to resolve itself.
|
|
9064
|
+
* Re-validates status/exhaustion/cap itself (same race-guard shape as the
|
|
9065
|
+
* failed-autoreset loop above) so a stale target computed before this
|
|
9066
|
+
* mutate() pass can never double-apply. Returns the outcome, or null if the
|
|
9067
|
+
* race guard rejected it.
|
|
9068
|
+
*
|
|
9069
|
+
* A row can reach here through either door (isEligibleForNeedsReviewAutoResolve):
|
|
9070
|
+
* auto-fix genuinely exhausted, or parked directly by a GUARD_VERDICT_EVIDENCE_
|
|
9071
|
+
* ELIGIBLE verdict with no auto-fix history at all. `originIsGuardParked` picks
|
|
9072
|
+
* which door this particular row came through, purely to make the requeue/skip
|
|
9073
|
+
* reason text (and the Queue UI's job.error) name the RIGHT evidence — a
|
|
9074
|
+
* guard-parked row was never "exhausted auto-fix" and must never claim to be.
|
|
9075
|
+
*/
|
|
9076
|
+
function applyNeedsReviewAutoResolve(j) {
|
|
9077
|
+
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
|
|
9078
|
+
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
|
|
9079
|
+
const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
|
|
9080
|
+
|
|
9081
|
+
if (j.looksDone) {
|
|
9082
|
+
const attempt = j.exhaustedResolveAttempts ?? 0;
|
|
9083
|
+
const reason = originIsGuardParked
|
|
9084
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with landed commit and looksDone evidence (${j.looksDone.rule}) — `
|
|
9085
|
+
+ `${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths, work landed`
|
|
9086
|
+
: `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.rule} — ${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths)`;
|
|
9087
|
+
transitionJob(j, 'completed', { reason, source: 'needsReviewAutoResolve' });
|
|
9088
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
|
|
9089
|
+
return 'completed';
|
|
9090
|
+
}
|
|
9091
|
+
|
|
9092
|
+
const attemptsSoFar = j.exhaustedResolveAttempts ?? 0;
|
|
9093
|
+
if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
|
|
9094
|
+
const attempt = attemptsSoFar + 1;
|
|
9095
|
+
j.exhaustedResolveAttempts = attempt;
|
|
9096
|
+
const reason = originIsGuardParked
|
|
9097
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence yet — `
|
|
9098
|
+
+ `requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`
|
|
9099
|
+
: `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`;
|
|
9100
|
+
transitionJob(j, 'pending', { reason, source: 'needsReviewAutoResolve' });
|
|
9101
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
|
|
9102
|
+
return 'requeued';
|
|
9103
|
+
}
|
|
9104
|
+
|
|
9105
|
+
j.needsReviewAutoResolvedSkip = true;
|
|
9106
|
+
j.error = originIsGuardParked
|
|
9107
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence after `
|
|
9108
|
+
+ `${NEEDS_REVIEW_RESOLVE_CAP} requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`
|
|
9109
|
+
: `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
|
|
9110
|
+
+ `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
|
|
9111
|
+
+ `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
|
|
9112
|
+
transitionJob(j, 'skipped', {
|
|
9113
|
+
reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
|
|
9114
|
+
source: 'needsReviewAutoResolve',
|
|
9115
|
+
});
|
|
9116
|
+
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'skipped', attempt: attemptsSoFar });
|
|
9117
|
+
return 'skipped';
|
|
9118
|
+
}
|
|
9119
|
+
|
|
7825
9120
|
/**
|
|
7826
9121
|
* Self-healing pass over needs_review jobs. The verifier runs in-process, so a
|
|
7827
9122
|
* fix to runVerify.cjs only takes effect for jobs verified AFTER an app
|
|
@@ -7879,6 +9174,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
7879
9174
|
// fix-plan investigation to diagnose — there is no code defect to
|
|
7880
9175
|
// author a PRD against, only another job's still-uncommitted tree.
|
|
7881
9176
|
if (job.blockedByForeignWip === true) return false;
|
|
9177
|
+
// A budget-killed job parks for a human/ladder decision, never an
|
|
9178
|
+
// auto-fix investigation or auto-retry — the run didn't fail, it simply
|
|
9179
|
+
// overran its own estimate; there's no code defect to diagnose (Out of
|
|
9180
|
+
// scope: "retrying or auto-resuming a budget-killed job" for this PRD).
|
|
9181
|
+
if (job.verifierVerdict === 'budget_exceeded') return false;
|
|
7882
9182
|
// A stale re-run whose work already shipped (rcaReport's 'already-shipped'
|
|
7883
9183
|
// class) must never buy a fix-plan PRD — there is nothing to fix, and the
|
|
7884
9184
|
// correct recovery (archiving the PRD) is a human/reconcile action, not
|
|
@@ -7938,35 +9238,153 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
|
|
|
7938
9238
|
}
|
|
7939
9239
|
|
|
7940
9240
|
/**
|
|
7941
|
-
*
|
|
7942
|
-
*
|
|
7943
|
-
*
|
|
7944
|
-
*
|
|
9241
|
+
* attributeLandedCommits(job, pathCommits, cwd) → { commits, rule } | null
|
|
9242
|
+
*
|
|
9243
|
+
* Narrows a set of PATH-overlapping commits (computeLooksDone's
|
|
9244
|
+
* `landedSinceRun` result — any commit touching the PRD's declared paths,
|
|
9245
|
+
* regardless of who authored it) down to the subset actually attributable to
|
|
9246
|
+
* THIS job's own run. Path overlap alone is not attribution: inside one Epic,
|
|
9247
|
+
* sibling PRDs routinely declare the same hot file, so a sibling's commit is
|
|
9248
|
+
* indistinguishable from this job's own by path alone (the incident this
|
|
9249
|
+
* function exists to close — PRD 1204's parked row cited PRD 1205's commit
|
|
9250
|
+
* 5dadf3c as its own evidence).
|
|
9251
|
+
*
|
|
9252
|
+
* Three rules, tried strongest-first, first match wins:
|
|
9253
|
+
*
|
|
9254
|
+
* 1. 'landedCommit' — the row's own `job.landedCommit`, re-verified here via
|
|
9255
|
+
* `resolveLandedCommitEvidence` against THIS job's `startedAt`. This is
|
|
9256
|
+
* the strongest signal because it is not inferred from `git log` at all:
|
|
9257
|
+
* it is the sha spawnJob's own finalize step observed THIS dispatch's
|
|
9258
|
+
* worktree/branch landing (see resolveLandedCommitEvidence's own header
|
|
9259
|
+
* for why it also guards against a stale sha surviving a reset). Trusted
|
|
9260
|
+
* independent of whether it appears in `pathCommits` — it is definitionally
|
|
9261
|
+
* this job's own work, not something discovered by scanning history.
|
|
9262
|
+
* 2. 'job branch' — a path-overlapping commit reachable from (an ancestor of
|
|
9263
|
+
* or equal to) this job's own `sm-job/<slug>` branch tip. Still
|
|
9264
|
+
* job-specific even though it IS a `git log` scan: a sibling's commit can
|
|
9265
|
+
* never be an ancestor of THIS job's own branch ref. In practice this
|
|
9266
|
+
* branch is deleted on successful integration (gitWorktree.cjs's
|
|
9267
|
+
* `cleanupWorktree`), so this mainly fires when integration failed and
|
|
9268
|
+
* the branch was deliberately kept for recovery, or reverify runs before
|
|
9269
|
+
* cleanup — a narrower window than rule 1, hence checked second.
|
|
9270
|
+
* 3. 'slug trailer' — a path-overlapping commit whose message contains this
|
|
9271
|
+
* job's slug verbatim. Weakest of the three (a coincidental substring
|
|
9272
|
+
* match is possible, and nothing stamps this automatically today), so it
|
|
9273
|
+
* is the last resort when the two structural signals above found
|
|
9274
|
+
* nothing.
|
|
9275
|
+
*
|
|
9276
|
+
* Deliberately NOT a rule: raw path overlap by itself (the bug this function
|
|
9277
|
+
* fixes) and `committedInWindow`-style time-window-only evidence — a sibling
|
|
9278
|
+
* job running concurrently in the very same window is exactly as invisible to
|
|
9279
|
+
* a time bound as it is to a path filter, so neither narrows attribution.
|
|
9280
|
+
*
|
|
9281
|
+
* Never throws: a missing ref, an unresolvable sha, or any git failure for a
|
|
9282
|
+
* given commit/rule is treated as "that commit doesn't satisfy this rule",
|
|
9283
|
+
* never as a fabricated match.
|
|
9284
|
+
*/
|
|
9285
|
+
async function attributeLandedCommits(job, pathCommits, cwd) {
|
|
9286
|
+
if (job?.landedCommit && await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)) {
|
|
9287
|
+
return { commits: [job.landedCommit], rule: 'landedCommit' };
|
|
9288
|
+
}
|
|
9289
|
+
|
|
9290
|
+
const branch = `sm-job/${job?.slug}`;
|
|
9291
|
+
const branchCommits = [];
|
|
9292
|
+
for (const sha of pathCommits) {
|
|
9293
|
+
try {
|
|
9294
|
+
await execGitAt(cwd, ['merge-base', '--is-ancestor', sha, branch], { timeout: 10_000 });
|
|
9295
|
+
branchCommits.push(sha);
|
|
9296
|
+
} catch { /* not an ancestor of this job's own branch, or branch doesn't exist */ }
|
|
9297
|
+
}
|
|
9298
|
+
if (branchCommits.length) return { commits: branchCommits, rule: 'job branch' };
|
|
9299
|
+
|
|
9300
|
+
if (job?.slug) {
|
|
9301
|
+
const trailerCommits = [];
|
|
9302
|
+
for (const sha of pathCommits) {
|
|
9303
|
+
try {
|
|
9304
|
+
const msg = await execGitAt(cwd, ['log', '-1', '--format=%B', sha], { timeout: 10_000 });
|
|
9305
|
+
if (msg.includes(job.slug)) trailerCommits.push(sha);
|
|
9306
|
+
} catch { /* unresolvable sha */ }
|
|
9307
|
+
}
|
|
9308
|
+
if (trailerCommits.length) return { commits: trailerCommits, rule: 'slug trailer' };
|
|
9309
|
+
}
|
|
9310
|
+
|
|
9311
|
+
return null;
|
|
9312
|
+
}
|
|
9313
|
+
|
|
9314
|
+
/**
|
|
9315
|
+
* Widened evidence check (PRD 1102, narrowed to per-job attribution by a
|
|
9316
|
+
* later PRD): does at least one commit ATTRIBUTABLE TO THIS JOB land AFTER
|
|
9317
|
+
* its run window and touch a path the PRD itself declares? Scoped to the
|
|
9318
|
+
* PRD's own declared paths (never the whole repo) so a sibling job's
|
|
9319
|
+
* unrelated commit is never even considered — see healRefusalReason's own
|
|
7945
9320
|
* rationale for why unscoped, repo-wide evidence is not attribution.
|
|
7946
9321
|
*
|
|
9322
|
+
* Path overlap alone is NOT evidence (see attributeLandedCommits's header):
|
|
9323
|
+
* a sibling PRD in the same Epic routinely declares the same hot file, so
|
|
9324
|
+
* `landedSinceRun`'s raw result is only a candidate list — the returned
|
|
9325
|
+
* annotation is null unless `attributeLandedCommits` narrows it to at least
|
|
9326
|
+
* one commit this job can actually claim.
|
|
9327
|
+
*
|
|
7947
9328
|
* Returns null (no annotation, never fabricated) when the PRD names no
|
|
7948
|
-
* paths
|
|
7949
|
-
*
|
|
9329
|
+
* paths, when no commit touches a declared path at all, or when
|
|
9330
|
+
* path-overlapping commits exist but none are attributable to this job — the
|
|
9331
|
+
* caller then has only the existing, already-computed committedInWindow
|
|
9332
|
+
* signal to go on, same as before this PRD.
|
|
7950
9333
|
*
|
|
7951
|
-
*
|
|
9334
|
+
* `fetchedCwds` (optional) lets a caller iterating many candidates in one
|
|
9335
|
+
* pass (reverifyNeedsReview) dedupe the `git fetch --all --prune` across
|
|
9336
|
+
* candidates that share a `cwd` — several `needs_review` rows for the same
|
|
9337
|
+
* project is the common case a backlog produces, and each fetch is up to
|
|
9338
|
+
* ~20s, so re-fetching the same repo once per row multiplies that pass's
|
|
9339
|
+
* wall-clock cost for zero new evidence. Omitted (or a fresh Set per call)
|
|
9340
|
+
* simply always fetches, unchanged from before this cache existed.
|
|
9341
|
+
*
|
|
9342
|
+
* @returns {Promise<{commits: string[], paths: string[], detectedAt: string, rule: string} | null>}
|
|
7952
9343
|
*/
|
|
7953
|
-
async function computeLooksDone(job) {
|
|
9344
|
+
async function computeLooksDone(job, fetchedCwds) {
|
|
7954
9345
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
7955
9346
|
const paths = declaredPathsForPrd(prdPath);
|
|
7956
9347
|
if (!paths.length) return null;
|
|
7957
|
-
|
|
9348
|
+
if (!fetchedCwds || !fetchedCwds.has(job.cwd)) {
|
|
9349
|
+
await fetchAllRefs(job.cwd);
|
|
9350
|
+
if (fetchedCwds) fetchedCwds.add(job.cwd);
|
|
9351
|
+
}
|
|
7958
9352
|
const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
|
|
7959
9353
|
if (!commits.length) return null;
|
|
7960
|
-
|
|
9354
|
+
const attributed = await attributeLandedCommits(job, commits, job.cwd);
|
|
9355
|
+
if (!attributed) return null;
|
|
9356
|
+
return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
|
|
7961
9357
|
}
|
|
7962
9358
|
|
|
7963
9359
|
async function reverifyNeedsReview() {
|
|
7964
9360
|
const snap = await readQueue();
|
|
7965
|
-
|
|
9361
|
+
// isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
|
|
9362
|
+
// verifierVerdict is a commit-guard/shared-tree-guard verdict, not a
|
|
9363
|
+
// RESCANNABLE_VERDICTS transcript-verifier one) — included here so this
|
|
9364
|
+
// pass also computes their looksDone evidence, the widened half of the
|
|
9365
|
+
// guard-verdict auto-resolve gap this PRD closes. Handled in its own
|
|
9366
|
+
// branch below (no transcript rescan — there is no transcript verdict to
|
|
9367
|
+
// rescan) rather than through the isRescanCandidate machinery.
|
|
9368
|
+
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
|
|
7966
9369
|
const healed = [];
|
|
7967
9370
|
const leftForReview = [];
|
|
7968
9371
|
const looksDoneUpdates = [];
|
|
9372
|
+
// Shared across every computeLooksDone call in this one pass — dedupes
|
|
9373
|
+
// the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
|
|
9374
|
+
// header) rather than re-fetching the same repo once per candidate row.
|
|
9375
|
+
const fetchedCwds = new Set();
|
|
7969
9376
|
for (const job of candidates) {
|
|
9377
|
+
if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
|
|
9378
|
+
// Guard-verdict park, never auto-fixed: only evidence gathering, never
|
|
9379
|
+
// a transcript rescan (there was never a transcript-verifier verdict
|
|
9380
|
+
// here) and never a direct heal — applyNeedsReviewAutoResolve is the
|
|
9381
|
+
// sole place that turns this annotation into a status change.
|
|
9382
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
9383
|
+
if (looksDone) {
|
|
9384
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
9385
|
+
}
|
|
9386
|
+
continue;
|
|
9387
|
+
}
|
|
7970
9388
|
if (job.status === 'failed') {
|
|
7971
9389
|
// A failed row never runs the transcript-verifier rescan below — that
|
|
7972
9390
|
// machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
|
|
@@ -7975,7 +9393,7 @@ async function reverifyNeedsReview() {
|
|
|
7975
9393
|
// completing-direction constraint). The only thing a failed candidate
|
|
7976
9394
|
// can gain here is a looksDone annotation + a failed → needs_review
|
|
7977
9395
|
// transition, for a human to confirm.
|
|
7978
|
-
const looksDone = await computeLooksDone(job);
|
|
9396
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
7979
9397
|
if (looksDone) {
|
|
7980
9398
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
|
|
7981
9399
|
} else {
|
|
@@ -8030,7 +9448,7 @@ async function reverifyNeedsReview() {
|
|
|
8030
9448
|
// always before this periodic/boot pass can run against the same row, so
|
|
8031
9449
|
// this check reliably catches the only order that can occur.
|
|
8032
9450
|
if (stillOpen && job.autoFixAttempted !== true) {
|
|
8033
|
-
const looksDone = await computeLooksDone(job);
|
|
9451
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
8034
9452
|
if (looksDone) {
|
|
8035
9453
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
8036
9454
|
}
|
|
@@ -8044,14 +9462,14 @@ async function reverifyNeedsReview() {
|
|
|
8044
9462
|
if (!u) continue;
|
|
8045
9463
|
if (u.fromFailed) {
|
|
8046
9464
|
transitionJob(j, 'needs_review', {
|
|
8047
|
-
reason:
|
|
9465
|
+
reason: `looks done (${u.looksDone.rule}) — commit(s) attributable to this job's own run touch this PRD's declared paths; confirm before archiving`,
|
|
8048
9466
|
source: 'reverifyNeedsReview:looksDone',
|
|
8049
9467
|
});
|
|
8050
9468
|
}
|
|
8051
9469
|
if (j.status !== 'needs_review') continue;
|
|
8052
9470
|
j.looksDone = u.looksDone;
|
|
8053
9471
|
const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
|
|
8054
|
-
j.error = `looks done — ${u.looksDone.commits.length} commit(s)
|
|
9472
|
+
j.error = `looks done (${u.looksDone.rule}) — ${u.looksDone.commits.length} commit(s) attributable to this job's own run touch this PRD's paths (${shaList}); confirm before archiving`;
|
|
8055
9473
|
}
|
|
8056
9474
|
});
|
|
8057
9475
|
console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
|
|
@@ -8323,11 +9741,21 @@ function registerScheduleHandlers() {
|
|
|
8323
9741
|
ensureDirs();
|
|
8324
9742
|
supervisor.registerHandlers();
|
|
8325
9743
|
|
|
9744
|
+
// Cheap read only — no reconcile(), no writeQueue(). The renderer treats
|
|
9745
|
+
// this as a fast call behind a 5s deadline (scheduleState.ts's
|
|
9746
|
+
// withTimeout), but reconcile() does a cross-project PRD discovery walk
|
|
9747
|
+
// plus a disk write, which could blow that deadline and, worse, throw
|
|
9748
|
+
// outright on a torn queue.json (reconcile refuses to run against
|
|
9749
|
+
// `state.unreadable`) — turning a recoverable read into a rejected IPC and
|
|
9750
|
+
// an error toast. Discovery still runs on a fixed cadence elsewhere:
|
|
9751
|
+
// tickQueue (every POLL_INTERVAL_MS, 60s), rescheduleTimer, broadcast()'s
|
|
9752
|
+
// coalescer (getPayload), schedule:rescan and schedule:adopt-prd. Worst
|
|
9753
|
+
// case, a PRD dropped on disk while the Scheduler tab is open surfaces
|
|
9754
|
+
// here within ~POLL_INTERVAL_MS + BROADCAST_COALESCE_MS (~60.2s) — via
|
|
9755
|
+
// tickQueue's reconcile + its trailing broadcast() — not via this handler.
|
|
8326
9756
|
ipcMain.handle('schedule:state', async () => {
|
|
8327
9757
|
const state = await readQueue();
|
|
8328
|
-
|
|
8329
|
-
await writeQueue(state);
|
|
8330
|
-
return buildScheduleStatePayload(state, { withPaths: true });
|
|
9758
|
+
return buildScheduleStatePayload(state);
|
|
8331
9759
|
});
|
|
8332
9760
|
|
|
8333
9761
|
// Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
|
|
@@ -8370,6 +9798,57 @@ function registerScheduleHandlers() {
|
|
|
8370
9798
|
};
|
|
8371
9799
|
});
|
|
8372
9800
|
|
|
9801
|
+
// Queue-health header (PRD): the one honest read of "why does the queue
|
|
9802
|
+
// look stale" — reuses classifyQueueHealth so the UI and the starvation
|
|
9803
|
+
// watchdog can never disagree. `cwd` is optional (null = machine-wide,
|
|
9804
|
+
// matching WindowStrip's own scopeCwd fallback).
|
|
9805
|
+
ipcMain.handle('schedule:queue-health', async (_e, payload) => {
|
|
9806
|
+
const cwd = (payload && typeof payload.cwd === 'string') ? payload.cwd : null;
|
|
9807
|
+
const state = await readQueue();
|
|
9808
|
+
if (state.unreadable) {
|
|
9809
|
+
return { unknown: true, reason: state.unreadable };
|
|
9810
|
+
}
|
|
9811
|
+
const now = Date.now();
|
|
9812
|
+
const slotSnapshot = sessionSlots.snapshot();
|
|
9813
|
+
const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
|
|
9814
|
+
const verdict = classifyQueueHealth({
|
|
9815
|
+
jobs: state.jobs,
|
|
9816
|
+
paused: state.paused,
|
|
9817
|
+
launchBlocks: state.launchBlocks,
|
|
9818
|
+
runningSet,
|
|
9819
|
+
freeSlots,
|
|
9820
|
+
totalSlots: slotSnapshot.total,
|
|
9821
|
+
lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
|
|
9822
|
+
now,
|
|
9823
|
+
cwd,
|
|
9824
|
+
});
|
|
9825
|
+
// Oldest running job across the whole machine (any project) — the
|
|
9826
|
+
// number that actually explains slot saturation, alongside the
|
|
9827
|
+
// machine-wide slot pool itself.
|
|
9828
|
+
let oldestRunningAgeMs = null;
|
|
9829
|
+
for (const j of state.jobs) {
|
|
9830
|
+
if (j.status !== 'running' && !runningSet.has(j.slug)) continue;
|
|
9831
|
+
const startedAtMs = j.startedAt ? Date.parse(j.startedAt) : NaN;
|
|
9832
|
+
if (!Number.isFinite(startedAtMs)) continue;
|
|
9833
|
+
const age = now - startedAtMs;
|
|
9834
|
+
if (oldestRunningAgeMs === null || age > oldestRunningAgeMs) oldestRunningAgeMs = age;
|
|
9835
|
+
}
|
|
9836
|
+
return {
|
|
9837
|
+
unknown: false,
|
|
9838
|
+
now,
|
|
9839
|
+
verdict,
|
|
9840
|
+
slots: {
|
|
9841
|
+
inUse: slotSnapshot.inUse,
|
|
9842
|
+
total: slotSnapshot.total,
|
|
9843
|
+
free: freeSlots,
|
|
9844
|
+
source: slotSnapshot.envOverride ? 'env' : 'pool',
|
|
9845
|
+
},
|
|
9846
|
+
oldestRunningAgeMs,
|
|
9847
|
+
lastRunAt: state.lastRunAt ?? null,
|
|
9848
|
+
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
9849
|
+
};
|
|
9850
|
+
});
|
|
9851
|
+
|
|
8373
9852
|
ipcMain.handle('schedule:force-tick', async () => {
|
|
8374
9853
|
// Bypass the billing-poll gate entirely — fire pending jobs immediately regardless of meter state.
|
|
8375
9854
|
// Clears any existing pause first (same semantics as run-now).
|
|
@@ -8448,6 +9927,21 @@ function registerScheduleHandlers() {
|
|
|
8448
9927
|
return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
|
|
8449
9928
|
}));
|
|
8450
9929
|
|
|
9930
|
+
// Scheduler UI's "change disposition" action (scheduler wave-disposition
|
|
9931
|
+
// PRD): promotes an appended wave to its own head, or re-attaches a head
|
|
9932
|
+
// behind another chain. Thin wrapper over remote.setPrdDisposition, which
|
|
9933
|
+
// validates the rewrite (cycle-safety, running/completed rows untouched)
|
|
9934
|
+
// before delegating to the same remote.updatePrd every other PRD edit
|
|
9935
|
+
// path uses — see that method's own comment in this file.
|
|
9936
|
+
ipcMain.handle('schedule:set-prd-disposition', validated(schemas.scheduleSetPrdDisposition, async ({ slug, cwd, disposition, dependsOn }) => {
|
|
9937
|
+
if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
|
|
9938
|
+
const result = await remote.setPrdDisposition({ slug, cwd, disposition, dependsOn });
|
|
9939
|
+
if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'disposition change failed' };
|
|
9940
|
+
appendAuditEvent('scheduler_prd_disposition_set', { slug, cwd: cwd ?? null, disposition, source: 'ipc:schedule:set-prd-disposition' });
|
|
9941
|
+
await broadcast({ flush: true });
|
|
9942
|
+
return { ok: true, kind: 'info', message: `${slug} is now ${disposition === 'new-head' ? 'an independent head' : 'attached behind the chosen chain'}` };
|
|
9943
|
+
}));
|
|
9944
|
+
|
|
8451
9945
|
ipcMain.handle('schedule:run-now', async () => {
|
|
8452
9946
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
8453
9947
|
await clearPause('run-now');
|
|
@@ -8460,9 +9954,11 @@ function registerScheduleHandlers() {
|
|
|
8460
9954
|
return { ok: true };
|
|
8461
9955
|
});
|
|
8462
9956
|
|
|
8463
|
-
// Re-scan prds/ folder and merge into queue.json.
|
|
8464
|
-
//
|
|
8465
|
-
// explicit
|
|
9957
|
+
// Re-scan prds/ folder and merge into queue.json. `schedule:state` is a
|
|
9958
|
+
// cheap read with no reconcile of its own — this is the renderer's
|
|
9959
|
+
// explicit, immediate discovery path (mutate() + reconcile() + broadcast())
|
|
9960
|
+
// for "I just dropped a PRD on disk and want it to show up now" rather than
|
|
9961
|
+
// waiting for tickQueue's next ~60s pass.
|
|
8466
9962
|
ipcMain.handle('schedule:rescan', async () => {
|
|
8467
9963
|
const { added, removed } = await mutate(async (state) => {
|
|
8468
9964
|
const before = new Set(state.jobs.map((j) => j.slug));
|
|
@@ -8691,6 +10187,18 @@ async function init() {
|
|
|
8691
10187
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
8692
10188
|
bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
|
|
8693
10189
|
}
|
|
10190
|
+
// Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
|
|
10191
|
+
// BEFORE mutate() for the same reason (git spawn work must never run
|
|
10192
|
+
// inside mutate()'s single global serialization chain) — an orphaned job
|
|
10193
|
+
// classified 'failed'/'unknown' from its log tail alone can still have
|
|
10194
|
+
// actually landed a real commit before the app restarted mid-run.
|
|
10195
|
+
const bootLandedCommitEvidence = new Map();
|
|
10196
|
+
await Promise.all(bootSnap.jobs.map(async (j) => {
|
|
10197
|
+
if (!immediateSlugs.includes(j.slug) || j.status !== 'running') return;
|
|
10198
|
+
if (bootOutcomes.get(j.slug) === 'success' || !j.landedCommit) return;
|
|
10199
|
+
const resolved = await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt);
|
|
10200
|
+
if (resolved) bootLandedCommitEvidence.set(j.slug, j.landedCommit);
|
|
10201
|
+
}));
|
|
8694
10202
|
const bootReconciledCompletions = [];
|
|
8695
10203
|
await mutate((state) => {
|
|
8696
10204
|
for (const j of state.jobs) {
|
|
@@ -8698,7 +10206,7 @@ async function init() {
|
|
|
8698
10206
|
const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
|
|
8699
10207
|
const pid = j.runtime?.pid;
|
|
8700
10208
|
const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
|
|
8701
|
-
applyOrphanOutcome(j, outcome, killNote);
|
|
10209
|
+
applyOrphanOutcome(j, outcome, killNote, bootLandedCommitEvidence.get(j.slug) || null);
|
|
8702
10210
|
if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
|
|
8703
10211
|
console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
|
|
8704
10212
|
}
|
|
@@ -8722,9 +10230,18 @@ async function init() {
|
|
|
8722
10230
|
if (result === 'killed') {
|
|
8723
10231
|
console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
|
|
8724
10232
|
}
|
|
8725
|
-
setTimeout(() => {
|
|
10233
|
+
setTimeout(async () => {
|
|
8726
10234
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
8727
10235
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
10236
|
+
// Same evidence-before-failure gate as the immediate-orphan path
|
|
10237
|
+
// above, resolved before mutate() for the same reason (git spawn
|
|
10238
|
+
// work must never run inside mutate()'s serialization chain). Uses
|
|
10239
|
+
// the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
|
|
10240
|
+
// race guard below already confirms `cur` is still this same run
|
|
10241
|
+
// (runId === bootRunId) before this evidence is applied.
|
|
10242
|
+
const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
|
|
10243
|
+
? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
|
|
10244
|
+
: null;
|
|
8728
10245
|
let deferredCompletedCwd;
|
|
8729
10246
|
mutate((state) => {
|
|
8730
10247
|
const cur = state.jobs.find((x) => x.slug === slug);
|
|
@@ -8733,7 +10250,7 @@ async function init() {
|
|
|
8733
10250
|
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
8734
10251
|
// touched by this stale classification.
|
|
8735
10252
|
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
8736
|
-
applyOrphanOutcome(cur, outcome, killNote);
|
|
10253
|
+
applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
|
|
8737
10254
|
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
8738
10255
|
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
8739
10256
|
}).then(() => {
|
|
@@ -8879,7 +10396,8 @@ async function init() {
|
|
|
8879
10396
|
// else distinguishes "no pending work" from "pending work, never
|
|
8880
10397
|
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
8881
10398
|
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
8882
|
-
|
|
10399
|
+
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
10400
|
+
for (const sp of starvedProjects) {
|
|
8883
10401
|
console.warn(
|
|
8884
10402
|
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
8885
10403
|
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
@@ -8887,30 +10405,104 @@ async function init() {
|
|
|
8887
10405
|
);
|
|
8888
10406
|
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
8889
10407
|
}
|
|
8890
|
-
|
|
8891
|
-
//
|
|
8892
|
-
//
|
|
8893
|
-
//
|
|
8894
|
-
//
|
|
8895
|
-
//
|
|
8896
|
-
|
|
8897
|
-
|
|
8898
|
-
|
|
8899
|
-
|
|
8900
|
-
|
|
8901
|
-
|
|
8902
|
-
|
|
8903
|
-
|
|
8904
|
-
|
|
10408
|
+
// Bounded, automated consequence for a starve that outlives the WARN
|
|
10409
|
+
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
10410
|
+
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
10411
|
+
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
10412
|
+
// a subset of the rows already reported above — same verdict, no
|
|
10413
|
+
// re-derivation.
|
|
10414
|
+
runStarveEscalationSweep(starvedProjects);
|
|
10415
|
+
|
|
10416
|
+
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
10417
|
+
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
10418
|
+
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
10419
|
+
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
10420
|
+
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
10421
|
+
// above it can never observe two different snapshots of the same row.
|
|
10422
|
+
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
10423
|
+
const autoResetTargets = failedAutoResetDisabled()
|
|
10424
|
+
? []
|
|
10425
|
+
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
10426
|
+
const stuckFailed = stuckFailedEscalationDisabled()
|
|
10427
|
+
? []
|
|
10428
|
+
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
10429
|
+
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
10430
|
+
// (this PRD): computed alongside the failed-row passes above and acted
|
|
10431
|
+
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
10432
|
+
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
10433
|
+
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
10434
|
+
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
10435
|
+
? []
|
|
10436
|
+
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
10437
|
+
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
10438
|
+
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
10439
|
+
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
10440
|
+
// never be read from one snapshot and written from another, and the
|
|
10441
|
+
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
10442
|
+
// same turn as the transition it gates. Kill-switch:
|
|
10443
|
+
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
10444
|
+
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
10445
|
+
? []
|
|
10446
|
+
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
10447
|
+
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
10448
|
+
mutate(async (ms) => {
|
|
10449
|
+
for (const target of autoResetTargets) {
|
|
10450
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
10451
|
+
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
10452
|
+
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
10453
|
+
j.failedAutoResetAttempts = attempt;
|
|
10454
|
+
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
10455
|
+
// resetJobFields is the same field-clearing list the admin
|
|
10456
|
+
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
10457
|
+
// it rather than inventing a second list. It also sets job.error to
|
|
10458
|
+
// the reason text passed in; we clear that back to null right
|
|
10459
|
+
// after since this is a clean auto-reset, not a recorded error.
|
|
10460
|
+
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
10461
|
+
j.error = null;
|
|
10462
|
+
delete j.stuckFailedNotified;
|
|
10463
|
+
console.warn(
|
|
10464
|
+
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
10465
|
+
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
10466
|
+
);
|
|
10467
|
+
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
10468
|
+
}
|
|
10469
|
+
for (const stuck of stuckFailed) {
|
|
10470
|
+
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
10471
|
+
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
10472
|
+
// Still has auto-reset attempts left — it will be (or already was,
|
|
10473
|
+
// earlier this same pass) picked up by the loop above instead.
|
|
10474
|
+
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
10475
|
+
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
10476
|
+
j.stuckFailedNotified = true;
|
|
10477
|
+
console.warn(
|
|
10478
|
+
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
10479
|
+
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
10480
|
+
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
10481
|
+
);
|
|
10482
|
+
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
10483
|
+
}
|
|
10484
|
+
for (const target of exhaustedNeedsReviewTargets) {
|
|
10485
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
10486
|
+
const outcome = applyNeedsReviewAutoResolve(j);
|
|
10487
|
+
if (outcome) {
|
|
8905
10488
|
console.warn(
|
|
8906
|
-
`[scheduler]
|
|
8907
|
-
+ `
|
|
8908
|
-
+ `no automated recovery reaches a failed row; reset it by hand via scheduler_reset_job`,
|
|
10489
|
+
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
10490
|
+
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
8909
10491
|
);
|
|
8910
|
-
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
8911
10492
|
}
|
|
8912
|
-
}
|
|
8913
|
-
|
|
10493
|
+
}
|
|
10494
|
+
for (const target of quarantineTargets) {
|
|
10495
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
10496
|
+
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
10497
|
+
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
10498
|
+
if (outcome) {
|
|
10499
|
+
console.warn(
|
|
10500
|
+
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
10501
|
+
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
10502
|
+
);
|
|
10503
|
+
}
|
|
10504
|
+
}
|
|
10505
|
+
}).catch(() => {});
|
|
8914
10506
|
}
|
|
8915
10507
|
}, 10 * 60_000);
|
|
8916
10508
|
|
|
@@ -9094,7 +10686,9 @@ async function listPrdsInternal() {
|
|
|
9094
10686
|
estimateMinutes: parsed.estimateMinutes,
|
|
9095
10687
|
sourcePromptId: parsed.sourcePromptId,
|
|
9096
10688
|
epicId: parsed.epicId ?? null,
|
|
10689
|
+
dependsOn: parsed.dependsOn ?? null,
|
|
9097
10690
|
agentType: parsed.agentType ?? null,
|
|
10691
|
+
disposition: parsed.disposition ?? null,
|
|
9098
10692
|
mtimeMs: stat.mtimeMs,
|
|
9099
10693
|
archived,
|
|
9100
10694
|
};
|
|
@@ -9491,6 +11085,33 @@ const remote = {
|
|
|
9491
11085
|
}
|
|
9492
11086
|
},
|
|
9493
11087
|
|
|
11088
|
+
// Backs the Scheduler UI's "change disposition" action (scheduler
|
|
11089
|
+
// wave-disposition PRD): promoting an appended wave to its own head, or
|
|
11090
|
+
// re-attaching a head behind another chain. `dependsOn` for a 'new-head'
|
|
11091
|
+
// disposition is ignored (cleared unconditionally); for 'append' it's the
|
|
11092
|
+
// caller's chosen target chain's terminal slug(s) — the renderer computes
|
|
11093
|
+
// that from the SAME backlog tree (lib/backlogTree.ts) it already renders,
|
|
11094
|
+
// so this function only has to validate the rewrite is safe, never
|
|
11095
|
+
// re-derive "the" terminal itself.
|
|
11096
|
+
//
|
|
11097
|
+
// Validates via prdDisposition.cjs's computeDispositionRewrite (row not
|
|
11098
|
+
// running/completed, no already-satisfied blocker being rewritten out from
|
|
11099
|
+
// under it, no dependsOn cycle) BEFORE delegating the actual write to this
|
|
11100
|
+
// SAME updatePrd — so a rejected rewrite never reaches the filesystem, and
|
|
11101
|
+
// an accepted one gets updatePrd's own dependsOn FK re-validation for free.
|
|
11102
|
+
async setPrdDisposition({ slug, cwd, disposition, dependsOn }) {
|
|
11103
|
+
let listing;
|
|
11104
|
+
try {
|
|
11105
|
+
listing = await this.listPrds({ cwd, fields: 'full', limit: Number.MAX_SAFE_INTEGER });
|
|
11106
|
+
} catch (e) {
|
|
11107
|
+
return { ok: false, error: `could not read project PRDs: ${e?.message ?? e}` };
|
|
11108
|
+
}
|
|
11109
|
+
const rows = listing.prds ?? [];
|
|
11110
|
+
const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
|
|
11111
|
+
if (!rewrite.ok) return rewrite;
|
|
11112
|
+
return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition } });
|
|
11113
|
+
},
|
|
11114
|
+
|
|
9494
11115
|
// Cancels a job that hasn't finished yet. A 'running' job's process group
|
|
9495
11116
|
// is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
|
|
9496
11117
|
// reconciliation uses for an orphaned running job) before its queue row is
|
|
@@ -9518,18 +11139,38 @@ const remote = {
|
|
|
9518
11139
|
if (wasRunning && pid) {
|
|
9519
11140
|
killOrphanClaudePid(pid);
|
|
9520
11141
|
}
|
|
11142
|
+
// Evidence-before-failure guard, scoped to an actually-running job being
|
|
11143
|
+
// killed here (a 'pending' cancel has no live process, so nothing new
|
|
11144
|
+
// could have landed since its last stamp — and 'needs_review' is not
|
|
11145
|
+
// even a legal transition from 'pending', see LEGAL_TRANSITIONS): the
|
|
11146
|
+
// same reapDeadRunningJobs evidence gate (job 1192 — a landedCommit
|
|
11147
|
+
// being non-empty is not proof by itself, but discarding proof of real
|
|
11148
|
+
// landed work with no check at all is worse) applies here too. A
|
|
11149
|
+
// dead-pid reap of a job that landed a commit (e.g. via the
|
|
11150
|
+
// dispatch-time sidecar backfill) is routed to needs_review/completed;
|
|
11151
|
+
// a deliberate cancel of that same state deserves no less.
|
|
11152
|
+
const confirmedLandedCommit = (wasRunning && job.landedCommit)
|
|
11153
|
+
? ((await resolveLandedCommitEvidence(job.cwd || DEFAULT_PROJECT_CWD, job.landedCommit, job.startedAt))
|
|
11154
|
+
? job.landedCommit
|
|
11155
|
+
: null)
|
|
11156
|
+
: null;
|
|
11157
|
+
const targetStatus = confirmedLandedCommit ? 'needs_review' : 'failed';
|
|
11158
|
+
const cancelReason = confirmedLandedCommit
|
|
11159
|
+
? `cancelled via admin API, but landedCommit ${confirmedLandedCommit} resolves — verify before treating as done`
|
|
11160
|
+
: 'cancelled via admin API';
|
|
9521
11161
|
await mutate((s) => {
|
|
9522
11162
|
const idx = s.jobs.findIndex((j) => j.slug === slug);
|
|
9523
11163
|
if (idx < 0) return;
|
|
9524
11164
|
const j = s.jobs[idx];
|
|
9525
|
-
transitionJob(j,
|
|
9526
|
-
j.error =
|
|
11165
|
+
transitionJob(j, targetStatus, { reason: cancelReason, source: 'remote:cancelJob' });
|
|
11166
|
+
j.error = cancelReason;
|
|
9527
11167
|
j.finishedAt = new Date().toISOString();
|
|
9528
11168
|
j.exitCode = j.exitCode ?? null;
|
|
11169
|
+
if (confirmedLandedCommit) j.verifierVerdict = 'cancelled_with_landed_commit';
|
|
9529
11170
|
delete j.runtime;
|
|
9530
11171
|
});
|
|
9531
11172
|
await broadcast({ flush: true });
|
|
9532
|
-
return { ok: true, slug, status:
|
|
11173
|
+
return { ok: true, slug, status: targetStatus, wasRunning, cwd: job.cwd ?? null };
|
|
9533
11174
|
},
|
|
9534
11175
|
|
|
9535
11176
|
// Exposes the module-level allocateParallelGroup (PRD 548) to callers that
|
|
@@ -9575,13 +11216,27 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
9575
11216
|
|
|
9576
11217
|
module.exports = {
|
|
9577
11218
|
classifyQueueStarvation,
|
|
11219
|
+
classifyQueueStarvationByProject,
|
|
11220
|
+
classifyQueueHealth,
|
|
9578
11221
|
runQueueStarvationWatchdog,
|
|
9579
11222
|
QUEUE_STARVATION_MS,
|
|
11223
|
+
selectStarveEscalations,
|
|
11224
|
+
runStarveEscalationSweep,
|
|
11225
|
+
STARVE_ESCALATION_MS,
|
|
9580
11226
|
computeBlockedChains,
|
|
9581
11227
|
stripAppOwnedChurn,
|
|
9582
11228
|
findOverrunningJobs,
|
|
9583
11229
|
JOB_OVERRUN_FACTOR,
|
|
9584
11230
|
JOB_OVERRUN_FLOOR_MS,
|
|
11231
|
+
computeJobBudgetMs,
|
|
11232
|
+
classifyBudgetKill,
|
|
11233
|
+
isJobBudgetExempt,
|
|
11234
|
+
shouldKillForBudget,
|
|
11235
|
+
resolveBudgetKillOutcome,
|
|
11236
|
+
JOB_BUDGET_FACTOR,
|
|
11237
|
+
JOB_BUDGET_FLOOR_MS,
|
|
11238
|
+
JOB_BUDGET_CEILING_MS,
|
|
11239
|
+
BUDGET_WARNING_FRACTION,
|
|
9585
11240
|
registerScheduleHandlers,
|
|
9586
11241
|
attachWindow,
|
|
9587
11242
|
init,
|
|
@@ -9595,10 +11250,12 @@ module.exports = {
|
|
|
9595
11250
|
healRefusalReason,
|
|
9596
11251
|
writeQueue,
|
|
9597
11252
|
reconcile,
|
|
11253
|
+
broadcast,
|
|
9598
11254
|
reconcileSourcePromptId,
|
|
9599
11255
|
allocateParallelGroup,
|
|
9600
11256
|
selectHistoryJobs,
|
|
9601
11257
|
parsePorcelain,
|
|
11258
|
+
parsePorcelainEntries,
|
|
9602
11259
|
FINISH_PROTOCOL,
|
|
9603
11260
|
IDLE_OUTPUT_KILL_MS,
|
|
9604
11261
|
BASH_DEFAULT_TIMEOUT_MS,
|
|
@@ -9616,9 +11273,19 @@ module.exports = {
|
|
|
9616
11273
|
findStuckFailedJobs,
|
|
9617
11274
|
STUCK_FAILED_ESCALATE_MS,
|
|
9618
11275
|
stuckFailedEscalationDisabled,
|
|
11276
|
+
selectFailedAutoResetTargets,
|
|
11277
|
+
FAILED_AUTORESET_CAP,
|
|
11278
|
+
FAILED_AUTORESET_MS,
|
|
11279
|
+
failedAutoResetDisabled,
|
|
11280
|
+
selectExhaustedNeedsReviewTargets,
|
|
11281
|
+
applyNeedsReviewAutoResolve,
|
|
11282
|
+
NEEDS_REVIEW_RESOLVE_CAP,
|
|
11283
|
+
NEEDS_REVIEW_RESOLVE_MS,
|
|
11284
|
+
needsReviewAutoResolveDisabled,
|
|
9619
11285
|
isRescanCandidate,
|
|
9620
11286
|
isFailedUnverifiedShaped,
|
|
9621
11287
|
computeLooksDone,
|
|
11288
|
+
attributeLandedCommits,
|
|
9622
11289
|
isPromotableOriginal,
|
|
9623
11290
|
selectAutoFixTargets,
|
|
9624
11291
|
applyRcaClassification,
|
|
@@ -9626,6 +11293,9 @@ module.exports = {
|
|
|
9626
11293
|
resolveRunId,
|
|
9627
11294
|
isUnresolvableNeedsReview,
|
|
9628
11295
|
isExhaustedAutoFix,
|
|
11296
|
+
GUARD_VERDICT_EVIDENCE_ELIGIBLE,
|
|
11297
|
+
isGuardParkedWithoutAutoFix,
|
|
11298
|
+
isEligibleForNeedsReviewAutoResolve,
|
|
9629
11299
|
isPlanUnqueued,
|
|
9630
11300
|
isFixPlanDead,
|
|
9631
11301
|
fixSlugFor,
|
|
@@ -9646,6 +11316,7 @@ module.exports = {
|
|
|
9646
11316
|
MAX_INVESTIGATION_DEPTH,
|
|
9647
11317
|
forceTickOutcome,
|
|
9648
11318
|
applyPauseCleared,
|
|
11319
|
+
formatLoadGateDetail,
|
|
9649
11320
|
detectNetworkErrorInLog,
|
|
9650
11321
|
detectRateLimitInLog,
|
|
9651
11322
|
classifyFailureOutcome,
|
|
@@ -9695,6 +11366,10 @@ module.exports = {
|
|
|
9695
11366
|
computeStallSummary,
|
|
9696
11367
|
findStaleQuarantinedJobs,
|
|
9697
11368
|
QUARANTINE_ESCALATE_MS,
|
|
11369
|
+
selectQuarantineAutoResolveTargets,
|
|
11370
|
+
autoResolveQuarantine,
|
|
11371
|
+
QUARANTINE_RESOLVE_CAP,
|
|
11372
|
+
quarantineAutoResolveDisabled,
|
|
9698
11373
|
applyClearQueueVictims,
|
|
9699
11374
|
PIDLESS_SPAWN_GRACE_MS,
|
|
9700
11375
|
findStrandedInvestigations,
|
|
@@ -9706,6 +11381,7 @@ module.exports = {
|
|
|
9706
11381
|
evaluateSharedTreeGuard,
|
|
9707
11382
|
checkSharedTreeGuard,
|
|
9708
11383
|
uncommittedChanges,
|
|
11384
|
+
uncommittedChangesWithStatus,
|
|
9709
11385
|
gitHead,
|
|
9710
11386
|
isBranchAlreadyIntegrated,
|
|
9711
11387
|
selectResumeRecoveryTarget,
|