claude-code-session-manager 0.86.0 → 0.87.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/AgentLibrary-DyLWzZDf.js +3 -0
- package/dist/assets/{DataModel-Q4jhl24R.js → DataModel--mISIJ6h.js} +1 -1
- package/dist/assets/{History-Cj2FejEo.js → History-C2ahUXTg.js} +2 -2
- package/dist/assets/{Hooks-CaelQI6t.js → Hooks-BiC6oyR2.js} +3 -3
- package/dist/assets/{HostBilko--v7cMR8I.js → HostBilko-BPleEOld.js} +1 -1
- package/dist/assets/{Library-DgI9oCCZ.js → Library-Dc8Qst1R.js} +1 -1
- package/dist/assets/{ListDetail-DYUZN-x-.js → ListDetail-DIXh-OLX.js} +1 -1
- package/dist/assets/MarkdownEditor-C90bkLXK.js +1 -0
- package/dist/assets/{McpServers-ypCYURh3.js → McpServers-DqcbLOLZ.js} +2 -2
- package/dist/assets/{Memory-C2qYp-3M.js → Memory-CW62MXlh.js} +4 -4
- package/dist/assets/{Panel-Cj2kw-Zv.js → Panel-Bw1FhRuF.js} +1 -1
- package/dist/assets/Permissions-BcUC-5y8.js +3 -0
- package/dist/assets/{Plugins-C1Vj8_dU.js → Plugins-BnKx9flD.js} +2 -2
- package/dist/assets/{ProvenanceBadge-DczPNM5U.js → ProvenanceBadge-Bw5vNVPT.js} +1 -1
- package/dist/assets/{SaveBar-Cd_7U6Gb.js → SaveBar-CWr0O_w-.js} +1 -1
- package/dist/assets/Scheduler-DYdLuUqq.js +14 -0
- package/dist/assets/{ScopeSwitcher-DVSyI44-.js → ScopeSwitcher-CrBLbg8s.js} +1 -1
- package/dist/assets/Settings-DluB-vN1.js +3 -0
- package/dist/assets/{SkillReferenceGraph-CUv1_Q2c.js → SkillReferenceGraph-CHLSseay.js} +1 -1
- package/dist/assets/{Skills-C_YHkAy-.js → Skills-gNdo_HNK.js} +2 -2
- package/dist/assets/{SystemPrompt-B8R7T9xn.js → SystemPrompt-Cru05-Ia.js} +1 -1
- package/dist/assets/{TagLibrary-dj9YHWyy.js → TagLibrary-DNHY0xou.js} +1 -1
- package/dist/assets/{TiptapBody-DnSBUjHE.js → TiptapBody-I4lmbCgP.js} +1 -1
- package/dist/assets/{Toggle-CjV_BJn6.js → Toggle-bWMHjmRh.js} +1 -1
- package/dist/assets/{index-CDo9xBR9.css → index-DV3PorRY.css} +1 -1
- package/dist/assets/{index-CXFQIPhO.js → index-fc_JjdxL.js} +724 -724
- package/dist/assets/settingsSchema-BfhtZnGD.js +3 -0
- package/dist/index.html +2 -2
- package/package.json +14 -14
- package/plugins/CLAUDE.md +61 -0
- package/plugins/session-manager-dev/.claude-plugin/plugin.json +1 -1
- package/plugins/session-manager-dev/skills/builder/4-manual/SKILL.md +1 -1
- package/plugins/session-manager-dev/skills/ops-sweep/SKILL.md +1 -1
- package/scripts/scheduler-mcp-server.cjs +7 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +100 -9
- package/src/main/__tests__/broadcastCoalescer.test.cjs +18 -0
- package/src/main/__tests__/epicMint.test.cjs +2 -2
- package/src/main/__tests__/health-delegation-chain.test.cjs +2 -1
- package/src/main/__tests__/needsReviewLedger.test.cjs +162 -0
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +3 -3
- package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +15 -1
- package/src/main/__tests__/prdCreateDisposition.test.cjs +201 -0
- package/src/main/__tests__/prdFrontmatterDisposition.test.cjs +125 -0
- package/src/main/__tests__/prdLocations.test.cjs +100 -2
- package/src/main/__tests__/prdLocationsArchived.test.cjs +43 -1
- package/src/main/__tests__/prdSetDisposition.test.cjs +222 -0
- package/src/main/__tests__/queue-health-verdict.test.cjs +170 -0
- package/src/main/__tests__/queue-starvation-per-project.test.cjs +14 -2
- package/src/main/__tests__/queueHistory.test.cjs +63 -0
- package/src/main/__tests__/reconcileTiming.test.cjs +135 -0
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +101 -1
- package/src/main/__tests__/scheduler-boot-orphans.test.cjs +2 -2
- package/src/main/__tests__/scheduler-broadcast-reconcile.test.cjs +121 -0
- package/src/main/__tests__/scheduler-cross-project-batch.test.cjs +43 -0
- package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +344 -0
- package/src/main/__tests__/scheduler-job-budget.test.cjs +172 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +93 -3
- package/src/main/__tests__/scheduler-porcelain-rename.test.cjs +164 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +160 -0
- package/src/main/__tests__/scheduler-reaper-helpers-basics.test.cjs +87 -0
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +88 -0
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +14 -4
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +14 -0
- package/src/main/chatRunner.cjs +8 -5
- package/src/main/health.cjs +1 -1
- package/src/main/historyAggregator.cjs +5 -0
- package/src/main/index.cjs +95 -44
- package/src/main/ipcSchemas.cjs +47 -0
- package/src/main/lib/__tests__/active-sessions.test.cjs +251 -0
- package/src/main/lib/__tests__/bootSelfHeal.test.cjs +107 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +322 -43
- package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +239 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +89 -0
- package/src/main/lib/__tests__/guardShims.test.cjs +151 -0
- package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +5 -5
- package/src/main/lib/__tests__/prdDisposition.test.cjs +224 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +179 -1
- package/src/main/lib/__tests__/usageCircuit.test.cjs +224 -0
- package/src/main/lib/__tests__/watchdog-helpers.test.cjs +312 -0
- package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +193 -0
- package/{scripts → src/main}/lib/activeSessions.cjs +50 -4
- package/src/main/lib/agentModelResolve.cjs +65 -27
- package/src/main/lib/bootSelfHeal.cjs +88 -0
- package/src/main/lib/delegationReadiness.cjs +290 -225
- package/src/main/lib/effectiveModelInfo.cjs +333 -0
- package/src/main/lib/ephemeralCwd.cjs +1 -1
- package/src/main/lib/epicMint.cjs +3 -3
- package/src/main/lib/gitWorktree.cjs +42 -12
- package/src/main/lib/guardShims.cjs +156 -0
- package/src/main/lib/jobDirtFilter.cjs +7 -2
- package/src/main/lib/launchFailure.cjs +2 -1
- package/src/main/lib/mcpToolCatalog.cjs +4 -1
- package/src/main/lib/needsReviewLedger.cjs +205 -0
- package/src/main/lib/opsErrorLog.cjs +1 -1
- package/src/main/lib/opsOwnership.cjs +1 -1
- package/src/main/lib/prdCreate.cjs +56 -1
- package/src/main/lib/prdDisposition.cjs +199 -0
- package/src/main/lib/prdFrontmatter.cjs +8 -2
- package/src/main/lib/prdLocations.cjs +167 -45
- package/src/main/lib/projectHomeAdminRoutes.cjs +4 -4
- package/src/main/lib/projectPageSummarySchema.cjs +1 -1
- package/src/main/lib/projectRootResolve.cjs +1 -1
- package/src/main/lib/queueHistory.cjs +19 -1
- package/src/main/lib/queueStore.cjs +6 -1
- package/src/main/lib/reaperHelpers.cjs +181 -15
- package/src/main/lib/scheduleJobSchema.cjs +8 -0
- package/src/main/lib/scheduleJobTransitions.cjs +33 -0
- package/src/main/lib/schedulerConfig.cjs +24 -0
- package/src/main/lib/usageCircuit.cjs +159 -0
- package/{scripts → src/main}/lib/watchdogHelpers.cjs +1 -1
- package/src/main/scheduler/prdParser.cjs +13 -0
- package/src/main/scheduler.cjs +1285 -143
- package/src/main/templates/PRD_AUTHORING.md +50 -0
- package/src/main/templates/project-pages-catalog.json +1 -1
- package/src/main/usage.cjs +21 -3
- package/src/preload/api.d.ts +92 -1
- package/src/preload/index.cjs +10 -0
- package/web/README.md +41 -0
- package/{scripts/render-project-pages.cjs → web/project-pages/render.cjs} +4 -4
- package/{scripts/render-project-pages → web/project-pages/renderer}/dist/renderer.cjs +1 -1
- package/{scripts/validate-project-pages-summary.cjs → web/project-pages/validate-summary.cjs} +5 -5
- package/dist/assets/AgentLibrary-DTFL7y8G.js +0 -3
- package/dist/assets/MarkdownEditor-DCIubYWf.js +0 -1
- package/dist/assets/Permissions-BiYZNGYW.js +0 -3
- package/dist/assets/Scheduler-DcLBiJBq.js +0 -14
- package/dist/assets/Settings-Cv-pRyms.js +0 -3
- package/dist/assets/settingsSchema-BJVciriw.js +0 -3
- /package/{scripts/project-pages-logic → web/project-pages/logic}/dist/logic.cjs +0 -0
package/src/main/scheduler.cjs
CHANGED
|
@@ -60,7 +60,9 @@ const { readTail } = require('./lib/fileTail.cjs');
|
|
|
60
60
|
const {
|
|
61
61
|
claudePidAlive, classifyRunOutcome, mapOutcomeToGateOutcome, ORPHAN_REQUEUE_CAP, selectReapableJobs,
|
|
62
62
|
findLiveProcessForJob, logHasOutput, resolvePidlessGateOutcome, resolveCommitGuardOutcome,
|
|
63
|
+
readSpawnedPidFromLog, readLogMtimeMs,
|
|
63
64
|
} = require('./lib/reaperHelpers.cjs');
|
|
65
|
+
const { resolveProjectRoot } = require('./lib/opsOwnership.cjs');
|
|
64
66
|
const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
|
|
65
67
|
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
66
68
|
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
@@ -89,9 +91,13 @@ const {
|
|
|
89
91
|
USAGE_REFRESH_INTERVAL_MS,
|
|
90
92
|
MAX_JOB_DURATION_MS,
|
|
91
93
|
BROADCAST_COALESCE_MS,
|
|
94
|
+
RECONCILE_SLOW_PASS_MS,
|
|
92
95
|
QUARANTINE_ESCALATE_MS: QUARANTINE_ESCALATE_MS_DEFAULT,
|
|
93
96
|
JOB_OVERRUN_FACTOR: JOB_OVERRUN_FACTOR_DEFAULT,
|
|
94
97
|
JOB_OVERRUN_FLOOR_MS: JOB_OVERRUN_FLOOR_MS_DEFAULT,
|
|
98
|
+
JOB_BUDGET_FACTOR: JOB_BUDGET_FACTOR_DEFAULT,
|
|
99
|
+
JOB_BUDGET_FLOOR_MS: JOB_BUDGET_FLOOR_MS_DEFAULT,
|
|
100
|
+
JOB_BUDGET_CEILING_MS: JOB_BUDGET_CEILING_MS_DEFAULT,
|
|
95
101
|
PIDLESS_SPAWN_GRACE_MS,
|
|
96
102
|
INVESTIGATION_MAX_MS,
|
|
97
103
|
STARVATION_ESCALATE_MS,
|
|
@@ -100,12 +106,27 @@ const {
|
|
|
100
106
|
const QUARANTINE_ESCALATE_MS = process.env.SM_QUARANTINE_ESCALATE_HOURS
|
|
101
107
|
? Number(process.env.SM_QUARANTINE_ESCALATE_HOURS) * 60 * 60_000
|
|
102
108
|
: QUARANTINE_ESCALATE_MS_DEFAULT;
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
:
|
|
109
|
+
// Shared by every SM_*-env-overridable numeric constant below (bare factors
|
|
110
|
+
// use unitMs=1; minute-denominated knobs use unitMs=60_000) — one parse rule
|
|
111
|
+
// instead of one hand-copied ternary per constant.
|
|
112
|
+
function numEnvOverride(envVar, unitMs, fallback) {
|
|
113
|
+
const raw = process.env[envVar];
|
|
114
|
+
return raw ? Number(raw) * unitMs : fallback;
|
|
115
|
+
}
|
|
116
|
+
const JOB_OVERRUN_FACTOR = numEnvOverride('SM_JOB_OVERRUN_FACTOR', 1, JOB_OVERRUN_FACTOR_DEFAULT);
|
|
117
|
+
const JOB_OVERRUN_FLOOR_MS = numEnvOverride('SM_JOB_OVERRUN_FLOOR_MINUTES', 60_000, JOB_OVERRUN_FLOOR_MS_DEFAULT);
|
|
118
|
+
// Same three numbers as JOB_OVERRUN_FACTOR/JOB_OVERRUN_FLOOR_MS today (3x,
|
|
119
|
+
// 45min) is coincidental, not structural — this triad ACTS (kills) where
|
|
120
|
+
// JOB_OVERRUN_* only ever escalates (see JOB_OVERRUN_FACTOR's own header);
|
|
121
|
+
// tune them independently, don't re-couple on a future pass just because the
|
|
122
|
+
// defaults happen to match right now.
|
|
123
|
+
const JOB_BUDGET_FACTOR = numEnvOverride('SM_JOB_BUDGET_FACTOR', 1, JOB_BUDGET_FACTOR_DEFAULT);
|
|
124
|
+
const JOB_BUDGET_FLOOR_MS = numEnvOverride('SM_JOB_BUDGET_FLOOR_MINUTES', 60_000, JOB_BUDGET_FLOOR_MS_DEFAULT);
|
|
125
|
+
const JOB_BUDGET_CEILING_MS = numEnvOverride('SM_JOB_BUDGET_CEILING_MINUTES', 60_000, JOB_BUDGET_CEILING_MS_DEFAULT);
|
|
126
|
+
// A running job past this fraction of its own budget gets a durable
|
|
127
|
+
// `budgetWarning` stamp on its row (see the budget watchdog below) so the
|
|
128
|
+
// renderer can warn BEFORE the kill, not only after.
|
|
129
|
+
const BUDGET_WARNING_FRACTION = 0.75;
|
|
109
130
|
const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD, DEP_HISTORY_FAIL_OPEN } = require('./lib/schedulerBatch.cjs');
|
|
110
131
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
111
132
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
@@ -151,8 +172,9 @@ const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
|
151
172
|
const queueStore = require('./lib/queueStore.cjs');
|
|
152
173
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
153
174
|
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
175
|
+
const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
|
|
154
176
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
155
|
-
const { allProjectCwds } = require('
|
|
177
|
+
const { allProjectCwds } = require('./lib/activeSessions.cjs');
|
|
156
178
|
|
|
157
179
|
// Captured once at module load so every run's meta sidecar can record how
|
|
158
180
|
// stale the running process is relative to on-disk source (incident: PRD
|
|
@@ -311,22 +333,95 @@ only post-AC work. If a review finding can't be fixed within scope, commit what
|
|
|
311
333
|
you have, describe the finding in the commit body, and note the follow-up in your
|
|
312
334
|
final result.`;
|
|
313
335
|
|
|
314
|
-
//
|
|
315
|
-
//
|
|
316
|
-
//
|
|
317
|
-
//
|
|
318
|
-
|
|
336
|
+
// Unquote a single git porcelain v1 path token. Defined once in
|
|
337
|
+
// gitWorktree.cjs (which this file already requires — the reverse would be
|
|
338
|
+
// circular, since gitWorktree.cjs's own salvageDirtyDelta needs the exact
|
|
339
|
+
// same unquoting) and reused here rather than re-implemented, so the two
|
|
340
|
+
// porcelain consumers in this codebase can never drift apart.
|
|
341
|
+
const { unquotePorcelainPath } = gitWorktree;
|
|
342
|
+
|
|
343
|
+
// Split a rename/copy porcelain path field ("old -> new") into its two real
|
|
344
|
+
// paths. Each side is independently quoted per unquotePorcelainPath's rule —
|
|
345
|
+
// only the side that needs escaping is wrapped in quotes, the literal " -> "
|
|
346
|
+
// arrow between them never is. Returns null when no " -> " separator is
|
|
347
|
+
// found (a malformed/unexpected line) so the caller can fall back to treating
|
|
348
|
+
// the whole field as one opaque path rather than guessing.
|
|
349
|
+
function splitRenamePorcelainField(field) {
|
|
350
|
+
const arrow = ' -> ';
|
|
351
|
+
let head;
|
|
352
|
+
let rest;
|
|
353
|
+
if (field[0] === '"') {
|
|
354
|
+
let end = -1;
|
|
355
|
+
for (let i = 1; i < field.length; i += 1) {
|
|
356
|
+
if (field[i] === '\\') { i += 1; continue; }
|
|
357
|
+
if (field[i] === '"') { end = i; break; }
|
|
358
|
+
}
|
|
359
|
+
if (end === -1) return null;
|
|
360
|
+
head = field.slice(0, end + 1);
|
|
361
|
+
rest = field.slice(end + 1);
|
|
362
|
+
} else {
|
|
363
|
+
const idx = field.indexOf(arrow);
|
|
364
|
+
if (idx === -1) return null;
|
|
365
|
+
// An unquoted path containing a literal " -> " substring (git only
|
|
366
|
+
// quotes for a quote/backslash/control-byte/non-ASCII byte — a plain
|
|
367
|
+
// ASCII arrow inside a filename is never quoted) makes the true
|
|
368
|
+
// old/new boundary genuinely ambiguous from this text alone: the first
|
|
369
|
+
// occurrence could be the real separator, or it could be sitting
|
|
370
|
+
// inside the old path with the real separator later in the field.
|
|
371
|
+
// Guessing wrong silently corrupts oldPath/path for downstream
|
|
372
|
+
// fs.existsSync/Set-membership checks, which is worse than the
|
|
373
|
+
// existing "malformed line" fallback below — so more than one
|
|
374
|
+
// occurrence falls back to treating the whole field as one opaque
|
|
375
|
+
// path, same as any other line this function can't confidently parse.
|
|
376
|
+
if (field.indexOf(arrow, idx + arrow.length) !== -1) return null;
|
|
377
|
+
head = field.slice(0, idx);
|
|
378
|
+
rest = field.slice(idx);
|
|
379
|
+
}
|
|
380
|
+
if (!rest.startsWith(arrow)) return null;
|
|
381
|
+
return { oldPath: unquotePorcelainPath(head), path: unquotePorcelainPath(rest.slice(arrow.length)) };
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
// Parse `git status --porcelain` output into `{ code, path }` entries (plus
|
|
385
|
+
// `oldPath` for a rename/copy). Pure + exported for unit testing. Each
|
|
386
|
+
// porcelain line is "XY<space>PATH"; a staged rename/copy line is
|
|
387
|
+
// "XY<space>OLD -> NEW" instead — X (index status) is 'R' or 'C' — and NEW is
|
|
388
|
+
// the path git will report in any later `git status` call, so callers that
|
|
389
|
+
// key off `.path` (dirtyAfter membership, pathsCommittedDuringRun membership,
|
|
390
|
+
// fs.existsSync) must compare against NEW, never the fused "OLD -> NEW"
|
|
391
|
+
// string. `oldPath` is retained on the entry for callers that need the
|
|
392
|
+
// original path too. `code` is the raw 2-char status (e.g. '??' for
|
|
393
|
+
// untracked) — callers that need to distinguish "untracked" from
|
|
394
|
+
// "tracked-but-modified" (the shared-tree guard's revert-vs-now-ignored
|
|
395
|
+
// split) read it off the entry instead of re-deriving it later.
|
|
396
|
+
function parsePorcelainEntries(stdout) {
|
|
319
397
|
return String(stdout || '')
|
|
320
398
|
.split('\n')
|
|
321
399
|
.filter((l) => l.length > 0)
|
|
322
|
-
.map((l) =>
|
|
323
|
-
|
|
400
|
+
.map((l) => {
|
|
401
|
+
const code = l.slice(0, 2);
|
|
402
|
+
const field = l.slice(3);
|
|
403
|
+
if (code.includes('R') || code.includes('C')) {
|
|
404
|
+
const split = splitRenamePorcelainField(field);
|
|
405
|
+
if (split) return { code, path: split.path, oldPath: split.oldPath };
|
|
406
|
+
}
|
|
407
|
+
return { code, path: unquotePorcelainPath(field) };
|
|
408
|
+
})
|
|
409
|
+
.filter((e) => e.path);
|
|
324
410
|
}
|
|
325
411
|
|
|
326
|
-
//
|
|
327
|
-
//
|
|
328
|
-
|
|
329
|
-
|
|
412
|
+
// Parse `git status --porcelain` output into a list of changed paths. Pure +
|
|
413
|
+
// exported for unit testing.
|
|
414
|
+
function parsePorcelain(stdout) {
|
|
415
|
+
return parsePorcelainEntries(stdout).map((e) => e.path);
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
// Same as uncommittedChanges but keeps each path's porcelain status code —
|
|
419
|
+
// the shared-tree guard's baseline needs this to tell "was untracked" apart
|
|
420
|
+
// from "was tracked-and-modified" (see evaluateSharedTreeGuard). Returns null
|
|
421
|
+
// when the guard does not apply (cwd is not a git work tree, git is missing,
|
|
422
|
+
// or the call errors); never throws — a guard failure must not fail an
|
|
423
|
+
// otherwise-successful job.
|
|
424
|
+
function uncommittedChangesWithStatus(cwd) {
|
|
330
425
|
return new Promise((resolve) => {
|
|
331
426
|
if (!cwd) { resolve(null); return; }
|
|
332
427
|
execFile(
|
|
@@ -334,13 +429,24 @@ function uncommittedChanges(cwd) {
|
|
|
334
429
|
['-C', cwd, 'status', '--porcelain'],
|
|
335
430
|
{ timeout: 10_000, windowsHide: true },
|
|
336
431
|
(err, stdout) => {
|
|
337
|
-
if (err) { resolve(null); return; }
|
|
338
|
-
resolve(
|
|
432
|
+
if (err) { resolve(null); return; }
|
|
433
|
+
resolve(parsePorcelainEntries(stdout));
|
|
339
434
|
},
|
|
340
435
|
);
|
|
341
436
|
});
|
|
342
437
|
}
|
|
343
438
|
|
|
439
|
+
// Return the list of uncommitted paths in cwd, or null under the same
|
|
440
|
+
// conditions as uncommittedChangesWithStatus (never throws). Kept as a thin
|
|
441
|
+
// path-only projection of that call rather than its own execFile, so a future
|
|
442
|
+
// fix to the git invocation (timeout, error handling) can't land in one and
|
|
443
|
+
// silently miss the other.
|
|
444
|
+
function uncommittedChanges(cwd) {
|
|
445
|
+
return uncommittedChangesWithStatus(cwd).then((entries) => (
|
|
446
|
+
entries === null ? null : entries.map((e) => e.path)
|
|
447
|
+
));
|
|
448
|
+
}
|
|
449
|
+
|
|
344
450
|
// Return the current HEAD commit sha in cwd, or null on any error. Used by the
|
|
345
451
|
// commit-guard to detect whether the job self-committed during its run (HEAD
|
|
346
452
|
// moved) — in which case leftover working-tree dirt is presumptively from a
|
|
@@ -429,17 +535,52 @@ function restoreSpecificStash(cwd, ref) {
|
|
|
429
535
|
// - reverted: a path that was dirty in the baseline, is clean now, and was
|
|
430
536
|
// not touched by any commit landed during the run — the job reset/
|
|
431
537
|
// checked-out over pre-existing uncommitted work without stashing it.
|
|
432
|
-
//
|
|
433
|
-
//
|
|
434
|
-
|
|
538
|
+
//
|
|
539
|
+
// A THIRD outcome is not a revert at all: an untracked path can drop out of
|
|
540
|
+
// `git status` because the run committed a `.gitignore` change that now
|
|
541
|
+
// matches it — the file is untouched on disk, just no longer visible to git
|
|
542
|
+
// (Incident: 2026-09-12, PRD 1181 added a bare `logs/` ignore pattern, eleven
|
|
543
|
+
// untracked `session-manager-operations/logs/*` paths vanished from status,
|
|
544
|
+
// and an otherwise-perfect run was parked in needs_review for a human who had
|
|
545
|
+
// nothing to decide). `dirtyBefore` entries therefore carry each path's
|
|
546
|
+
// porcelain status code (`{ code, path }`, from parsePorcelainEntries) so this
|
|
547
|
+
// function can tell "was untracked" apart from "was tracked-and-modified":
|
|
548
|
+
// - a TRACKED path (any code other than '??') leaving the dirty set always
|
|
549
|
+
// means its content was restored to HEAD — still `reverted`, even though
|
|
550
|
+
// the file still exists on disk, because for a tracked file "exists" is
|
|
551
|
+
// not the question; "matches what the human left uncommitted" is.
|
|
552
|
+
// - an UNTRACKED path ('??') leaving the dirty set is `reverted` only if it
|
|
553
|
+
// no longer exists on disk; if it still exists, it merely became ignored
|
|
554
|
+
// and is reported separately as `nowIgnored`.
|
|
555
|
+
// Plain path strings are still accepted in `dirtyBefore` for callers that
|
|
556
|
+
// have no status code (e.g. the stash-detection pass, which always passes an
|
|
557
|
+
// empty array) — an entry with no `code` is treated as tracked, matching the
|
|
558
|
+
// old behavior exactly.
|
|
559
|
+
//
|
|
560
|
+
// Pure/no I/O — status-code parsing and on-disk existence checks both happen
|
|
561
|
+
// at the call site (checkSharedTreeGuard); this function never stats the
|
|
562
|
+
// filesystem. Exported for unit testing.
|
|
563
|
+
function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAfter, pathsCommittedDuringRun, existsAfter }) {
|
|
435
564
|
const beforeHashes = new Set((stashBefore || []).map((l) => parseStashLine(l)?.hash).filter(Boolean));
|
|
436
565
|
const newStashes = (stashAfter || [])
|
|
437
566
|
.map(parseStashLine)
|
|
438
567
|
.filter((e) => e && !beforeHashes.has(e.hash));
|
|
439
568
|
const dirtyAfterSet = new Set(dirtyAfter || []);
|
|
440
569
|
const committedSet = new Set(pathsCommittedDuringRun || []);
|
|
441
|
-
const
|
|
442
|
-
|
|
570
|
+
const existsSet = new Set(existsAfter || []);
|
|
571
|
+
const reverted = [];
|
|
572
|
+
const nowIgnored = [];
|
|
573
|
+
for (const entry of dirtyBefore || []) {
|
|
574
|
+
const p = typeof entry === 'string' ? entry : entry.path;
|
|
575
|
+
const code = typeof entry === 'string' ? undefined : entry.code;
|
|
576
|
+
if (dirtyAfterSet.has(p) || committedSet.has(p)) continue;
|
|
577
|
+
if (code === '??' && existsSet.has(p)) {
|
|
578
|
+
nowIgnored.push(p);
|
|
579
|
+
} else {
|
|
580
|
+
reverted.push(p);
|
|
581
|
+
}
|
|
582
|
+
}
|
|
583
|
+
return { newStashes, reverted, nowIgnored };
|
|
443
584
|
}
|
|
444
585
|
|
|
445
586
|
// Post-run shared-tree guard for an IN-PLACE job (worktree.ok === false —
|
|
@@ -489,18 +630,40 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
|
|
|
489
630
|
// restored stash is not ALSO reported as an unexplained revert (it was
|
|
490
631
|
// explained — by the stash this guard just restored).
|
|
491
632
|
const dirtyAfter = await module.exports.uncommittedChanges(cwd);
|
|
492
|
-
|
|
633
|
+
// Existence check for the "now ignored, not reverted" split (2026-09-12
|
|
634
|
+
// incident) — only untracked baseline entries need it; a tracked path
|
|
635
|
+
// leaving the dirty set is always a revert regardless of disk state (see
|
|
636
|
+
// evaluateSharedTreeGuard). Scoped to entries carrying a status code —
|
|
637
|
+
// plain path strings (no code) fall back to the old always-reverted path.
|
|
638
|
+
const untrackedBaselinePaths = (dirtyBaseline || [])
|
|
639
|
+
.filter((e) => e && typeof e === 'object' && e.code === '??')
|
|
640
|
+
.map((e) => e.path);
|
|
641
|
+
// A large untracked baseline (the 2026-09-12 incident's shared tree had
|
|
642
|
+
// ~240 such paths) makes this a lot of stat calls — fs.promises.access
|
|
643
|
+
// run concurrently instead of fs.existsSync run synchronously one at a
|
|
644
|
+
// time keeps this off the event loop instead of blocking every other
|
|
645
|
+
// in-flight scheduler/IPC task for the duration.
|
|
646
|
+
const existsChecks = await Promise.all(
|
|
647
|
+
untrackedBaselinePaths.map((p) => fsp.access(path.join(cwd, p)).then(() => true, () => false)),
|
|
648
|
+
);
|
|
649
|
+
const existsAfter = untrackedBaselinePaths.filter((_, i) => existsChecks[i]);
|
|
650
|
+
const { reverted, nowIgnored } = module.exports.evaluateSharedTreeGuard({
|
|
493
651
|
stashBefore: stashBaseline,
|
|
494
652
|
stashAfter,
|
|
495
653
|
dirtyBefore: dirtyBaseline,
|
|
496
654
|
dirtyAfter,
|
|
497
655
|
pathsCommittedDuringRun,
|
|
656
|
+
existsAfter,
|
|
498
657
|
});
|
|
499
658
|
if (reverted.length) {
|
|
500
659
|
result.reverted = reverted;
|
|
501
660
|
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
502
661
|
}
|
|
503
|
-
|
|
662
|
+
if (nowIgnored.length) {
|
|
663
|
+
result.nowIgnored = nowIgnored;
|
|
664
|
+
console.log(`[scheduler] ${slug}: shared-tree guard: ${nowIgnored.length} path(s) no longer shown by git status but still present on disk — likely a new ignore rule, not a revert (${nowIgnored.slice(0, 3).join(', ')})`);
|
|
665
|
+
}
|
|
666
|
+
return (result.restoredStash || result.restoreFailed || result.ambiguousStashes || result.reverted || result.nowIgnored) ? result : null;
|
|
504
667
|
} catch (e) {
|
|
505
668
|
console.error(`[scheduler] ${slug}: shared-tree guard error`, e);
|
|
506
669
|
return null;
|
|
@@ -1152,8 +1315,9 @@ function ensureDirs() {
|
|
|
1152
1315
|
* reconcile()-level call is what makes "anything written to the retired flat
|
|
1153
1316
|
* prds/ dir is swept into prds-archived/ without being executed" actually
|
|
1154
1317
|
* true regardless of which of reconcile's several callers (tickQueue's poll,
|
|
1155
|
-
* job completion, the schedule:
|
|
1156
|
-
* rescheduleTimer) triggers the pass
|
|
1318
|
+
* job completion, the schedule:rescan/schedule:adopt-prd IPC handlers,
|
|
1319
|
+
* broadcast()'s coalescer, rescheduleTimer) triggers the pass — schedule:state
|
|
1320
|
+
* no longer reconciles on read. A PRD dropped in the flat dir has no
|
|
1157
1321
|
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
1158
1322
|
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
1159
1323
|
*/
|
|
@@ -1217,7 +1381,22 @@ async function runPrdMigration() {
|
|
|
1217
1381
|
// on every pass, but stays here so a fresh boot's very first log line
|
|
1218
1382
|
// still reports the initial sweep — see consolidateAllFlatPrds's own
|
|
1219
1383
|
// comment for why reconcile() is the load-bearing call site.)
|
|
1220
|
-
|
|
1384
|
+
//
|
|
1385
|
+
// Deferred off scheduler.init()'s synchronous critical path: allProjectCwds()
|
|
1386
|
+
// is a synchronous ~270ms directory scan, and awaiting it inline here
|
|
1387
|
+
// competed with the renderer's first IPC round trips (schedule.state,
|
|
1388
|
+
// billing.fetch, teams.list) for the event loop during boot. Dropping the
|
|
1389
|
+
// await doesn't weaken the consolidation guarantee — consolidateAllFlatPrds
|
|
1390
|
+
// also runs at the top of every reconcile() (see its own comment above),
|
|
1391
|
+
// and a flat PRD can only ever execute via tickQueue, which always
|
|
1392
|
+
// reconciles first, so nothing dropped in the flat dir can run before a
|
|
1393
|
+
// reconcile() pass sweeps it regardless of whether this boot-time pass has
|
|
1394
|
+
// finished yet.
|
|
1395
|
+
setImmediate(() => {
|
|
1396
|
+
consolidateAllFlatPrds(allProjectCwds()).catch((e) => {
|
|
1397
|
+
logs.writeLine({ level: 'warn', scope: 'scheduler', message: 'deferred flat-PRD consolidation failed', meta: { error: e?.message } });
|
|
1398
|
+
});
|
|
1399
|
+
});
|
|
1221
1400
|
|
|
1222
1401
|
// Rollout migration for the PRD-authoring-lockdown feature: stamp every
|
|
1223
1402
|
// pre-existing PRD as legacy-adopted BEFORE reconcile() ever runs its
|
|
@@ -1655,6 +1834,99 @@ function findOverrunningJobs(jobs, now, { factor, floorMs } = {}) {
|
|
|
1655
1834
|
return out;
|
|
1656
1835
|
}
|
|
1657
1836
|
|
|
1837
|
+
/**
|
|
1838
|
+
* computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs }) → number
|
|
1839
|
+
*
|
|
1840
|
+
* Pure. `budgetMs = clamp(estimateMinutes * factor, floorMs, ceilingMs)` — see
|
|
1841
|
+
* JOB_BUDGET_FACTOR's header comment (schedulerConfig.cjs) for the measured
|
|
1842
|
+
* p50/p90/max this is calibrated against. A missing/zero/non-finite estimate
|
|
1843
|
+
* is treated as 0, which the floor clamp then dominates — "jobs with a
|
|
1844
|
+
* missing estimate get the floor" falls straight out of the clamp, no
|
|
1845
|
+
* special-casing needed.
|
|
1846
|
+
*/
|
|
1847
|
+
function computeJobBudgetMs(estimateMinutes, { factor, floorMs, ceilingMs } = {}) {
|
|
1848
|
+
const f = typeof factor === 'number' && factor > 0 ? factor : JOB_BUDGET_FACTOR;
|
|
1849
|
+
const floor = typeof floorMs === 'number' && floorMs >= 0 ? floorMs : JOB_BUDGET_FLOOR_MS;
|
|
1850
|
+
const ceiling = typeof ceilingMs === 'number' && ceilingMs > 0 ? ceilingMs : JOB_BUDGET_CEILING_MS;
|
|
1851
|
+
const est = Number(estimateMinutes);
|
|
1852
|
+
const minutes = Number.isFinite(est) && est > 0 ? est : 0;
|
|
1853
|
+
return Math.min(Math.max(minutes * f * 60_000, floor), ceiling);
|
|
1854
|
+
}
|
|
1855
|
+
|
|
1856
|
+
/**
|
|
1857
|
+
* classifyBudgetKill(res, landedCommitEvidence) → { status, reason, landedCommit } | null
|
|
1858
|
+
*
|
|
1859
|
+
* Pure. `res` is executeJob's resolved outcome — only fires when
|
|
1860
|
+
* `res.killedByWatchdog === 'budget'` (stamped by the budget watchdog inside
|
|
1861
|
+
* executeJob, never inferred from exit code/duration alone, so it can never
|
|
1862
|
+
* collide with an ordinary idle-tail/deadman/external kill). ALWAYS routes to
|
|
1863
|
+
* needs_review — never 'failed' (spawnJob's ordinary non-zero-exit default)
|
|
1864
|
+
* and never silently 'completed' (executeJob's onExit excludes
|
|
1865
|
+
* killedByWatchdog === 'budget' from the result=success → exit 0 mapping
|
|
1866
|
+
* idle-tail/deadman get) — so a budget kill is a visible, actionable park,
|
|
1867
|
+
* never a retry (classifyFailureOutcome/selectAutoFixTargets only ever see
|
|
1868
|
+
* 'failed'/ordinary needs_review rows, not this one — see
|
|
1869
|
+
* selectAutoFixTargets' own budget_exceeded exclusion) and never a discard:
|
|
1870
|
+
* `landedCommitEvidence`, when the caller resolved one via the SAME
|
|
1871
|
+
* commit-guard evidence check the plain sigterm/exit paths already use, is
|
|
1872
|
+
* threaded onto the row as `landedCommit` so a job that HAD already
|
|
1873
|
+
* committed before overrunning is still adjudicated on its git evidence.
|
|
1874
|
+
*/
|
|
1875
|
+
function classifyBudgetKill(res, landedCommitEvidence) {
|
|
1876
|
+
if (!res || res.killedByWatchdog !== 'budget') return null;
|
|
1877
|
+
return {
|
|
1878
|
+
status: 'needs_review',
|
|
1879
|
+
reason: res.budgetKillReason || `wall-clock budget exceeded (exit ${res.exitCode})`,
|
|
1880
|
+
landedCommit: landedCommitEvidence || null,
|
|
1881
|
+
};
|
|
1882
|
+
}
|
|
1883
|
+
|
|
1884
|
+
/**
|
|
1885
|
+
* isJobBudgetExempt(job) → boolean
|
|
1886
|
+
*
|
|
1887
|
+
* Pure. `quietMachine: true` PRDs (their whole point is running alone,
|
|
1888
|
+
* un-contended, for a timing-sensitive measurement) and any PRD with an
|
|
1889
|
+
* explicit `budgetExempt: true` opt-out have no wall-clock kill ceiling.
|
|
1890
|
+
*/
|
|
1891
|
+
function isJobBudgetExempt(job) {
|
|
1892
|
+
return job?.quietMachine === true || job?.budgetExempt === true;
|
|
1893
|
+
}
|
|
1894
|
+
|
|
1895
|
+
/** Pure predicate the budget watchdog's shouldFire calls — single source of
|
|
1896
|
+
* truth for "has this job run past its own budget" so it's unit-testable
|
|
1897
|
+
* without spinning up real timers. */
|
|
1898
|
+
function shouldKillForBudget(elapsedMs, budgetMs) {
|
|
1899
|
+
return elapsedMs >= budgetMs;
|
|
1900
|
+
}
|
|
1901
|
+
|
|
1902
|
+
/**
|
|
1903
|
+
* resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes })
|
|
1904
|
+
* → { killedByWatchdog: 'budget'|null, budgetKillReason: string|null }
|
|
1905
|
+
*
|
|
1906
|
+
* Pure. `ctx.killedByWatchdog` is stamped by the budget watchdog's action()
|
|
1907
|
+
* the instant its periodic shouldFire() observes elapsedMs >= jobBudgetMs —
|
|
1908
|
+
* but that setInterval tick and the child's real 'exit' event both run on
|
|
1909
|
+
* the SAME single-threaded event loop, so it's possible to observe the
|
|
1910
|
+
* budget threshold crossed and call ctx.killTree() in the same window the
|
|
1911
|
+
* agent happens to exit cleanly (exit 0) or fails on its own for an
|
|
1912
|
+
* unrelated reason — ctx.killTree() against an already-exited pid is a
|
|
1913
|
+
* silent no-op (ESRCH, caught), but the flag would still read 'budget'
|
|
1914
|
+
* unless gated here. Only trusted when the exit SHAPE actually looks like a
|
|
1915
|
+
* signal kill (killedBySignal — mirrors the exact same check onExit already
|
|
1916
|
+
* uses for its own mappedToSuccess exclusion), so a clean exit=0 or an
|
|
1917
|
+
* ordinary unrelated non-zero failure racing the watchdog's tick is never
|
|
1918
|
+
* misclassified as a budget kill downstream.
|
|
1919
|
+
*/
|
|
1920
|
+
function resolveBudgetKillOutcome({ killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes }) {
|
|
1921
|
+
if (killedByWatchdog !== 'budget' || !killedBySignal) {
|
|
1922
|
+
return { killedByWatchdog: killedByWatchdog === 'budget' ? null : (killedByWatchdog ?? null), budgetKillReason: null };
|
|
1923
|
+
}
|
|
1924
|
+
return {
|
|
1925
|
+
killedByWatchdog: 'budget',
|
|
1926
|
+
budgetKillReason: `wall-clock budget exceeded: ran ${Math.round(durationMs / 60_000)}m against a ${Math.round(jobBudgetMs / 60_000)}m budget (estimateMinutes=${estimateMinutes ?? 0})`,
|
|
1927
|
+
};
|
|
1928
|
+
}
|
|
1929
|
+
|
|
1658
1930
|
/**
|
|
1659
1931
|
* findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
1660
1932
|
* → [{ slug, cwd, ageMs, restoreStatus }]
|
|
@@ -1874,7 +2146,7 @@ async function listPrdFiles() {
|
|
|
1874
2146
|
ensureDirs();
|
|
1875
2147
|
const dirs = candidatePrdsDirs();
|
|
1876
2148
|
const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
|
|
1877
|
-
return perDir.flat().sort();
|
|
2149
|
+
return { files: perDir.flat().sort(), dirCount: dirs.length };
|
|
1878
2150
|
}
|
|
1879
2151
|
|
|
1880
2152
|
/**
|
|
@@ -2018,16 +2290,31 @@ async function reconcile(state) {
|
|
|
2018
2290
|
if (state && state.unreadable) {
|
|
2019
2291
|
throw new Error(`reconcile skipped: queue.json unreadable (${state.unreadable})`);
|
|
2020
2292
|
}
|
|
2293
|
+
// Per-phase timing (PRD: reconcile evidence trail) — plain Date.now() diffs,
|
|
2294
|
+
// matching the ad-hoc elapsedMs idiom already used in health.cjs/
|
|
2295
|
+
// definitionOfDone.cjs. Only logged when the total exceeds
|
|
2296
|
+
// RECONCILE_SLOW_PASS_MS (see the warn emission at the bottom of this
|
|
2297
|
+
// function); a normal-speed pass logs nothing.
|
|
2298
|
+
const reconcileStartMs = Date.now();
|
|
2299
|
+
const phaseMs = {};
|
|
2300
|
+
|
|
2021
2301
|
// Sweep the retired flat prds/ dir BEFORE scanning it below. reconcile()
|
|
2022
|
-
// has several callers besides tickQueue's ~60s poll (broadcast
|
|
2023
|
-
// rescheduleTimer,
|
|
2302
|
+
// has several callers besides tickQueue's ~60s poll (broadcast()'s
|
|
2303
|
+
// coalescer, rescheduleTimer, schedule:rescan, schedule:adopt-prd) — this
|
|
2024
2304
|
// lives here, not in any one caller, so the "a hand-written PRD in the flat
|
|
2025
2305
|
// dir is swept before it can become a job" guarantee holds regardless of
|
|
2026
2306
|
// which caller triggers this reconcile pass. A freshly hand-written file
|
|
2027
2307
|
// has no queue row yet, so it is never "live" and gets archived here
|
|
2028
2308
|
// instead of ever reaching the onDisk scan below.
|
|
2309
|
+
let phaseStartMs = Date.now();
|
|
2029
2310
|
await consolidateAllFlatPrds(allProjectCwds());
|
|
2030
|
-
|
|
2311
|
+
phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
|
|
2312
|
+
|
|
2313
|
+
phaseStartMs = Date.now();
|
|
2314
|
+
const { files, dirCount } = await listPrdFiles();
|
|
2315
|
+
phaseMs.prdDirResolve = Date.now() - phaseStartMs;
|
|
2316
|
+
|
|
2317
|
+
phaseStartMs = Date.now();
|
|
2031
2318
|
const onDisk = new Map();
|
|
2032
2319
|
for (const f of files) {
|
|
2033
2320
|
try {
|
|
@@ -2039,6 +2326,7 @@ async function reconcile(state) {
|
|
|
2039
2326
|
console.warn('[scheduler] failed to parse', f, e?.message);
|
|
2040
2327
|
}
|
|
2041
2328
|
}
|
|
2329
|
+
phaseMs.parseLoop = Date.now() - phaseStartMs;
|
|
2042
2330
|
|
|
2043
2331
|
const next = [];
|
|
2044
2332
|
const seen = new Set();
|
|
@@ -2104,7 +2392,9 @@ async function reconcile(state) {
|
|
|
2104
2392
|
// membership, so moving the file between Epic dirs must re-point the row.
|
|
2105
2393
|
epicId: p.epicId ?? job.epicId ?? null,
|
|
2106
2394
|
dependsOn: p.dependsOn,
|
|
2395
|
+
disposition: p.disposition ?? null,
|
|
2107
2396
|
quietMachine: p.quietMachine === true,
|
|
2397
|
+
budgetExempt: p.budgetExempt === true,
|
|
2108
2398
|
originSessionId: job.originSessionId
|
|
2109
2399
|
?? resolveOriginSessionId(p.cwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
2110
2400
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
@@ -2159,9 +2449,11 @@ async function reconcile(state) {
|
|
|
2159
2449
|
// ScheduleJobSchema (e.g. the 1021/1022 incident's `"status": "queued"`) —
|
|
2160
2450
|
// see the repair pass below, right after historyBySlug is available.
|
|
2161
2451
|
const invalidJobs = Array.isArray(state.invalidJobs) ? state.invalidJobs : [];
|
|
2452
|
+
phaseStartMs = Date.now();
|
|
2162
2453
|
const historyBySlug = (unmatchedSlugs.length > 0 || terminalDroppedNeedingHistoryCheck.length > 0 || invalidJobs.length > 0)
|
|
2163
2454
|
? await queueHistory.historyTerminalBySlug()
|
|
2164
2455
|
: new Map();
|
|
2456
|
+
phaseMs.historyLookup = Date.now() - phaseStartMs;
|
|
2165
2457
|
|
|
2166
2458
|
// Backfill: any terminal job dropped above whose slug isn't already in
|
|
2167
2459
|
// history.jsonl gets written now, before its row is gone for good. This is
|
|
@@ -2217,7 +2509,9 @@ async function reconcile(state) {
|
|
|
2217
2509
|
sourceTabId: p.sourceTabId ?? inv.row?.sourceTabId ?? null,
|
|
2218
2510
|
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
2219
2511
|
dependsOn: p.dependsOn,
|
|
2512
|
+
disposition: p.disposition ?? null,
|
|
2220
2513
|
quietMachine: p.quietMachine === true,
|
|
2514
|
+
budgetExempt: p.budgetExempt === true,
|
|
2221
2515
|
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
2222
2516
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2223
2517
|
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
@@ -2339,7 +2633,9 @@ async function reconcile(state) {
|
|
|
2339
2633
|
sourceTabId: p.sourceTabId,
|
|
2340
2634
|
epicId: p.epicId ?? null,
|
|
2341
2635
|
dependsOn: p.dependsOn,
|
|
2636
|
+
disposition: p.disposition ?? null,
|
|
2342
2637
|
quietMachine: p.quietMachine === true,
|
|
2638
|
+
budgetExempt: p.budgetExempt === true,
|
|
2343
2639
|
originSessionId: resolveOriginSessionId(p.cwd, p.epicId ?? p.sourcePromptId),
|
|
2344
2640
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2345
2641
|
agentType: p.agentType ?? null,
|
|
@@ -2436,7 +2732,8 @@ async function reconcile(state) {
|
|
|
2436
2732
|
// small. Append BEFORE dropping so a crash between the two can't lose a
|
|
2437
2733
|
// record — appendHistory dedupes by slug+runId, so a replay of the same
|
|
2438
2734
|
// batch on next boot is a safe no-op.
|
|
2439
|
-
|
|
2735
|
+
phaseStartMs = Date.now();
|
|
2736
|
+
const nowMs = phaseStartMs;
|
|
2440
2737
|
const { hot, toArchive } = queueHistory.partitionJobs(sorted, nowMs);
|
|
2441
2738
|
if (toArchive.length > 0) {
|
|
2442
2739
|
await queueHistory.appendHistory(toArchive);
|
|
@@ -2472,6 +2769,17 @@ async function reconcile(state) {
|
|
|
2472
2769
|
} catch (e) {
|
|
2473
2770
|
console.warn('[scheduler] autoArchiveCompleted failed', e?.message);
|
|
2474
2771
|
}
|
|
2772
|
+
phaseMs.queueWrite = Date.now() - phaseStartMs;
|
|
2773
|
+
|
|
2774
|
+
const totalMs = Date.now() - reconcileStartMs;
|
|
2775
|
+
if (totalMs > RECONCILE_SLOW_PASS_MS) {
|
|
2776
|
+
logs.writeLine({
|
|
2777
|
+
level: 'warn',
|
|
2778
|
+
scope: 'scheduler',
|
|
2779
|
+
message: `reconcile() pass took ${totalMs}ms (threshold ${RECONCILE_SLOW_PASS_MS}ms)`,
|
|
2780
|
+
meta: { totalMs, phaseMs, prdFileCount: files.length, resolvedDirCount: dirCount },
|
|
2781
|
+
});
|
|
2782
|
+
}
|
|
2475
2783
|
|
|
2476
2784
|
return state;
|
|
2477
2785
|
}
|
|
@@ -2678,16 +2986,19 @@ function attachWindow(w) { mainWindow = w; }
|
|
|
2678
2986
|
|
|
2679
2987
|
/**
|
|
2680
2988
|
* Build the snapshot payload consumed by both the `schedule:state` IPC
|
|
2681
|
-
* handler and the `schedule:state` broadcast event.
|
|
2682
|
-
* `paths` map (renderer uses it for "open folder" actions); broadcast omits
|
|
2683
|
-
* it because subscribers don't need to re-derive paths on every tick.
|
|
2989
|
+
* handler and the `schedule:state` broadcast event.
|
|
2684
2990
|
*/
|
|
2685
|
-
function buildScheduleStatePayload(state
|
|
2991
|
+
function buildScheduleStatePayload(state) {
|
|
2686
2992
|
const payload = {
|
|
2687
2993
|
config: state.config,
|
|
2688
2994
|
jobs: state.jobs,
|
|
2689
2995
|
scheduledFor: state.scheduledFor,
|
|
2690
2996
|
lastRunAt: state.lastRunAt,
|
|
2997
|
+
// Distinct from lastRunAt (only stamped when a batch actually launches):
|
|
2998
|
+
// stamped every time tickQueue reaches the picker at all. See
|
|
2999
|
+
// classifyQueueHealth/classifyQueueStarvation's header comments for why
|
|
3000
|
+
// the two must never merge.
|
|
3001
|
+
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
2691
3002
|
nextReset: getNextResetCached(),
|
|
2692
3003
|
paused: state.paused,
|
|
2693
3004
|
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
@@ -2718,9 +3029,6 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
2718
3029
|
};
|
|
2719
3030
|
})(),
|
|
2720
3031
|
};
|
|
2721
|
-
if (withPaths) {
|
|
2722
|
-
payload.paths = { root: ROOT, prds: PRDS_DIR, runs: RUNS_DIR, queue: queueStore.MACHINE_STATE_PATH };
|
|
2723
|
-
}
|
|
2724
3032
|
return payload;
|
|
2725
3033
|
}
|
|
2726
3034
|
|
|
@@ -2731,20 +3039,33 @@ function buildScheduleStatePayload(state, { withPaths = false } = {}) {
|
|
|
2731
3039
|
// per mutation. Callers where latency matters (pause/resume, job
|
|
2732
3040
|
// start/finish/reap/reset) pass `{ flush: true }` to bypass the window and
|
|
2733
3041
|
// send immediately.
|
|
3042
|
+
// getPayload is the coalescer's ONLY entry point back into queue state, so
|
|
3043
|
+
// routing the reconcile+write pair through it (rather than broadcast() doing
|
|
3044
|
+
// its own bare readQueue/reconcile/writeQueue) is what makes a burst of
|
|
3045
|
+
// broadcast() calls cost exactly one reconcile + one write per coalesce
|
|
3046
|
+
// window. It goes through mutate() (not a bare read/write pair) so it
|
|
3047
|
+
// serializes against every other concurrent mutation and inherits mutate's
|
|
3048
|
+
// pre-fn `state.unreadable` bail — never write a state derived from a failed
|
|
3049
|
+
// read. `module.exports.reconcile` (not the bare local binding) is the seam
|
|
3050
|
+
// tests spy on, matching this file's existing testable-seam convention (see
|
|
3051
|
+
// module.exports.stashList/evaluateSharedTreeGuard/committedInWindow above).
|
|
2734
3052
|
const broadcastCoalescer = createBroadcastCoalescer({
|
|
2735
3053
|
delayMs: BROADCAST_COALESCE_MS,
|
|
2736
3054
|
send: (payload) => {
|
|
2737
3055
|
if (!mainWindow || mainWindow.isDestroyed()) return;
|
|
2738
3056
|
sendIfAlive(mainWindow, 'schedule:state', payload);
|
|
2739
3057
|
},
|
|
2740
|
-
getPayload:
|
|
3058
|
+
getPayload: () => mutate(async (state) => {
|
|
3059
|
+
await module.exports.reconcile(state);
|
|
3060
|
+
return buildScheduleStatePayload(state);
|
|
3061
|
+
}),
|
|
2741
3062
|
});
|
|
2742
3063
|
|
|
3064
|
+
// Reconcile is unconditional — even with no window attached (a
|
|
3065
|
+
// scheduler-passive or headless instance), discovery must still run so
|
|
3066
|
+
// on-disk PRDs get onboarded. Only the actual IPC push is window-gated,
|
|
3067
|
+
// inside the coalescer's own `send`.
|
|
2743
3068
|
async function broadcast(opts = {}) {
|
|
2744
|
-
if (!mainWindow || mainWindow.isDestroyed()) return;
|
|
2745
|
-
const state = await readQueue();
|
|
2746
|
-
await reconcile(state);
|
|
2747
|
-
await writeQueue(state);
|
|
2748
3069
|
if (opts.flush) {
|
|
2749
3070
|
await broadcastCoalescer.flush();
|
|
2750
3071
|
} else {
|
|
@@ -3044,7 +3365,7 @@ const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
|
|
|
3044
3365
|
* process may still be writing to it, so reading now risks misclassifying a
|
|
3045
3366
|
* job that is about to emit result:success as no_result and double-running it.
|
|
3046
3367
|
* Ported from reconcileQueueOffline's cross-tick escalation (see
|
|
3047
|
-
*
|
|
3368
|
+
* src/main/lib/watchdogHelpers.cjs) — here it's a single deferred window since
|
|
3048
3369
|
* this process stays up to revisit it, rather than a separate short-lived
|
|
3049
3370
|
* watchdog process needing another tick.
|
|
3050
3371
|
*/
|
|
@@ -3064,17 +3385,26 @@ function partitionBootOrphans(jobs, isAlive = claudePidAlive) {
|
|
|
3064
3385
|
}
|
|
3065
3386
|
|
|
3066
3387
|
/**
|
|
3067
|
-
* applyOrphanOutcome(job, outcome, killNote?) → void
|
|
3388
|
+
* applyOrphanOutcome(job, outcome, killNote?, confirmedLandedCommit?) → void
|
|
3068
3389
|
*
|
|
3069
3390
|
* Mutates `job` in place to finalize a boot-orphaned 'running' job given its
|
|
3070
3391
|
* classified run outcome: success/failed finalize terminally; no_result/unknown
|
|
3071
3392
|
* re-queues to pending bounded by ORPHAN_REQUEUE_CAP. The status-mutation
|
|
3072
3393
|
* semantics (and the cap-exhaustion boundary) match the now-deleted
|
|
3073
|
-
* reconcileQueueOffline (
|
|
3394
|
+
* reconcileQueueOffline (src/main/lib/watchdogHelpers.cjs) verbatim; killNote
|
|
3074
3395
|
* plumbing differs slightly (see call sites) since this path always knows
|
|
3075
3396
|
* pid liveness up front rather than re-checking per tick.
|
|
3397
|
+
*
|
|
3398
|
+
* `confirmedLandedCommit` is the same evidence-before-failure gate
|
|
3399
|
+
* reapDeadRunningJobs applies (see resolveLandedCommitEvidence): a job that
|
|
3400
|
+
* dies while the app itself is offline is classified 'failed' from its log
|
|
3401
|
+
* tail alone, exactly like the pre-fix reap path was — so without this, an
|
|
3402
|
+
* orphaned job that actually landed a real commit is reachable via boot
|
|
3403
|
+
* reconciliation even though the live reap path is now guarded. Callers
|
|
3404
|
+
* must resolve this (a git spawn) BEFORE calling mutate(), never inside it —
|
|
3405
|
+
* pass null to skip the gate (e.g. when the outcome isn't 'failed').
|
|
3076
3406
|
*/
|
|
3077
|
-
function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
3407
|
+
function applyOrphanOutcome(job, outcome, killNote = '', confirmedLandedCommit = null) {
|
|
3078
3408
|
const now = new Date().toISOString();
|
|
3079
3409
|
if (outcome === 'success') {
|
|
3080
3410
|
transitionJob(job, 'completed', { reason: 'boot orphan reconciliation: run succeeded', source: 'applyOrphanOutcome' });
|
|
@@ -3083,9 +3413,16 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
3083
3413
|
job.finishedAt = now;
|
|
3084
3414
|
delete job.runtime;
|
|
3085
3415
|
} else if (outcome === 'failed') {
|
|
3086
|
-
|
|
3087
|
-
|
|
3088
|
-
|
|
3416
|
+
if (confirmedLandedCommit) {
|
|
3417
|
+
transitionJob(job, 'completed', { reason: `orphaned: app restarted while running${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
|
|
3418
|
+
job.exitCode = 0;
|
|
3419
|
+
job.error = null;
|
|
3420
|
+
job.landedCommit = confirmedLandedCommit;
|
|
3421
|
+
} else {
|
|
3422
|
+
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running${killNote}`, source: 'applyOrphanOutcome' });
|
|
3423
|
+
job.exitCode = job.exitCode ?? 1;
|
|
3424
|
+
job.error = `orphaned: app restarted while running${killNote}`;
|
|
3425
|
+
}
|
|
3089
3426
|
job.finishedAt = now;
|
|
3090
3427
|
delete job.runtime;
|
|
3091
3428
|
} else {
|
|
@@ -3093,6 +3430,13 @@ function applyOrphanOutcome(job, outcome, killNote = '') {
|
|
|
3093
3430
|
if (tries < ORPHAN_REQUEUE_CAP) {
|
|
3094
3431
|
resetJobFields(job, `orphaned: app restarted mid-run, re-queued (attempt ${tries + 1}/${ORPHAN_REQUEUE_CAP})${killNote}`, { source: 'applyOrphanOutcome' });
|
|
3095
3432
|
job.orphanRetries = tries + 1;
|
|
3433
|
+
} else if (confirmedLandedCommit) {
|
|
3434
|
+
transitionJob(job, 'completed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence`, source: 'applyOrphanOutcome:landed' });
|
|
3435
|
+
job.exitCode = 0;
|
|
3436
|
+
job.error = null;
|
|
3437
|
+
job.landedCommit = confirmedLandedCommit;
|
|
3438
|
+
job.finishedAt = now;
|
|
3439
|
+
delete job.runtime;
|
|
3096
3440
|
} else {
|
|
3097
3441
|
transitionJob(job, 'failed', { reason: `orphaned: app restarted while running, exhausted ${ORPHAN_REQUEUE_CAP} re-queue attempts${killNote}`, source: 'applyOrphanOutcome' });
|
|
3098
3442
|
job.exitCode = job.exitCode ?? 1;
|
|
@@ -3980,6 +4324,60 @@ async function pathExistsInTree(cwd, treeish, p) {
|
|
|
3980
4324
|
}
|
|
3981
4325
|
}
|
|
3982
4326
|
|
|
4327
|
+
/**
|
|
4328
|
+
* resolveLandedCommitEvidence(cwd, sha, sinceIso) → Promise<boolean>
|
|
4329
|
+
*
|
|
4330
|
+
* Bounded, non-fatal proof that `sha` is a real, resolvable commit in the
|
|
4331
|
+
* repo at `cwd`, committed no earlier than `sinceIso` — `git cat-file -e
|
|
4332
|
+
* <sha>^{commit}` plus a `git log -1 --format=%cI` timestamp check, both via
|
|
4333
|
+
* execGitAt's existing spawn+timeout bound (never shell:true, never an
|
|
4334
|
+
* unbounded execSync). This is the evidence gate reapDeadRunningJobs (PRD:
|
|
4335
|
+
* reaper must consult completion evidence) adds ahead of stamping a reaped
|
|
4336
|
+
* row 'failed': a landedCommit field being non-empty is not proof by itself
|
|
4337
|
+
* (job 1192 had one and still got reaped 'failed') — only a git-verified
|
|
4338
|
+
* resolution is.
|
|
4339
|
+
*
|
|
4340
|
+
* The timestamp bound matters because `landedCommit` deliberately survives
|
|
4341
|
+
* resetJobFields (see the comment there) so a re-fired run can consult it as
|
|
4342
|
+
* priorLandedCommit — which means a STALE landedCommit from an earlier
|
|
4343
|
+
* dispatch of the same slug can still be sitting on the row when a LATER
|
|
4344
|
+
* dispatch dies for real. Without `sinceIso`, that stale-but-real sha would
|
|
4345
|
+
* satisfy `cat-file -e` and wrongly promote a genuine failure to
|
|
4346
|
+
* 'completed'. Passing the current dispatch's `row.startedAt` as `sinceIso`
|
|
4347
|
+
* closes that: only a commit landed during THIS run counts as evidence.
|
|
4348
|
+
*
|
|
4349
|
+
* `cwd` is normalized through opsOwnership's resolveProjectRoot first (the
|
|
4350
|
+
* same "never trust a raw agent cwd" reasoning delegationReadiness.cjs
|
|
4351
|
+
* already relies on) so a row reaped while its cwd is an ephemeral worktree
|
|
4352
|
+
* checkout resolves the commit against the real project root instead.
|
|
4353
|
+
*
|
|
4354
|
+
* Never throws: a missing sha, a resolveProjectRoot failure (ephemeral cwd,
|
|
4355
|
+
* thrown error), a spawn failure, a timeout, or a cwd that no longer exists
|
|
4356
|
+
* on disk all resolve to `false` — the caller's safe default is 'failed',
|
|
4357
|
+
* exactly like today, whenever this can't positively prove landing.
|
|
4358
|
+
*/
|
|
4359
|
+
async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
|
|
4360
|
+
if (!sha || typeof sha !== 'string') return false;
|
|
4361
|
+
try {
|
|
4362
|
+
const root = resolveProjectRoot(cwd);
|
|
4363
|
+
await execGitAt(root, ['cat-file', '-e', `${sha}^{commit}`], { timeout: 10_000 });
|
|
4364
|
+
if (sinceIso) {
|
|
4365
|
+
const since = new Date(sinceIso).getTime();
|
|
4366
|
+
if (Number.isFinite(since)) {
|
|
4367
|
+
const committedIso = (await execGitAt(root, ['log', '-1', '--format=%cI', sha], { timeout: 10_000 })).trim();
|
|
4368
|
+
const committedAt = new Date(committedIso).getTime();
|
|
4369
|
+
// A commit dated before this dispatch even started can only be a
|
|
4370
|
+
// stale sha surviving from an earlier life of the row — never
|
|
4371
|
+
// evidence that THIS dispatch landed anything.
|
|
4372
|
+
if (Number.isFinite(committedAt) && committedAt < since) return false;
|
|
4373
|
+
}
|
|
4374
|
+
}
|
|
4375
|
+
return true;
|
|
4376
|
+
} catch {
|
|
4377
|
+
return false;
|
|
4378
|
+
}
|
|
4379
|
+
}
|
|
4380
|
+
|
|
3983
4381
|
/**
|
|
3984
4382
|
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
3985
4383
|
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
@@ -4522,6 +4920,56 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4522
4920
|
},
|
|
4523
4921
|
};
|
|
4524
4922
|
|
|
4923
|
+
// Wall-clock budget watchdog: unlike idleTailWatchdog above (which only
|
|
4924
|
+
// fires when the log mtime STALLS), this fires on total elapsed wall-
|
|
4925
|
+
// clock time regardless of whether the job keeps writing output — the
|
|
4926
|
+
// gap a chatty-but-runaway executor slips through (see
|
|
4927
|
+
// computeJobBudgetMs's header for the measured p50/p90/max this budget
|
|
4928
|
+
// is calibrated against). `quietMachine` jobs and any PRD with an
|
|
4929
|
+
// explicit `budgetExempt: true` opt out entirely — logged once here so
|
|
4930
|
+
// an unbounded job is never silently unbounded.
|
|
4931
|
+
const jobBudgetMs = computeJobBudgetMs(job.estimateMinutes);
|
|
4932
|
+
const budgetExempt = isJobBudgetExempt(job);
|
|
4933
|
+
if (budgetExempt) {
|
|
4934
|
+
safeLog(`[scheduler] wall-clock budget watchdog EXEMPT for ${job.slug} ` +
|
|
4935
|
+
`(${job.quietMachine === true ? 'quietMachine' : 'budgetExempt'}) — no wall-clock kill ceiling this run\n`);
|
|
4936
|
+
}
|
|
4937
|
+
let budgetWarningStamped = false;
|
|
4938
|
+
const budgetWatchdog = {
|
|
4939
|
+
label: 'budget',
|
|
4940
|
+
intervalMs: IDLE_CHECK_INTERVAL_MS,
|
|
4941
|
+
shouldFire(ctx) {
|
|
4942
|
+
if (budgetExempt) return false;
|
|
4943
|
+
const elapsedMs = Date.now() - ctx.startedAt;
|
|
4944
|
+
if (!budgetWarningStamped && elapsedMs >= jobBudgetMs * BUDGET_WARNING_FRACTION) {
|
|
4945
|
+
budgetWarningStamped = true;
|
|
4946
|
+
// Fire-and-forget (side effect inside a sync predicate, same pattern
|
|
4947
|
+
// resultTailWatchdog's shouldFire already uses for agentResultSubtype)
|
|
4948
|
+
// — exposes the warning on the row well before the kill fires, so
|
|
4949
|
+
// the renderer can show it without waiting for the next tick.
|
|
4950
|
+
mutate((state) => {
|
|
4951
|
+
const j = state.jobs.find((x) => x.slug === job.slug);
|
|
4952
|
+
if (!j) return;
|
|
4953
|
+
j.budgetWarning = { budgetMs: jobBudgetMs, elapsedMs, at: new Date().toISOString() };
|
|
4954
|
+
}).catch((e) => console.warn('[scheduler] budget-warning stamp failed', job.slug, e?.message));
|
|
4955
|
+
}
|
|
4956
|
+
return shouldKillForBudget(elapsedMs, jobBudgetMs);
|
|
4957
|
+
},
|
|
4958
|
+
action(ctx) {
|
|
4959
|
+
const elapsedMs = Date.now() - ctx.startedAt;
|
|
4960
|
+
ctx.safeLog(`\n[scheduler] wall-clock budget watchdog: ran ${Math.round(elapsedMs / 60_000)}m ` +
|
|
4961
|
+
`(> ${Math.round(jobBudgetMs / 60_000)}m budget, estimateMinutes=${job.estimateMinutes ?? 0}) — SIGTERM process group\n`);
|
|
4962
|
+
ctx.killedByWatchdog = 'budget';
|
|
4963
|
+
ctx.killTree('SIGTERM');
|
|
4964
|
+
const budgetKillTimer = setTimeout(() => {
|
|
4965
|
+
ctx.safeLog(`\n[scheduler] budget watchdog: still alive ${Math.round(POST_RESULT_KILL_MS/1000)}s after SIGTERM — SIGKILL\n`);
|
|
4966
|
+
ctx.killTree('SIGKILL');
|
|
4967
|
+
}, POST_RESULT_KILL_MS);
|
|
4968
|
+
if (budgetKillTimer.unref) budgetKillTimer.unref();
|
|
4969
|
+
ctx.addTimer(budgetKillTimer);
|
|
4970
|
+
},
|
|
4971
|
+
};
|
|
4972
|
+
|
|
4525
4973
|
// ---------- spawn ----------
|
|
4526
4974
|
|
|
4527
4975
|
const { child } = withChildAndLog({
|
|
@@ -4552,8 +5000,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4552
5000
|
detached: true,
|
|
4553
5001
|
},
|
|
4554
5002
|
},
|
|
4555
|
-
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog],
|
|
4556
|
-
onExit({ exitCode, signal, killedByWatchdog
|
|
5003
|
+
watchdogs: [resultTailWatchdog, deadmanWatchdog, idleTailWatchdog, budgetWatchdog],
|
|
5004
|
+
onExit({ exitCode, signal, killedByWatchdog, error, spawnFailed, leakedDescendants, safeLog: sl }) {
|
|
4557
5005
|
const durationMs = Date.now() - startedAt;
|
|
4558
5006
|
const leaked = leakedDescendants ?? [];
|
|
4559
5007
|
if (leaked.length > 0) {
|
|
@@ -4583,7 +5031,11 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4583
5031
|
// and 137 (128+SIGKILL) in case the process exited via signal-as-code.
|
|
4584
5032
|
let effectiveCode = exitCode;
|
|
4585
5033
|
const killedBySignal = signal === 'SIGTERM' || signal === 'SIGKILL' || exitCode === 143 || exitCode === 137 || exitCode === null;
|
|
4586
|
-
|
|
5034
|
+
// A budget kill must NEVER be laundered into a clean exit=0, even when
|
|
5035
|
+
// the agent had already emitted result=success before it fired — the
|
|
5036
|
+
// AC requires it always park needs_review, never silently 'completed'.
|
|
5037
|
+
// idle-tail/deadman/result-tail kills keep the existing success-mapping.
|
|
5038
|
+
const mappedToSuccess = agentResultSubtype === 'success' && killedBySignal && killedByWatchdog !== 'budget';
|
|
4587
5039
|
if (mappedToSuccess) {
|
|
4588
5040
|
effectiveCode = 0;
|
|
4589
5041
|
sl(`\n[scheduler] mapping exit code=${exitCode} signal=${signal} → 0 ` +
|
|
@@ -4606,6 +5058,16 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4606
5058
|
sl(`\n[scheduler] LAUNCH FAILURE (${launchFailed.kind}${launchFailed.httpStatus ? ` HTTP ${launchFailed.httpStatus}` : ''}): ` +
|
|
4607
5059
|
`${launchFailed.message} — no turn was taken; this is not a PRD failure\n`);
|
|
4608
5060
|
}
|
|
5061
|
+
// Formatted once, here, off the FINAL durationMs (more accurate than
|
|
5062
|
+
// the watchdog action's own snapshot at kill time) — matches the
|
|
5063
|
+
// reason string format the AC requires verbatim. See
|
|
5064
|
+
// resolveBudgetKillOutcome's own header for why this is gated on
|
|
5065
|
+
// killedBySignal, not on killedByWatchdog alone.
|
|
5066
|
+
const budgetKillOutcome = resolveBudgetKillOutcome({
|
|
5067
|
+
killedByWatchdog, killedBySignal, durationMs, jobBudgetMs, estimateMinutes: job.estimateMinutes,
|
|
5068
|
+
});
|
|
5069
|
+
const { budgetKillReason } = budgetKillOutcome;
|
|
5070
|
+
const effectiveKilledByWatchdog = budgetKillOutcome.killedByWatchdog;
|
|
4609
5071
|
// Sync write: child 'exit' handler must flush meta before resolve()
|
|
4610
5072
|
// so the spawnJob mutate() that follows sees the persisted exit code.
|
|
4611
5073
|
config.writeJsonSync(metaPath, {
|
|
@@ -4616,10 +5078,14 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4616
5078
|
launchEnvApplied: launchEnv && Object.keys(launchEnv).length ? Object.keys(launchEnv) : [],
|
|
4617
5079
|
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
4618
5080
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
5081
|
+
killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
4619
5082
|
schedulerBootedAt: SCHEDULER_BOOTED_AT, schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
4620
5083
|
originSessionId, contextDigestApplied,
|
|
4621
5084
|
});
|
|
4622
|
-
resolve({
|
|
5085
|
+
resolve({
|
|
5086
|
+
exitCode: effectiveCode, durationMs, rateLimited, networkError, launchFailure: launchFailed, resultStats,
|
|
5087
|
+
leakedDescendants: leaked, sessionId, killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
5088
|
+
});
|
|
4623
5089
|
},
|
|
4624
5090
|
});
|
|
4625
5091
|
|
|
@@ -4627,8 +5093,27 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4627
5093
|
safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
|
|
4628
5094
|
// Make this job the OOM killer's preferred victim over Electron.
|
|
4629
5095
|
biasJobOomScore(child.pid);
|
|
4630
|
-
//
|
|
4631
|
-
|
|
5096
|
+
// Persist runtime.pid with one retry — still fire-and-forget (must
|
|
5097
|
+
// never block the spawn), but a final failure is now loud instead of
|
|
5098
|
+
// silently swallowed. A silent failure here is exactly what let the
|
|
5099
|
+
// pidless-grace reaper terminalize a live, working job (runtime.pid
|
|
5100
|
+
// never landed, so selectReapableJobs had no way to tell "never
|
|
5101
|
+
// spawned" from "spawned but unrecorded").
|
|
5102
|
+
if (onPid) {
|
|
5103
|
+
(async () => {
|
|
5104
|
+
try {
|
|
5105
|
+
await onPid(child.pid, sessionId, cwd);
|
|
5106
|
+
} catch (firstErr) {
|
|
5107
|
+
try {
|
|
5108
|
+
await onPid(child.pid, sessionId, cwd);
|
|
5109
|
+
} catch (finalErr) {
|
|
5110
|
+
const message = finalErr?.message ?? String(finalErr);
|
|
5111
|
+
console.error(`[scheduler] FAILED to persist runtime.pid for ${job.slug} pid=${child.pid}: ${message}`);
|
|
5112
|
+
appendAuditEvent('job_pid_persist_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
|
|
5113
|
+
}
|
|
5114
|
+
}
|
|
5115
|
+
})();
|
|
5116
|
+
}
|
|
4632
5117
|
}
|
|
4633
5118
|
});
|
|
4634
5119
|
}
|
|
@@ -5550,8 +6035,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5550
6035
|
|
|
5551
6036
|
// Commit-guard baseline: snapshot the working tree BEFORE the run so the
|
|
5552
6037
|
// post-run check flags only paths THIS job left dirty, not pre-existing WIP.
|
|
6038
|
+
// Captured once, with status codes, so the shared-tree guard below can
|
|
6039
|
+
// tell "was untracked" apart from "was tracked-and-modified" (2026-09-12
|
|
6040
|
+
// incident) without a second `git status` call; every other consumer of
|
|
6041
|
+
// `guardBaseline` still gets the plain path-string array it always did.
|
|
5553
6042
|
const guardCwd = job.cwd || defaultCwd;
|
|
5554
|
-
const
|
|
6043
|
+
const guardBaselineEntries = await uncommittedChangesWithStatus(guardCwd);
|
|
6044
|
+
const guardBaseline = guardBaselineEntries ? guardBaselineEntries.map((e) => e.path) : guardBaselineEntries;
|
|
5555
6045
|
const guardHeadBefore = await gitHead(guardCwd);
|
|
5556
6046
|
// Shared-tree stash guard baseline (incident 2026-09-01): captured
|
|
5557
6047
|
// unconditionally, before worktree isolation is even attempted, so an
|
|
@@ -5603,6 +6093,22 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5603
6093
|
console.log(`[scheduler] ${job.slug}: isolated in worktree ${worktree.dir} (branch ${worktree.branch})`);
|
|
5604
6094
|
} else {
|
|
5605
6095
|
console.log(`[scheduler] ${job.slug}: running in main tree (worktree not used: ${worktree.reason})`);
|
|
6096
|
+
// A job losing worktree isolation must never be a silent downgrade
|
|
6097
|
+
// discoverable only by reading queue.json afterwards — every genuine
|
|
6098
|
+
// fallback (never the deliberate SM_JOB_WORKTREE_DISABLE opt-out) is
|
|
6099
|
+
// logged at warn in the durable ops error log, with the job slug, cwd,
|
|
6100
|
+
// and specific reason attached.
|
|
6101
|
+
if (!jobWorktree.isWorktreeDisabled()) {
|
|
6102
|
+
try {
|
|
6103
|
+
appendError({
|
|
6104
|
+
cwd: job.cwd || defaultCwd,
|
|
6105
|
+
scope: 'scheduler',
|
|
6106
|
+
level: 'warn',
|
|
6107
|
+
message: `${job.slug}: worktree isolation fell back to the SHARED working tree — ${worktree.reason}`,
|
|
6108
|
+
meta: { slug: job.slug, cwd: job.cwd || defaultCwd, reason: worktree.reason },
|
|
6109
|
+
});
|
|
6110
|
+
} catch { /* durable logging must never break dispatch */ }
|
|
6111
|
+
}
|
|
5606
6112
|
}
|
|
5607
6113
|
// dispatchPhase stamp folded into a single unconditional mutate covering
|
|
5608
6114
|
// both branches above — the degraded-isolation fallback flag (skipped
|
|
@@ -5997,7 +6503,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5997
6503
|
sharedTreeGuard = await module.exports.checkSharedTreeGuard({
|
|
5998
6504
|
cwd: guardCwd,
|
|
5999
6505
|
stashBaseline,
|
|
6000
|
-
dirtyBaseline:
|
|
6506
|
+
dirtyBaseline: guardBaselineEntries,
|
|
6001
6507
|
headBefore: guardHeadBefore,
|
|
6002
6508
|
slug: job.slug,
|
|
6003
6509
|
});
|
|
@@ -6024,6 +6530,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6024
6530
|
// narrowly to exit 143 — never applied to other non-zero exit codes or to
|
|
6025
6531
|
// rateLimited (already handled separately, above).
|
|
6026
6532
|
let sigtermCommitFound = false;
|
|
6533
|
+
// Verified SHA twin of sigtermCommitFound's boolean — only the budget-kill
|
|
6534
|
+
// path (below) threads this onto the row's landedCommit; classifySigtermWithCommit's
|
|
6535
|
+
// own needs_review branch is unchanged and keeps using the boolean alone.
|
|
6536
|
+
let sigtermLandedCommitEvidence = null;
|
|
6027
6537
|
if (res.exitCode === 143 && !res.rateLimited) {
|
|
6028
6538
|
const guardHeadAtSigterm = await gitHead(guardCwd);
|
|
6029
6539
|
sigtermCommitFound = await computeCommittedDuringRun(
|
|
@@ -6033,6 +6543,10 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6033
6543
|
job.startedAt,
|
|
6034
6544
|
new Date().toISOString(),
|
|
6035
6545
|
);
|
|
6546
|
+
if (guardHeadBefore && guardHeadAtSigterm && guardHeadAtSigterm !== guardHeadBefore) {
|
|
6547
|
+
const verified = await resolveLandedCommitEvidence(guardCwd, guardHeadAtSigterm, job.startedAt);
|
|
6548
|
+
if (verified) sigtermLandedCommitEvidence = guardHeadAtSigterm;
|
|
6549
|
+
}
|
|
6036
6550
|
}
|
|
6037
6551
|
|
|
6038
6552
|
// BLOCKED_BY_FOREIGN_WIP claim scan: the executor exits non-zero for this
|
|
@@ -6054,6 +6568,27 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6054
6568
|
}
|
|
6055
6569
|
}
|
|
6056
6570
|
|
|
6571
|
+
// Evidence gate for a PLAIN non-zero exit — not SIGTERM-with-commit
|
|
6572
|
+
// (classifySigtermWithCommit above already routes exit 143 to
|
|
6573
|
+
// needs_review when a commit landed) and not rate-limited (handled
|
|
6574
|
+
// separately). A process that dies non-zero for any OTHER reason
|
|
6575
|
+
// (crash during post-commit cleanup, an overrun watchdog's SIGKILL) is
|
|
6576
|
+
// observed directly by THIS exit handler — it never reaches
|
|
6577
|
+
// reapDeadRunningJobs' own git-verified landedCommit evidence gate, so
|
|
6578
|
+
// without this check the exact bug that gate exists to prevent (job
|
|
6579
|
+
// 1192: a landedCommit non-empty is not proof by itself, but discarding
|
|
6580
|
+
// proof of real landed work with no evidence check at all is worse)
|
|
6581
|
+
// recurs here, one call site over. Computed outside mutate() (I/O) like
|
|
6582
|
+
// every other pre-finalize git check above.
|
|
6583
|
+
let plainExitLandedCommitEvidence = null;
|
|
6584
|
+
if (res.exitCode !== 0 && res.exitCode !== 143 && !res.rateLimited) {
|
|
6585
|
+
const headAtPlainExit = await gitHead(guardCwd);
|
|
6586
|
+
if (guardHeadBefore && headAtPlainExit && headAtPlainExit !== guardHeadBefore) {
|
|
6587
|
+
const verified = await resolveLandedCommitEvidence(guardCwd, headAtPlainExit, job.startedAt);
|
|
6588
|
+
if (verified) plainExitLandedCommitEvidence = headAtPlainExit;
|
|
6589
|
+
}
|
|
6590
|
+
}
|
|
6591
|
+
|
|
6057
6592
|
let actuallyFailed = false;
|
|
6058
6593
|
let failedJobSnapshot = null;
|
|
6059
6594
|
let needsInvestigationNow = false;
|
|
@@ -6124,7 +6659,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6124
6659
|
// Determine effective status, applying the verifier verdict for exit=0 runs.
|
|
6125
6660
|
let effectiveStatus;
|
|
6126
6661
|
let sigtermOverrideReason = null;
|
|
6127
|
-
|
|
6662
|
+
// Wall-clock budget kill — checked FIRST and unconditionally wins:
|
|
6663
|
+
// never 'failed', never silently 'completed', and (via
|
|
6664
|
+
// sigtermLandedCommitEvidence/plainExitLandedCommitEvidence, whichever
|
|
6665
|
+
// this exit code populated) still adjudicated on git evidence rather
|
|
6666
|
+
// than discarded. See classifyBudgetKill's own header.
|
|
6667
|
+
const budgetKill = classifyBudgetKill(res, sigtermLandedCommitEvidence || plainExitLandedCommitEvidence);
|
|
6668
|
+
const sigtermOverride = (!budgetKill && res.exitCode !== 0)
|
|
6128
6669
|
? classifySigtermWithCommit(res.exitCode, sigtermCommitFound)
|
|
6129
6670
|
: null;
|
|
6130
6671
|
// Validated against the LIVE row's own foreign-WIP manifest — never
|
|
@@ -6132,7 +6673,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6132
6673
|
// launder a real regression into a block (PRD: give the executor a
|
|
6133
6674
|
// first-class verdict for "the gate failed on a sibling's in-flight
|
|
6134
6675
|
// file", but VALIDATE the claim rather than trust it).
|
|
6135
|
-
const foreignWipValidation = (!sigtermOverride && foreignWipClaimedPaths !== null)
|
|
6676
|
+
const foreignWipValidation = (!budgetKill && !sigtermOverride && foreignWipClaimedPaths !== null)
|
|
6136
6677
|
? validateForeignWipBlockClaim(foreignWipClaimedPaths, s.jobs[i2])
|
|
6137
6678
|
: null;
|
|
6138
6679
|
// Consecutive-block streak: cleared by default on every outcome and
|
|
@@ -6141,7 +6682,12 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6141
6682
|
// between two blocks always resets "in a row" back to zero.
|
|
6142
6683
|
const priorForeignWipBlockCount = s.jobs[i2].foreignWipBlockCount ?? 0;
|
|
6143
6684
|
delete s.jobs[i2].foreignWipBlockCount;
|
|
6144
|
-
if (
|
|
6685
|
+
if (budgetKill) {
|
|
6686
|
+
effectiveStatus = budgetKill.status;
|
|
6687
|
+
sigtermOverrideReason = budgetKill.reason;
|
|
6688
|
+
s.jobs[i2].verifierVerdict = 'budget_exceeded';
|
|
6689
|
+
if (budgetKill.landedCommit) jobLandedCommitThisRun = budgetKill.landedCommit;
|
|
6690
|
+
} else if (sigtermOverride) {
|
|
6145
6691
|
effectiveStatus = sigtermOverride.status;
|
|
6146
6692
|
sigtermOverrideReason = sigtermOverride.reason;
|
|
6147
6693
|
} else if (foreignWipValidation && foreignWipValidation.ok) {
|
|
@@ -6176,7 +6722,17 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6176
6722
|
effectiveStatus = 'failed';
|
|
6177
6723
|
sigtermOverrideReason = `SCHEDULER_VERDICT: BLOCKED_BY_FOREIGN_WIP rejected — unlisted path(s) not in the disclosed foreign-WIP manifest: ${foreignWipValidation.invalidPaths.join(', ') || '(no FOREIGN_WIP_PATHS line)'}`;
|
|
6178
6724
|
} else if (res.exitCode !== 0) {
|
|
6179
|
-
|
|
6725
|
+
if (plainExitLandedCommitEvidence) {
|
|
6726
|
+
// Same conservative posture as the SIGTERM+commit case above:
|
|
6727
|
+
// a landed commit doesn't prove every AC line passed, so this
|
|
6728
|
+
// still routes to needs_review for a human/reverify pass,
|
|
6729
|
+
// never silently to completed.
|
|
6730
|
+
effectiveStatus = 'needs_review';
|
|
6731
|
+
sigtermOverrideReason = `exited ${res.exitCode} after landing a git-verified commit — verify AC before treating as done`;
|
|
6732
|
+
jobLandedCommitThisRun = plainExitLandedCommitEvidence;
|
|
6733
|
+
} else {
|
|
6734
|
+
effectiveStatus = 'failed';
|
|
6735
|
+
}
|
|
6180
6736
|
} else if (
|
|
6181
6737
|
!verifyResult
|
|
6182
6738
|
|| COMPLETED_EQUIVALENT_VERDICTS.has(verifyResult.verdict)
|
|
@@ -6205,15 +6761,14 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6205
6761
|
const finalizeReason = (effectiveStatus === 'completed' && verifyResult?.verdict === 'already_satisfied_on_main')
|
|
6206
6762
|
? verifyResult.reason
|
|
6207
6763
|
: (sigtermOverrideReason ?? `run finished with exit ${res.exitCode}`);
|
|
6208
|
-
|
|
6209
|
-
s
|
|
6210
|
-
|
|
6211
|
-
s
|
|
6212
|
-
|
|
6213
|
-
|
|
6214
|
-
|
|
6215
|
-
|
|
6216
|
-
}
|
|
6764
|
+
// error/verifierVerdict are stamped BEFORE transitionJob() below —
|
|
6765
|
+
// needsReviewLedger's buildNeedsReviewEntryLine reads job.
|
|
6766
|
+
// verifierVerdict/heldReason/error synchronously off `job` the
|
|
6767
|
+
// instant transitionJob() runs (it's called inside transitionJob,
|
|
6768
|
+
// not deferred), so setting these after that call fed the durable
|
|
6769
|
+
// needs_review ledger stale/leftover values from before this run,
|
|
6770
|
+
// defeating its whole `byReason` rollup for the two escalation
|
|
6771
|
+
// paths that land here.
|
|
6217
6772
|
s.jobs[i2].error = (effectiveStatus === 'needs_review' || s.jobs[i2].blockedByForeignWip === true)
|
|
6218
6773
|
? (verifyResult?.reason ?? sigtermOverrideReason ?? null)
|
|
6219
6774
|
// A failed job (non-zero exit) never consults verifyResult above,
|
|
@@ -6231,18 +6786,28 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6231
6786
|
s.jobs[i2].landedCommit = jobLandedCommitThisRun;
|
|
6232
6787
|
}
|
|
6233
6788
|
// Persist the verifier's verdict string so the renderer can show it.
|
|
6234
|
-
// 'blocked_by_foreign_wip_streak'
|
|
6235
|
-
// exit-code path, never from verifyResult (which
|
|
6236
|
-
// non-zero exit) — never clobber
|
|
6789
|
+
// 'blocked_by_foreign_wip_streak'/'budget_exceeded' are set above
|
|
6790
|
+
// from the sigterm/exit-code path, never from verifyResult (which
|
|
6791
|
+
// stays null on a non-zero exit) — never clobber either here.
|
|
6237
6792
|
if (verifyResult?.verdict && verifyResult.verdict !== 'clean') {
|
|
6238
6793
|
s.jobs[i2].verifierVerdict = verifyResult.verdict;
|
|
6239
|
-
} else if (s.jobs[i2].verifierVerdict
|
|
6794
|
+
} else if (!['blocked_by_foreign_wip_streak', 'budget_exceeded'].includes(s.jobs[i2].verifierVerdict)) {
|
|
6240
6795
|
delete s.jobs[i2].verifierVerdict;
|
|
6241
6796
|
}
|
|
6797
|
+
transitionJob(s.jobs[i2], effectiveStatus, { reason: finalizeReason, source: 'spawnJob:finalize' });
|
|
6798
|
+
s.jobs[i2].finishedAt = new Date().toISOString();
|
|
6799
|
+
s.jobs[i2].exitCode = res.exitCode;
|
|
6800
|
+
s.jobs[i2].leakedDescendants = res.leakedDescendants ?? [];
|
|
6801
|
+
if (salvagePatch) {
|
|
6802
|
+
s.jobs[i2].salvagePatch = salvagePatch;
|
|
6803
|
+
} else {
|
|
6804
|
+
delete s.jobs[i2].salvagePatch;
|
|
6805
|
+
}
|
|
6242
6806
|
// Closed-set outcome taxonomy (issue #11 list A2) so a queue row
|
|
6243
6807
|
// says WHY it ended without anyone opening the transcript.
|
|
6244
6808
|
s.jobs[i2].terminalReason = launchFailure.deriveTerminalReason({
|
|
6245
6809
|
effectiveStatus, exitCode: res.exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure,
|
|
6810
|
+
budgetKill: !!budgetKill,
|
|
6246
6811
|
});
|
|
6247
6812
|
delete s.jobs[i2].launchFailure;
|
|
6248
6813
|
delete s.jobs[i2].heldReason;
|
|
@@ -7023,6 +7588,119 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7023
7588
|
return verdicts;
|
|
7024
7589
|
}
|
|
7025
7590
|
|
|
7591
|
+
/**
|
|
7592
|
+
* classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
|
|
7593
|
+
* totalSlots, lastDispatchAttemptAtMs, now, cwd, thresholdMs })
|
|
7594
|
+
* → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
|
|
7595
|
+
*
|
|
7596
|
+
* Single source of truth for the Scheduler page's queue-health header: the
|
|
7597
|
+
* one thing a human staring at a stale-looking queue needs is "which of the
|
|
7598
|
+
* genuinely different causes is this" (all slots busy? every pending row
|
|
7599
|
+
* blocked on a dependency? the dispatch driver itself never ticked?) — this
|
|
7600
|
+
* function names that cause instead of leaving the renderer to re-derive it.
|
|
7601
|
+
*
|
|
7602
|
+
* Reuses classifyQueueStarvation for the blocked/stalled read so the header
|
|
7603
|
+
* can never disagree with runQueueStarvationWatchdog's own decision to force
|
|
7604
|
+
* a tick: both are handed the same lastDispatchAttemptAt-based idle clock and
|
|
7605
|
+
* the same computeBlockedChains walk under the hood. Called here with
|
|
7606
|
+
* `thresholdMs: 0` first (a live header must say "blocked" the instant every
|
|
7607
|
+
* pending row is dependency-stuck, not wait out the watchdog's own 10-minute
|
|
7608
|
+
* grace period) — the returned `idleMs` is then compared against the REAL
|
|
7609
|
+
* `thresholdMs` to decide 'stalled' vs the healthy 'running' default, which
|
|
7610
|
+
* is exactly the comparison classifyQueueStarvation would make internally.
|
|
7611
|
+
*
|
|
7612
|
+
* `pending`/`dispatchable`/`blockedChains`/`needsReviewCount` are always
|
|
7613
|
+
* populated (via computeBlockedChains — the exact primitive
|
|
7614
|
+
* classifyQueueStarvation itself calls) regardless of kind, so a 'saturated'
|
|
7615
|
+
* or 'running' header can still say how much of the backlog is dependency-
|
|
7616
|
+
* blocked, not just the kinds where that's the headline cause.
|
|
7617
|
+
*
|
|
7618
|
+
* Kinds, in the priority order they're checked (paused is a decision, not a
|
|
7619
|
+
* stall; an open launch breaker explains an otherwise-inexplicable
|
|
7620
|
+
* non-dispatch before slot/dependency causes are even considered):
|
|
7621
|
+
* 'paused' — the scheduler itself is paused.
|
|
7622
|
+
* 'launch-blocked' — a pending row's persona has an active circuit-breaker
|
|
7623
|
+
* entry (lib/launchFailure.cjs).
|
|
7624
|
+
* 'idle' — nothing pending in this scope.
|
|
7625
|
+
* 'saturated' — pending work exists but every session slot is in use.
|
|
7626
|
+
* 'blocked' — nothing running, slots free, every pending row's
|
|
7627
|
+
* dependsOn chain terminates in a non-completed row.
|
|
7628
|
+
* 'stalled' — nothing running, slots free, at least one row is
|
|
7629
|
+
* dispatchable right now, and the dispatch driver has
|
|
7630
|
+
* been idle >= thresholdMs (agrees with the watchdog).
|
|
7631
|
+
* 'running' — the healthy default: work is flowing, or the driver
|
|
7632
|
+
* hasn't been idle long enough to call a stall yet.
|
|
7633
|
+
*
|
|
7634
|
+
* Pure, no IO. `cwd` scopes jobs/pending/blocked/needsReview to one project
|
|
7635
|
+
* (the Scheduler nav row is PROJECT-face — see CLAUDE.md); `freeSlots` /
|
|
7636
|
+
* `totalSlots` / `launchBlocks` stay machine-wide inputs by design, same as
|
|
7637
|
+
* WindowStrip's existing scopeCwd split.
|
|
7638
|
+
*/
|
|
7639
|
+
function classifyQueueHealth({
|
|
7640
|
+
jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
|
|
7641
|
+
lastDispatchAttemptAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
|
|
7642
|
+
} = {}) {
|
|
7643
|
+
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7644
|
+
const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
|
|
7645
|
+
const pendingRows = projectJobs.filter((j) => j.status === 'pending');
|
|
7646
|
+
const runningRows = projectJobs.filter((j) => j.status === 'running' || runningSlugs?.has?.(j.slug));
|
|
7647
|
+
const needsReviewCount = projectJobs.filter((j) => j.status === 'needs_review').length;
|
|
7648
|
+
|
|
7649
|
+
// computeBlockedChains is the SAME primitive classifyQueueStarvation calls
|
|
7650
|
+
// internally, computed once here so EVERY kind (not just 'blocked'/
|
|
7651
|
+
// 'stalled') carries real dispatchable-vs-blocked counts instead of a null.
|
|
7652
|
+
const blockedChains = computeBlockedChains(projectJobs);
|
|
7653
|
+
const blockedTotal = blockedChains.reduce((n, c) => n + c.blocked, 0);
|
|
7654
|
+
const dispatchable = Math.max(0, pendingRows.length - blockedTotal);
|
|
7655
|
+
const base = {
|
|
7656
|
+
cwd, pending: pendingRows.length, dispatchable, blockedChains, needsReviewCount, runningCount: runningRows.length,
|
|
7657
|
+
};
|
|
7658
|
+
|
|
7659
|
+
if (paused) return { ...base, kind: 'paused', reason: paused.reason ?? null };
|
|
7660
|
+
|
|
7661
|
+
// launch-blocked: only a persona a PENDING row in this scope actually uses
|
|
7662
|
+
// — a breaker open for a persona nothing here needs is not this scope's
|
|
7663
|
+
// problem (matches WindowStrip's own unconditional-banner-per-block read).
|
|
7664
|
+
const neededAgentTypes = new Set(pendingRows.map((j) => launchFailure.launchBlockKeyFor(j)));
|
|
7665
|
+
for (const [key, block] of Object.entries(launchBlocks ?? {})) {
|
|
7666
|
+
if (block && neededAgentTypes.has(key)) {
|
|
7667
|
+
return { ...base, kind: 'launch-blocked', agentType: key, block };
|
|
7668
|
+
}
|
|
7669
|
+
}
|
|
7670
|
+
|
|
7671
|
+
if (pendingRows.length === 0) return { ...base, kind: 'idle' };
|
|
7672
|
+
|
|
7673
|
+
// classifyQueueStarvation only ever classifies while nothing is running
|
|
7674
|
+
// (its own runningCount > 0 guard) — that boundary is also exactly where
|
|
7675
|
+
// slot saturation, not dependency shape, is the honest cause.
|
|
7676
|
+
if (runningRows.length > 0) {
|
|
7677
|
+
if (Number.isFinite(freeSlots) && freeSlots <= 0) {
|
|
7678
|
+
return { ...base, kind: 'saturated', totalSlots: totalSlots ?? null };
|
|
7679
|
+
}
|
|
7680
|
+
return { ...base, kind: 'running' };
|
|
7681
|
+
}
|
|
7682
|
+
|
|
7683
|
+
// Nothing running: hand the SAME rows + idle clock to classifyQueueStarvation
|
|
7684
|
+
// (thresholdMs: 0 — a live header must say "blocked" the instant every
|
|
7685
|
+
// pending row is dependency-stuck, not wait out the watchdog's own grace
|
|
7686
|
+
// period) purely for its idleMs reading; its own dispatchable/blockedChains
|
|
7687
|
+
// are mathematically identical to `base`'s (same computeBlockedChains walk
|
|
7688
|
+
// over the same rows), so `base` already carries them.
|
|
7689
|
+
const immediate = classifyQueueStarvation({
|
|
7690
|
+
jobs: projectJobs, paused: false, runningCount: 0,
|
|
7691
|
+
lastRunAtMs: lastDispatchAttemptAtMs, now, thresholdMs: 0,
|
|
7692
|
+
});
|
|
7693
|
+
// pending.length is already > 0 above, so `immediate` can only be null when
|
|
7694
|
+
// lastDispatchAttemptAtMs is itself in the future (clock skew) — fall back
|
|
7695
|
+
// to computing idleMs the same way rather than asserting a kind we can't
|
|
7696
|
+
// back up with a real number.
|
|
7697
|
+
const idleMs = immediate ? immediate.idleMs
|
|
7698
|
+
: (Number.isFinite(lastDispatchAttemptAtMs) ? now - lastDispatchAttemptAtMs : Infinity);
|
|
7699
|
+
if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
|
|
7700
|
+
const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
|
|
7701
|
+
return { ...base, kind, idleMs };
|
|
7702
|
+
}
|
|
7703
|
+
|
|
7026
7704
|
/**
|
|
7027
7705
|
* The watchdog half: acts on classifyQueueStarvationByProject. Called from
|
|
7028
7706
|
* the heartbeat, which already runs on its own timer independent of the
|
|
@@ -7254,12 +7932,19 @@ async function reapDeadRunningJobs() {
|
|
|
7254
7932
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
7255
7933
|
// queue.json is the source of truth for which jobs are actually running.
|
|
7256
7934
|
const state = await readQueue();
|
|
7935
|
+
// Shared by the log-evidence injections below and the reapable-processing
|
|
7936
|
+
// loop further down — same `j.runId` → run log path formula either way.
|
|
7937
|
+
const logPathForJob = (j) => (j?.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null);
|
|
7257
7938
|
const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
|
|
7258
7939
|
pidAlive: claudePidAlive,
|
|
7259
7940
|
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
7260
7941
|
findLiveProcess: (j) => findLiveProcessForJob(j, {
|
|
7261
7942
|
worktreeDir: jobWorktree.worktreeDirFor(j.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD, j.slug),
|
|
7943
|
+
runCwd: j.runtime?.cwd || j.cwd,
|
|
7262
7944
|
}),
|
|
7945
|
+
getLogPid: (j) => readSpawnedPidFromLog(logPathForJob(j)),
|
|
7946
|
+
getLogMtimeMs: (j) => readLogMtimeMs(logPathForJob(j)),
|
|
7947
|
+
logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
|
|
7263
7948
|
});
|
|
7264
7949
|
for (const w of warnings) {
|
|
7265
7950
|
console.warn(`[scheduler] reapDeadRunningJobs: ${w.reason} slug=${w.slug} — leaving row alone`);
|
|
@@ -7284,11 +7969,9 @@ async function reapDeadRunningJobs() {
|
|
|
7284
7969
|
}
|
|
7285
7970
|
|
|
7286
7971
|
const dead = [];
|
|
7287
|
-
for (const { slug, pid, pidless, reason } of reapable) {
|
|
7972
|
+
for (const { slug, pid, pidless, reason, failureOverride } of reapable) {
|
|
7288
7973
|
const j = state.jobs.find((x) => x.slug === slug);
|
|
7289
|
-
const logPath = j
|
|
7290
|
-
? path.join(RUNS_DIR, j.runId, `${j.slug}.log`)
|
|
7291
|
-
: null;
|
|
7974
|
+
const logPath = logPathForJob(j);
|
|
7292
7975
|
// Absent/empty run dir → classifyRunOutcome finds no result event →
|
|
7293
7976
|
// 'no_result' → non-success below → filed as failed, never completed.
|
|
7294
7977
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
@@ -7307,7 +7990,7 @@ async function reapDeadRunningJobs() {
|
|
|
7307
7990
|
// re-derived so a phantom link never survives the reap.
|
|
7308
7991
|
const hasOwnArtifact = pidless ? logHasOutput(logPath) : true;
|
|
7309
7992
|
const gateOutcome = pidless ? resolvePidlessGateOutcome(outcome, hasOwnArtifact) : mapOutcomeToGateOutcome(outcome);
|
|
7310
|
-
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact });
|
|
7993
|
+
dead.push({ slug, pid, outcome, gateOutcome, pidless, reason, logPath, noOwnArtifact: pidless && !hasOwnArtifact, failureOverride });
|
|
7311
7994
|
}
|
|
7312
7995
|
|
|
7313
7996
|
queueHealthSweepCycle += 1;
|
|
@@ -7405,8 +8088,44 @@ async function reapDeadRunningJobs() {
|
|
|
7405
8088
|
integrationResults.set(d.slug, { effectiveSuccess, landedCommit, notLandedInfo });
|
|
7406
8089
|
}
|
|
7407
8090
|
|
|
8091
|
+
// Evidence-before-failure guard for a row about to be stamped 'failed'
|
|
8092
|
+
// (this PRD — job 1192 shipped a real 3-file commit and was still
|
|
8093
|
+
// reaped 'failed' because this check did not exist): a `landedCommit`
|
|
8094
|
+
// already recorded on the row is only ever stamped from an actual HEAD
|
|
8095
|
+
// advance or a proven branch-integration (jobLandedCommitThisRun / the
|
|
8096
|
+
// dead-pid integration proof above / the dispatch-time sidecar
|
|
8097
|
+
// backfill) — never speculative — but it can still be STALE by the time
|
|
8098
|
+
// this row is reaped (the branch it named could have been force-pushed
|
|
8099
|
+
// over, or the row could be carrying a sidecar-backfilled sha from a
|
|
8100
|
+
// run that was later discarded). git-resolving it here is what turns
|
|
8101
|
+
// "the field is non-empty" into "this sha is a real commit in this
|
|
8102
|
+
// repo right now". Computed OUTSIDE mutate() for the same reason
|
|
8103
|
+
// integrationResults is above: git spawn work must never run inside
|
|
8104
|
+
// mutate()'s single global serialization chain.
|
|
8105
|
+
//
|
|
8106
|
+
// Scoped to exactly the rows that would otherwise fall through to
|
|
8107
|
+
// 'failed' below: a 'success' outcome is already resolved by
|
|
8108
|
+
// integrationResults above (never reaches 'failed'), a rate-limited
|
|
8109
|
+
// death is retryable and never terminal, and a pidless row that already
|
|
8110
|
+
// carries a `failureOverride` (PRD 1173) is already diverted to
|
|
8111
|
+
// needs_review — this gate must never re-litigate either of those.
|
|
8112
|
+
const landedCommitEvidence = new Map();
|
|
8113
|
+
// Each row's evidence check is an independent read-only `git cat-file`/
|
|
8114
|
+
// `git log` pair with no shared mutable state between iterations, so
|
|
8115
|
+
// this runs the whole dead-job batch concurrently rather than one
|
|
8116
|
+
// dispatch's git-spawn latency at a time.
|
|
8117
|
+
await Promise.all(dead.map(async (d) => {
|
|
8118
|
+
if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
|
|
8119
|
+
if (d.pidless && d.failureOverride) return;
|
|
8120
|
+
const row = state.jobs.find((x) => x.slug === d.slug);
|
|
8121
|
+
if (!row?.landedCommit) return;
|
|
8122
|
+
const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
8123
|
+
const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
|
|
8124
|
+
if (resolved) landedCommitEvidence.set(d.slug, row.landedCommit);
|
|
8125
|
+
}));
|
|
8126
|
+
|
|
7408
8127
|
await mutate(async (s) => {
|
|
7409
|
-
for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact } of dead) {
|
|
8128
|
+
for (const { slug, pid, outcome, gateOutcome, pidless, reason, noOwnArtifact, failureOverride } of dead) {
|
|
7410
8129
|
const idx = s.jobs.findIndex((x) => x.slug === slug);
|
|
7411
8130
|
if (idx < 0 || s.jobs[idx].status !== 'running') continue; // race guard
|
|
7412
8131
|
const rateLimited = outcome === 'rate_limited';
|
|
@@ -7465,6 +8184,20 @@ async function reapDeadRunningJobs() {
|
|
|
7465
8184
|
notLandedInfo = ir.notLandedInfo;
|
|
7466
8185
|
}
|
|
7467
8186
|
}
|
|
8187
|
+
// Evidence-before-failure guard for the pidless-reap path (PRD 1173):
|
|
8188
|
+
// a pidless reap about to stamp 'failed' purely because runtime.pid
|
|
8189
|
+
// was never recorded must first check whether this row already
|
|
8190
|
+
// carries a landedCommit from an earlier dispatch of the same slug
|
|
8191
|
+
// (landedCommit survives a reset — see the comment near
|
|
8192
|
+
// resetJobFields). Diverted to needs_review, never silently
|
|
8193
|
+
// 'completed' — see resolvePidlessFailureOverride's header in
|
|
8194
|
+
// reaperHelpers.cjs for why needs_review is the correct destination.
|
|
8195
|
+
// Scoped strictly to the pidless branch: dead-pid and rate-limited
|
|
8196
|
+
// rows are untouched, and a pidless row that already resolved to
|
|
8197
|
+
// effectiveSuccess/notLandedInfo above is left alone too.
|
|
8198
|
+
if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
|
|
8199
|
+
notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
|
|
8200
|
+
}
|
|
7468
8201
|
|
|
7469
8202
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
7470
8203
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
@@ -7472,9 +8205,20 @@ async function reapDeadRunningJobs() {
|
|
|
7472
8205
|
const baseReason = notLandedInfo
|
|
7473
8206
|
? `reaped: ${notLandedInfo.reason}`
|
|
7474
8207
|
: (pidless ? reason : `reaped: process gone (outcome=${outcome})`);
|
|
8208
|
+
// Evidence gate (this PRD): a row that would otherwise fall through
|
|
8209
|
+
// to 'failed' below, but whose landedCommit was proven to resolve
|
|
8210
|
+
// via git cat-file BEFORE this mutate() ran (see landedCommitEvidence
|
|
8211
|
+
// above), gets promoted to 'completed' instead — the row already
|
|
8212
|
+
// shipped real work, so a bookkeeping gap (no runtime.pid recorded)
|
|
8213
|
+
// must never override git-verified evidence with a false failure.
|
|
8214
|
+
const confirmedLandedCommit = (!effectiveSuccess && !notLandedInfo && !rateLimited)
|
|
8215
|
+
? (landedCommitEvidence.get(slug) || null)
|
|
8216
|
+
: null;
|
|
7475
8217
|
const transitionReason = rateLimited
|
|
7476
8218
|
? `reaped: rate limit detected — reset to pending, not failed (outcome=${outcome})${leftoverSuffix}`
|
|
7477
|
-
:
|
|
8219
|
+
: confirmedLandedCommit
|
|
8220
|
+
? `${baseReason}, but landedCommit ${confirmedLandedCommit} resolves — completed on evidence${leftoverSuffix}`
|
|
8221
|
+
: baseReason + leftoverSuffix;
|
|
7478
8222
|
|
|
7479
8223
|
if (rateLimited) {
|
|
7480
8224
|
// Retryable, never terminal (PRD 1117) — same resetJobFields path
|
|
@@ -7483,17 +8227,27 @@ async function reapDeadRunningJobs() {
|
|
|
7483
8227
|
// paused-for-rate-limit reset: fresh runId/startedAt/exitCode.
|
|
7484
8228
|
resetJobFields(s.jobs[idx], transitionReason, { source: 'reapDeadRunningJobs:rate-limit' });
|
|
7485
8229
|
} else {
|
|
7486
|
-
const
|
|
7487
|
-
|
|
7488
|
-
|
|
7489
|
-
|
|
7490
|
-
|
|
7491
|
-
|
|
8230
|
+
const landed = effectiveSuccess || Boolean(confirmedLandedCommit);
|
|
8231
|
+
const targetStatus = effectiveSuccess
|
|
8232
|
+
? 'completed'
|
|
8233
|
+
: (notLandedInfo ? 'needs_review' : (confirmedLandedCommit ? 'completed' : 'failed'));
|
|
8234
|
+
const source = confirmedLandedCommit ? 'reapDeadRunningJobs:landed' : 'reapDeadRunningJobs';
|
|
8235
|
+
// error/verifierVerdict are stamped BEFORE transitionJob() below —
|
|
8236
|
+
// see the identical ordering fix (and its rationale) in spawnJob's
|
|
8237
|
+
// finalize path: transitionJob's needs_review ledger entry reads
|
|
8238
|
+
// these fields off `job` synchronously the instant it runs, so
|
|
8239
|
+
// setting them after fed the ledger a stale/leftover reason.
|
|
8240
|
+
s.jobs[idx].error = landed ? null : `${transitionReason} (outcome=${outcome})`;
|
|
7492
8241
|
if (notLandedInfo) {
|
|
7493
8242
|
s.jobs[idx].verifierVerdict = notLandedInfo.verdict;
|
|
7494
8243
|
} else {
|
|
7495
8244
|
delete s.jobs[idx].verifierVerdict;
|
|
7496
8245
|
}
|
|
8246
|
+
transitionJob(s.jobs[idx], targetStatus, { reason: transitionReason, source });
|
|
8247
|
+
s.jobs[idx].exitCode = landed ? 0 : (s.jobs[idx].exitCode ?? 1);
|
|
8248
|
+
s.jobs[idx].finishedAt = new Date().toISOString();
|
|
8249
|
+
s.jobs[idx].gateOutcome = gateOutcome;
|
|
8250
|
+
if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
|
|
7497
8251
|
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
7498
8252
|
}
|
|
7499
8253
|
// A pidless spawn that never wrote its own '<slug>.log' into the
|
|
@@ -7529,7 +8283,14 @@ async function reapDeadRunningJobs() {
|
|
|
7529
8283
|
appendAuditEvent('job_reaped_rate_limited', { slug, cwd: s.jobs[idx].cwd ?? null });
|
|
7530
8284
|
} else if (pidless) {
|
|
7531
8285
|
console.log(`[scheduler] reaped pidless zombie job slug=${slug} outcome=${outcome}`);
|
|
7532
|
-
appendAuditEvent('job_reaped_pidless', {
|
|
8286
|
+
appendAuditEvent('job_reaped_pidless', {
|
|
8287
|
+
slug,
|
|
8288
|
+
cwd: s.jobs[idx].cwd ?? null,
|
|
8289
|
+
outcome,
|
|
8290
|
+
graceMs: PIDLESS_SPAWN_GRACE_MS,
|
|
8291
|
+
landedCommit: s.jobs[idx].landedCommit ?? null,
|
|
8292
|
+
verifierVerdict: s.jobs[idx].verifierVerdict ?? null,
|
|
8293
|
+
});
|
|
7533
8294
|
} else {
|
|
7534
8295
|
console.log(`[scheduler] reaped dead job slug=${slug} pid=${pid} outcome=${outcome}`);
|
|
7535
8296
|
}
|
|
@@ -7724,6 +8485,12 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
7724
8485
|
const seen = new Set(hot.map((j) => `${j.slug}|${j.runId ?? ''}`));
|
|
7725
8486
|
const archived = (Array.isArray(historyEntries) ? historyEntries : []).filter((j) => {
|
|
7726
8487
|
if (!j) return false;
|
|
8488
|
+
// needs_review_entry/needs_review_resolution lines (needsReviewLedger.cjs)
|
|
8489
|
+
// share history.jsonl with terminal job rows but carry no `status` — the
|
|
8490
|
+
// History view (SchedulerHistoryView.tsx) renders ScheduleJob rows, so a
|
|
8491
|
+
// ledger line slipping through here would show up as a statusless,
|
|
8492
|
+
// meaningless row in that table.
|
|
8493
|
+
if (j.kind && j.kind !== 'terminal') return false;
|
|
7727
8494
|
const key = `${j.slug}|${j.runId ?? ''}`;
|
|
7728
8495
|
if (seen.has(key)) return false;
|
|
7729
8496
|
seen.add(key);
|
|
@@ -7849,6 +8616,69 @@ function isExhaustedAutoFix(job) {
|
|
|
7849
8616
|
return job.autoFixOutcome !== 'plan' && (job.autoFixRetries ?? 0) >= 1;
|
|
7850
8617
|
}
|
|
7851
8618
|
|
|
8619
|
+
/**
|
|
8620
|
+
* Verdicts from the POST-RUN GUARDS (commit-guard / shared-tree guard) that a
|
|
8621
|
+
* later, independently-checkable commit/looksDone signal can meaningfully
|
|
8622
|
+
* confirm or refute. Auto-fix investigations only ever launch for FAILING
|
|
8623
|
+
* runs (see isExhaustedAutoFix above and selectAutoFixTargets) — a job parked
|
|
8624
|
+
* by one of these GUARD verdicts exits 0 and never has autoFixAttempted set,
|
|
8625
|
+
* so it is invisible to isExhaustedAutoFix and can sit in needs_review
|
|
8626
|
+
* forever with nothing to spend and nothing to exhaust (PRD 1181, 2026-09-12:
|
|
8627
|
+
* exit 0, commit aff5607 landed, parked on a shared-tree verdict, cleared
|
|
8628
|
+
* only by a human).
|
|
8629
|
+
*
|
|
8630
|
+
* 'worktree_integration_failed' is deliberately EXCLUDED — its damage IS a
|
|
8631
|
+
* commit: one stranded on an unmerged `sm-job/<slug>` branch. A `landedCommit`
|
|
8632
|
+
* existing is not evidence against that verdict, it is a restatement of it,
|
|
8633
|
+
* so admitting it here would auto-complete a row whose work never actually
|
|
8634
|
+
* reached the target branch. It already has its own dedicated, git-native
|
|
8635
|
+
* resolution path (selectMechanicalRecoveryTarget / performMechanicalRecovery
|
|
8636
|
+
* — a real re-attempted merge) and must never be pulled into this ladder.
|
|
8637
|
+
*
|
|
8638
|
+
* 'pidless_reap_with_landed_commit' (PRD 1173, resolvePidlessFailureOverride
|
|
8639
|
+
* in reaperHelpers.cjs) is included: it parks on the exact same shape (exit
|
|
8640
|
+
* never observed / no autoFixAttempted, real landedCommit evidence) as
|
|
8641
|
+
* 'silent_no_op' and 'shared_tree_reverted', and this ladder never trusts
|
|
8642
|
+
* landedCommit alone anyway — applyNeedsReviewAutoResolve only resolves once
|
|
8643
|
+
* job.looksDone independently reconfirms via a fresh commits-since-this-run
|
|
8644
|
+
* scan, which is exactly the "does the commit correspond to THIS dispatch"
|
|
8645
|
+
* re-verification resolvePidlessFailureOverride's own header says the
|
|
8646
|
+
* pidless-reap path itself cannot do. Omitting it here reproduces the same
|
|
8647
|
+
* "nothing to spend, nothing to exhaust" needs_review stall this PRD exists
|
|
8648
|
+
* to fix, just for a third verdict.
|
|
8649
|
+
*/
|
|
8650
|
+
const GUARD_VERDICT_EVIDENCE_ELIGIBLE = new Set(['silent_no_op', 'shared_tree_reverted', 'pidless_reap_with_landed_commit']);
|
|
8651
|
+
|
|
8652
|
+
/**
|
|
8653
|
+
* Pure predicate, no I/O: a needs_review row parked directly by one of the
|
|
8654
|
+
* GUARD_VERDICT_EVIDENCE_ELIGIBLE verdicts, that never went through an
|
|
8655
|
+
* auto-fix investigation at all (autoFixAttempted is not true) — the
|
|
8656
|
+
* structural gap this PRD closes, distinct from isExhaustedAutoFix's "went
|
|
8657
|
+
* through auto-fix and spent it" case. A job that DID get an auto-fix
|
|
8658
|
+
* investigation is left to isExhaustedAutoFix's own ladder rather than this
|
|
8659
|
+
* one, even if its verifierVerdict happens to also be in the eligible set.
|
|
8660
|
+
* Exported for tests.
|
|
8661
|
+
*/
|
|
8662
|
+
function isGuardParkedWithoutAutoFix(job) {
|
|
8663
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
8664
|
+
if (job.autoFixAttempted === true) return false;
|
|
8665
|
+
return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
|
|
8666
|
+
}
|
|
8667
|
+
|
|
8668
|
+
/**
|
|
8669
|
+
* Pure predicate, no I/O: is this needs_review row eligible for the bounded
|
|
8670
|
+
* auto-resolve ladder at all — either because its auto-fix path is genuinely
|
|
8671
|
+
* spent (isExhaustedAutoFix), or because it was parked by a GUARD verdict
|
|
8672
|
+
* that never entered auto-fix in the first place (isGuardParkedWithoutAutoFix).
|
|
8673
|
+
* Both classes share ONE ladder (applyNeedsReviewAutoResolve) rather than a
|
|
8674
|
+
* duplicated one — the ladder itself doesn't care which door a row came
|
|
8675
|
+
* through, only whether it now carries completion evidence (job.looksDone).
|
|
8676
|
+
* Exported for tests.
|
|
8677
|
+
*/
|
|
8678
|
+
function isEligibleForNeedsReviewAutoResolve(job) {
|
|
8679
|
+
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
|
|
8680
|
+
}
|
|
8681
|
+
|
|
7852
8682
|
/**
|
|
7853
8683
|
* Pure predicate: an investigation produced a fix plan (autoFixOutcome ===
|
|
7854
8684
|
* 'plan') but its fix-plan slug is not present among `queuedSlugs` — the
|
|
@@ -7996,17 +8826,26 @@ function isRescanCandidate(job) {
|
|
|
7996
8826
|
* selectResumeRecoveryTarget / selectAutoFixTargets) so the guard can never
|
|
7997
8827
|
* again be narrower than the work reverifyNeedsReview performs.
|
|
7998
8828
|
*
|
|
7999
|
-
*
|
|
8000
|
-
*
|
|
8001
|
-
*
|
|
8002
|
-
*
|
|
8003
|
-
*
|
|
8004
|
-
*
|
|
8829
|
+
* Widened again (this PRD): a `needs_review` row parked directly by a GUARD
|
|
8830
|
+
* verdict with no auto-fix history (isGuardParkedWithoutAutoFix) is not an
|
|
8831
|
+
* isRescanCandidate either — RESCANNABLE_VERDICTS covers transcript-verifier
|
|
8832
|
+
* verdicts, not commit-guard/shared-tree-guard verdicts — but
|
|
8833
|
+
* reverifyNeedsReview's looksDone-annotation pass now runs for it too (see
|
|
8834
|
+
* that function). Same rule as always: never let this guard be narrower than
|
|
8835
|
+
* the work reverifyNeedsReview actually performs.
|
|
8836
|
+
*
|
|
8837
|
+
* Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
|
|
8838
|
+
* isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
|
|
8839
|
+
* called with an injected fixSlugExists that always returns false — cheap
|
|
8840
|
+
* and deliberately over-inclusive (a false positive here just means one
|
|
8841
|
+
* extra periodic pass, never a missed one) so this guard never pays
|
|
8842
|
+
* selectAutoFixTargets's production fs.existsSync scan per tick.
|
|
8843
|
+
* resolveRunId's IO only fires for rows missing job.runId, same as
|
|
8005
8844
|
* isRescanCandidate already incurs above.
|
|
8006
8845
|
*/
|
|
8007
8846
|
function shouldRunPeriodicReverify(jobs) {
|
|
8008
8847
|
if (!Array.isArray(jobs)) return false;
|
|
8009
|
-
if (jobs.some((j) => isRescanCandidate(j))) return true;
|
|
8848
|
+
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
|
|
8010
8849
|
if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
|
|
8011
8850
|
return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
|
|
8012
8851
|
}
|
|
@@ -8172,9 +9011,11 @@ function needsReviewAutoResolveDisabled() {
|
|
|
8172
9011
|
* selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) →
|
|
8173
9012
|
* [{ slug, cwd, ageMs, attempts }]
|
|
8174
9013
|
*
|
|
8175
|
-
* Pure selector — no IO. Selects `needs_review` rows
|
|
8176
|
-
*
|
|
8177
|
-
*
|
|
9014
|
+
* Pure selector — no IO. Selects `needs_review` rows eligible for the
|
|
9015
|
+
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — either
|
|
9016
|
+
* auto-fix genuinely spent, or parked by a GUARD verdict that never entered
|
|
9017
|
+
* auto-fix at all), whose newest statusHistory entry with `to ===
|
|
9018
|
+
* 'needs_review'` is older than `thresholdMs`, and whose
|
|
8178
9019
|
* exhaustedResolveAttempts counter has not yet spent its cap.
|
|
8179
9020
|
*
|
|
8180
9021
|
* The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
|
|
@@ -8188,7 +9029,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
8188
9029
|
const targets = [];
|
|
8189
9030
|
for (const j of jobs ?? []) {
|
|
8190
9031
|
if (j.status !== 'needs_review') continue;
|
|
8191
|
-
if (!
|
|
9032
|
+
if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
|
|
8192
9033
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
|
|
8193
9034
|
const history = j.statusHistory || [];
|
|
8194
9035
|
let entry = null;
|
|
@@ -8224,17 +9065,26 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
8224
9065
|
* failed-autoreset loop above) so a stale target computed before this
|
|
8225
9066
|
* mutate() pass can never double-apply. Returns the outcome, or null if the
|
|
8226
9067
|
* race guard rejected it.
|
|
9068
|
+
*
|
|
9069
|
+
* A row can reach here through either door (isEligibleForNeedsReviewAutoResolve):
|
|
9070
|
+
* auto-fix genuinely exhausted, or parked directly by a GUARD_VERDICT_EVIDENCE_
|
|
9071
|
+
* ELIGIBLE verdict with no auto-fix history at all. `originIsGuardParked` picks
|
|
9072
|
+
* which door this particular row came through, purely to make the requeue/skip
|
|
9073
|
+
* reason text (and the Queue UI's job.error) name the RIGHT evidence — a
|
|
9074
|
+
* guard-parked row was never "exhausted auto-fix" and must never claim to be.
|
|
8227
9075
|
*/
|
|
8228
9076
|
function applyNeedsReviewAutoResolve(j) {
|
|
8229
|
-
if (!j || j.status !== 'needs_review' || !
|
|
9077
|
+
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
|
|
8230
9078
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
|
|
9079
|
+
const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
|
|
8231
9080
|
|
|
8232
9081
|
if (j.looksDone) {
|
|
8233
9082
|
const attempt = j.exhaustedResolveAttempts ?? 0;
|
|
8234
|
-
|
|
8235
|
-
|
|
8236
|
-
|
|
8237
|
-
|
|
9083
|
+
const reason = originIsGuardParked
|
|
9084
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with landed commit and looksDone evidence (${j.looksDone.rule}) — `
|
|
9085
|
+
+ `${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths, work landed`
|
|
9086
|
+
: `needs_review auto-resolve: verifier annotation shows work landed (${j.looksDone.rule} — ${j.looksDone.commits.length} commit(s) attributable to this job's own run touch the PRD's declared paths)`;
|
|
9087
|
+
transitionJob(j, 'completed', { reason, source: 'needsReviewAutoResolve' });
|
|
8238
9088
|
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'completed', attempt });
|
|
8239
9089
|
return 'completed';
|
|
8240
9090
|
}
|
|
@@ -8243,18 +9093,22 @@ function applyNeedsReviewAutoResolve(j) {
|
|
|
8243
9093
|
if (attemptsSoFar < NEEDS_REVIEW_RESOLVE_CAP) {
|
|
8244
9094
|
const attempt = attemptsSoFar + 1;
|
|
8245
9095
|
j.exhaustedResolveAttempts = attempt;
|
|
8246
|
-
|
|
8247
|
-
|
|
8248
|
-
|
|
8249
|
-
|
|
9096
|
+
const reason = originIsGuardParked
|
|
9097
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence yet — `
|
|
9098
|
+
+ `requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`
|
|
9099
|
+
: `needs_review auto-resolve: exhausted auto-fix, no completion evidence — requeued for one more run (attempt ${attempt}/${NEEDS_REVIEW_RESOLVE_CAP})`;
|
|
9100
|
+
transitionJob(j, 'pending', { reason, source: 'needsReviewAutoResolve' });
|
|
8250
9101
|
appendAuditEvent('needs_review_auto_resolved', { slug: j.slug, cwd: j.cwd ?? null, outcome: 'requeued', attempt });
|
|
8251
9102
|
return 'requeued';
|
|
8252
9103
|
}
|
|
8253
9104
|
|
|
8254
9105
|
j.needsReviewAutoResolvedSkip = true;
|
|
8255
|
-
j.error =
|
|
8256
|
-
|
|
8257
|
-
|
|
9106
|
+
j.error = originIsGuardParked
|
|
9107
|
+
? `needs_review auto-resolve: guard verdict '${j.verifierVerdict}' with no completion evidence after `
|
|
9108
|
+
+ `${NEEDS_REVIEW_RESOLVE_CAP} requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`
|
|
9109
|
+
: `needs_review auto-resolve: exhausted auto-fix path (autoFixOutcome=${j.autoFixOutcome ?? 'none'}, `
|
|
9110
|
+
+ `autoFixRetries=${j.autoFixRetries ?? 0}) with no completion evidence after ${NEEDS_REVIEW_RESOLVE_CAP} `
|
|
9111
|
+
+ `requeue attempt(s) — auto-skipped to unblock downstream dependsOn rows`;
|
|
8258
9112
|
transitionJob(j, 'skipped', {
|
|
8259
9113
|
reason: `needs_review auto-resolve: cap exhausted (${NEEDS_REVIEW_RESOLVE_CAP}/${NEEDS_REVIEW_RESOLVE_CAP} requeue attempts) — auto-skipped`,
|
|
8260
9114
|
source: 'needsReviewAutoResolve',
|
|
@@ -8320,6 +9174,11 @@ function selectAutoFixTargets(jobs, { fixSlugExists, resolveJobRunId = resolveRu
|
|
|
8320
9174
|
// fix-plan investigation to diagnose — there is no code defect to
|
|
8321
9175
|
// author a PRD against, only another job's still-uncommitted tree.
|
|
8322
9176
|
if (job.blockedByForeignWip === true) return false;
|
|
9177
|
+
// A budget-killed job parks for a human/ladder decision, never an
|
|
9178
|
+
// auto-fix investigation or auto-retry — the run didn't fail, it simply
|
|
9179
|
+
// overran its own estimate; there's no code defect to diagnose (Out of
|
|
9180
|
+
// scope: "retrying or auto-resuming a budget-killed job" for this PRD).
|
|
9181
|
+
if (job.verifierVerdict === 'budget_exceeded') return false;
|
|
8323
9182
|
// A stale re-run whose work already shipped (rcaReport's 'already-shipped'
|
|
8324
9183
|
// class) must never buy a fix-plan PRD — there is nothing to fix, and the
|
|
8325
9184
|
// correct recovery (archiving the PRD) is a human/reconcile action, not
|
|
@@ -8379,35 +9238,153 @@ function isEligibleForImmediateAutoFix(job, allJobs, fixSlugExists) {
|
|
|
8379
9238
|
}
|
|
8380
9239
|
|
|
8381
9240
|
/**
|
|
8382
|
-
*
|
|
8383
|
-
*
|
|
8384
|
-
*
|
|
8385
|
-
*
|
|
9241
|
+
* attributeLandedCommits(job, pathCommits, cwd) → { commits, rule } | null
|
|
9242
|
+
*
|
|
9243
|
+
* Narrows a set of PATH-overlapping commits (computeLooksDone's
|
|
9244
|
+
* `landedSinceRun` result — any commit touching the PRD's declared paths,
|
|
9245
|
+
* regardless of who authored it) down to the subset actually attributable to
|
|
9246
|
+
* THIS job's own run. Path overlap alone is not attribution: inside one Epic,
|
|
9247
|
+
* sibling PRDs routinely declare the same hot file, so a sibling's commit is
|
|
9248
|
+
* indistinguishable from this job's own by path alone (the incident this
|
|
9249
|
+
* function exists to close — PRD 1204's parked row cited PRD 1205's commit
|
|
9250
|
+
* 5dadf3c as its own evidence).
|
|
9251
|
+
*
|
|
9252
|
+
* Three rules, tried strongest-first, first match wins:
|
|
9253
|
+
*
|
|
9254
|
+
* 1. 'landedCommit' — the row's own `job.landedCommit`, re-verified here via
|
|
9255
|
+
* `resolveLandedCommitEvidence` against THIS job's `startedAt`. This is
|
|
9256
|
+
* the strongest signal because it is not inferred from `git log` at all:
|
|
9257
|
+
* it is the sha spawnJob's own finalize step observed THIS dispatch's
|
|
9258
|
+
* worktree/branch landing (see resolveLandedCommitEvidence's own header
|
|
9259
|
+
* for why it also guards against a stale sha surviving a reset). Trusted
|
|
9260
|
+
* independent of whether it appears in `pathCommits` — it is definitionally
|
|
9261
|
+
* this job's own work, not something discovered by scanning history.
|
|
9262
|
+
* 2. 'job branch' — a path-overlapping commit reachable from (an ancestor of
|
|
9263
|
+
* or equal to) this job's own `sm-job/<slug>` branch tip. Still
|
|
9264
|
+
* job-specific even though it IS a `git log` scan: a sibling's commit can
|
|
9265
|
+
* never be an ancestor of THIS job's own branch ref. In practice this
|
|
9266
|
+
* branch is deleted on successful integration (gitWorktree.cjs's
|
|
9267
|
+
* `cleanupWorktree`), so this mainly fires when integration failed and
|
|
9268
|
+
* the branch was deliberately kept for recovery, or reverify runs before
|
|
9269
|
+
* cleanup — a narrower window than rule 1, hence checked second.
|
|
9270
|
+
* 3. 'slug trailer' — a path-overlapping commit whose message contains this
|
|
9271
|
+
* job's slug verbatim. Weakest of the three (a coincidental substring
|
|
9272
|
+
* match is possible, and nothing stamps this automatically today), so it
|
|
9273
|
+
* is the last resort when the two structural signals above found
|
|
9274
|
+
* nothing.
|
|
9275
|
+
*
|
|
9276
|
+
* Deliberately NOT a rule: raw path overlap by itself (the bug this function
|
|
9277
|
+
* fixes) and `committedInWindow`-style time-window-only evidence — a sibling
|
|
9278
|
+
* job running concurrently in the very same window is exactly as invisible to
|
|
9279
|
+
* a time bound as it is to a path filter, so neither narrows attribution.
|
|
9280
|
+
*
|
|
9281
|
+
* Never throws: a missing ref, an unresolvable sha, or any git failure for a
|
|
9282
|
+
* given commit/rule is treated as "that commit doesn't satisfy this rule",
|
|
9283
|
+
* never as a fabricated match.
|
|
9284
|
+
*/
|
|
9285
|
+
async function attributeLandedCommits(job, pathCommits, cwd) {
|
|
9286
|
+
if (job?.landedCommit && await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)) {
|
|
9287
|
+
return { commits: [job.landedCommit], rule: 'landedCommit' };
|
|
9288
|
+
}
|
|
9289
|
+
|
|
9290
|
+
const branch = `sm-job/${job?.slug}`;
|
|
9291
|
+
const branchCommits = [];
|
|
9292
|
+
for (const sha of pathCommits) {
|
|
9293
|
+
try {
|
|
9294
|
+
await execGitAt(cwd, ['merge-base', '--is-ancestor', sha, branch], { timeout: 10_000 });
|
|
9295
|
+
branchCommits.push(sha);
|
|
9296
|
+
} catch { /* not an ancestor of this job's own branch, or branch doesn't exist */ }
|
|
9297
|
+
}
|
|
9298
|
+
if (branchCommits.length) return { commits: branchCommits, rule: 'job branch' };
|
|
9299
|
+
|
|
9300
|
+
if (job?.slug) {
|
|
9301
|
+
const trailerCommits = [];
|
|
9302
|
+
for (const sha of pathCommits) {
|
|
9303
|
+
try {
|
|
9304
|
+
const msg = await execGitAt(cwd, ['log', '-1', '--format=%B', sha], { timeout: 10_000 });
|
|
9305
|
+
if (msg.includes(job.slug)) trailerCommits.push(sha);
|
|
9306
|
+
} catch { /* unresolvable sha */ }
|
|
9307
|
+
}
|
|
9308
|
+
if (trailerCommits.length) return { commits: trailerCommits, rule: 'slug trailer' };
|
|
9309
|
+
}
|
|
9310
|
+
|
|
9311
|
+
return null;
|
|
9312
|
+
}
|
|
9313
|
+
|
|
9314
|
+
/**
|
|
9315
|
+
* Widened evidence check (PRD 1102, narrowed to per-job attribution by a
|
|
9316
|
+
* later PRD): does at least one commit ATTRIBUTABLE TO THIS JOB land AFTER
|
|
9317
|
+
* its run window and touch a path the PRD itself declares? Scoped to the
|
|
9318
|
+
* PRD's own declared paths (never the whole repo) so a sibling job's
|
|
9319
|
+
* unrelated commit is never even considered — see healRefusalReason's own
|
|
8386
9320
|
* rationale for why unscoped, repo-wide evidence is not attribution.
|
|
8387
9321
|
*
|
|
9322
|
+
* Path overlap alone is NOT evidence (see attributeLandedCommits's header):
|
|
9323
|
+
* a sibling PRD in the same Epic routinely declares the same hot file, so
|
|
9324
|
+
* `landedSinceRun`'s raw result is only a candidate list — the returned
|
|
9325
|
+
* annotation is null unless `attributeLandedCommits` narrows it to at least
|
|
9326
|
+
* one commit this job can actually claim.
|
|
9327
|
+
*
|
|
8388
9328
|
* Returns null (no annotation, never fabricated) when the PRD names no
|
|
8389
|
-
* paths
|
|
8390
|
-
*
|
|
9329
|
+
* paths, when no commit touches a declared path at all, or when
|
|
9330
|
+
* path-overlapping commits exist but none are attributable to this job — the
|
|
9331
|
+
* caller then has only the existing, already-computed committedInWindow
|
|
9332
|
+
* signal to go on, same as before this PRD.
|
|
9333
|
+
*
|
|
9334
|
+
* `fetchedCwds` (optional) lets a caller iterating many candidates in one
|
|
9335
|
+
* pass (reverifyNeedsReview) dedupe the `git fetch --all --prune` across
|
|
9336
|
+
* candidates that share a `cwd` — several `needs_review` rows for the same
|
|
9337
|
+
* project is the common case a backlog produces, and each fetch is up to
|
|
9338
|
+
* ~20s, so re-fetching the same repo once per row multiplies that pass's
|
|
9339
|
+
* wall-clock cost for zero new evidence. Omitted (or a fresh Set per call)
|
|
9340
|
+
* simply always fetches, unchanged from before this cache existed.
|
|
8391
9341
|
*
|
|
8392
|
-
* @returns {Promise<{commits: string[], paths: string[], detectedAt: string} | null>}
|
|
9342
|
+
* @returns {Promise<{commits: string[], paths: string[], detectedAt: string, rule: string} | null>}
|
|
8393
9343
|
*/
|
|
8394
|
-
async function computeLooksDone(job) {
|
|
9344
|
+
async function computeLooksDone(job, fetchedCwds) {
|
|
8395
9345
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
8396
9346
|
const paths = declaredPathsForPrd(prdPath);
|
|
8397
9347
|
if (!paths.length) return null;
|
|
8398
|
-
|
|
9348
|
+
if (!fetchedCwds || !fetchedCwds.has(job.cwd)) {
|
|
9349
|
+
await fetchAllRefs(job.cwd);
|
|
9350
|
+
if (fetchedCwds) fetchedCwds.add(job.cwd);
|
|
9351
|
+
}
|
|
8399
9352
|
const commits = await landedSinceRun(job.cwd, job.startedAt, paths);
|
|
8400
9353
|
if (!commits.length) return null;
|
|
8401
|
-
|
|
9354
|
+
const attributed = await attributeLandedCommits(job, commits, job.cwd);
|
|
9355
|
+
if (!attributed) return null;
|
|
9356
|
+
return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
|
|
8402
9357
|
}
|
|
8403
9358
|
|
|
8404
9359
|
async function reverifyNeedsReview() {
|
|
8405
9360
|
const snap = await readQueue();
|
|
8406
|
-
|
|
9361
|
+
// isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
|
|
9362
|
+
// verifierVerdict is a commit-guard/shared-tree-guard verdict, not a
|
|
9363
|
+
// RESCANNABLE_VERDICTS transcript-verifier one) — included here so this
|
|
9364
|
+
// pass also computes their looksDone evidence, the widened half of the
|
|
9365
|
+
// guard-verdict auto-resolve gap this PRD closes. Handled in its own
|
|
9366
|
+
// branch below (no transcript rescan — there is no transcript verdict to
|
|
9367
|
+
// rescan) rather than through the isRescanCandidate machinery.
|
|
9368
|
+
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
|
|
8407
9369
|
const healed = [];
|
|
8408
9370
|
const leftForReview = [];
|
|
8409
9371
|
const looksDoneUpdates = [];
|
|
9372
|
+
// Shared across every computeLooksDone call in this one pass — dedupes
|
|
9373
|
+
// the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
|
|
9374
|
+
// header) rather than re-fetching the same repo once per candidate row.
|
|
9375
|
+
const fetchedCwds = new Set();
|
|
8410
9376
|
for (const job of candidates) {
|
|
9377
|
+
if (!isRescanCandidate(job) && isGuardParkedWithoutAutoFix(job)) {
|
|
9378
|
+
// Guard-verdict park, never auto-fixed: only evidence gathering, never
|
|
9379
|
+
// a transcript rescan (there was never a transcript-verifier verdict
|
|
9380
|
+
// here) and never a direct heal — applyNeedsReviewAutoResolve is the
|
|
9381
|
+
// sole place that turns this annotation into a status change.
|
|
9382
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
9383
|
+
if (looksDone) {
|
|
9384
|
+
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
9385
|
+
}
|
|
9386
|
+
continue;
|
|
9387
|
+
}
|
|
8411
9388
|
if (job.status === 'failed') {
|
|
8412
9389
|
// A failed row never runs the transcript-verifier rescan below — that
|
|
8413
9390
|
// machinery (verifyRun/COMPLETED_EQUIVALENT_VERDICTS) exists to
|
|
@@ -8416,7 +9393,7 @@ async function reverifyNeedsReview() {
|
|
|
8416
9393
|
// completing-direction constraint). The only thing a failed candidate
|
|
8417
9394
|
// can gain here is a looksDone annotation + a failed → needs_review
|
|
8418
9395
|
// transition, for a human to confirm.
|
|
8419
|
-
const looksDone = await computeLooksDone(job);
|
|
9396
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
8420
9397
|
if (looksDone) {
|
|
8421
9398
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: true });
|
|
8422
9399
|
} else {
|
|
@@ -8471,7 +9448,7 @@ async function reverifyNeedsReview() {
|
|
|
8471
9448
|
// always before this periodic/boot pass can run against the same row, so
|
|
8472
9449
|
// this check reliably catches the only order that can occur.
|
|
8473
9450
|
if (stillOpen && job.autoFixAttempted !== true) {
|
|
8474
|
-
const looksDone = await computeLooksDone(job);
|
|
9451
|
+
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
8475
9452
|
if (looksDone) {
|
|
8476
9453
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
8477
9454
|
}
|
|
@@ -8485,14 +9462,14 @@ async function reverifyNeedsReview() {
|
|
|
8485
9462
|
if (!u) continue;
|
|
8486
9463
|
if (u.fromFailed) {
|
|
8487
9464
|
transitionJob(j, 'needs_review', {
|
|
8488
|
-
reason:
|
|
9465
|
+
reason: `looks done (${u.looksDone.rule}) — commit(s) attributable to this job's own run touch this PRD's declared paths; confirm before archiving`,
|
|
8489
9466
|
source: 'reverifyNeedsReview:looksDone',
|
|
8490
9467
|
});
|
|
8491
9468
|
}
|
|
8492
9469
|
if (j.status !== 'needs_review') continue;
|
|
8493
9470
|
j.looksDone = u.looksDone;
|
|
8494
9471
|
const shaList = u.looksDone.commits.slice(0, 5).map((c) => c.slice(0, 7)).join(', ');
|
|
8495
|
-
j.error = `looks done — ${u.looksDone.commits.length} commit(s)
|
|
9472
|
+
j.error = `looks done (${u.looksDone.rule}) — ${u.looksDone.commits.length} commit(s) attributable to this job's own run touch this PRD's paths (${shaList}); confirm before archiving`;
|
|
8496
9473
|
}
|
|
8497
9474
|
});
|
|
8498
9475
|
console.log(`[scheduler] boot reverify: looksDone annotated for ${looksDoneUpdates.length} row(s): ${looksDoneUpdates.map((u) => u.slug).join(', ')}`);
|
|
@@ -8764,11 +9741,21 @@ function registerScheduleHandlers() {
|
|
|
8764
9741
|
ensureDirs();
|
|
8765
9742
|
supervisor.registerHandlers();
|
|
8766
9743
|
|
|
9744
|
+
// Cheap read only — no reconcile(), no writeQueue(). The renderer treats
|
|
9745
|
+
// this as a fast call behind a 5s deadline (scheduleState.ts's
|
|
9746
|
+
// withTimeout), but reconcile() does a cross-project PRD discovery walk
|
|
9747
|
+
// plus a disk write, which could blow that deadline and, worse, throw
|
|
9748
|
+
// outright on a torn queue.json (reconcile refuses to run against
|
|
9749
|
+
// `state.unreadable`) — turning a recoverable read into a rejected IPC and
|
|
9750
|
+
// an error toast. Discovery still runs on a fixed cadence elsewhere:
|
|
9751
|
+
// tickQueue (every POLL_INTERVAL_MS, 60s), rescheduleTimer, broadcast()'s
|
|
9752
|
+
// coalescer (getPayload), schedule:rescan and schedule:adopt-prd. Worst
|
|
9753
|
+
// case, a PRD dropped on disk while the Scheduler tab is open surfaces
|
|
9754
|
+
// here within ~POLL_INTERVAL_MS + BROADCAST_COALESCE_MS (~60.2s) — via
|
|
9755
|
+
// tickQueue's reconcile + its trailing broadcast() — not via this handler.
|
|
8767
9756
|
ipcMain.handle('schedule:state', async () => {
|
|
8768
9757
|
const state = await readQueue();
|
|
8769
|
-
|
|
8770
|
-
await writeQueue(state);
|
|
8771
|
-
return buildScheduleStatePayload(state, { withPaths: true });
|
|
9758
|
+
return buildScheduleStatePayload(state);
|
|
8772
9759
|
});
|
|
8773
9760
|
|
|
8774
9761
|
// Session-Manager-wide claude -p slot pool (lib/sessionSlots.cjs) —
|
|
@@ -8811,6 +9798,57 @@ function registerScheduleHandlers() {
|
|
|
8811
9798
|
};
|
|
8812
9799
|
});
|
|
8813
9800
|
|
|
9801
|
+
// Queue-health header (PRD): the one honest read of "why does the queue
|
|
9802
|
+
// look stale" — reuses classifyQueueHealth so the UI and the starvation
|
|
9803
|
+
// watchdog can never disagree. `cwd` is optional (null = machine-wide,
|
|
9804
|
+
// matching WindowStrip's own scopeCwd fallback).
|
|
9805
|
+
ipcMain.handle('schedule:queue-health', async (_e, payload) => {
|
|
9806
|
+
const cwd = (payload && typeof payload.cwd === 'string') ? payload.cwd : null;
|
|
9807
|
+
const state = await readQueue();
|
|
9808
|
+
if (state.unreadable) {
|
|
9809
|
+
return { unknown: true, reason: state.unreadable };
|
|
9810
|
+
}
|
|
9811
|
+
const now = Date.now();
|
|
9812
|
+
const slotSnapshot = sessionSlots.snapshot();
|
|
9813
|
+
const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
|
|
9814
|
+
const verdict = classifyQueueHealth({
|
|
9815
|
+
jobs: state.jobs,
|
|
9816
|
+
paused: state.paused,
|
|
9817
|
+
launchBlocks: state.launchBlocks,
|
|
9818
|
+
runningSet,
|
|
9819
|
+
freeSlots,
|
|
9820
|
+
totalSlots: slotSnapshot.total,
|
|
9821
|
+
lastDispatchAttemptAtMs: Date.parse(state.lastDispatchAttemptAt ?? ''),
|
|
9822
|
+
now,
|
|
9823
|
+
cwd,
|
|
9824
|
+
});
|
|
9825
|
+
// Oldest running job across the whole machine (any project) — the
|
|
9826
|
+
// number that actually explains slot saturation, alongside the
|
|
9827
|
+
// machine-wide slot pool itself.
|
|
9828
|
+
let oldestRunningAgeMs = null;
|
|
9829
|
+
for (const j of state.jobs) {
|
|
9830
|
+
if (j.status !== 'running' && !runningSet.has(j.slug)) continue;
|
|
9831
|
+
const startedAtMs = j.startedAt ? Date.parse(j.startedAt) : NaN;
|
|
9832
|
+
if (!Number.isFinite(startedAtMs)) continue;
|
|
9833
|
+
const age = now - startedAtMs;
|
|
9834
|
+
if (oldestRunningAgeMs === null || age > oldestRunningAgeMs) oldestRunningAgeMs = age;
|
|
9835
|
+
}
|
|
9836
|
+
return {
|
|
9837
|
+
unknown: false,
|
|
9838
|
+
now,
|
|
9839
|
+
verdict,
|
|
9840
|
+
slots: {
|
|
9841
|
+
inUse: slotSnapshot.inUse,
|
|
9842
|
+
total: slotSnapshot.total,
|
|
9843
|
+
free: freeSlots,
|
|
9844
|
+
source: slotSnapshot.envOverride ? 'env' : 'pool',
|
|
9845
|
+
},
|
|
9846
|
+
oldestRunningAgeMs,
|
|
9847
|
+
lastRunAt: state.lastRunAt ?? null,
|
|
9848
|
+
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
9849
|
+
};
|
|
9850
|
+
});
|
|
9851
|
+
|
|
8814
9852
|
ipcMain.handle('schedule:force-tick', async () => {
|
|
8815
9853
|
// Bypass the billing-poll gate entirely — fire pending jobs immediately regardless of meter state.
|
|
8816
9854
|
// Clears any existing pause first (same semantics as run-now).
|
|
@@ -8889,6 +9927,21 @@ function registerScheduleHandlers() {
|
|
|
8889
9927
|
return { ok: true, kind: 'info', message: `Adopted ${slug} — it will run as a normal pending job` };
|
|
8890
9928
|
}));
|
|
8891
9929
|
|
|
9930
|
+
// Scheduler UI's "change disposition" action (scheduler wave-disposition
|
|
9931
|
+
// PRD): promotes an appended wave to its own head, or re-attaches a head
|
|
9932
|
+
// behind another chain. Thin wrapper over remote.setPrdDisposition, which
|
|
9933
|
+
// validates the rewrite (cycle-safety, running/completed rows untouched)
|
|
9934
|
+
// before delegating to the same remote.updatePrd every other PRD edit
|
|
9935
|
+
// path uses — see that method's own comment in this file.
|
|
9936
|
+
ipcMain.handle('schedule:set-prd-disposition', validated(schemas.scheduleSetPrdDisposition, async ({ slug, cwd, disposition, dependsOn }) => {
|
|
9937
|
+
if (!(await safeSlugPath(slug))) return { ok: false, kind: 'error', message: 'invalid slug' };
|
|
9938
|
+
const result = await remote.setPrdDisposition({ slug, cwd, disposition, dependsOn });
|
|
9939
|
+
if (!result.ok) return { ok: false, kind: 'error', message: result.error ?? 'disposition change failed' };
|
|
9940
|
+
appendAuditEvent('scheduler_prd_disposition_set', { slug, cwd: cwd ?? null, disposition, source: 'ipc:schedule:set-prd-disposition' });
|
|
9941
|
+
await broadcast({ flush: true });
|
|
9942
|
+
return { ok: true, kind: 'info', message: `${slug} is now ${disposition === 'new-head' ? 'an independent head' : 'attached behind the chosen chain'}` };
|
|
9943
|
+
}));
|
|
9944
|
+
|
|
8892
9945
|
ipcMain.handle('schedule:run-now', async () => {
|
|
8893
9946
|
// Manual run-now overrides any auto-pause. Clear it first.
|
|
8894
9947
|
await clearPause('run-now');
|
|
@@ -8901,9 +9954,11 @@ function registerScheduleHandlers() {
|
|
|
8901
9954
|
return { ok: true };
|
|
8902
9955
|
});
|
|
8903
9956
|
|
|
8904
|
-
// Re-scan prds/ folder and merge into queue.json.
|
|
8905
|
-
//
|
|
8906
|
-
// explicit
|
|
9957
|
+
// Re-scan prds/ folder and merge into queue.json. `schedule:state` is a
|
|
9958
|
+
// cheap read with no reconcile of its own — this is the renderer's
|
|
9959
|
+
// explicit, immediate discovery path (mutate() + reconcile() + broadcast())
|
|
9960
|
+
// for "I just dropped a PRD on disk and want it to show up now" rather than
|
|
9961
|
+
// waiting for tickQueue's next ~60s pass.
|
|
8907
9962
|
ipcMain.handle('schedule:rescan', async () => {
|
|
8908
9963
|
const { added, removed } = await mutate(async (state) => {
|
|
8909
9964
|
const before = new Set(state.jobs.map((j) => j.slug));
|
|
@@ -9132,6 +10187,18 @@ async function init() {
|
|
|
9132
10187
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
9133
10188
|
bootOutcomes.set(j.slug, logPath ? classifyRunOutcome(logPath) : 'unknown');
|
|
9134
10189
|
}
|
|
10190
|
+
// Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
|
|
10191
|
+
// BEFORE mutate() for the same reason (git spawn work must never run
|
|
10192
|
+
// inside mutate()'s single global serialization chain) — an orphaned job
|
|
10193
|
+
// classified 'failed'/'unknown' from its log tail alone can still have
|
|
10194
|
+
// actually landed a real commit before the app restarted mid-run.
|
|
10195
|
+
const bootLandedCommitEvidence = new Map();
|
|
10196
|
+
await Promise.all(bootSnap.jobs.map(async (j) => {
|
|
10197
|
+
if (!immediateSlugs.includes(j.slug) || j.status !== 'running') return;
|
|
10198
|
+
if (bootOutcomes.get(j.slug) === 'success' || !j.landedCommit) return;
|
|
10199
|
+
const resolved = await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt);
|
|
10200
|
+
if (resolved) bootLandedCommitEvidence.set(j.slug, j.landedCommit);
|
|
10201
|
+
}));
|
|
9135
10202
|
const bootReconciledCompletions = [];
|
|
9136
10203
|
await mutate((state) => {
|
|
9137
10204
|
for (const j of state.jobs) {
|
|
@@ -9139,7 +10206,7 @@ async function init() {
|
|
|
9139
10206
|
const outcome = bootOutcomes.get(j.slug) ?? 'unknown';
|
|
9140
10207
|
const pid = j.runtime?.pid;
|
|
9141
10208
|
const killNote = pid ? ` (orphan pid=${pid}: dead)` : '';
|
|
9142
|
-
applyOrphanOutcome(j, outcome, killNote);
|
|
10209
|
+
applyOrphanOutcome(j, outcome, killNote, bootLandedCommitEvidence.get(j.slug) || null);
|
|
9143
10210
|
if (j.status === 'completed') bootReconciledCompletions.push({ slug: j.slug, cwd: j.cwd });
|
|
9144
10211
|
console.log(`[scheduler] boot reconcile: slug=${j.slug} outcome=${outcome} → status=${j.status}`);
|
|
9145
10212
|
}
|
|
@@ -9163,9 +10230,18 @@ async function init() {
|
|
|
9163
10230
|
if (result === 'killed') {
|
|
9164
10231
|
console.log(`[scheduler] boot: SIGTERM'd orphan claude pid=${pid} for ${slug} — deferring finalize ${BOOT_ORPHAN_KILL_GRACE_MS}ms`);
|
|
9165
10232
|
}
|
|
9166
|
-
setTimeout(() => {
|
|
10233
|
+
setTimeout(async () => {
|
|
9167
10234
|
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
9168
10235
|
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
10236
|
+
// Same evidence-before-failure gate as the immediate-orphan path
|
|
10237
|
+
// above, resolved before mutate() for the same reason (git spawn
|
|
10238
|
+
// work must never run inside mutate()'s serialization chain). Uses
|
|
10239
|
+
// the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
|
|
10240
|
+
// race guard below already confirms `cur` is still this same run
|
|
10241
|
+
// (runId === bootRunId) before this evidence is applied.
|
|
10242
|
+
const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
|
|
10243
|
+
? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
|
|
10244
|
+
: null;
|
|
9169
10245
|
let deferredCompletedCwd;
|
|
9170
10246
|
mutate((state) => {
|
|
9171
10247
|
const cur = state.jobs.find((x) => x.slug === slug);
|
|
@@ -9174,7 +10250,7 @@ async function init() {
|
|
|
9174
10250
|
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
9175
10251
|
// touched by this stale classification.
|
|
9176
10252
|
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
9177
|
-
applyOrphanOutcome(cur, outcome, killNote);
|
|
10253
|
+
applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
|
|
9178
10254
|
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
9179
10255
|
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
9180
10256
|
}).then(() => {
|
|
@@ -9610,7 +10686,9 @@ async function listPrdsInternal() {
|
|
|
9610
10686
|
estimateMinutes: parsed.estimateMinutes,
|
|
9611
10687
|
sourcePromptId: parsed.sourcePromptId,
|
|
9612
10688
|
epicId: parsed.epicId ?? null,
|
|
10689
|
+
dependsOn: parsed.dependsOn ?? null,
|
|
9613
10690
|
agentType: parsed.agentType ?? null,
|
|
10691
|
+
disposition: parsed.disposition ?? null,
|
|
9614
10692
|
mtimeMs: stat.mtimeMs,
|
|
9615
10693
|
archived,
|
|
9616
10694
|
};
|
|
@@ -10007,6 +11085,33 @@ const remote = {
|
|
|
10007
11085
|
}
|
|
10008
11086
|
},
|
|
10009
11087
|
|
|
11088
|
+
// Backs the Scheduler UI's "change disposition" action (scheduler
|
|
11089
|
+
// wave-disposition PRD): promoting an appended wave to its own head, or
|
|
11090
|
+
// re-attaching a head behind another chain. `dependsOn` for a 'new-head'
|
|
11091
|
+
// disposition is ignored (cleared unconditionally); for 'append' it's the
|
|
11092
|
+
// caller's chosen target chain's terminal slug(s) — the renderer computes
|
|
11093
|
+
// that from the SAME backlog tree (lib/backlogTree.ts) it already renders,
|
|
11094
|
+
// so this function only has to validate the rewrite is safe, never
|
|
11095
|
+
// re-derive "the" terminal itself.
|
|
11096
|
+
//
|
|
11097
|
+
// Validates via prdDisposition.cjs's computeDispositionRewrite (row not
|
|
11098
|
+
// running/completed, no already-satisfied blocker being rewritten out from
|
|
11099
|
+
// under it, no dependsOn cycle) BEFORE delegating the actual write to this
|
|
11100
|
+
// SAME updatePrd — so a rejected rewrite never reaches the filesystem, and
|
|
11101
|
+
// an accepted one gets updatePrd's own dependsOn FK re-validation for free.
|
|
11102
|
+
async setPrdDisposition({ slug, cwd, disposition, dependsOn }) {
|
|
11103
|
+
let listing;
|
|
11104
|
+
try {
|
|
11105
|
+
listing = await this.listPrds({ cwd, fields: 'full', limit: Number.MAX_SAFE_INTEGER });
|
|
11106
|
+
} catch (e) {
|
|
11107
|
+
return { ok: false, error: `could not read project PRDs: ${e?.message ?? e}` };
|
|
11108
|
+
}
|
|
11109
|
+
const rows = listing.prds ?? [];
|
|
11110
|
+
const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
|
|
11111
|
+
if (!rewrite.ok) return rewrite;
|
|
11112
|
+
return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition } });
|
|
11113
|
+
},
|
|
11114
|
+
|
|
10010
11115
|
// Cancels a job that hasn't finished yet. A 'running' job's process group
|
|
10011
11116
|
// is SIGTERM'd (reusing killOrphanClaudePid — the same kill path boot
|
|
10012
11117
|
// reconciliation uses for an orphaned running job) before its queue row is
|
|
@@ -10034,18 +11139,38 @@ const remote = {
|
|
|
10034
11139
|
if (wasRunning && pid) {
|
|
10035
11140
|
killOrphanClaudePid(pid);
|
|
10036
11141
|
}
|
|
11142
|
+
// Evidence-before-failure guard, scoped to an actually-running job being
|
|
11143
|
+
// killed here (a 'pending' cancel has no live process, so nothing new
|
|
11144
|
+
// could have landed since its last stamp — and 'needs_review' is not
|
|
11145
|
+
// even a legal transition from 'pending', see LEGAL_TRANSITIONS): the
|
|
11146
|
+
// same reapDeadRunningJobs evidence gate (job 1192 — a landedCommit
|
|
11147
|
+
// being non-empty is not proof by itself, but discarding proof of real
|
|
11148
|
+
// landed work with no check at all is worse) applies here too. A
|
|
11149
|
+
// dead-pid reap of a job that landed a commit (e.g. via the
|
|
11150
|
+
// dispatch-time sidecar backfill) is routed to needs_review/completed;
|
|
11151
|
+
// a deliberate cancel of that same state deserves no less.
|
|
11152
|
+
const confirmedLandedCommit = (wasRunning && job.landedCommit)
|
|
11153
|
+
? ((await resolveLandedCommitEvidence(job.cwd || DEFAULT_PROJECT_CWD, job.landedCommit, job.startedAt))
|
|
11154
|
+
? job.landedCommit
|
|
11155
|
+
: null)
|
|
11156
|
+
: null;
|
|
11157
|
+
const targetStatus = confirmedLandedCommit ? 'needs_review' : 'failed';
|
|
11158
|
+
const cancelReason = confirmedLandedCommit
|
|
11159
|
+
? `cancelled via admin API, but landedCommit ${confirmedLandedCommit} resolves — verify before treating as done`
|
|
11160
|
+
: 'cancelled via admin API';
|
|
10037
11161
|
await mutate((s) => {
|
|
10038
11162
|
const idx = s.jobs.findIndex((j) => j.slug === slug);
|
|
10039
11163
|
if (idx < 0) return;
|
|
10040
11164
|
const j = s.jobs[idx];
|
|
10041
|
-
transitionJob(j,
|
|
10042
|
-
j.error =
|
|
11165
|
+
transitionJob(j, targetStatus, { reason: cancelReason, source: 'remote:cancelJob' });
|
|
11166
|
+
j.error = cancelReason;
|
|
10043
11167
|
j.finishedAt = new Date().toISOString();
|
|
10044
11168
|
j.exitCode = j.exitCode ?? null;
|
|
11169
|
+
if (confirmedLandedCommit) j.verifierVerdict = 'cancelled_with_landed_commit';
|
|
10045
11170
|
delete j.runtime;
|
|
10046
11171
|
});
|
|
10047
11172
|
await broadcast({ flush: true });
|
|
10048
|
-
return { ok: true, slug, status:
|
|
11173
|
+
return { ok: true, slug, status: targetStatus, wasRunning, cwd: job.cwd ?? null };
|
|
10049
11174
|
},
|
|
10050
11175
|
|
|
10051
11176
|
// Exposes the module-level allocateParallelGroup (PRD 548) to callers that
|
|
@@ -10092,6 +11217,7 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
10092
11217
|
module.exports = {
|
|
10093
11218
|
classifyQueueStarvation,
|
|
10094
11219
|
classifyQueueStarvationByProject,
|
|
11220
|
+
classifyQueueHealth,
|
|
10095
11221
|
runQueueStarvationWatchdog,
|
|
10096
11222
|
QUEUE_STARVATION_MS,
|
|
10097
11223
|
selectStarveEscalations,
|
|
@@ -10102,6 +11228,15 @@ module.exports = {
|
|
|
10102
11228
|
findOverrunningJobs,
|
|
10103
11229
|
JOB_OVERRUN_FACTOR,
|
|
10104
11230
|
JOB_OVERRUN_FLOOR_MS,
|
|
11231
|
+
computeJobBudgetMs,
|
|
11232
|
+
classifyBudgetKill,
|
|
11233
|
+
isJobBudgetExempt,
|
|
11234
|
+
shouldKillForBudget,
|
|
11235
|
+
resolveBudgetKillOutcome,
|
|
11236
|
+
JOB_BUDGET_FACTOR,
|
|
11237
|
+
JOB_BUDGET_FLOOR_MS,
|
|
11238
|
+
JOB_BUDGET_CEILING_MS,
|
|
11239
|
+
BUDGET_WARNING_FRACTION,
|
|
10105
11240
|
registerScheduleHandlers,
|
|
10106
11241
|
attachWindow,
|
|
10107
11242
|
init,
|
|
@@ -10115,10 +11250,12 @@ module.exports = {
|
|
|
10115
11250
|
healRefusalReason,
|
|
10116
11251
|
writeQueue,
|
|
10117
11252
|
reconcile,
|
|
11253
|
+
broadcast,
|
|
10118
11254
|
reconcileSourcePromptId,
|
|
10119
11255
|
allocateParallelGroup,
|
|
10120
11256
|
selectHistoryJobs,
|
|
10121
11257
|
parsePorcelain,
|
|
11258
|
+
parsePorcelainEntries,
|
|
10122
11259
|
FINISH_PROTOCOL,
|
|
10123
11260
|
IDLE_OUTPUT_KILL_MS,
|
|
10124
11261
|
BASH_DEFAULT_TIMEOUT_MS,
|
|
@@ -10148,6 +11285,7 @@ module.exports = {
|
|
|
10148
11285
|
isRescanCandidate,
|
|
10149
11286
|
isFailedUnverifiedShaped,
|
|
10150
11287
|
computeLooksDone,
|
|
11288
|
+
attributeLandedCommits,
|
|
10151
11289
|
isPromotableOriginal,
|
|
10152
11290
|
selectAutoFixTargets,
|
|
10153
11291
|
applyRcaClassification,
|
|
@@ -10155,6 +11293,9 @@ module.exports = {
|
|
|
10155
11293
|
resolveRunId,
|
|
10156
11294
|
isUnresolvableNeedsReview,
|
|
10157
11295
|
isExhaustedAutoFix,
|
|
11296
|
+
GUARD_VERDICT_EVIDENCE_ELIGIBLE,
|
|
11297
|
+
isGuardParkedWithoutAutoFix,
|
|
11298
|
+
isEligibleForNeedsReviewAutoResolve,
|
|
10158
11299
|
isPlanUnqueued,
|
|
10159
11300
|
isFixPlanDead,
|
|
10160
11301
|
fixSlugFor,
|
|
@@ -10240,6 +11381,7 @@ module.exports = {
|
|
|
10240
11381
|
evaluateSharedTreeGuard,
|
|
10241
11382
|
checkSharedTreeGuard,
|
|
10242
11383
|
uncommittedChanges,
|
|
11384
|
+
uncommittedChangesWithStatus,
|
|
10243
11385
|
gitHead,
|
|
10244
11386
|
isBranchAlreadyIntegrated,
|
|
10245
11387
|
selectResumeRecoveryTarget,
|