claude-code-session-manager 0.92.1 → 0.94.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/DataModel-B0LDnnSL.js +1 -0
- package/dist/assets/{History-BOR_fJNr.js → History-C685Kytt.js} +2 -2
- package/dist/assets/{Hooks-BtD436iL.js → Hooks-CwLnp0Z_.js} +3 -3
- package/dist/assets/HostBilko-CwEHEKYk.js +1 -0
- package/dist/assets/{Library-DeKdY-tD.js → Library-BU9np05E.js} +25 -25
- package/dist/assets/{MarkdownEditor-CbzC-Oqf.js → MarkdownEditor-B74ogK2f.js} +1 -1
- package/dist/assets/McpServers-BLM3ftzP.js +2 -0
- package/dist/assets/{Memory-B68AJjg2.js → Memory-DAP1t8bA.js} +6 -6
- package/dist/assets/{Permissions-Cr0RRvNy.js → Permissions-3pGLJmxQ.js} +3 -3
- package/dist/assets/Plugins-Q1KoLEzn.js +2 -0
- package/dist/assets/ProvenanceBadge-BnjBQiV9.js +1 -0
- package/dist/assets/{SaveBar-87ZJfscX.js → SaveBar-Zq2NfcPf.js} +1 -1
- package/dist/assets/Scheduler-D488ebKm.js +16 -0
- package/dist/assets/{ScopeSwitcher-CKzpqjJ_.js → ScopeSwitcher-Cnojv22D.js} +1 -1
- package/dist/assets/Settings-D5Lhj7ga.js +3 -0
- package/dist/assets/{SkillReferenceGraph-D_-wg4r9.js → SkillReferenceGraph-BRDiE6Hv.js} +2 -2
- package/dist/assets/Skills-BQDqp-EN.js +3 -0
- package/dist/assets/SystemPrompt-tFr19Od6.js +1 -0
- package/dist/assets/TagLibrary-BjyrnaRE.js +1 -0
- package/dist/assets/{TiptapBody-DYXIp8aG.js → TiptapBody-DGzSg3BO.js} +1 -1
- package/dist/assets/{Toggle-CDXR5l3m.js → Toggle-DzoROibf.js} +1 -1
- package/dist/assets/index-CCl4tz-u.js +3079 -0
- package/dist/assets/{index-D3P_jldk.css → index-CKH5Uxik.css} +1 -1
- package/dist/assets/settingsSchema-BYKVe1WM.js +1 -0
- package/dist/index.html +2 -2
- package/package.json +6 -3
- package/scripts/README.md +3 -0
- package/src/main/__tests__/agentEffortResolve.test.cjs +117 -0
- package/src/main/__tests__/agentLibrary.test.cjs +21 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +16 -0
- package/src/main/__tests__/agentOverlayWrite.test.cjs +95 -0
- package/src/main/__tests__/chat-cancel-terminal.test.cjs +5 -2
- package/src/main/__tests__/chat-exit-close-race.test.cjs +8 -2
- package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +5 -2
- package/src/main/__tests__/chatRunner-session-flag-retry.test.cjs +44 -0
- package/src/main/__tests__/intradayRefresh.test.cjs +39 -0
- package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +25 -0
- package/src/main/__tests__/opsErrorLog.test.cjs +22 -0
- package/src/main/__tests__/personaMerge.test.cjs +169 -0
- package/src/main/__tests__/prdCreatePlanId.test.cjs +132 -0
- package/src/main/__tests__/promptSessionsCreateEpicHandler.test.cjs +13 -0
- package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +21 -0
- package/src/main/__tests__/rateLimitPollerStreak.test.cjs +14 -0
- package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +26 -0
- package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +11 -1
- package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +11 -1
- package/src/main/__tests__/runVerify.test.cjs +12 -1
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +47 -2
- package/src/main/__tests__/transcriptsUsageFor.test.cjs +112 -1
- package/src/main/agentLibrary.cjs +86 -40
- package/src/main/build-info.json +4 -4
- package/src/main/chatRunner.cjs +50 -10
- package/src/main/git.cjs +9 -2
- package/src/main/historyAggregator.cjs +2 -19
- package/src/main/index.cjs +17 -8
- package/src/main/ipcSchemas.cjs +25 -3
- package/src/main/lib/__tests__/childWithLog.test.cjs +180 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +19 -0
- package/src/main/lib/__tests__/effectiveModelInfo.test.cjs +10 -5
- package/src/main/lib/__tests__/gitCacheBound.test.cjs +69 -0
- package/src/main/lib/__tests__/modelCatalog.test.cjs +202 -0
- package/src/main/lib/agentEffortResolve.cjs +74 -0
- package/src/main/lib/agentModelResolve.cjs +52 -59
- package/src/main/lib/agentPersonaSchema.cjs +5 -0
- package/src/main/lib/childWithLog.cjs +92 -51
- package/src/main/lib/delegationReadiness.cjs +3 -1
- package/src/main/lib/effectiveModelInfo.cjs +38 -20
- package/src/main/lib/epicMint.cjs +5 -2
- package/src/main/lib/epicSpawnPlan.cjs +27 -8
- package/src/main/lib/epicTranscriptPath.cjs +5 -1
- package/src/main/lib/epicWorktreeMerge.cjs +7 -6
- package/src/main/lib/headTailBuffer.cjs +43 -0
- package/src/main/lib/intradayRefresh.cjs +33 -0
- package/src/main/lib/lruCache.cjs +39 -0
- package/src/main/lib/modelCatalog.cjs +243 -0
- package/src/main/lib/openExternalApp.cjs +27 -9
- package/src/main/lib/opsErrorLog.cjs +22 -0
- package/src/main/lib/personaMerge.cjs +166 -0
- package/src/main/lib/prdCreate.cjs +23 -5
- package/src/main/lib/prdDisposition.cjs +46 -0
- package/src/main/lib/prdFrontmatter.cjs +6 -2
- package/src/main/lib/promptSessionSchema.cjs +5 -0
- package/src/main/lib/promptSessionsCreateEpic.cjs +5 -3
- package/src/main/lib/rendererRecovery.cjs +141 -0
- package/src/main/lib/scheduleJobSchema.cjs +3 -0
- package/src/main/runVerify.cjs +3 -1
- package/src/main/scheduler/prdParser.cjs +2 -0
- package/src/main/scheduler.cjs +381 -262
- package/src/main/templates/PRD_AUTHORING.md +4 -0
- package/src/main/transcripts.cjs +54 -11
- package/src/preload/api.d.ts +56 -3
- package/src/preload/index.cjs +4 -0
- package/dist/assets/AgentLibrary-Ci6u03bq.js +0 -3
- package/dist/assets/DataModel-BIqcjsv2.js +0 -1
- package/dist/assets/HostBilko-CMe5cH3H.js +0 -1
- package/dist/assets/ListDetail-CtVjrHKB.js +0 -1
- package/dist/assets/McpServers-DKN75Shn.js +0 -2
- package/dist/assets/Panel-C2YTW-qP.js +0 -1
- package/dist/assets/Plugins-DbZkJs5l.js +0 -2
- package/dist/assets/ProvenanceBadge-kJ__BOQv.js +0 -1
- package/dist/assets/Scheduler-DdEaxku8.js +0 -16
- package/dist/assets/Settings-CjywVsNB.js +0 -3
- package/dist/assets/Skills-Cd_yPO_l.js +0 -3
- package/dist/assets/SystemPrompt-2z6eGPSF.js +0 -1
- package/dist/assets/TagLibrary-DcaWQSp4.js +0 -1
- package/dist/assets/index-D5H_H5wC.js +0 -3074
- package/dist/assets/settingsSchema-JRDfjgHb.js +0 -3
package/src/main/scheduler.cjs
CHANGED
|
@@ -145,6 +145,7 @@ const queueOps = require('./queueOps.cjs');
|
|
|
145
145
|
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
|
|
146
146
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
147
147
|
const agentModelResolve = require('./lib/agentModelResolve.cjs');
|
|
148
|
+
const { resolveEpicEffort, effortArgs } = require('./lib/agentEffortResolve.cjs');
|
|
148
149
|
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
149
150
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
150
151
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
@@ -181,7 +182,7 @@ const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
|
|
|
181
182
|
const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
|
|
182
183
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
183
184
|
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
184
|
-
const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
|
|
185
|
+
const { computeDispositionRewrite, mintPlanId, resolveInheritedPlanId } = require('./lib/prdDisposition.cjs');
|
|
185
186
|
const { migratePrds, consolidateFlatPrds, legacyAdoptExistingPrds } = require('./lib/prdMigration.cjs');
|
|
186
187
|
const { allProjectCwds } = require('./lib/activeSessions.cjs');
|
|
187
188
|
|
|
@@ -1506,6 +1507,7 @@ function loadSchedulerState() {
|
|
|
1506
1507
|
if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
|
|
1507
1508
|
if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
|
|
1508
1509
|
if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
|
|
1510
|
+
failureStreakWarnedAt = restoreFailureStreakWarnedAt(failureStreakWarned, failureStreakWarnedAt, Date.now());
|
|
1509
1511
|
if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
|
|
1510
1512
|
} catch { /* first boot or corrupt — start fresh */ }
|
|
1511
1513
|
}
|
|
@@ -2811,6 +2813,7 @@ async function reconcile(state) {
|
|
|
2811
2813
|
epicId: p.epicId ?? job.epicId ?? null,
|
|
2812
2814
|
dependsOn: p.dependsOn,
|
|
2813
2815
|
disposition: p.disposition ?? null,
|
|
2816
|
+
planId: p.planId ?? null,
|
|
2814
2817
|
quietMachine: p.quietMachine === true,
|
|
2815
2818
|
budgetExempt: p.budgetExempt === true,
|
|
2816
2819
|
originSessionId: job.originSessionId
|
|
@@ -2933,6 +2936,7 @@ async function reconcile(state) {
|
|
|
2933
2936
|
epicId: p.epicId ?? inv.row?.epicId ?? null,
|
|
2934
2937
|
dependsOn: p.dependsOn,
|
|
2935
2938
|
disposition: p.disposition ?? null,
|
|
2939
|
+
planId: p.planId ?? null,
|
|
2936
2940
|
quietMachine: p.quietMachine === true,
|
|
2937
2941
|
budgetExempt: p.budgetExempt === true,
|
|
2938
2942
|
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
|
|
@@ -3067,6 +3071,7 @@ async function reconcile(state) {
|
|
|
3067
3071
|
epicId: p.epicId ?? null,
|
|
3068
3072
|
dependsOn: p.dependsOn,
|
|
3069
3073
|
disposition: p.disposition ?? null,
|
|
3074
|
+
planId: p.planId ?? null,
|
|
3070
3075
|
quietMachine: p.quietMachine === true,
|
|
3071
3076
|
budgetExempt: p.budgetExempt === true,
|
|
3072
3077
|
originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
|
|
@@ -3393,6 +3398,21 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
|
|
|
3393
3398
|
return consecutiveFailures >= threshold && !alreadyWarned;
|
|
3394
3399
|
}
|
|
3395
3400
|
|
|
3401
|
+
/**
|
|
3402
|
+
* Pure: `failureStreakWarned === true` must always carry a numeric
|
|
3403
|
+
* `failureStreakWarnedAt` (a state file may hold one without the other), so
|
|
3404
|
+
* the escalation message never renders "after nullm". Backfills `nowMs`.
|
|
3405
|
+
*/
|
|
3406
|
+
function restoreFailureStreakWarnedAt(warned, warnedAt, nowMs) {
|
|
3407
|
+
if (!warned) return typeof warnedAt === 'number' ? warnedAt : null;
|
|
3408
|
+
return typeof warnedAt === 'number' ? warnedAt : nowMs;
|
|
3409
|
+
}
|
|
3410
|
+
|
|
3411
|
+
/** Pure: whole minutes a warned streak has persisted; never null/NaN. */
|
|
3412
|
+
function persistedStreakMinutes(warnedAt, nowMs) {
|
|
3413
|
+
return typeof warnedAt === 'number' ? Math.round((nowMs - warnedAt) / 60_000) : 0;
|
|
3414
|
+
}
|
|
3415
|
+
|
|
3396
3416
|
/**
|
|
3397
3417
|
* Pure: does a PERSISTING failure streak warrant another escalation (audit
|
|
3398
3418
|
* event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
|
|
@@ -3434,7 +3454,7 @@ function warnFailureStreakIfNeeded() {
|
|
|
3434
3454
|
}
|
|
3435
3455
|
if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
|
|
3436
3456
|
lastEscalationAtMs = nowMs;
|
|
3437
|
-
const persistedMinutes = failureStreakWarnedAt
|
|
3457
|
+
const persistedMinutes = persistedStreakMinutes(failureStreakWarnedAt, nowMs);
|
|
3438
3458
|
try {
|
|
3439
3459
|
appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
|
|
3440
3460
|
appendError({
|
|
@@ -5249,15 +5269,17 @@ async function performLeftoverQuarantine(job, paths, headBefore = null) {
|
|
|
5249
5269
|
* selects `--resume <sessionId>` (reconnect) INSTEAD of `--session-id
|
|
5250
5270
|
* <sessionId>` (mint) — the two flags are mutually exclusive, never both.
|
|
5251
5271
|
* `--model` is always explicit (never left to the CLI's drifting default —
|
|
5252
|
-
* see conventions.md). `
|
|
5272
|
+
* see conventions.md). `effort`, when a level (persona `effort:` via
|
|
5273
|
+
* agentEffortResolve.cjs), appends `--effort <level>`; null/inherit → no flag. `systemPrompt`, when given (the PRD's `agentType`
|
|
5253
5274
|
* persona body, resolved by agentModelResolve.cjs's resolvePrdPersonaForSpawn),
|
|
5254
5275
|
* is passed as `--append-system-prompt` so the executor IS that persona at
|
|
5255
5276
|
* launch rather than being asked in prose to adopt one.
|
|
5256
5277
|
*/
|
|
5257
|
-
function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }) {
|
|
5278
|
+
function buildClaudeSpawnArgs({ prompt, model, effort, sessionId, resume, systemPrompt }) {
|
|
5258
5279
|
return [
|
|
5259
5280
|
'-p', prompt,
|
|
5260
5281
|
'--model', model,
|
|
5282
|
+
...effortArgs(effort),
|
|
5261
5283
|
...(systemPrompt ? ['--append-system-prompt', systemPrompt] : []),
|
|
5262
5284
|
'--dangerously-skip-permissions',
|
|
5263
5285
|
'--output-format', 'stream-json',
|
|
@@ -5510,7 +5532,8 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5510
5532
|
// a dangling/absent agentType falls back to no persona + FALLBACK_MODEL
|
|
5511
5533
|
// and is logged once by resolvePrdPersonaForSpawn itself.
|
|
5512
5534
|
const personaResolution = await agentModelResolve.resolvePrdPersonaForSpawn({ cwd, agentType: job.agentType });
|
|
5513
|
-
|
|
5535
|
+
const personaEffort = resolveEpicEffort({ cwd, agentType: job.agentType }).effort;
|
|
5536
|
+
safeLog(`[scheduler] agentType=${job.agentType || '(none)'} persona=${personaResolution.personaPath || '(fallback — no persona applied)'} model=${personaResolution.model}${personaEffort ? ` effort=${personaEffort}` : ''}\n`);
|
|
5514
5537
|
|
|
5515
5538
|
return await new Promise((resolve) => {
|
|
5516
5539
|
const claudeBin = resolveClaudeBin();
|
|
@@ -5710,6 +5733,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5710
5733
|
args: buildClaudeSpawnArgs({
|
|
5711
5734
|
prompt,
|
|
5712
5735
|
model: personaResolution.model,
|
|
5736
|
+
effort: personaEffort,
|
|
5713
5737
|
sessionId,
|
|
5714
5738
|
resume: !!resumeTarget,
|
|
5715
5739
|
systemPrompt: personaResolution.systemPrompt,
|
|
@@ -6588,6 +6612,117 @@ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchE
|
|
|
6588
6612
|
await broadcast({ flush: true });
|
|
6589
6613
|
}
|
|
6590
6614
|
|
|
6615
|
+
// Scheduler-scoped error sink for failures that would otherwise be invisible
|
|
6616
|
+
// in packaged/npx builds (stdout unread). Never throws.
|
|
6617
|
+
function reportSchedulerError(message, slug, e) {
|
|
6618
|
+
try {
|
|
6619
|
+
logs.writeLine({
|
|
6620
|
+
scope: 'scheduler',
|
|
6621
|
+
level: 'error',
|
|
6622
|
+
message,
|
|
6623
|
+
meta: { slug, error: e?.message || String(e), stack: e?.stack },
|
|
6624
|
+
});
|
|
6625
|
+
} catch { /* logging must never be the thing that fails */ }
|
|
6626
|
+
try {
|
|
6627
|
+
appendAuditEvent('scheduler_error', { slug, message, error: e?.message || String(e), stack: e?.stack });
|
|
6628
|
+
} catch { /* same */ }
|
|
6629
|
+
}
|
|
6630
|
+
|
|
6631
|
+
/**
|
|
6632
|
+
* Salvage, integrate and clean up a job's throwaway worktree once its run has
|
|
6633
|
+
* ended. NEVER throws: any rejection (salvage / integrate / cleanup) is
|
|
6634
|
+
* reported through deps.reportSchedulerError and surfaces as
|
|
6635
|
+
* `worktreeIntegrationFailure`, so spawnJob's finalize mutate always runs and
|
|
6636
|
+
* the job can never be left `running`. The branch is kept on every failure.
|
|
6637
|
+
* @returns {Promise<{worktreeLeftoverDirty: string[], salvagePatch: string|null,
|
|
6638
|
+
* worktreeIntegrationFailure: string|null, worktreeIntegrationDetail: object|null,
|
|
6639
|
+
* mergeAutoResolved: string|null, mergeAutoResolvedPaths: string[]|null}>}
|
|
6640
|
+
*/
|
|
6641
|
+
async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths, deps = {} }) {
|
|
6642
|
+
const jw = deps.jobWorktree || jobWorktree;
|
|
6643
|
+
const uncommitted = deps.uncommittedChanges || uncommittedChanges;
|
|
6644
|
+
const report = deps.reportSchedulerError || reportSchedulerError;
|
|
6645
|
+
let worktreeLeftoverDirty = [];
|
|
6646
|
+
let salvagePatch = null;
|
|
6647
|
+
let worktreeIntegrationFailure = null;
|
|
6648
|
+
let worktreeIntegrationDetail = null;
|
|
6649
|
+
let mergeAutoResolved = null;
|
|
6650
|
+
let mergeAutoResolvedPaths = null;
|
|
6651
|
+
try {
|
|
6652
|
+
worktreeLeftoverDirty = (await uncommitted(worktree.dir)) || [];
|
|
6653
|
+
// Salvage the worktree's full diff (tracked + untracked) to the run
|
|
6654
|
+
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
6655
|
+
// before its finish-protocol commit loses that work outright, with
|
|
6656
|
+
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
6657
|
+
// integration/cleanup and never changes the job's verdict.
|
|
6658
|
+
if (worktreeLeftoverDirty.length) {
|
|
6659
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
6660
|
+
const salvage = await jw.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
6661
|
+
if (salvage && salvage.ok) {
|
|
6662
|
+
salvagePatch = salvagePath;
|
|
6663
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
6664
|
+
}
|
|
6665
|
+
}
|
|
6666
|
+
const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
6667
|
+
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
6668
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
6669
|
+
}
|
|
6670
|
+
if (!integration.ok) {
|
|
6671
|
+
worktreeIntegrationFailure = integration.reason;
|
|
6672
|
+
worktreeIntegrationDetail = integration;
|
|
6673
|
+
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
6674
|
+
} else if (integration.integrated) {
|
|
6675
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
6676
|
+
if (integration.autoResolved) {
|
|
6677
|
+
mergeAutoResolved = integration.autoResolved;
|
|
6678
|
+
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
6679
|
+
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
6680
|
+
}
|
|
6681
|
+
}
|
|
6682
|
+
await jw.cleanupJobWorktree({
|
|
6683
|
+
cwd: guardCwd,
|
|
6684
|
+
dir: worktree.dir,
|
|
6685
|
+
branch: worktree.branch,
|
|
6686
|
+
keepBranch: !integration.ok,
|
|
6687
|
+
});
|
|
6688
|
+
} catch (e) {
|
|
6689
|
+
worktreeIntegrationFailure = e?.message || String(e);
|
|
6690
|
+
report('spawnJob worktree finalize failed', job.slug, e);
|
|
6691
|
+
// Best-effort: release the checkout + worktree-cap slot, keep the branch.
|
|
6692
|
+
try {
|
|
6693
|
+
await jw.cleanupJobWorktree({ cwd: guardCwd, dir: worktree.dir, branch: worktree.branch, keepBranch: true });
|
|
6694
|
+
} catch { /* already reported above */ }
|
|
6695
|
+
}
|
|
6696
|
+
return { worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths };
|
|
6697
|
+
}
|
|
6698
|
+
|
|
6699
|
+
/**
|
|
6700
|
+
* Map a worktree integration failure onto the verifier verdict spawnJob stamps
|
|
6701
|
+
* (pure). Null failure -> null (no override). Always downgrades to needs_review.
|
|
6702
|
+
*/
|
|
6703
|
+
function worktreeIntegrationVerdict({ failure, detail, slug }) {
|
|
6704
|
+
if (!failure) return null;
|
|
6705
|
+
return {
|
|
6706
|
+
verdict: 'worktree_integration_failed',
|
|
6707
|
+
reason: detail && detail.failureKind === 'content_conflict'
|
|
6708
|
+
? `Integration blocked by a content conflict in ${(detail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(slug)} preserved; needs a manual merge.`
|
|
6709
|
+
: `worktree branch integration failed: ${failure} — branch preserved for manual merge`,
|
|
6710
|
+
downgradeTo: 'needs_review',
|
|
6711
|
+
};
|
|
6712
|
+
}
|
|
6713
|
+
|
|
6714
|
+
/**
|
|
6715
|
+
* Run one interval tick; a throw is reported and swallowed so the interval
|
|
6716
|
+
* keeps firing.
|
|
6717
|
+
*/
|
|
6718
|
+
function guardedTick(fn, label) {
|
|
6719
|
+
try {
|
|
6720
|
+
fn();
|
|
6721
|
+
} catch (e) {
|
|
6722
|
+
reportSchedulerError(label, null, e);
|
|
6723
|
+
}
|
|
6724
|
+
}
|
|
6725
|
+
|
|
6591
6726
|
async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
6592
6727
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
6593
6728
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
@@ -6963,42 +7098,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6963
7098
|
});
|
|
6964
7099
|
} finally {
|
|
6965
7100
|
if (worktree.ok) {
|
|
6966
|
-
worktreeLeftoverDirty
|
|
6967
|
-
|
|
6968
|
-
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
6969
|
-
// before its finish-protocol commit loses that work outright, with
|
|
6970
|
-
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
6971
|
-
// integration/cleanup and never changes the job's verdict.
|
|
6972
|
-
if (worktreeLeftoverDirty.length) {
|
|
6973
|
-
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
6974
|
-
const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
6975
|
-
if (salvage && salvage.ok) {
|
|
6976
|
-
salvagePatch = salvagePath;
|
|
6977
|
-
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
6978
|
-
}
|
|
6979
|
-
}
|
|
6980
|
-
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
6981
|
-
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
6982
|
-
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
6983
|
-
}
|
|
6984
|
-
if (!integration.ok) {
|
|
6985
|
-
worktreeIntegrationFailure = integration.reason;
|
|
6986
|
-
worktreeIntegrationDetail = integration;
|
|
6987
|
-
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
6988
|
-
} else if (integration.integrated) {
|
|
6989
|
-
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
6990
|
-
if (integration.autoResolved) {
|
|
6991
|
-
mergeAutoResolved = integration.autoResolved;
|
|
6992
|
-
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
6993
|
-
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
6994
|
-
}
|
|
6995
|
-
}
|
|
6996
|
-
await jobWorktree.cleanupJobWorktree({
|
|
6997
|
-
cwd: guardCwd,
|
|
6998
|
-
dir: worktree.dir,
|
|
6999
|
-
branch: worktree.branch,
|
|
7000
|
-
keepBranch: !integration.ok,
|
|
7001
|
-
});
|
|
7101
|
+
({ worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths } =
|
|
7102
|
+
await finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths }));
|
|
7002
7103
|
} else {
|
|
7003
7104
|
// In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
|
|
7004
7105
|
// failure) — there is no throwaway checkout to diff, so salvage only
|
|
@@ -7271,13 +7372,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
7271
7372
|
// guard AC explicitly requires this failure be surfaced as an explicit job
|
|
7272
7373
|
// outcome, never silently dropped alongside the branch it's stranded on.
|
|
7273
7374
|
if (worktreeIntegrationFailure) {
|
|
7274
|
-
verifyResult = {
|
|
7275
|
-
verdict: 'worktree_integration_failed',
|
|
7276
|
-
reason: worktreeIntegrationDetail && worktreeIntegrationDetail.failureKind === 'content_conflict'
|
|
7277
|
-
? `Integration blocked by a content conflict in ${(worktreeIntegrationDetail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(job.slug)} preserved; needs a manual merge.`
|
|
7278
|
-
: `worktree branch integration failed: ${worktreeIntegrationFailure} — branch preserved for manual merge`,
|
|
7279
|
-
downgradeTo: 'needs_review',
|
|
7280
|
-
};
|
|
7375
|
+
verifyResult = worktreeIntegrationVerdict({ failure: worktreeIntegrationFailure, detail: worktreeIntegrationDetail, slug: job.slug });
|
|
7281
7376
|
}
|
|
7282
7377
|
|
|
7283
7378
|
// Shared-tree stash guard (incident 2026-09-01): only meaningful for an
|
|
@@ -7935,6 +8030,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
7935
8030
|
}
|
|
7936
8031
|
} catch (e) {
|
|
7937
8032
|
console.error('[scheduler] spawnJob error', job.slug, e);
|
|
8033
|
+
reportSchedulerError('spawnJob error', job.slug, e);
|
|
7938
8034
|
} finally {
|
|
7939
8035
|
runningSet.delete(job.slug);
|
|
7940
8036
|
// Slot release notifies subscribed pumps (chat lane) machine-wide.
|
|
@@ -8294,7 +8390,7 @@ async function tickBody(gen, { bypassLoadGate }) {
|
|
|
8294
8390
|
for (const job of gatedBatch) {
|
|
8295
8391
|
if (cancelToken.cancelled || stale()) break;
|
|
8296
8392
|
// spawnJob is fire-and-forget; it calls tickQueue() on completion.
|
|
8297
|
-
spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() =>
|
|
8393
|
+
spawnJob(job, runId, runDir, state.config.defaultCwd).catch((e) => reportSchedulerError('spawnJob dispatch rejected', job.slug, e));
|
|
8298
8394
|
}
|
|
8299
8395
|
return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
|
|
8300
8396
|
}
|
|
@@ -11418,6 +11514,223 @@ function stop() {
|
|
|
11418
11514
|
stopDispatchLoop();
|
|
11419
11515
|
}
|
|
11420
11516
|
|
|
11517
|
+
// Body of the 10-minute maintenance interval (self-heal, escalations, restores).
|
|
11518
|
+
// Extracted so a throw is testable through guardedTick.
|
|
11519
|
+
function rescheduleIntervalTick() {
|
|
11520
|
+
rescheduleTimer().catch(() => {});
|
|
11521
|
+
const s = readQueueSync();
|
|
11522
|
+
// Periodic self-heal: re-run the verifier over stale needs_review jobs so a
|
|
11523
|
+
// job whose work actually landed (committed in-window, no FAIL sentinel)
|
|
11524
|
+
// auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
|
|
11525
|
+
// shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
|
|
11526
|
+
// and the candidate filter can never drift apart again (they did once —
|
|
11527
|
+
// see that function's comment). Kill-switch:
|
|
11528
|
+
// SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
|
|
11529
|
+
// reverifyNeedsReview's auto-fix loop is capped downstream by
|
|
11530
|
+
// MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
|
|
11531
|
+
// past it), so this interval firing cannot fan out investigations.
|
|
11532
|
+
if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
|
|
11533
|
+
if (shouldRunPeriodicReverify(s.jobs)) {
|
|
11534
|
+
reverifyNeedsReview().catch(() => {});
|
|
11535
|
+
}
|
|
11536
|
+
// A quarantined row only ever promotes to 'pending' through
|
|
11537
|
+
// reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
|
|
11538
|
+
// it re-checks the PRD file's createdVia stamp every pass. broadcast()
|
|
11539
|
+
// already runs reconcile+writeQueue on every normal poll tick, but an
|
|
11540
|
+
// idle queue (nothing pending/running to fire) can back off that
|
|
11541
|
+
// cadence for a long time; this guarantees an adopted-but-still-
|
|
11542
|
+
// quarantined row is re-checked within 10 minutes regardless.
|
|
11543
|
+
if (s.jobs.some((j) => j.status === 'quarantined')) {
|
|
11544
|
+
broadcast().catch(() => {});
|
|
11545
|
+
}
|
|
11546
|
+
}
|
|
11547
|
+
// Age-based escalation (independent of the self-heal kill-switch above —
|
|
11548
|
+
// this is a monitoring signal, not an auto-fix action): a quarantined
|
|
11549
|
+
// row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
|
|
11550
|
+
// warn-logged by project + slug + age so it cannot sit stranded and
|
|
11551
|
+
// silent (the four burrow-project rows this PRD was written against).
|
|
11552
|
+
for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
|
|
11553
|
+
console.warn(
|
|
11554
|
+
`[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
|
|
11555
|
+
+ `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11556
|
+
+ `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
|
|
11557
|
+
);
|
|
11558
|
+
appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
|
|
11559
|
+
}
|
|
11560
|
+
|
|
11561
|
+
// Estimate-relative overrun escalation. Sits in the blind spot between
|
|
11562
|
+
// the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
|
|
11563
|
+
// producing output while looping trips neither, so nothing noticed a PRD
|
|
11564
|
+
// running 9x its own estimate until a human went looking. Escalate loudly;
|
|
11565
|
+
// never kill on an estimate (see JOB_OVERRUN_FACTOR).
|
|
11566
|
+
for (const over of findOverrunningJobs(s.jobs, Date.now())) {
|
|
11567
|
+
console.warn(
|
|
11568
|
+
`[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
|
|
11569
|
+
+ `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
|
|
11570
|
+
+ `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
|
|
11571
|
+
+ `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
|
|
11572
|
+
+ `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
|
|
11573
|
+
);
|
|
11574
|
+
appendAuditEvent('job_overrunning_estimate', {
|
|
11575
|
+
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
11576
|
+
});
|
|
11577
|
+
// Durable stamp so schedule:state (and therefore the renderer) can see
|
|
11578
|
+
// this without re-deriving it — the console.warn/audit event above are
|
|
11579
|
+
// visible only in the log, never on the row itself. Display-only
|
|
11580
|
+
// advisory field; re-stamped in place every sweep, never appended.
|
|
11581
|
+
mutate((state) => {
|
|
11582
|
+
const j = state.jobs.find((x) => x.slug === over.slug);
|
|
11583
|
+
if (!j) return;
|
|
11584
|
+
j.overrun = {
|
|
11585
|
+
ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
|
|
11586
|
+
};
|
|
11587
|
+
}).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
|
|
11588
|
+
}
|
|
11589
|
+
|
|
11590
|
+
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
11591
|
+
// one ACTS: 'investigating' is a transient status whose restore
|
|
11592
|
+
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
11593
|
+
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
11594
|
+
// there forever (see findStrandedInvestigations' header, and the
|
|
11595
|
+
// "'investigating' must never be the job's resting state" comment at
|
|
11596
|
+
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
11597
|
+
// exact terminal status it already carried before the probe was
|
|
11598
|
+
// spawned — it never re-runs or re-investigates anything.
|
|
11599
|
+
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
11600
|
+
if (stranded.length > 0) {
|
|
11601
|
+
mutate((ms) => {
|
|
11602
|
+
for (const st of stranded) {
|
|
11603
|
+
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
11604
|
+
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
11605
|
+
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
11606
|
+
delete j.runtime;
|
|
11607
|
+
console.warn(
|
|
11608
|
+
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
11609
|
+
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
11610
|
+
+ `restored to '${st.restoreStatus}'`,
|
|
11611
|
+
);
|
|
11612
|
+
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
11613
|
+
}
|
|
11614
|
+
})
|
|
11615
|
+
.then(() => broadcast({ flush: true }))
|
|
11616
|
+
.catch(() => {});
|
|
11617
|
+
}
|
|
11618
|
+
|
|
11619
|
+
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
11620
|
+
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
11621
|
+
// else distinguishes "no pending work" from "pending work, never
|
|
11622
|
+
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
11623
|
+
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
11624
|
+
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
11625
|
+
for (const sp of starvedProjects) {
|
|
11626
|
+
console.warn(
|
|
11627
|
+
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
11628
|
+
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
11629
|
+
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
11630
|
+
);
|
|
11631
|
+
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
11632
|
+
}
|
|
11633
|
+
// Bounded, automated consequence for a starve that outlives the WARN
|
|
11634
|
+
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
11635
|
+
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
11636
|
+
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
11637
|
+
// a subset of the rows already reported above — same verdict, no
|
|
11638
|
+
// re-derivation.
|
|
11639
|
+
runStarveEscalationSweep(starvedProjects);
|
|
11640
|
+
|
|
11641
|
+
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
11642
|
+
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
11643
|
+
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
11644
|
+
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
11645
|
+
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
11646
|
+
// above it can never observe two different snapshots of the same row.
|
|
11647
|
+
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
11648
|
+
const autoResetTargets = failedAutoResetDisabled()
|
|
11649
|
+
? []
|
|
11650
|
+
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
11651
|
+
const stuckFailed = stuckFailedEscalationDisabled()
|
|
11652
|
+
? []
|
|
11653
|
+
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
11654
|
+
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
11655
|
+
// (this PRD): computed alongside the failed-row passes above and acted
|
|
11656
|
+
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
11657
|
+
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
11658
|
+
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
11659
|
+
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
11660
|
+
? []
|
|
11661
|
+
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
11662
|
+
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
11663
|
+
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
11664
|
+
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
11665
|
+
// never be read from one snapshot and written from another, and the
|
|
11666
|
+
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
11667
|
+
// same turn as the transition it gates. Kill-switch:
|
|
11668
|
+
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
11669
|
+
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
11670
|
+
? []
|
|
11671
|
+
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
11672
|
+
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
11673
|
+
mutate(async (ms) => {
|
|
11674
|
+
for (const target of autoResetTargets) {
|
|
11675
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11676
|
+
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
11677
|
+
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
11678
|
+
j.failedAutoResetAttempts = attempt;
|
|
11679
|
+
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
11680
|
+
// resetJobFields is the same field-clearing list the admin
|
|
11681
|
+
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
11682
|
+
// it rather than inventing a second list. It also sets job.error to
|
|
11683
|
+
// the reason text passed in; we clear that back to null right
|
|
11684
|
+
// after since this is a clean auto-reset, not a recorded error.
|
|
11685
|
+
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
11686
|
+
j.error = null;
|
|
11687
|
+
delete j.stuckFailedNotified;
|
|
11688
|
+
console.warn(
|
|
11689
|
+
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11690
|
+
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
11691
|
+
);
|
|
11692
|
+
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
11693
|
+
}
|
|
11694
|
+
for (const stuck of stuckFailed) {
|
|
11695
|
+
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
11696
|
+
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
11697
|
+
// Still has auto-reset attempts left — it will be (or already was,
|
|
11698
|
+
// earlier this same pass) picked up by the loop above instead.
|
|
11699
|
+
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
11700
|
+
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
11701
|
+
j.stuckFailedNotified = true;
|
|
11702
|
+
console.warn(
|
|
11703
|
+
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
11704
|
+
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11705
|
+
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
11706
|
+
);
|
|
11707
|
+
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
11708
|
+
}
|
|
11709
|
+
for (const target of exhaustedNeedsReviewTargets) {
|
|
11710
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11711
|
+
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
11712
|
+
if (outcome) {
|
|
11713
|
+
console.warn(
|
|
11714
|
+
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11715
|
+
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
11716
|
+
);
|
|
11717
|
+
}
|
|
11718
|
+
}
|
|
11719
|
+
for (const target of quarantineTargets) {
|
|
11720
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11721
|
+
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
11722
|
+
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
11723
|
+
if (outcome) {
|
|
11724
|
+
console.warn(
|
|
11725
|
+
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11726
|
+
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
11727
|
+
);
|
|
11728
|
+
}
|
|
11729
|
+
}
|
|
11730
|
+
}).catch(() => {});
|
|
11731
|
+
}
|
|
11732
|
+
}
|
|
11733
|
+
|
|
11421
11734
|
async function init() {
|
|
11422
11735
|
ensureDirs();
|
|
11423
11736
|
// Boot phase — reconciliation, migrations, self-heal, first reset probe.
|
|
@@ -11618,218 +11931,9 @@ async function init() {
|
|
|
11618
11931
|
// resets early or the auth token rotates. Tracked so re-init doesn't leak.
|
|
11619
11932
|
if (rescheduleInterval) clearInterval(rescheduleInterval);
|
|
11620
11933
|
rescheduleInterval = setInterval(() => {
|
|
11621
|
-
|
|
11622
|
-
|
|
11623
|
-
|
|
11624
|
-
// job whose work actually landed (committed in-window, no FAIL sentinel)
|
|
11625
|
-
// auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
|
|
11626
|
-
// shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
|
|
11627
|
-
// and the candidate filter can never drift apart again (they did once —
|
|
11628
|
-
// see that function's comment). Kill-switch:
|
|
11629
|
-
// SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
|
|
11630
|
-
// reverifyNeedsReview's auto-fix loop is capped downstream by
|
|
11631
|
-
// MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
|
|
11632
|
-
// past it), so this interval firing cannot fan out investigations.
|
|
11633
|
-
if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
|
|
11634
|
-
if (shouldRunPeriodicReverify(s.jobs)) {
|
|
11635
|
-
reverifyNeedsReview().catch(() => {});
|
|
11636
|
-
}
|
|
11637
|
-
// A quarantined row only ever promotes to 'pending' through
|
|
11638
|
-
// reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
|
|
11639
|
-
// it re-checks the PRD file's createdVia stamp every pass. broadcast()
|
|
11640
|
-
// already runs reconcile+writeQueue on every normal poll tick, but an
|
|
11641
|
-
// idle queue (nothing pending/running to fire) can back off that
|
|
11642
|
-
// cadence for a long time; this guarantees an adopted-but-still-
|
|
11643
|
-
// quarantined row is re-checked within 10 minutes regardless.
|
|
11644
|
-
if (s.jobs.some((j) => j.status === 'quarantined')) {
|
|
11645
|
-
broadcast().catch(() => {});
|
|
11646
|
-
}
|
|
11647
|
-
}
|
|
11648
|
-
// Age-based escalation (independent of the self-heal kill-switch above —
|
|
11649
|
-
// this is a monitoring signal, not an auto-fix action): a quarantined
|
|
11650
|
-
// row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
|
|
11651
|
-
// warn-logged by project + slug + age so it cannot sit stranded and
|
|
11652
|
-
// silent (the four burrow-project rows this PRD was written against).
|
|
11653
|
-
for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
|
|
11654
|
-
console.warn(
|
|
11655
|
-
`[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
|
|
11656
|
-
+ `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11657
|
-
+ `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
|
|
11658
|
-
);
|
|
11659
|
-
appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
|
|
11660
|
-
}
|
|
11661
|
-
|
|
11662
|
-
// Estimate-relative overrun escalation. Sits in the blind spot between
|
|
11663
|
-
// the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
|
|
11664
|
-
// producing output while looping trips neither, so nothing noticed a PRD
|
|
11665
|
-
// running 9x its own estimate until a human went looking. Escalate loudly;
|
|
11666
|
-
// never kill on an estimate (see JOB_OVERRUN_FACTOR).
|
|
11667
|
-
for (const over of findOverrunningJobs(s.jobs, Date.now())) {
|
|
11668
|
-
console.warn(
|
|
11669
|
-
`[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
|
|
11670
|
-
+ `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
|
|
11671
|
-
+ `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
|
|
11672
|
-
+ `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
|
|
11673
|
-
+ `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
|
|
11674
|
-
);
|
|
11675
|
-
appendAuditEvent('job_overrunning_estimate', {
|
|
11676
|
-
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
11677
|
-
});
|
|
11678
|
-
// Durable stamp so schedule:state (and therefore the renderer) can see
|
|
11679
|
-
// this without re-deriving it — the console.warn/audit event above are
|
|
11680
|
-
// visible only in the log, never on the row itself. Display-only
|
|
11681
|
-
// advisory field; re-stamped in place every sweep, never appended.
|
|
11682
|
-
mutate((state) => {
|
|
11683
|
-
const j = state.jobs.find((x) => x.slug === over.slug);
|
|
11684
|
-
if (!j) return;
|
|
11685
|
-
j.overrun = {
|
|
11686
|
-
ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
|
|
11687
|
-
};
|
|
11688
|
-
}).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
|
|
11689
|
-
}
|
|
11690
|
-
|
|
11691
|
-
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
11692
|
-
// one ACTS: 'investigating' is a transient status whose restore
|
|
11693
|
-
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
11694
|
-
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
11695
|
-
// there forever (see findStrandedInvestigations' header, and the
|
|
11696
|
-
// "'investigating' must never be the job's resting state" comment at
|
|
11697
|
-
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
11698
|
-
// exact terminal status it already carried before the probe was
|
|
11699
|
-
// spawned — it never re-runs or re-investigates anything.
|
|
11700
|
-
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
11701
|
-
if (stranded.length > 0) {
|
|
11702
|
-
mutate((ms) => {
|
|
11703
|
-
for (const st of stranded) {
|
|
11704
|
-
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
11705
|
-
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
11706
|
-
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
11707
|
-
delete j.runtime;
|
|
11708
|
-
console.warn(
|
|
11709
|
-
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
11710
|
-
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
11711
|
-
+ `restored to '${st.restoreStatus}'`,
|
|
11712
|
-
);
|
|
11713
|
-
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
11714
|
-
}
|
|
11715
|
-
})
|
|
11716
|
-
.then(() => broadcast({ flush: true }))
|
|
11717
|
-
.catch(() => {});
|
|
11718
|
-
}
|
|
11719
|
-
|
|
11720
|
-
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
11721
|
-
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
11722
|
-
// else distinguishes "no pending work" from "pending work, never
|
|
11723
|
-
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
11724
|
-
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
11725
|
-
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
11726
|
-
for (const sp of starvedProjects) {
|
|
11727
|
-
console.warn(
|
|
11728
|
-
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
11729
|
-
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
11730
|
-
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
11731
|
-
);
|
|
11732
|
-
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
11733
|
-
}
|
|
11734
|
-
// Bounded, automated consequence for a starve that outlives the WARN
|
|
11735
|
-
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
11736
|
-
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
11737
|
-
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
11738
|
-
// a subset of the rows already reported above — same verdict, no
|
|
11739
|
-
// re-derivation.
|
|
11740
|
-
runStarveEscalationSweep(starvedProjects);
|
|
11741
|
-
|
|
11742
|
-
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
11743
|
-
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
11744
|
-
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
11745
|
-
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
11746
|
-
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
11747
|
-
// above it can never observe two different snapshots of the same row.
|
|
11748
|
-
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
11749
|
-
const autoResetTargets = failedAutoResetDisabled()
|
|
11750
|
-
? []
|
|
11751
|
-
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
11752
|
-
const stuckFailed = stuckFailedEscalationDisabled()
|
|
11753
|
-
? []
|
|
11754
|
-
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
11755
|
-
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
11756
|
-
// (this PRD): computed alongside the failed-row passes above and acted
|
|
11757
|
-
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
11758
|
-
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
11759
|
-
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
11760
|
-
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
11761
|
-
? []
|
|
11762
|
-
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
11763
|
-
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
11764
|
-
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
11765
|
-
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
11766
|
-
// never be read from one snapshot and written from another, and the
|
|
11767
|
-
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
11768
|
-
// same turn as the transition it gates. Kill-switch:
|
|
11769
|
-
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
11770
|
-
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
11771
|
-
? []
|
|
11772
|
-
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
11773
|
-
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
11774
|
-
mutate(async (ms) => {
|
|
11775
|
-
for (const target of autoResetTargets) {
|
|
11776
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11777
|
-
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
11778
|
-
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
11779
|
-
j.failedAutoResetAttempts = attempt;
|
|
11780
|
-
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
11781
|
-
// resetJobFields is the same field-clearing list the admin
|
|
11782
|
-
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
11783
|
-
// it rather than inventing a second list. It also sets job.error to
|
|
11784
|
-
// the reason text passed in; we clear that back to null right
|
|
11785
|
-
// after since this is a clean auto-reset, not a recorded error.
|
|
11786
|
-
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
11787
|
-
j.error = null;
|
|
11788
|
-
delete j.stuckFailedNotified;
|
|
11789
|
-
console.warn(
|
|
11790
|
-
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11791
|
-
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
11792
|
-
);
|
|
11793
|
-
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
11794
|
-
}
|
|
11795
|
-
for (const stuck of stuckFailed) {
|
|
11796
|
-
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
11797
|
-
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
11798
|
-
// Still has auto-reset attempts left — it will be (or already was,
|
|
11799
|
-
// earlier this same pass) picked up by the loop above instead.
|
|
11800
|
-
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
11801
|
-
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
11802
|
-
j.stuckFailedNotified = true;
|
|
11803
|
-
console.warn(
|
|
11804
|
-
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
11805
|
-
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11806
|
-
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
11807
|
-
);
|
|
11808
|
-
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
11809
|
-
}
|
|
11810
|
-
for (const target of exhaustedNeedsReviewTargets) {
|
|
11811
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11812
|
-
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
11813
|
-
if (outcome) {
|
|
11814
|
-
console.warn(
|
|
11815
|
-
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11816
|
-
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
11817
|
-
);
|
|
11818
|
-
}
|
|
11819
|
-
}
|
|
11820
|
-
for (const target of quarantineTargets) {
|
|
11821
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11822
|
-
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
11823
|
-
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
11824
|
-
if (outcome) {
|
|
11825
|
-
console.warn(
|
|
11826
|
-
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11827
|
-
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
11828
|
-
);
|
|
11829
|
-
}
|
|
11830
|
-
}
|
|
11831
|
-
}).catch(() => {});
|
|
11832
|
-
}
|
|
11934
|
+
// One throwing tick (e.g. readQueueSync on a torn queue.json) must skip
|
|
11935
|
+
// only itself — the interval keeps firing and the failure is logged.
|
|
11936
|
+
guardedTick(rescheduleIntervalTick, 'rescheduleInterval tick failed');
|
|
11833
11937
|
}, REVERIFY_INTERVAL_MS);
|
|
11834
11938
|
|
|
11835
11939
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
@@ -11946,6 +12050,7 @@ async function listPrdsInternal() {
|
|
|
11946
12050
|
dependsOn: parsed.dependsOn ?? null,
|
|
11947
12051
|
agentType: parsed.agentType ?? null,
|
|
11948
12052
|
disposition: parsed.disposition ?? null,
|
|
12053
|
+
planId: parsed.planId ?? null,
|
|
11949
12054
|
mtimeMs: stat.mtimeMs,
|
|
11950
12055
|
archived,
|
|
11951
12056
|
};
|
|
@@ -12379,7 +12484,14 @@ const remote = {
|
|
|
12379
12484
|
const rows = listing.prds ?? [];
|
|
12380
12485
|
const rewrite = computeDispositionRewrite({ slug, disposition, dependsOn: dependsOn ?? [], rows });
|
|
12381
12486
|
if (!rewrite.ok) return rewrite;
|
|
12382
|
-
|
|
12487
|
+
// Keep the durable planId in step with the new relationship: a promoted head
|
|
12488
|
+
// starts its own plan; a re-attached row joins the target chain's plan (cleared
|
|
12489
|
+
// when the target predates the stamp, so the derivation fallback applies).
|
|
12490
|
+
let planId = mintPlanId();
|
|
12491
|
+
if (disposition === 'append') {
|
|
12492
|
+
planId = resolveInheritedPlanId(rewrite.dependsOn, rows).planId;
|
|
12493
|
+
}
|
|
12494
|
+
return this.updatePrd({ slug, cwd, frontmatter: { dependsOn: rewrite.dependsOn, disposition, planId } });
|
|
12383
12495
|
},
|
|
12384
12496
|
|
|
12385
12497
|
// Cancels a job that hasn't finished yet. A 'running' job's process group
|
|
@@ -12493,6 +12605,11 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
12493
12605
|
}
|
|
12494
12606
|
|
|
12495
12607
|
module.exports = {
|
|
12608
|
+
reportSchedulerError,
|
|
12609
|
+
finalizeJobWorktree,
|
|
12610
|
+
worktreeIntegrationVerdict,
|
|
12611
|
+
guardedTick,
|
|
12612
|
+
rescheduleIntervalTick,
|
|
12496
12613
|
classifyQueueStarvation,
|
|
12497
12614
|
classifyQueueStarvationByProject,
|
|
12498
12615
|
dispatchIdleMs,
|
|
@@ -12526,6 +12643,8 @@ module.exports = {
|
|
|
12526
12643
|
nextBackoffMs,
|
|
12527
12644
|
shouldWarnFailureStreak,
|
|
12528
12645
|
shouldEscalateFailureStreak,
|
|
12646
|
+
restoreFailureStreakWarnedAt,
|
|
12647
|
+
persistedStreakMinutes,
|
|
12529
12648
|
computeDegradedBudget,
|
|
12530
12649
|
healRefusalReason,
|
|
12531
12650
|
writeQueue,
|