claude-code-session-manager 0.93.0 → 0.95.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{DataModel-DDGx3qrn.js → DataModel-DZK9QZpV.js} +1 -1
- package/dist/assets/{History-CHBjYZyD.js → History-BDyJUGtR.js} +2 -2
- package/dist/assets/{Hooks-DG4c-Rzz.js → Hooks-CmK6l0uf.js} +3 -3
- package/dist/assets/HostBilko-D8TIx-Bx.js +1 -0
- package/dist/assets/{Library-CN1h5rAE.js → Library-DTxpOjx8.js} +25 -25
- package/dist/assets/{MarkdownEditor-DmBxrhdn.js → MarkdownEditor-BPJyrevq.js} +1 -1
- package/dist/assets/McpServers-B_rvIYDs.js +2 -0
- package/dist/assets/{Memory-Dk4uDCjg.js → Memory-B-9GlHqe.js} +6 -6
- package/dist/assets/{Permissions-BovobeZw.js → Permissions-DhMMp27n.js} +3 -3
- package/dist/assets/{Plugins-DvOf_B62.js → Plugins-DKfAlmMY.js} +2 -2
- package/dist/assets/{ProvenanceBadge-Dutz_4OY.js → ProvenanceBadge-ptlItdCK.js} +1 -1
- package/dist/assets/{SaveBar-CT5u6ZCJ.js → SaveBar-CSbNSl2W.js} +1 -1
- package/dist/assets/Scheduler--rn1QBYc.js +16 -0
- package/dist/assets/{ScopeSwitcher-BUYDpz2v.js → ScopeSwitcher-BJof-GWH.js} +1 -1
- package/dist/assets/{Settings-DIRmr3tL.js → Settings-Dsx6XZP5.js} +3 -3
- package/dist/assets/{SkillReferenceGraph-CdGr6d3_.js → SkillReferenceGraph-CGl4B89-.js} +2 -2
- package/dist/assets/{Skills-BQOheQv4.js → Skills-D522Zfol.js} +2 -2
- package/dist/assets/{SystemPrompt-5LsimKbX.js → SystemPrompt-DKj-q39J.js} +1 -1
- package/dist/assets/{TagLibrary-DreLsT3n.js → TagLibrary-BH7CG7-6.js} +1 -1
- package/dist/assets/{TiptapBody-C9kmlBnI.js → TiptapBody-BbcJL9az.js} +1 -1
- package/dist/assets/{Toggle-BKLCBu9A.js → Toggle-DAslvnnv.js} +1 -1
- package/dist/assets/{index-Db_SK9rW.js → index-Yf8e7VCB.js} +821 -821
- package/dist/assets/{settingsSchema-B5jBhSUP.js → settingsSchema-pzp0TgvN.js} +1 -1
- package/dist/index.html +1 -1
- package/package.json +8 -3
- package/scripts/README.md +6 -0
- package/src/main/__tests__/chat-cancel-terminal.test.cjs +5 -2
- package/src/main/__tests__/chat-exit-close-race.test.cjs +8 -2
- package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +5 -2
- package/src/main/__tests__/health-tick-liveness.test.cjs +12 -0
- package/src/main/__tests__/intradayRefresh.test.cjs +39 -0
- package/src/main/__tests__/openExternalApp-spawn-error.test.cjs +25 -0
- package/src/main/__tests__/opsErrorLog.test.cjs +22 -0
- package/src/main/__tests__/pty-epic-worktree-spawn-cwd.test.cjs +21 -0
- package/src/main/__tests__/rateLimitPollerStreak.test.cjs +14 -0
- package/src/main/__tests__/runVerify-atomic-verdicts.test.cjs +26 -0
- package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +11 -1
- package/src/main/__tests__/runVerify-transcript-commit-evidence.test.cjs +11 -1
- package/src/main/__tests__/runVerify.test.cjs +12 -1
- package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +15 -0
- package/src/main/__tests__/scheduler-utilization-hold.test.cjs +89 -0
- package/src/main/__tests__/transcriptsUsageFor.test.cjs +112 -1
- package/src/main/build-info.json +4 -4
- package/src/main/chatRunner.cjs +45 -10
- package/src/main/crashDiagnostics.cjs +9 -0
- package/src/main/git.cjs +9 -2
- package/src/main/health.cjs +25 -1
- package/src/main/historyAggregator.cjs +2 -19
- package/src/main/index.cjs +7 -8
- package/src/main/ipcSchemas.cjs +13 -3
- package/src/main/lib/__tests__/childWithLog.test.cjs +180 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +19 -0
- package/src/main/lib/__tests__/gitCacheBound.test.cjs +69 -0
- package/src/main/lib/__tests__/loopDelay.test.cjs +68 -0
- package/src/main/lib/__tests__/usageCircuit.test.cjs +86 -17
- package/src/main/lib/agentPersonaSchema.cjs +2 -0
- package/src/main/lib/childWithLog.cjs +92 -51
- package/src/main/lib/delegationReadiness.cjs +3 -1
- package/src/main/lib/epicSpawnPlan.cjs +27 -8
- package/src/main/lib/epicTranscriptPath.cjs +5 -1
- package/src/main/lib/headTailBuffer.cjs +43 -0
- package/src/main/lib/intradayRefresh.cjs +33 -0
- package/src/main/lib/loopDelay.cjs +44 -0
- package/src/main/lib/lruCache.cjs +39 -0
- package/src/main/lib/openExternalApp.cjs +27 -9
- package/src/main/lib/opsErrorLog.cjs +22 -0
- package/src/main/lib/promptSessionSchema.cjs +2 -0
- package/src/main/lib/rendererRecovery.cjs +141 -0
- package/src/main/lib/scheduleJobSchema.cjs +3 -0
- package/src/main/lib/usageCircuit.cjs +40 -12
- package/src/main/runVerify.cjs +3 -1
- package/src/main/scheduler.cjs +409 -260
- package/src/main/transcripts.cjs +54 -11
- package/src/main/usage.cjs +9 -1
- package/src/preload/api.d.ts +20 -3
- package/dist/assets/HostBilko-DHXcpV2c.js +0 -1
- package/dist/assets/McpServers-Beb_HD6I.js +0 -2
- package/dist/assets/Scheduler-B7GSnZj5.js +0 -16
package/src/main/scheduler.cjs
CHANGED
|
@@ -955,7 +955,11 @@ const DEFAULT_CONFIG = {
|
|
|
955
955
|
// 'on-reset' = fire offsetMinutes after the next 5h reset (legacy).
|
|
956
956
|
// 'manual' = only fire on explicit Run now click.
|
|
957
957
|
firePolicy: 'when-available',
|
|
958
|
-
// For 'when-available'. Fire only when
|
|
958
|
+
// For 'when-available'. Fire only when BINDING-window utilization < this
|
|
959
|
+
// percent. Binding = the most-consumed unscoped window (five_hour OR weekly):
|
|
960
|
+
// dispatching into an exhausted weekly window would just produce 429s, so the
|
|
961
|
+
// gate holds on it — a hold that can last days, which is why it is surfaced
|
|
962
|
+
// loudly (utilizationHold snapshot field, heartbeat window name, health).
|
|
959
963
|
utilizationThreshold: 90,
|
|
960
964
|
schemaVersion: 1,
|
|
961
965
|
supervisor: {
|
|
@@ -1507,6 +1511,7 @@ function loadSchedulerState() {
|
|
|
1507
1511
|
if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
|
|
1508
1512
|
if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
|
|
1509
1513
|
if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
|
|
1514
|
+
failureStreakWarnedAt = restoreFailureStreakWarnedAt(failureStreakWarned, failureStreakWarnedAt, Date.now());
|
|
1510
1515
|
if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
|
|
1511
1516
|
} catch { /* first boot or corrupt — start fresh */ }
|
|
1512
1517
|
}
|
|
@@ -1803,6 +1808,9 @@ function heartbeatTick(deps = {}) {
|
|
|
1803
1808
|
quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
|
|
1804
1809
|
nextReset: cachedNextReset,
|
|
1805
1810
|
utilization: cachedUtilization,
|
|
1811
|
+
// Which window `utilization` (and nextReset) refer to — a reader of
|
|
1812
|
+
// this log can't otherwise tell a 5h hold from a weekly one.
|
|
1813
|
+
utilizationWindow: cachedBindingWindowName,
|
|
1806
1814
|
consecutiveFailures,
|
|
1807
1815
|
// State/consecutiveFailures/degraded-budget snapshot of the shared
|
|
1808
1816
|
// usage-meter breaker, so a human reading only the heartbeat log can
|
|
@@ -3225,6 +3233,10 @@ async function reconcile(state) {
|
|
|
3225
3233
|
|
|
3226
3234
|
let cachedNextReset = null; // bare ISO string or null
|
|
3227
3235
|
let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
|
|
3236
|
+
let cachedBindingWindowName = null; // name/kind of the window cachedUtilization reads (e.g. 'five_hour', 'weekly_all')
|
|
3237
|
+
// Non-null while maybeLaunchWhenAvailable is holding the queue on the
|
|
3238
|
+
// utilization gate: { window, percent, threshold, resetsAt }. Snapshot-level.
|
|
3239
|
+
let utilizationHold = null;
|
|
3228
3240
|
// ms timestamp of the last FRESH reset observation (see recordObservedReset)
|
|
3229
3241
|
// — distinct from Date.now(), so persistSchedulerState never re-stamps a
|
|
3230
3242
|
// stale cachedNextReset as "just observed" on every poll cycle.
|
|
@@ -3280,6 +3292,7 @@ function computeDegradedBudget() {
|
|
|
3280
3292
|
function applyDegradedBudget() {
|
|
3281
3293
|
const budget = computeDegradedBudget();
|
|
3282
3294
|
cachedUtilization = budget.utilization;
|
|
3295
|
+
cachedBindingWindowName = bindingWindow(lastGoodUsagePayload).name;
|
|
3283
3296
|
degradedConcurrencyCapValue = budget.concurrencyCap;
|
|
3284
3297
|
return budget;
|
|
3285
3298
|
}
|
|
@@ -3289,8 +3302,15 @@ async function refreshNextReset() {
|
|
|
3289
3302
|
const r = await billing.fetchUsage();
|
|
3290
3303
|
if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
|
|
3291
3304
|
const window = bindingWindow(r.data?.usage);
|
|
3305
|
+
if (!Number.isFinite(window.utilization)) {
|
|
3306
|
+
// Successful fetch with no readable percent = meter/contract problem, not a reading.
|
|
3307
|
+
billing.usageCircuit.recordFailure('no_utilization');
|
|
3308
|
+
throw new Error('usage fetch ok but payload yielded no finite utilization');
|
|
3309
|
+
}
|
|
3292
3310
|
recordObservedReset(window.resets_at ?? null);
|
|
3293
|
-
cachedUtilization =
|
|
3311
|
+
cachedUtilization = window.utilization;
|
|
3312
|
+
cachedBindingWindowName = window.name;
|
|
3313
|
+
lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
|
|
3294
3314
|
return cachedNextReset;
|
|
3295
3315
|
}
|
|
3296
3316
|
|
|
@@ -3397,6 +3417,21 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
|
|
|
3397
3417
|
return consecutiveFailures >= threshold && !alreadyWarned;
|
|
3398
3418
|
}
|
|
3399
3419
|
|
|
3420
|
+
/**
|
|
3421
|
+
* Pure: `failureStreakWarned === true` must always carry a numeric
|
|
3422
|
+
* `failureStreakWarnedAt` (a state file may hold one without the other), so
|
|
3423
|
+
* the escalation message never renders "after nullm". Backfills `nowMs`.
|
|
3424
|
+
*/
|
|
3425
|
+
function restoreFailureStreakWarnedAt(warned, warnedAt, nowMs) {
|
|
3426
|
+
if (!warned) return typeof warnedAt === 'number' ? warnedAt : null;
|
|
3427
|
+
return typeof warnedAt === 'number' ? warnedAt : nowMs;
|
|
3428
|
+
}
|
|
3429
|
+
|
|
3430
|
+
/** Pure: whole minutes a warned streak has persisted; never null/NaN. */
|
|
3431
|
+
function persistedStreakMinutes(warnedAt, nowMs) {
|
|
3432
|
+
return typeof warnedAt === 'number' ? Math.round((nowMs - warnedAt) / 60_000) : 0;
|
|
3433
|
+
}
|
|
3434
|
+
|
|
3400
3435
|
/**
|
|
3401
3436
|
* Pure: does a PERSISTING failure streak warrant another escalation (audit
|
|
3402
3437
|
* event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
|
|
@@ -3438,7 +3473,7 @@ function warnFailureStreakIfNeeded() {
|
|
|
3438
3473
|
}
|
|
3439
3474
|
if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
|
|
3440
3475
|
lastEscalationAtMs = nowMs;
|
|
3441
|
-
const persistedMinutes = failureStreakWarnedAt
|
|
3476
|
+
const persistedMinutes = persistedStreakMinutes(failureStreakWarnedAt, nowMs);
|
|
3442
3477
|
try {
|
|
3443
3478
|
appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
|
|
3444
3479
|
appendError({
|
|
@@ -3590,6 +3625,8 @@ function buildScheduleStatePayload(state) {
|
|
|
3590
3625
|
launchBlocks: state.launchBlocks ?? {},
|
|
3591
3626
|
launchMitigations: state.launchMitigations ?? {},
|
|
3592
3627
|
utilization: cachedUtilization,
|
|
3628
|
+
utilizationWindow: cachedBindingWindowName,
|
|
3629
|
+
utilizationHold,
|
|
3593
3630
|
pollHealth: {
|
|
3594
3631
|
lastPollAt,
|
|
3595
3632
|
lastPollOk,
|
|
@@ -6596,6 +6633,117 @@ async function handleLaunchFailure({ job, res, runId, runDir, launchKey, launchE
|
|
|
6596
6633
|
await broadcast({ flush: true });
|
|
6597
6634
|
}
|
|
6598
6635
|
|
|
6636
|
+
// Scheduler-scoped error sink for failures that would otherwise be invisible
|
|
6637
|
+
// in packaged/npx builds (stdout unread). Never throws.
|
|
6638
|
+
function reportSchedulerError(message, slug, e) {
|
|
6639
|
+
try {
|
|
6640
|
+
logs.writeLine({
|
|
6641
|
+
scope: 'scheduler',
|
|
6642
|
+
level: 'error',
|
|
6643
|
+
message,
|
|
6644
|
+
meta: { slug, error: e?.message || String(e), stack: e?.stack },
|
|
6645
|
+
});
|
|
6646
|
+
} catch { /* logging must never be the thing that fails */ }
|
|
6647
|
+
try {
|
|
6648
|
+
appendAuditEvent('scheduler_error', { slug, message, error: e?.message || String(e), stack: e?.stack });
|
|
6649
|
+
} catch { /* same */ }
|
|
6650
|
+
}
|
|
6651
|
+
|
|
6652
|
+
/**
|
|
6653
|
+
* Salvage, integrate and clean up a job's throwaway worktree once its run has
|
|
6654
|
+
* ended. NEVER throws: any rejection (salvage / integrate / cleanup) is
|
|
6655
|
+
* reported through deps.reportSchedulerError and surfaces as
|
|
6656
|
+
* `worktreeIntegrationFailure`, so spawnJob's finalize mutate always runs and
|
|
6657
|
+
* the job can never be left `running`. The branch is kept on every failure.
|
|
6658
|
+
* @returns {Promise<{worktreeLeftoverDirty: string[], salvagePatch: string|null,
|
|
6659
|
+
* worktreeIntegrationFailure: string|null, worktreeIntegrationDetail: object|null,
|
|
6660
|
+
* mergeAutoResolved: string|null, mergeAutoResolvedPaths: string[]|null}>}
|
|
6661
|
+
*/
|
|
6662
|
+
async function finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths, deps = {} }) {
|
|
6663
|
+
const jw = deps.jobWorktree || jobWorktree;
|
|
6664
|
+
const uncommitted = deps.uncommittedChanges || uncommittedChanges;
|
|
6665
|
+
const report = deps.reportSchedulerError || reportSchedulerError;
|
|
6666
|
+
let worktreeLeftoverDirty = [];
|
|
6667
|
+
let salvagePatch = null;
|
|
6668
|
+
let worktreeIntegrationFailure = null;
|
|
6669
|
+
let worktreeIntegrationDetail = null;
|
|
6670
|
+
let mergeAutoResolved = null;
|
|
6671
|
+
let mergeAutoResolvedPaths = null;
|
|
6672
|
+
try {
|
|
6673
|
+
worktreeLeftoverDirty = (await uncommitted(worktree.dir)) || [];
|
|
6674
|
+
// Salvage the worktree's full diff (tracked + untracked) to the run
|
|
6675
|
+
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
6676
|
+
// before its finish-protocol commit loses that work outright, with
|
|
6677
|
+
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
6678
|
+
// integration/cleanup and never changes the job's verdict.
|
|
6679
|
+
if (worktreeLeftoverDirty.length) {
|
|
6680
|
+
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
6681
|
+
const salvage = await jw.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
6682
|
+
if (salvage && salvage.ok) {
|
|
6683
|
+
salvagePatch = salvagePath;
|
|
6684
|
+
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
6685
|
+
}
|
|
6686
|
+
}
|
|
6687
|
+
const integration = await jw.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
6688
|
+
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
6689
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
6690
|
+
}
|
|
6691
|
+
if (!integration.ok) {
|
|
6692
|
+
worktreeIntegrationFailure = integration.reason;
|
|
6693
|
+
worktreeIntegrationDetail = integration;
|
|
6694
|
+
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
6695
|
+
} else if (integration.integrated) {
|
|
6696
|
+
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
6697
|
+
if (integration.autoResolved) {
|
|
6698
|
+
mergeAutoResolved = integration.autoResolved;
|
|
6699
|
+
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
6700
|
+
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
6701
|
+
}
|
|
6702
|
+
}
|
|
6703
|
+
await jw.cleanupJobWorktree({
|
|
6704
|
+
cwd: guardCwd,
|
|
6705
|
+
dir: worktree.dir,
|
|
6706
|
+
branch: worktree.branch,
|
|
6707
|
+
keepBranch: !integration.ok,
|
|
6708
|
+
});
|
|
6709
|
+
} catch (e) {
|
|
6710
|
+
worktreeIntegrationFailure = e?.message || String(e);
|
|
6711
|
+
report('spawnJob worktree finalize failed', job.slug, e);
|
|
6712
|
+
// Best-effort: release the checkout + worktree-cap slot, keep the branch.
|
|
6713
|
+
try {
|
|
6714
|
+
await jw.cleanupJobWorktree({ cwd: guardCwd, dir: worktree.dir, branch: worktree.branch, keepBranch: true });
|
|
6715
|
+
} catch { /* already reported above */ }
|
|
6716
|
+
}
|
|
6717
|
+
return { worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths };
|
|
6718
|
+
}
|
|
6719
|
+
|
|
6720
|
+
/**
|
|
6721
|
+
* Map a worktree integration failure onto the verifier verdict spawnJob stamps
|
|
6722
|
+
* (pure). Null failure -> null (no override). Always downgrades to needs_review.
|
|
6723
|
+
*/
|
|
6724
|
+
function worktreeIntegrationVerdict({ failure, detail, slug }) {
|
|
6725
|
+
if (!failure) return null;
|
|
6726
|
+
return {
|
|
6727
|
+
verdict: 'worktree_integration_failed',
|
|
6728
|
+
reason: detail && detail.failureKind === 'content_conflict'
|
|
6729
|
+
? `Integration blocked by a content conflict in ${(detail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(slug)} preserved; needs a manual merge.`
|
|
6730
|
+
: `worktree branch integration failed: ${failure} — branch preserved for manual merge`,
|
|
6731
|
+
downgradeTo: 'needs_review',
|
|
6732
|
+
};
|
|
6733
|
+
}
|
|
6734
|
+
|
|
6735
|
+
/**
|
|
6736
|
+
* Run one interval tick; a throw is reported and swallowed so the interval
|
|
6737
|
+
* keeps firing.
|
|
6738
|
+
*/
|
|
6739
|
+
function guardedTick(fn, label) {
|
|
6740
|
+
try {
|
|
6741
|
+
fn();
|
|
6742
|
+
} catch (e) {
|
|
6743
|
+
reportSchedulerError(label, null, e);
|
|
6744
|
+
}
|
|
6745
|
+
}
|
|
6746
|
+
|
|
6599
6747
|
async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
6600
6748
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
6601
6749
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
@@ -6971,42 +7119,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6971
7119
|
});
|
|
6972
7120
|
} finally {
|
|
6973
7121
|
if (worktree.ok) {
|
|
6974
|
-
worktreeLeftoverDirty
|
|
6975
|
-
|
|
6976
|
-
// dir BEFORE the checkout is removed below — otherwise a job killed
|
|
6977
|
-
// before its finish-protocol commit loses that work outright, with
|
|
6978
|
-
// no branch, no stash, no patch anywhere. Best-effort: never blocks
|
|
6979
|
-
// integration/cleanup and never changes the job's verdict.
|
|
6980
|
-
if (worktreeLeftoverDirty.length) {
|
|
6981
|
-
const salvagePath = path.join(runDir, `${job.slug}.uncommitted.patch`);
|
|
6982
|
-
const salvage = await jobWorktree.salvageJobWorktreeDiff({ dir: worktree.dir, outFile: salvagePath });
|
|
6983
|
-
if (salvage && salvage.ok) {
|
|
6984
|
-
salvagePatch = salvagePath;
|
|
6985
|
-
console.log(`[scheduler] ${job.slug}: salvaged ${salvage.bytes} byte(s) of uncommitted worktree diff to ${salvagePath}`);
|
|
6986
|
-
}
|
|
6987
|
-
}
|
|
6988
|
-
const integration = await jobWorktree.integrateJobBranch({ cwd: guardCwd, branch: worktree.branch, slug: job.slug, carriedPaths });
|
|
6989
|
-
if (integration.ok && integration.reason === 'carried-wip-only') {
|
|
6990
|
-
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} touched only carried base WIP paths — skipping merge (carried-wip-only)`);
|
|
6991
|
-
}
|
|
6992
|
-
if (!integration.ok) {
|
|
6993
|
-
worktreeIntegrationFailure = integration.reason;
|
|
6994
|
-
worktreeIntegrationDetail = integration;
|
|
6995
|
-
console.error(`[scheduler] ${job.slug}: worktree branch integration FAILED (${integration.reason}) — branch ${worktree.branch} preserved in ${guardCwd} for manual recovery`);
|
|
6996
|
-
} else if (integration.integrated) {
|
|
6997
|
-
console.log(`[scheduler] ${job.slug}: worktree branch ${worktree.branch} integrated into ${guardCwd}${integration.mergeCommit ? ' (merge commit)' : ' (fast-forward)'}`);
|
|
6998
|
-
if (integration.autoResolved) {
|
|
6999
|
-
mergeAutoResolved = integration.autoResolved;
|
|
7000
|
-
mergeAutoResolvedPaths = integration.resolvedPaths || [];
|
|
7001
|
-
console.log(`[scheduler] ${job.slug}: merge auto-resolved (${integration.autoResolved}) — discarded ${mergeAutoResolvedPaths.length} identical working-tree duplicate(s): ${mergeAutoResolvedPaths.join(', ')}`);
|
|
7002
|
-
}
|
|
7003
|
-
}
|
|
7004
|
-
await jobWorktree.cleanupJobWorktree({
|
|
7005
|
-
cwd: guardCwd,
|
|
7006
|
-
dir: worktree.dir,
|
|
7007
|
-
branch: worktree.branch,
|
|
7008
|
-
keepBranch: !integration.ok,
|
|
7009
|
-
});
|
|
7122
|
+
({ worktreeLeftoverDirty, salvagePatch, worktreeIntegrationFailure, worktreeIntegrationDetail, mergeAutoResolved, mergeAutoResolvedPaths } =
|
|
7123
|
+
await finalizeJobWorktree({ job, runDir, worktree, guardCwd, carriedPaths }));
|
|
7010
7124
|
} else {
|
|
7011
7125
|
// In-place run (non-git cwd, cap reached, env-disabled, or a carry-over
|
|
7012
7126
|
// failure) — there is no throwaway checkout to diff, so salvage only
|
|
@@ -7279,13 +7393,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
7279
7393
|
// guard AC explicitly requires this failure be surfaced as an explicit job
|
|
7280
7394
|
// outcome, never silently dropped alongside the branch it's stranded on.
|
|
7281
7395
|
if (worktreeIntegrationFailure) {
|
|
7282
|
-
verifyResult = {
|
|
7283
|
-
verdict: 'worktree_integration_failed',
|
|
7284
|
-
reason: worktreeIntegrationDetail && worktreeIntegrationDetail.failureKind === 'content_conflict'
|
|
7285
|
-
? `Integration blocked by a content conflict in ${(worktreeIntegrationDetail.conflictedPaths || []).join(', ') || 'unknown paths'} — branch ${jobWorktree.branchNameFor(job.slug)} preserved; needs a manual merge.`
|
|
7286
|
-
: `worktree branch integration failed: ${worktreeIntegrationFailure} — branch preserved for manual merge`,
|
|
7287
|
-
downgradeTo: 'needs_review',
|
|
7288
|
-
};
|
|
7396
|
+
verifyResult = worktreeIntegrationVerdict({ failure: worktreeIntegrationFailure, detail: worktreeIntegrationDetail, slug: job.slug });
|
|
7289
7397
|
}
|
|
7290
7398
|
|
|
7291
7399
|
// Shared-tree stash guard (incident 2026-09-01): only meaningful for an
|
|
@@ -7943,6 +8051,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
7943
8051
|
}
|
|
7944
8052
|
} catch (e) {
|
|
7945
8053
|
console.error('[scheduler] spawnJob error', job.slug, e);
|
|
8054
|
+
reportSchedulerError('spawnJob error', job.slug, e);
|
|
7946
8055
|
} finally {
|
|
7947
8056
|
runningSet.delete(job.slug);
|
|
7948
8057
|
// Slot release notifies subscribed pumps (chat lane) machine-wide.
|
|
@@ -8302,7 +8411,7 @@ async function tickBody(gen, { bypassLoadGate }) {
|
|
|
8302
8411
|
for (const job of gatedBatch) {
|
|
8303
8412
|
if (cancelToken.cancelled || stale()) break;
|
|
8304
8413
|
// spawnJob is fire-and-forget; it calls tickQueue() on completion.
|
|
8305
|
-
spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() =>
|
|
8414
|
+
spawnJob(job, runId, runDir, state.config.defaultCwd).catch((e) => reportSchedulerError('spawnJob dispatch rejected', job.slug, e));
|
|
8306
8415
|
}
|
|
8307
8416
|
return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
|
|
8308
8417
|
}
|
|
@@ -8361,12 +8470,20 @@ async function runDueJobs({ bypassLoadGate = false } = {}) {
|
|
|
8361
8470
|
// ---------- when-available launch logic ----------
|
|
8362
8471
|
|
|
8363
8472
|
async function maybeLaunchWhenAvailable(state) {
|
|
8473
|
+
utilizationHold = null;
|
|
8364
8474
|
if (state.config.firePolicy !== 'when-available') return;
|
|
8365
8475
|
if (state.paused) return;
|
|
8366
8476
|
const pending = state.jobs.filter((j) => j.status === 'pending' && !runningSet.has(j.slug));
|
|
8367
8477
|
if (pending.length === 0) return;
|
|
8368
8478
|
if (cachedUtilization === null || cachedUtilization === undefined) return;
|
|
8369
8479
|
if (cachedUtilization >= state.config.utilizationThreshold) {
|
|
8480
|
+
utilizationHold = {
|
|
8481
|
+
window: cachedBindingWindowName,
|
|
8482
|
+
percent: cachedUtilization,
|
|
8483
|
+
threshold: state.config.utilizationThreshold,
|
|
8484
|
+
resetsAt: cachedNextReset,
|
|
8485
|
+
};
|
|
8486
|
+
console.log(`[scheduler] when-available: utilization-held — ${cachedBindingWindowName ?? 'unknown'} window at ${cachedUtilization}% ≥ ${state.config.utilizationThreshold}%, resets ${cachedNextReset ?? 'unknown'}, ${pending.length} pending — holding, not ticking`);
|
|
8370
8487
|
await broadcast();
|
|
8371
8488
|
return;
|
|
8372
8489
|
}
|
|
@@ -9432,10 +9549,26 @@ async function pollLoop() {
|
|
|
9432
9549
|
|
|
9433
9550
|
const r = await billing.fetchUsage();
|
|
9434
9551
|
|
|
9552
|
+
if (r.kind === 'ok' && !Number.isFinite(bindingWindow(r.data?.usage).utilization)) {
|
|
9553
|
+
// Successful fetch but no finite percent: a meter/contract problem, not a
|
|
9554
|
+
// reading. Report it truthfully instead of leaving a stale number that
|
|
9555
|
+
// looks live; carry the conservative degraded budget forward.
|
|
9556
|
+
billing.usageCircuit.recordFailure('no_utilization');
|
|
9557
|
+
applyDegradedBudget();
|
|
9558
|
+
lastPollAt = Date.now();
|
|
9559
|
+
lastPollOk = false;
|
|
9560
|
+
persistSchedulerState();
|
|
9561
|
+
const cur = await readQueue();
|
|
9562
|
+
await maybeLaunchWhenAvailable(cur);
|
|
9563
|
+
await broadcast();
|
|
9564
|
+
return;
|
|
9565
|
+
}
|
|
9566
|
+
|
|
9435
9567
|
if (r.kind === 'ok') {
|
|
9436
9568
|
const window = bindingWindow(r.data?.usage);
|
|
9437
9569
|
recordObservedReset(window.resets_at ?? null);
|
|
9438
|
-
cachedUtilization =
|
|
9570
|
+
cachedUtilization = window.utilization;
|
|
9571
|
+
cachedBindingWindowName = window.name;
|
|
9439
9572
|
lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
|
|
9440
9573
|
degradedConcurrencyCapValue = null;
|
|
9441
9574
|
consecutiveFailures = 0;
|
|
@@ -11426,6 +11559,223 @@ function stop() {
|
|
|
11426
11559
|
stopDispatchLoop();
|
|
11427
11560
|
}
|
|
11428
11561
|
|
|
11562
|
+
// Body of the 10-minute maintenance interval (self-heal, escalations, restores).
|
|
11563
|
+
// Extracted so a throw is testable through guardedTick.
|
|
11564
|
+
function rescheduleIntervalTick() {
|
|
11565
|
+
rescheduleTimer().catch(() => {});
|
|
11566
|
+
const s = readQueueSync();
|
|
11567
|
+
// Periodic self-heal: re-run the verifier over stale needs_review jobs so a
|
|
11568
|
+
// job whose work actually landed (committed in-window, no FAIL sentinel)
|
|
11569
|
+
// auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
|
|
11570
|
+
// shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
|
|
11571
|
+
// and the candidate filter can never drift apart again (they did once —
|
|
11572
|
+
// see that function's comment). Kill-switch:
|
|
11573
|
+
// SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
|
|
11574
|
+
// reverifyNeedsReview's auto-fix loop is capped downstream by
|
|
11575
|
+
// MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
|
|
11576
|
+
// past it), so this interval firing cannot fan out investigations.
|
|
11577
|
+
if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
|
|
11578
|
+
if (shouldRunPeriodicReverify(s.jobs)) {
|
|
11579
|
+
reverifyNeedsReview().catch(() => {});
|
|
11580
|
+
}
|
|
11581
|
+
// A quarantined row only ever promotes to 'pending' through
|
|
11582
|
+
// reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
|
|
11583
|
+
// it re-checks the PRD file's createdVia stamp every pass. broadcast()
|
|
11584
|
+
// already runs reconcile+writeQueue on every normal poll tick, but an
|
|
11585
|
+
// idle queue (nothing pending/running to fire) can back off that
|
|
11586
|
+
// cadence for a long time; this guarantees an adopted-but-still-
|
|
11587
|
+
// quarantined row is re-checked within 10 minutes regardless.
|
|
11588
|
+
if (s.jobs.some((j) => j.status === 'quarantined')) {
|
|
11589
|
+
broadcast().catch(() => {});
|
|
11590
|
+
}
|
|
11591
|
+
}
|
|
11592
|
+
// Age-based escalation (independent of the self-heal kill-switch above —
|
|
11593
|
+
// this is a monitoring signal, not an auto-fix action): a quarantined
|
|
11594
|
+
// row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
|
|
11595
|
+
// warn-logged by project + slug + age so it cannot sit stranded and
|
|
11596
|
+
// silent (the four burrow-project rows this PRD was written against).
|
|
11597
|
+
for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
|
|
11598
|
+
console.warn(
|
|
11599
|
+
`[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
|
|
11600
|
+
+ `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11601
|
+
+ `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
|
|
11602
|
+
);
|
|
11603
|
+
appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
|
|
11604
|
+
}
|
|
11605
|
+
|
|
11606
|
+
// Estimate-relative overrun escalation. Sits in the blind spot between
|
|
11607
|
+
// the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
|
|
11608
|
+
// producing output while looping trips neither, so nothing noticed a PRD
|
|
11609
|
+
// running 9x its own estimate until a human went looking. Escalate loudly;
|
|
11610
|
+
// never kill on an estimate (see JOB_OVERRUN_FACTOR).
|
|
11611
|
+
for (const over of findOverrunningJobs(s.jobs, Date.now())) {
|
|
11612
|
+
console.warn(
|
|
11613
|
+
`[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
|
|
11614
|
+
+ `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
|
|
11615
|
+
+ `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
|
|
11616
|
+
+ `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
|
|
11617
|
+
+ `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
|
|
11618
|
+
);
|
|
11619
|
+
appendAuditEvent('job_overrunning_estimate', {
|
|
11620
|
+
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
11621
|
+
});
|
|
11622
|
+
// Durable stamp so schedule:state (and therefore the renderer) can see
|
|
11623
|
+
// this without re-deriving it — the console.warn/audit event above are
|
|
11624
|
+
// visible only in the log, never on the row itself. Display-only
|
|
11625
|
+
// advisory field; re-stamped in place every sweep, never appended.
|
|
11626
|
+
mutate((state) => {
|
|
11627
|
+
const j = state.jobs.find((x) => x.slug === over.slug);
|
|
11628
|
+
if (!j) return;
|
|
11629
|
+
j.overrun = {
|
|
11630
|
+
ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
|
|
11631
|
+
};
|
|
11632
|
+
}).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
|
|
11633
|
+
}
|
|
11634
|
+
|
|
11635
|
+
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
11636
|
+
// one ACTS: 'investigating' is a transient status whose restore
|
|
11637
|
+
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
11638
|
+
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
11639
|
+
// there forever (see findStrandedInvestigations' header, and the
|
|
11640
|
+
// "'investigating' must never be the job's resting state" comment at
|
|
11641
|
+
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
11642
|
+
// exact terminal status it already carried before the probe was
|
|
11643
|
+
// spawned — it never re-runs or re-investigates anything.
|
|
11644
|
+
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
11645
|
+
if (stranded.length > 0) {
|
|
11646
|
+
mutate((ms) => {
|
|
11647
|
+
for (const st of stranded) {
|
|
11648
|
+
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
11649
|
+
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
11650
|
+
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
11651
|
+
delete j.runtime;
|
|
11652
|
+
console.warn(
|
|
11653
|
+
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
11654
|
+
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
11655
|
+
+ `restored to '${st.restoreStatus}'`,
|
|
11656
|
+
);
|
|
11657
|
+
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
11658
|
+
}
|
|
11659
|
+
})
|
|
11660
|
+
.then(() => broadcast({ flush: true }))
|
|
11661
|
+
.catch(() => {});
|
|
11662
|
+
}
|
|
11663
|
+
|
|
11664
|
+
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
11665
|
+
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
11666
|
+
// else distinguishes "no pending work" from "pending work, never
|
|
11667
|
+
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
11668
|
+
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
11669
|
+
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
11670
|
+
for (const sp of starvedProjects) {
|
|
11671
|
+
console.warn(
|
|
11672
|
+
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
11673
|
+
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
11674
|
+
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
11675
|
+
);
|
|
11676
|
+
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
11677
|
+
}
|
|
11678
|
+
// Bounded, automated consequence for a starve that outlives the WARN
|
|
11679
|
+
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
11680
|
+
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
11681
|
+
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
11682
|
+
// a subset of the rows already reported above — same verdict, no
|
|
11683
|
+
// re-derivation.
|
|
11684
|
+
runStarveEscalationSweep(starvedProjects);
|
|
11685
|
+
|
|
11686
|
+
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
11687
|
+
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
11688
|
+
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
11689
|
+
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
11690
|
+
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
11691
|
+
// above it can never observe two different snapshots of the same row.
|
|
11692
|
+
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
11693
|
+
const autoResetTargets = failedAutoResetDisabled()
|
|
11694
|
+
? []
|
|
11695
|
+
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
11696
|
+
const stuckFailed = stuckFailedEscalationDisabled()
|
|
11697
|
+
? []
|
|
11698
|
+
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
11699
|
+
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
11700
|
+
// (this PRD): computed alongside the failed-row passes above and acted
|
|
11701
|
+
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
11702
|
+
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
11703
|
+
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
11704
|
+
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
11705
|
+
? []
|
|
11706
|
+
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
11707
|
+
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
11708
|
+
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
11709
|
+
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
11710
|
+
// never be read from one snapshot and written from another, and the
|
|
11711
|
+
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
11712
|
+
// same turn as the transition it gates. Kill-switch:
|
|
11713
|
+
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
11714
|
+
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
11715
|
+
? []
|
|
11716
|
+
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
11717
|
+
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
11718
|
+
mutate(async (ms) => {
|
|
11719
|
+
for (const target of autoResetTargets) {
|
|
11720
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11721
|
+
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
11722
|
+
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
11723
|
+
j.failedAutoResetAttempts = attempt;
|
|
11724
|
+
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
11725
|
+
// resetJobFields is the same field-clearing list the admin
|
|
11726
|
+
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
11727
|
+
// it rather than inventing a second list. It also sets job.error to
|
|
11728
|
+
// the reason text passed in; we clear that back to null right
|
|
11729
|
+
// after since this is a clean auto-reset, not a recorded error.
|
|
11730
|
+
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
11731
|
+
j.error = null;
|
|
11732
|
+
delete j.stuckFailedNotified;
|
|
11733
|
+
console.warn(
|
|
11734
|
+
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11735
|
+
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
11736
|
+
);
|
|
11737
|
+
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
11738
|
+
}
|
|
11739
|
+
for (const stuck of stuckFailed) {
|
|
11740
|
+
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
11741
|
+
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
11742
|
+
// Still has auto-reset attempts left — it will be (or already was,
|
|
11743
|
+
// earlier this same pass) picked up by the loop above instead.
|
|
11744
|
+
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
11745
|
+
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
11746
|
+
j.stuckFailedNotified = true;
|
|
11747
|
+
console.warn(
|
|
11748
|
+
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
11749
|
+
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11750
|
+
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
11751
|
+
);
|
|
11752
|
+
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
11753
|
+
}
|
|
11754
|
+
for (const target of exhaustedNeedsReviewTargets) {
|
|
11755
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11756
|
+
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
11757
|
+
if (outcome) {
|
|
11758
|
+
console.warn(
|
|
11759
|
+
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11760
|
+
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
11761
|
+
);
|
|
11762
|
+
}
|
|
11763
|
+
}
|
|
11764
|
+
for (const target of quarantineTargets) {
|
|
11765
|
+
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11766
|
+
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
11767
|
+
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
11768
|
+
if (outcome) {
|
|
11769
|
+
console.warn(
|
|
11770
|
+
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11771
|
+
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
11772
|
+
);
|
|
11773
|
+
}
|
|
11774
|
+
}
|
|
11775
|
+
}).catch(() => {});
|
|
11776
|
+
}
|
|
11777
|
+
}
|
|
11778
|
+
|
|
11429
11779
|
async function init() {
|
|
11430
11780
|
ensureDirs();
|
|
11431
11781
|
// Boot phase — reconciliation, migrations, self-heal, first reset probe.
|
|
@@ -11626,218 +11976,9 @@ async function init() {
|
|
|
11626
11976
|
// resets early or the auth token rotates. Tracked so re-init doesn't leak.
|
|
11627
11977
|
if (rescheduleInterval) clearInterval(rescheduleInterval);
|
|
11628
11978
|
rescheduleInterval = setInterval(() => {
|
|
11629
|
-
|
|
11630
|
-
|
|
11631
|
-
|
|
11632
|
-
// job whose work actually landed (committed in-window, no FAIL sentinel)
|
|
11633
|
-
// auto-clears WITHOUT waiting for the next app restart. Cheap-guarded by
|
|
11634
|
-
// shouldRunPeriodicReverify, which reuses isRescanCandidate so the guard
|
|
11635
|
-
// and the candidate filter can never drift apart again (they did once —
|
|
11636
|
-
// see that function's comment). Kill-switch:
|
|
11637
|
-
// SM_REVERIFY_PERIODIC_DISABLE=1 (boot reverify above stays always-on).
|
|
11638
|
-
// reverifyNeedsReview's auto-fix loop is capped downstream by
|
|
11639
|
-
// MAX_CONCURRENT_INVESTIGATIONS (spawnInvestigation queues/early-returns
|
|
11640
|
-
// past it), so this interval firing cannot fan out investigations.
|
|
11641
|
-
if (process.env.SM_REVERIFY_PERIODIC_DISABLE !== '1') {
|
|
11642
|
-
if (shouldRunPeriodicReverify(s.jobs)) {
|
|
11643
|
-
reverifyNeedsReview().catch(() => {});
|
|
11644
|
-
}
|
|
11645
|
-
// A quarantined row only ever promotes to 'pending' through
|
|
11646
|
-
// reconcile()'s adopt path (see reconcile()'s "Adopt path" comment) —
|
|
11647
|
-
// it re-checks the PRD file's createdVia stamp every pass. broadcast()
|
|
11648
|
-
// already runs reconcile+writeQueue on every normal poll tick, but an
|
|
11649
|
-
// idle queue (nothing pending/running to fire) can back off that
|
|
11650
|
-
// cadence for a long time; this guarantees an adopted-but-still-
|
|
11651
|
-
// quarantined row is re-checked within 10 minutes regardless.
|
|
11652
|
-
if (s.jobs.some((j) => j.status === 'quarantined')) {
|
|
11653
|
-
broadcast().catch(() => {});
|
|
11654
|
-
}
|
|
11655
|
-
}
|
|
11656
|
-
// Age-based escalation (independent of the self-heal kill-switch above —
|
|
11657
|
-
// this is a monitoring signal, not an auto-fix action): a quarantined
|
|
11658
|
-
// row nobody has adopted or archived past QUARANTINE_ESCALATE_MS is
|
|
11659
|
-
// warn-logged by project + slug + age so it cannot sit stranded and
|
|
11660
|
-
// silent (the four burrow-project rows this PRD was written against).
|
|
11661
|
-
for (const stale of findStaleQuarantinedJobs(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS)) {
|
|
11662
|
-
console.warn(
|
|
11663
|
-
`[scheduler] QUARANTINED PRD STALE: project=${stale.cwd ?? '(unknown)'} slug=${stale.slug} `
|
|
11664
|
-
+ `age=${Math.round(stale.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11665
|
-
+ `adopt it from the Scheduler tab's Quarantined filter, or archive it; nothing else will clear this`,
|
|
11666
|
-
);
|
|
11667
|
-
appendAuditEvent('prd_quarantine_stale', { slug: stale.slug, cwd: stale.cwd, ageMs: stale.ageMs });
|
|
11668
|
-
}
|
|
11669
|
-
|
|
11670
|
-
// Estimate-relative overrun escalation. Sits in the blind spot between
|
|
11671
|
-
// the 4h deadman and the 20-minute idle-output watchdog: a job that keeps
|
|
11672
|
-
// producing output while looping trips neither, so nothing noticed a PRD
|
|
11673
|
-
// running 9x its own estimate until a human went looking. Escalate loudly;
|
|
11674
|
-
// never kill on an estimate (see JOB_OVERRUN_FACTOR).
|
|
11675
|
-
for (const over of findOverrunningJobs(s.jobs, Date.now())) {
|
|
11676
|
-
console.warn(
|
|
11677
|
-
`[scheduler] JOB OVERRUNNING ESTIMATE: project=${over.cwd ?? '(unknown)'} slug=${over.slug} `
|
|
11678
|
-
+ `ran=${Math.round(over.ranMs / 60_000)}m vs estimate=${over.estimateMinutes}m `
|
|
11679
|
-
+ `(${over.ratio.toFixed(1)}x, threshold ${JOB_OVERRUN_FACTOR}x floor ${Math.round(JOB_OVERRUN_FLOOR_MS / 60_000)}m) — `
|
|
11680
|
-
+ `still running; the ${Math.round(MAX_JOB_DURATION_MS / 3_600_000)}h deadman has NOT fired yet. `
|
|
11681
|
-
+ `Check the run log, then let it finish or cancel it via scheduler_cancel_job`,
|
|
11682
|
-
);
|
|
11683
|
-
appendAuditEvent('job_overrunning_estimate', {
|
|
11684
|
-
slug: over.slug, cwd: over.cwd, estimateMinutes: over.estimateMinutes, ranMs: over.ranMs, ratio: over.ratio,
|
|
11685
|
-
});
|
|
11686
|
-
// Durable stamp so schedule:state (and therefore the renderer) can see
|
|
11687
|
-
// this without re-deriving it — the console.warn/audit event above are
|
|
11688
|
-
// visible only in the log, never on the row itself. Display-only
|
|
11689
|
-
// advisory field; re-stamped in place every sweep, never appended.
|
|
11690
|
-
mutate((state) => {
|
|
11691
|
-
const j = state.jobs.find((x) => x.slug === over.slug);
|
|
11692
|
-
if (!j) return;
|
|
11693
|
-
j.overrun = {
|
|
11694
|
-
ratio: over.ratio, ranMs: over.ranMs, estimateMinutes: over.estimateMinutes, at: new Date().toISOString(),
|
|
11695
|
-
};
|
|
11696
|
-
}).catch((e) => console.warn('[scheduler] overrun stamp failed', e?.message));
|
|
11697
|
-
}
|
|
11698
|
-
|
|
11699
|
-
// Stranded-investigation restore. Unlike the two escalations above, this
|
|
11700
|
-
// one ACTS: 'investigating' is a transient status whose restore
|
|
11701
|
-
// (spawnInvestigation's onExit/catch) only runs inside the process that
|
|
11702
|
-
// spawned the probe, so an app restart mid-probe leaves the row frozen
|
|
11703
|
-
// there forever (see findStrandedInvestigations' header, and the
|
|
11704
|
-
// "'investigating' must never be the job's resting state" comment at
|
|
11705
|
-
// spawnInvestigation's onExit). This restores each stranded row to the
|
|
11706
|
-
// exact terminal status it already carried before the probe was
|
|
11707
|
-
// spawned — it never re-runs or re-investigates anything.
|
|
11708
|
-
const stranded = findStrandedInvestigations(s.jobs, Date.now(), INVESTIGATION_MAX_MS);
|
|
11709
|
-
if (stranded.length > 0) {
|
|
11710
|
-
mutate((ms) => {
|
|
11711
|
-
for (const st of stranded) {
|
|
11712
|
-
const j = ms.jobs.find((x) => x.slug === st.slug);
|
|
11713
|
-
if (!j || j.status !== 'investigating') continue; // race guard — may have resolved since the scan above
|
|
11714
|
-
transitionJob(j, st.restoreStatus, { reason: `stranded investigation restored after ${Math.round(st.ageMs / 60_000)}m with no live probe behind it`, source: 'findStrandedInvestigations' });
|
|
11715
|
-
delete j.runtime;
|
|
11716
|
-
console.warn(
|
|
11717
|
-
`[scheduler] STRANDED INVESTIGATION RESTORED: project=${st.cwd ?? '(unknown)'} slug=${st.slug} `
|
|
11718
|
-
+ `age=${Math.round(st.ageMs / 3_600_000)}h (>= ${Math.round(INVESTIGATION_MAX_MS / 3_600_000)}h threshold), no live probe — `
|
|
11719
|
-
+ `restored to '${st.restoreStatus}'`,
|
|
11720
|
-
);
|
|
11721
|
-
appendAuditEvent('investigation_stranded_restored', { slug: st.slug, cwd: st.cwd, ageMs: st.ageMs, restoreStatus: st.restoreStatus });
|
|
11722
|
-
}
|
|
11723
|
-
})
|
|
11724
|
-
.then(() => broadcast({ flush: true }))
|
|
11725
|
-
.catch(() => {});
|
|
11726
|
-
}
|
|
11727
|
-
|
|
11728
|
-
// Per-project starvation (PRD 1087): a project with pending work that has
|
|
11729
|
-
// been passed over on every tick while OTHER projects dispatch. Nothing
|
|
11730
|
-
// else distinguishes "no pending work" from "pending work, never
|
|
11731
|
-
// started" — the 2026-09-01 NN-ordering starvation ran 3.5 h unnoticed.
|
|
11732
|
-
// Escalation only, same shape as the quarantine/overrun warnings above.
|
|
11733
|
-
const starvedProjects = findStarvedProjects(s.jobs, Date.now(), STARVATION_ESCALATE_MS);
|
|
11734
|
-
for (const sp of starvedProjects) {
|
|
11735
|
-
console.warn(
|
|
11736
|
-
`[scheduler] PROJECT STARVED: project=${sp.cwd} pending=${sp.pendingCount} oldest=${sp.oldestPendingSlug} `
|
|
11737
|
-
+ `waiting=${Math.round(sp.ageMs / 60_000)}m (>= ${Math.round(STARVATION_ESCALATE_MS / 60_000)}m threshold) `
|
|
11738
|
-
+ `while other projects are running — check the cross-project fairness rule in pickNextBatch`,
|
|
11739
|
-
);
|
|
11740
|
-
appendAuditEvent('project_starved', { cwd: sp.cwd, pendingCount: sp.pendingCount, oldestPendingSlug: sp.oldestPendingSlug, ageMs: sp.ageMs });
|
|
11741
|
-
}
|
|
11742
|
-
// Bounded, automated consequence for a starve that outlives the WARN
|
|
11743
|
-
// above (PRD: the 2026-09-12 19h Bilko starve had ~115 identical
|
|
11744
|
-
// project_starved rows and zero consequence). STARVE_ESCALATION_MS is
|
|
11745
|
-
// strictly later than STARVATION_ESCALATE_MS, so this only ever fires on
|
|
11746
|
-
// a subset of the rows already reported above — same verdict, no
|
|
11747
|
-
// re-derivation.
|
|
11748
|
-
runStarveEscalationSweep(starvedProjects);
|
|
11749
|
-
|
|
11750
|
-
// Bounded failed -> pending auto-reset (PRD 1151), plus the stuck-failed
|
|
11751
|
-
// escalation now narrowed to only the rows that auto-reset gave up on.
|
|
11752
|
-
// See selectFailedAutoResetTargets' + findStuckFailedJobs' headers.
|
|
11753
|
-
// Computed together, acted on in the SAME mutate(...) pass, so the
|
|
11754
|
-
// stuckFailedNotified race guard below and the auto-reset race guard
|
|
11755
|
-
// above it can never observe two different snapshots of the same row.
|
|
11756
|
-
// Kill-switches: SM_FAILED_AUTORESET_DISABLE=1 / SM_STUCK_FAILED_ESCALATE_DISABLE=1.
|
|
11757
|
-
const autoResetTargets = failedAutoResetDisabled()
|
|
11758
|
-
? []
|
|
11759
|
-
: selectFailedAutoResetTargets(s.jobs, Date.now(), FAILED_AUTORESET_MS);
|
|
11760
|
-
const stuckFailed = stuckFailedEscalationDisabled()
|
|
11761
|
-
? []
|
|
11762
|
-
: findStuckFailedJobs(s.jobs, Date.now(), STUCK_FAILED_ESCALATE_MS);
|
|
11763
|
-
// Bounded automatic terminal decision for exhausted needs_review rows
|
|
11764
|
-
// (this PRD): computed alongside the failed-row passes above and acted
|
|
11765
|
-
// on in the SAME mutate(...) pass below, for the same race-guard reason
|
|
11766
|
-
// — a row's exhaustedResolveAttempts counter must never be read from one
|
|
11767
|
-
// snapshot and written from another. Kill-switch: SM_NEEDS_REVIEW_AUTORESOLVE_DISABLE=1.
|
|
11768
|
-
const exhaustedNeedsReviewTargets = needsReviewAutoResolveDisabled()
|
|
11769
|
-
? []
|
|
11770
|
-
: selectExhaustedNeedsReviewTargets(s.jobs, Date.now(), NEEDS_REVIEW_RESOLVE_MS);
|
|
11771
|
-
// Bounded automatic exit for quarantined rows (this PRD): computed
|
|
11772
|
-
// alongside the passes above and acted on in the SAME mutate(...) pass
|
|
11773
|
-
// below, for the same race-guard reason — quarantineResolveAttempts must
|
|
11774
|
-
// never be read from one snapshot and written from another, and the
|
|
11775
|
-
// createdVia re-check inside autoResolveQuarantine must happen in the
|
|
11776
|
-
// same turn as the transition it gates. Kill-switch:
|
|
11777
|
-
// SM_QUARANTINE_AUTORESOLVE_DISABLE=1.
|
|
11778
|
-
const quarantineTargets = quarantineAutoResolveDisabled()
|
|
11779
|
-
? []
|
|
11780
|
-
: selectQuarantineAutoResolveTargets(s.jobs, Date.now(), QUARANTINE_ESCALATE_MS);
|
|
11781
|
-
if (autoResetTargets.length > 0 || stuckFailed.length > 0 || exhaustedNeedsReviewTargets.length > 0 || quarantineTargets.length > 0) {
|
|
11782
|
-
mutate(async (ms) => {
|
|
11783
|
-
for (const target of autoResetTargets) {
|
|
11784
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11785
|
-
if (!j || j.status !== 'failed' || (j.failedAutoResetAttempts ?? 0) >= FAILED_AUTORESET_CAP) continue; // race guard
|
|
11786
|
-
const attempt = (j.failedAutoResetAttempts ?? 0) + 1;
|
|
11787
|
-
j.failedAutoResetAttempts = attempt;
|
|
11788
|
-
const reason = `auto-reset after ${Math.round(FAILED_AUTORESET_MS / 60_000)}m failed (attempt ${attempt}/${FAILED_AUTORESET_CAP})`;
|
|
11789
|
-
// resetJobFields is the same field-clearing list the admin
|
|
11790
|
-
// scheduler_reset_job handler uses (ipc:schedule:reset-job) — reuse
|
|
11791
|
-
// it rather than inventing a second list. It also sets job.error to
|
|
11792
|
-
// the reason text passed in; we clear that back to null right
|
|
11793
|
-
// after since this is a clean auto-reset, not a recorded error.
|
|
11794
|
-
if (!resetJobFields(j, reason, { source: 'autoResetFailed' })) continue;
|
|
11795
|
-
j.error = null;
|
|
11796
|
-
delete j.stuckFailedNotified;
|
|
11797
|
-
console.warn(
|
|
11798
|
-
`[scheduler] FAILED PRD AUTO-RESET: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11799
|
-
+ `failed=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(FAILED_AUTORESET_MS / 60_000)}m threshold) — ${reason}`,
|
|
11800
|
-
);
|
|
11801
|
-
appendAuditEvent('job_auto_reset_failed', { slug: j.slug, cwd: j.cwd, ageMs: target.ageMs, attempt });
|
|
11802
|
-
}
|
|
11803
|
-
for (const stuck of stuckFailed) {
|
|
11804
|
-
const j = ms.jobs.find((x) => x.slug === stuck.slug);
|
|
11805
|
-
if (!j || j.status !== 'failed' || j.stuckFailedNotified === true) continue; // race guard
|
|
11806
|
-
// Still has auto-reset attempts left — it will be (or already was,
|
|
11807
|
-
// earlier this same pass) picked up by the loop above instead.
|
|
11808
|
-
// Never log "reset it by hand" for a row that isn't actually stuck.
|
|
11809
|
-
if ((j.failedAutoResetAttempts ?? 0) < FAILED_AUTORESET_CAP) continue;
|
|
11810
|
-
j.stuckFailedNotified = true;
|
|
11811
|
-
console.warn(
|
|
11812
|
-
`[scheduler] FAILED PRD STUCK: project=${stuck.cwd ?? '(unknown)'} slug=${stuck.slug} `
|
|
11813
|
-
+ `failed=${Math.round(stuck.ageMs / 3_600_000)}h (>= ${Math.round(STUCK_FAILED_ESCALATE_MS / 3_600_000)}h threshold) — `
|
|
11814
|
-
+ `auto-reset cap exhausted (${FAILED_AUTORESET_CAP}/${FAILED_AUTORESET_CAP} attempts); reset it by hand via scheduler_reset_job`,
|
|
11815
|
-
);
|
|
11816
|
-
appendAuditEvent('job_stuck_failed', { slug: stuck.slug, cwd: stuck.cwd, ageMs: stuck.ageMs });
|
|
11817
|
-
}
|
|
11818
|
-
for (const target of exhaustedNeedsReviewTargets) {
|
|
11819
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11820
|
-
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
11821
|
-
if (outcome) {
|
|
11822
|
-
console.warn(
|
|
11823
|
-
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11824
|
-
+ `exhausted=${Math.round(target.ageMs / 60_000)}m (>= ${Math.round(NEEDS_REVIEW_RESOLVE_MS / 60_000)}m threshold) — outcome=${outcome}`,
|
|
11825
|
-
);
|
|
11826
|
-
}
|
|
11827
|
-
}
|
|
11828
|
-
for (const target of quarantineTargets) {
|
|
11829
|
-
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
11830
|
-
if (!j || j.status !== 'quarantined' || (j.quarantineResolveAttempts ?? 0) >= QUARANTINE_RESOLVE_CAP) continue; // race guard
|
|
11831
|
-
const outcome = await autoResolveQuarantine(j, target.ageMs);
|
|
11832
|
-
if (outcome) {
|
|
11833
|
-
console.warn(
|
|
11834
|
-
`[scheduler] QUARANTINED PRD AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
11835
|
-
+ `age=${Math.round(target.ageMs / 3_600_000)}h (>= ${Math.round(QUARANTINE_ESCALATE_MS / 3_600_000)}h threshold) — outcome=${outcome}`,
|
|
11836
|
-
);
|
|
11837
|
-
}
|
|
11838
|
-
}
|
|
11839
|
-
}).catch(() => {});
|
|
11840
|
-
}
|
|
11979
|
+
// One throwing tick (e.g. readQueueSync on a torn queue.json) must skip
|
|
11980
|
+
// only itself — the interval keeps firing and the failure is logged.
|
|
11981
|
+
guardedTick(rescheduleIntervalTick, 'rescheduleInterval tick failed');
|
|
11841
11982
|
}, REVERIFY_INTERVAL_MS);
|
|
11842
11983
|
|
|
11843
11984
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
@@ -12509,6 +12650,11 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
12509
12650
|
}
|
|
12510
12651
|
|
|
12511
12652
|
module.exports = {
|
|
12653
|
+
reportSchedulerError,
|
|
12654
|
+
finalizeJobWorktree,
|
|
12655
|
+
worktreeIntegrationVerdict,
|
|
12656
|
+
guardedTick,
|
|
12657
|
+
rescheduleIntervalTick,
|
|
12512
12658
|
classifyQueueStarvation,
|
|
12513
12659
|
classifyQueueStarvationByProject,
|
|
12514
12660
|
dispatchIdleMs,
|
|
@@ -12542,6 +12688,8 @@ module.exports = {
|
|
|
12542
12688
|
nextBackoffMs,
|
|
12543
12689
|
shouldWarnFailureStreak,
|
|
12544
12690
|
shouldEscalateFailureStreak,
|
|
12691
|
+
restoreFailureStreakWarnedAt,
|
|
12692
|
+
persistedStreakMinutes,
|
|
12545
12693
|
computeDegradedBudget,
|
|
12546
12694
|
healRefusalReason,
|
|
12547
12695
|
writeQueue,
|
|
@@ -12640,6 +12788,7 @@ module.exports = {
|
|
|
12640
12788
|
FOREIGN_WIP_END_DELIMITER,
|
|
12641
12789
|
TRANSIENT_RETRY_CAP,
|
|
12642
12790
|
buildScheduleStatePayload,
|
|
12791
|
+
refreshNextReset,
|
|
12643
12792
|
partitionBootOrphans,
|
|
12644
12793
|
applyOrphanOutcome,
|
|
12645
12794
|
registerAdminRoutes,
|