claude-code-session-manager 0.87.0 → 0.88.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -53
- package/dist/assets/{AgentLibrary-DyLWzZDf.js → AgentLibrary-kQF8_wAw.js} +1 -1
- package/dist/assets/{DataModel--mISIJ6h.js → DataModel-CY_mMnsr.js} +1 -1
- package/dist/assets/{History-C2ahUXTg.js → History-Sji2brb0.js} +1 -1
- package/dist/assets/{Hooks-BiC6oyR2.js → Hooks-CtqpO-p5.js} +1 -1
- package/dist/assets/{HostBilko-BPleEOld.js → HostBilko-DN0ydOCr.js} +1 -1
- package/dist/assets/{Library-Dc8Qst1R.js → Library-CoiLZTKh.js} +1 -1
- package/dist/assets/{ListDetail-DIXh-OLX.js → ListDetail-CmqiQ_io.js} +1 -1
- package/dist/assets/{MarkdownEditor-C90bkLXK.js → MarkdownEditor-HPoNeKhT.js} +1 -1
- package/dist/assets/{McpServers-DqcbLOLZ.js → McpServers-Bj1X3_NT.js} +1 -1
- package/dist/assets/{Memory-CW62MXlh.js → Memory-1PKjsM3b.js} +1 -1
- package/dist/assets/{Panel-Bw1FhRuF.js → Panel-I1s6ghPo.js} +1 -1
- package/dist/assets/{Permissions-BcUC-5y8.js → Permissions-BhiiZjHQ.js} +1 -1
- package/dist/assets/{Plugins-BnKx9flD.js → Plugins-gv3_fkPX.js} +2 -2
- package/dist/assets/{ProvenanceBadge-Bw5vNVPT.js → ProvenanceBadge-BGcgBd8V.js} +1 -1
- package/dist/assets/{SaveBar-CWr0O_w-.js → SaveBar-BGr9eZ4g.js} +1 -1
- package/dist/assets/{Scheduler-DYdLuUqq.js → Scheduler-Bdo_awPq.js} +7 -7
- package/dist/assets/{ScopeSwitcher-CrBLbg8s.js → ScopeSwitcher-ukl0qWaA.js} +1 -1
- package/dist/assets/{Settings-DluB-vN1.js → Settings-DJkFj22G.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-CHLSseay.js → SkillReferenceGraph-0t1Qblp6.js} +1 -1
- package/dist/assets/{Skills-gNdo_HNK.js → Skills-CeVdBGE-.js} +1 -1
- package/dist/assets/{SystemPrompt-Cru05-Ia.js → SystemPrompt-CfKvNHi3.js} +1 -1
- package/dist/assets/{TagLibrary-DNHY0xou.js → TagLibrary-DAzNaUPd.js} +1 -1
- package/dist/assets/{TiptapBody-I4lmbCgP.js → TiptapBody--S_rGXoA.js} +1 -1
- package/dist/assets/{Toggle-bWMHjmRh.js → Toggle-Dszm0xnp.js} +1 -1
- package/dist/assets/{index-DV3PorRY.css → index-BHTX4OTc.css} +1 -1
- package/dist/assets/{index-fc_JjdxL.js → index-Dp4Rc--q.js} +356 -356
- package/dist/assets/{settingsSchema-BfhtZnGD.js → settingsSchema-BLbssMYo.js} +1 -1
- package/dist/assets/{whisperWorker-Dbia1OpC.js → whisperWorker-C7ZGQwKg.js} +7 -7
- package/dist/index.html +2 -2
- package/dist/vad/ort-wasm-simd-threaded.asyncify.mjs +106 -110
- package/dist/vad/ort-wasm-simd-threaded.asyncify.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.jsep.mjs +98 -98
- package/dist/vad/ort-wasm-simd-threaded.jsep.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.jspi.mjs +99 -102
- package/dist/vad/ort-wasm-simd-threaded.jspi.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.mjs +46 -46
- package/dist/vad/ort-wasm-simd-threaded.wasm +0 -0
- package/package.json +10 -13
- package/scripts/README.md +59 -0
- package/scripts/audit-ops-hygiene.cjs +350 -0
- package/scripts/hooks/guard-destructive-git.cjs +10 -34
- package/scripts/hooks/guard-inline-implementation.cjs +14 -8
- package/scripts/hooks/guard-prd-writes.cjs +7 -36
- package/scripts/hooks/guard-self-schedule.cjs +175 -0
- package/scripts/ops-sweep.cjs +355 -0
- package/scripts/scheduler-mcp-server.cjs +28 -71
- package/src/main/__tests__/bilkoHost-integration.test.cjs +3 -3
- package/src/main/__tests__/chat-cancel-terminal.test.cjs +6 -9
- package/src/main/__tests__/chat-exit-close-race.test.cjs +3 -3
- package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +4 -5
- package/src/main/__tests__/chat-queue.test.cjs +2 -2
- package/src/main/__tests__/chat-stop-signal.test.cjs +2 -2
- package/src/main/__tests__/dep-orphan-archive-health.test.cjs +77 -0
- package/src/main/__tests__/dod-batchkey.test.cjs +2 -2
- package/src/main/__tests__/dod-drain-hook.test.cjs +2 -2
- package/src/main/__tests__/dod-report.test.cjs +2 -2
- package/src/main/__tests__/dod-reverify.test.cjs +2 -2
- package/src/main/__tests__/epicMint.test.cjs +2 -2
- package/src/main/__tests__/exchanges.test.cjs +2 -2
- package/src/main/__tests__/extractJson.test.cjs +2 -2
- package/src/main/__tests__/files-reject-credentials.test.cjs +1 -1
- package/src/main/__tests__/fixtures/1218-fo-01-move-scripts-lib-into-src-main-lib.log +556 -0
- package/src/main/__tests__/health-build-freshness.test.cjs +39 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +15 -1
- package/src/main/__tests__/health-queue-dispatch.test.cjs +58 -7
- package/src/main/__tests__/health-tick-liveness.test.cjs +27 -3
- package/src/main/__tests__/health-usage-poller.test.cjs +70 -23
- package/src/main/__tests__/historyRollup.test.cjs +2 -2
- package/src/main/__tests__/kg-augment.test.cjs +2 -2
- package/src/main/__tests__/mcpStatus.test.cjs +2 -2
- package/src/main/__tests__/memoryAggregate.test.cjs +1 -1
- package/src/main/__tests__/memoryStale.test.cjs +1 -1
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +25 -1
- package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +33 -3
- package/src/main/__tests__/prd-group-allocator.test.cjs +2 -2
- package/src/main/__tests__/prdAdminRouteParity.test.cjs +2 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +14 -2
- package/src/main/__tests__/prdAuthoringSeed.test.cjs +39 -0
- package/src/main/__tests__/prdLocationsArchived.test.cjs +20 -13
- package/src/main/__tests__/proc-role-env.test.cjs +125 -0
- package/src/main/__tests__/procname-claude-spawn-sites.test.cjs +304 -0
- package/src/main/__tests__/procname-sm-processes.test.cjs +127 -0
- package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +81 -401
- package/src/main/__tests__/projectPages.test.cjs +63 -149
- package/src/main/__tests__/queue-health-verdict.test.cjs +9 -9
- package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +117 -20
- package/src/main/__tests__/queueHistory.test.cjs +2 -2
- package/src/main/__tests__/rateLimitPollerStreak.test.cjs +38 -3
- package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +181 -0
- package/src/main/__tests__/runVerify.test.cjs +5 -5
- package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +3 -3
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +2 -2
- package/src/main/__tests__/scheduler-adopted-run-supervision.test.cjs +143 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +2 -2
- package/src/main/__tests__/scheduler-autopromote.test.cjs +2 -2
- package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +3 -5
- package/src/main/__tests__/scheduler-boot-orphans.test.cjs +78 -97
- package/src/main/__tests__/scheduler-default-eligible-heal.test.cjs +161 -0
- package/src/main/__tests__/scheduler-dispatch-loop.test.cjs +58 -0
- package/src/main/__tests__/scheduler-epic-digest.test.cjs +3 -5
- package/src/main/__tests__/scheduler-force-tick-outcome.test.cjs +2 -2
- package/src/main/__tests__/scheduler-gate-shadow.test.cjs +119 -0
- package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +46 -0
- package/src/main/__tests__/scheduler-heartbeat-payload.test.cjs +80 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +25 -18
- package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +4 -4
- package/src/main/__tests__/scheduler-launch-failure.test.cjs +3 -5
- package/src/main/__tests__/scheduler-looks-done.test.cjs +26 -7
- package/src/main/__tests__/scheduler-manual-pause.test.cjs +118 -0
- package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +26 -3
- package/src/main/__tests__/scheduler-prd-missing-skip.test.cjs +17 -2
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +3 -5
- package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +41 -6
- package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +62 -6
- package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +3 -5
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +23 -4
- package/src/main/__tests__/scheduler-reconcile-cwd-preserve.test.cjs +100 -0
- package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +3 -3
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +4 -4
- package/src/main/__tests__/scheduler-shard-quarantine.test.cjs +110 -0
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +81 -1
- package/src/main/__tests__/scheduler-stall-per-project.test.cjs +14 -0
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +4 -4
- package/src/main/__tests__/scheduler-stranded-autofix-park.test.cjs +245 -0
- package/src/main/__tests__/scheduler-supervisor-record.test.cjs +81 -0
- package/src/main/__tests__/scheduler-tick-cancel-token.test.cjs +2 -2
- package/src/main/__tests__/scheduler-tick-wedge.test.cjs +172 -0
- package/src/main/__tests__/scheduler-transient-failure.test.cjs +2 -2
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +37 -11
- package/src/main/__tests__/scheduler-worktree-exec-cwd.test.cjs +3 -5
- package/src/main/__tests__/usageSingleFlight.test.cjs +157 -0
- package/src/main/__tests__/workTypeLibrary.test.cjs +1 -1
- package/src/main/bilkoHost.cjs +16 -7
- package/src/main/build-info.json +8 -0
- package/src/main/chatRunner.cjs +4 -2
- package/src/main/config.cjs +2 -3
- package/src/main/docEdit.cjs +4 -2
- package/src/main/health.cjs +276 -52
- package/src/main/heapSnapshot.cjs +2 -2
- package/src/main/historyAggregator.cjs +1 -1
- package/src/main/index.cjs +61 -10
- package/src/main/ipcSchemas.cjs +6 -17
- package/src/main/lib/__tests__/auditLog.test.cjs +38 -0
- package/src/main/lib/__tests__/buildIdentity.test.cjs +121 -0
- package/src/main/lib/__tests__/cwdClassify.test.cjs +111 -0
- package/src/main/lib/__tests__/definitionOfDoneSequence.test.cjs +95 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +1 -1
- package/src/main/lib/__tests__/dispatchLoop.test.cjs +63 -0
- package/src/main/lib/__tests__/gateFixtures.json +20 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +8 -1
- package/src/main/lib/__tests__/instanceLock.test.cjs +93 -7
- package/src/main/lib/__tests__/jobSupervisorRecord.test.cjs +78 -0
- package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +11 -0
- package/src/main/lib/__tests__/localAdminHttp.test.cjs +1 -1
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +6 -1
- package/src/main/lib/__tests__/procIdentity.test.cjs +119 -0
- package/src/main/lib/__tests__/procName.test.cjs +92 -0
- package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +67 -0
- package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +6 -7
- package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +37 -204
- package/src/main/lib/__tests__/schedulerPaths.test.cjs +226 -0
- package/src/main/lib/__tests__/schedulerPathsWorktree.test.cjs +94 -0
- package/src/main/lib/__tests__/schedulerRuntimeState.test.cjs +56 -0
- package/src/main/lib/__tests__/sessionSlots.test.cjs +45 -0
- package/src/main/lib/__tests__/telemetryBacklog.test.cjs +1 -1
- package/src/main/lib/__tests__/upgradeDrain.test.cjs +130 -0
- package/src/main/lib/__tests__/usageCircuit.test.cjs +61 -0
- package/src/main/lib/__tests__/watchdog-helpers.test.cjs +63 -0
- package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +73 -0
- package/src/main/lib/activeIndexRebuild.cjs +1 -3
- package/src/main/lib/activeSessions.cjs +20 -88
- package/src/main/lib/adoptedRunSupervisor.cjs +136 -0
- package/src/main/lib/agentModelResolve.cjs +33 -1
- package/src/main/lib/agentPersonaSchema.cjs +2 -2
- package/src/main/lib/auditLog.cjs +30 -5
- package/src/main/lib/buildIdentity.cjs +113 -0
- package/src/main/lib/classifyPromptTicket.cjs +3 -2
- package/src/main/lib/claudeBin.cjs +37 -2
- package/src/main/lib/cleanEnv.cjs +27 -1
- package/src/main/lib/credentials.cjs +4 -2
- package/src/main/lib/cwdClassify.cjs +185 -0
- package/src/main/lib/definitionOfDone.cjs +296 -52
- package/src/main/lib/dispatchLoop.cjs +40 -0
- package/src/main/lib/effectiveModelInfo.cjs +9 -10
- package/src/main/lib/ephemeralCwd.cjs +7 -28
- package/src/main/lib/gitWorktree.cjs +95 -8
- package/src/main/lib/guardShims.cjs +3 -3
- package/src/main/lib/historyRollup.cjs +6 -7
- package/src/main/lib/instanceLock.cjs +32 -5
- package/src/main/lib/jobSupervisorRecord.cjs +147 -0
- package/src/main/lib/jobWorktreeBootLive.cjs +9 -5
- package/src/main/lib/localAdminHttp.cjs +9 -27
- package/src/main/lib/mcpToolCatalog.cjs +29 -77
- package/src/main/lib/opsOwnership.cjs +27 -19
- package/src/main/lib/prdAuthoringSeed.cjs +36 -0
- package/src/main/lib/prdLocations.cjs +48 -1
- package/src/main/lib/procIdentity.cjs +126 -0
- package/src/main/lib/procName.cjs +98 -0
- package/src/main/lib/projectHomeAdminRoutes.cjs +56 -327
- package/src/main/lib/queueHistory.cjs +8 -7
- package/src/main/lib/queueStore.cjs +57 -37
- package/src/main/lib/quietMachineLease.cjs +19 -2
- package/src/main/lib/reservationExpiry.cjs +30 -0
- package/src/main/lib/runClaudeP.cjs +4 -2
- package/src/main/lib/runLogRetention.cjs +3 -2
- package/src/main/lib/scheduleJobSchema.cjs +2 -2
- package/src/main/lib/scheduleJobTransitions.cjs +29 -2
- package/src/main/lib/schedulerBatch.cjs +27 -2
- package/src/main/lib/schedulerPaths.cjs +175 -0
- package/src/main/lib/schedulerRuntimeState.cjs +59 -0
- package/src/main/lib/sessionSlots.cjs +47 -9
- package/src/main/lib/smProcNames.cjs +51 -0
- package/src/main/lib/upgradeDrain.cjs +188 -0
- package/src/main/lib/usageCircuit.cjs +53 -6
- package/src/main/lib/watchdogHelpers.cjs +70 -37
- package/src/main/lib/withTimeout.cjs +33 -0
- package/src/main/mcpStatus.cjs +4 -2
- package/src/main/pluginInstall.cjs +6 -2
- package/src/main/projectPages.cjs +41 -145
- package/src/main/pty.cjs +8 -1
- package/src/main/queueOps.cjs +10 -10
- package/src/main/runVerify.cjs +69 -3
- package/src/main/scheduler.cjs +1607 -383
- package/src/main/seedAgentPersonas.cjs +1 -1
- package/src/main/seedDevPlugin.cjs +1 -1
- package/src/main/seedSchedulerMcp.cjs +7 -6
- package/src/main/seedStatus.cjs +1 -1
- package/src/main/supervisor.cjs +5 -3
- package/src/main/usage.cjs +126 -62
- package/src/preload/api.d.ts +18 -64
- package/src/preload/index.cjs +2 -4
- package/src/seed/agents/project-home-builder.md +31 -48
- package/screenshots/.gitkeep +0 -0
- package/screenshots/README-screenshots.md +0 -13
- package/src/main/lib/projectPageSummarySchema.cjs +0 -181
- package/src/main/teams.cjs +0 -95
- package/src/main/templates/project-pages-catalog.json +0 -741
- package/src/main/templates/project-pages-default-home.html +0 -123
- package/src/main/templates/project-pages-pipeline.md +0 -417
- package/src/seed/prompts/code-review/ac-coverage-check.md +0 -8
- package/src/seed/prompts/code-review/correctness-only.md +0 -8
- package/src/seed/prompts/code-review/full-spectrum-high.md +0 -8
- package/src/seed/prompts/code-review/hallucination-check.md +0 -8
- package/src/seed/prompts/code-review/public-api-compat.md +0 -8
- package/src/seed/prompts/code-review/readability-naming.md +0 -8
- package/src/seed/prompts/debugging/bug-as-failing-test.md +0 -8
- package/src/seed/prompts/debugging/git-bisect-regression.md +0 -8
- package/src/seed/prompts/debugging/instrument-intermittent-bug.md +0 -8
- package/src/seed/prompts/debugging/localize-pipeline-failure.md +0 -8
- package/src/seed/prompts/debugging/reproduce-then-diagnose.md +0 -8
- package/src/seed/prompts/documentation/adr-from-change.md +0 -8
- package/src/seed/prompts/documentation/module-readme.md +0 -8
- package/src/seed/prompts/documentation/onboarding-plan.md +0 -8
- package/src/seed/prompts/documentation/refresh-claude-md.md +0 -8
- package/src/seed/prompts/documentation/tsdoc-public-exports.md +0 -8
- package/src/seed/prompts/git-pr/conventional-commit.md +0 -8
- package/src/seed/prompts/git-pr/draft-pr-title-body.md +0 -8
- package/src/seed/prompts/git-pr/pre-commit-safety-sweep.md +0 -8
- package/src/seed/prompts/git-pr/release-notes-block.md +0 -8
- package/src/seed/prompts/git-pr/split-large-pr.md +0 -8
- package/src/seed/prompts/performance/bundle-startup-audit.md +0 -8
- package/src/seed/prompts/performance/complexity-audit.md +0 -8
- package/src/seed/prompts/performance/cpu-profile-hot-path.md +0 -8
- package/src/seed/prompts/performance/db-query-plan-review.md +0 -8
- package/src/seed/prompts/performance/memory-leak-hunt.md +0 -8
- package/src/seed/prompts/qa/api-contract-tests.md +0 -8
- package/src/seed/prompts/qa/e2e-critical-path.md +0 -8
- package/src/seed/prompts/qa/failing-test-for-bug.md +0 -8
- package/src/seed/prompts/qa/find-missing-test-coverage.md +0 -8
- package/src/seed/prompts/qa/stabilize-flaky-test.md +0 -8
- package/src/seed/prompts/qa/tdd-red-first.md +0 -8
- package/src/seed/prompts/qa/visual-regression-review.md +0 -8
- package/src/seed/prompts/qa/wcag-axe-scan.md +0 -8
- package/src/seed/prompts/refactoring/dead-code-sweep.md +0 -8
- package/src/seed/prompts/refactoring/extract-duplicated-pattern.md +0 -8
- package/src/seed/prompts/refactoring/modernize-legacy-file.md +0 -8
- package/src/seed/prompts/refactoring/reduce-cyclomatic-complexity.md +0 -8
- package/src/seed/prompts/refactoring/tighten-module-boundaries.md +0 -8
- package/src/seed/prompts/security/authz-audit.md +0 -8
- package/src/seed/prompts/security/crypto-correctness.md +0 -8
- package/src/seed/prompts/security/cwe-top-25-hunt.md +0 -8
- package/src/seed/prompts/security/dependency-audit.md +0 -8
- package/src/seed/prompts/security/ipc-boundary-hardening.md +0 -8
- package/src/seed/prompts/security/owasp-top-10-staged-diff.md +0 -8
- package/src/seed/prompts/security/secret-credential-scan.md +0 -8
- package/web/README.md +0 -41
- package/web/project-pages/logic/dist/logic.cjs +0 -4709
- package/web/project-pages/render.cjs +0 -70
- package/web/project-pages/renderer/dist/renderer.cjs +0 -18900
- package/web/project-pages/validate-summary.cjs +0 -62
package/src/main/scheduler.cjs
CHANGED
|
@@ -47,13 +47,15 @@ const fs = require('node:fs');
|
|
|
47
47
|
const fsp = require('node:fs/promises');
|
|
48
48
|
const path = require('node:path');
|
|
49
49
|
const os = require('node:os');
|
|
50
|
+
const { startDispatchLoop } = require('./lib/dispatchLoop.cjs');
|
|
51
|
+
const schedulerPaths = require('./lib/schedulerPaths.cjs');
|
|
50
52
|
const { randomUUID } = require('node:crypto');
|
|
51
53
|
const { execFile, execFileSync } = require('node:child_process');
|
|
52
54
|
const { ipcMain } = require('electron');
|
|
53
55
|
const billing = require('./usage.cjs');
|
|
54
56
|
const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
55
57
|
const supervisor = require('./supervisor.cjs');
|
|
56
|
-
const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
58
|
+
const { resolveClaudeBin, claudeSpawnTarget, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
57
59
|
const launchFailure = require('./lib/launchFailure.cjs');
|
|
58
60
|
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
59
61
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
@@ -67,6 +69,7 @@ const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
|
|
|
67
69
|
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
68
70
|
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
69
71
|
const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
|
|
72
|
+
const { isResetFresh, bindingWindow, degradedBudget } = require('./lib/usageCircuit.cjs');
|
|
70
73
|
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
71
74
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
72
75
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
@@ -83,6 +86,7 @@ const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require(
|
|
|
83
86
|
const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
|
|
84
87
|
const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
|
|
85
88
|
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
89
|
+
const { identity: procIdentityOf, isDifferentProcess } = require('./lib/procIdentity.cjs');
|
|
86
90
|
const logs = require('./logs.cjs');
|
|
87
91
|
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
88
92
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
@@ -131,19 +135,21 @@ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD,
|
|
|
131
135
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
132
136
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
133
137
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
138
|
+
const { resolveGate, runGateSequence } = require('./lib/definitionOfDone.cjs');
|
|
134
139
|
const queueOps = require('./queueOps.cjs');
|
|
135
140
|
// Feedback-auto-PRD sweep — formerly only run by the external scheduler-watchdog
|
|
136
141
|
// while the app was down (PRD 686 moved it in-app so it also runs while alive).
|
|
137
142
|
// Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
|
|
138
143
|
// match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
|
|
139
144
|
// home-dir layout.
|
|
140
|
-
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
145
|
+
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
|
|
141
146
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
142
147
|
const agentModelResolve = require('./lib/agentModelResolve.cjs');
|
|
143
148
|
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
144
149
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
145
150
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
146
151
|
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
152
|
+
const { withTimeout } = require('./lib/withTimeout.cjs');
|
|
147
153
|
|
|
148
154
|
// ---------- origin session resolution (PRD 832) ----------
|
|
149
155
|
// An Epic IS a tagged claude session — job rows carry the originating
|
|
@@ -164,12 +170,15 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
164
170
|
}
|
|
165
171
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
166
172
|
const quietMachineLease = require('./lib/quietMachineLease.cjs');
|
|
173
|
+
const runtimeState = require('./lib/schedulerRuntimeState.cjs');
|
|
167
174
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
168
175
|
const gitWorktree = require('./lib/gitWorktree.cjs');
|
|
169
176
|
const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
|
|
170
177
|
const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
|
|
171
178
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
172
179
|
const queueStore = require('./lib/queueStore.cjs');
|
|
180
|
+
const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
|
|
181
|
+
const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
|
|
173
182
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
174
183
|
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
175
184
|
const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
|
|
@@ -182,20 +191,26 @@ const { allProjectCwds } = require('./lib/activeSessions.cjs');
|
|
|
182
191
|
// an exemption it should have applied landed on disk, and nothing in the
|
|
183
192
|
// run record showed that; this is the fix).
|
|
184
193
|
const SCHEDULER_BOOTED_AT = new Date().toISOString();
|
|
185
|
-
//
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
194
|
+
// A production npx install ships no .git at all, so a runtime `git
|
|
195
|
+
// rev-parse` from __dirname was structurally always null there — every
|
|
196
|
+
// production run-meta sidecar recorded schedulerCodeSha: null. buildIdentity
|
|
197
|
+
// resolves build-info.json (baked at publish time) first, falling back to a
|
|
198
|
+
// non-walking git read only in a dev checkout / job worktree — see
|
|
199
|
+
// src/main/lib/buildIdentity.cjs's header.
|
|
200
|
+
const { resolveBuildIdentity, readInstalledBuildInfo } = require('./lib/buildIdentity.cjs');
|
|
201
|
+
const upgradeDrain = require('./lib/upgradeDrain.cjs');
|
|
202
|
+
const SCHEDULER_BUILD_IDENTITY = resolveBuildIdentity({ bootedAt: SCHEDULER_BOOTED_AT });
|
|
203
|
+
const SCHEDULER_CODE_SHA = SCHEDULER_BUILD_IDENTITY.codeSha;
|
|
204
|
+
// Spread into EVERY metaPath writer below (grep `metaPath` for the full
|
|
205
|
+
// list) — single source so a future field never lands in some sidecars and
|
|
206
|
+
// not others, the exact gap that left 3 of 5 writers silently missing
|
|
207
|
+
// schedulerBootedAt/schedulerCodeSha before this constant existed.
|
|
208
|
+
const SCHEDULER_META_IDENTITY = {
|
|
209
|
+
schedulerBootedAt: SCHEDULER_BOOTED_AT,
|
|
210
|
+
schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
211
|
+
schedulerVersion: SCHEDULER_BUILD_IDENTITY.version,
|
|
212
|
+
schedulerBuiltAt: SCHEDULER_BUILD_IDENTITY.builtAt,
|
|
213
|
+
};
|
|
199
214
|
|
|
200
215
|
const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
|
|
201
216
|
|
|
@@ -590,7 +605,7 @@ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAf
|
|
|
590
605
|
// executor-created stash (never guesses when there are 2+); reports anything
|
|
591
606
|
// it can't safely resolve on the returned object so the caller can surface it
|
|
592
607
|
// on the job row instead of finishing silently green.
|
|
593
|
-
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
|
|
608
|
+
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug, landedCommit }) {
|
|
594
609
|
try {
|
|
595
610
|
const [stashAfter, headAfter] = await Promise.all([
|
|
596
611
|
module.exports.stashList(cwd),
|
|
@@ -655,7 +670,16 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
|
|
|
655
670
|
pathsCommittedDuringRun,
|
|
656
671
|
existsAfter,
|
|
657
672
|
});
|
|
658
|
-
|
|
673
|
+
// Ground truth outranks the baseline diff (2026-09-18, 1229-fo-03): the
|
|
674
|
+
// dirty baseline is invalidated by ANY later writer (a human commit that
|
|
675
|
+
// sweeps the same paths), so it can't prove a revert on its own. The
|
|
676
|
+
// job's own landedCommit still being an ancestor of HEAD proves its work
|
|
677
|
+
// was not discarded — anchored to that sha, not to the baseline.
|
|
678
|
+
const workSurvives = reverted.length > 0
|
|
679
|
+
&& await module.exports.landedCommitIsAncestorOfHead(cwd, landedCommit);
|
|
680
|
+
if (workSurvives) {
|
|
681
|
+
console.log(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} baseline path(s) went clean but landed commit ${String(landedCommit).slice(0, 7)} is still an ancestor of HEAD — not a revert`);
|
|
682
|
+
} else if (reverted.length) {
|
|
659
683
|
result.reverted = reverted;
|
|
660
684
|
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
661
685
|
}
|
|
@@ -895,13 +919,9 @@ function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, his
|
|
|
895
919
|
return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
|
|
896
920
|
}
|
|
897
921
|
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
const PRDS_ARCHIVE_DIR = path.join(ROOT, 'prds-archived');
|
|
902
|
-
const QUEUE_PATH = path.join(ROOT, 'queue.json');
|
|
903
|
-
const SCHEDULER_STATE_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-state.json');
|
|
904
|
-
const HEARTBEAT_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-heartbeat.log');
|
|
922
|
+
// Machine-wide roots resolve lazily via lib/schedulerPaths.cjs (SM_SCHEDULER_HOME
|
|
923
|
+
// override) — never module-scope consts. The ROOT/PRDS_DIR/RUNS_DIR/
|
|
924
|
+
// SCHEDULER_STATE_PATH exports below are lazy getters over the same resolver.
|
|
905
925
|
const HEARTBEAT_MAX_BYTES = 1024 * 1024;
|
|
906
926
|
// DEFAULT_PROJECT_CWD imported from lib/schedulerBatch.cjs (single source of truth).
|
|
907
927
|
|
|
@@ -1020,7 +1040,7 @@ function biasJobOomScore(pid) {
|
|
|
1020
1040
|
* (reconcile, list-prds, lint, rescan).
|
|
1021
1041
|
*/
|
|
1022
1042
|
function candidatePrdsDirs() {
|
|
1023
|
-
return [
|
|
1043
|
+
return [schedulerPaths.prdsRoot(), ...resolvePrdsDirs()];
|
|
1024
1044
|
}
|
|
1025
1045
|
|
|
1026
1046
|
/**
|
|
@@ -1141,7 +1161,7 @@ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd,
|
|
|
1141
1161
|
const finishedAt = Date.now();
|
|
1142
1162
|
config.writeJsonSync(metaPath, {
|
|
1143
1163
|
slug: job.slug, cwd, sessionId, exitCode: 0, skipped: reason,
|
|
1144
|
-
note: msg, startedAt, finishedAt, durationMs: 0,
|
|
1164
|
+
note: msg, startedAt, finishedAt, durationMs: 0, ...SCHEDULER_META_IDENTITY,
|
|
1145
1165
|
});
|
|
1146
1166
|
return { exitCode: 0, durationMs: 0, skipped: reason, note: msg, sessionId };
|
|
1147
1167
|
}
|
|
@@ -1283,17 +1303,21 @@ async function retireCompletedSlugs(slugs) {
|
|
|
1283
1303
|
// Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
|
|
1284
1304
|
// plugin's /develop and /prd skills — which reference this stable `~`-absolute
|
|
1285
1305
|
// path — work on any user's machine, not just the author's.
|
|
1306
|
+
// Line 1 of the template is `<!-- PRD_AUTHORING.md vN -->`: bump vN whenever the
|
|
1307
|
+
// template changes, or existing installs never receive the update.
|
|
1286
1308
|
const PRD_AUTHORING_TEMPLATE = path.join(__dirname, 'templates', 'PRD_AUTHORING.md');
|
|
1287
|
-
const PRD_AUTHORING_DEST = path.join(ROOT, 'PRD_AUTHORING.md');
|
|
1288
1309
|
|
|
1289
1310
|
function ensureDirs() {
|
|
1290
|
-
fs.mkdirSync(
|
|
1291
|
-
fs.mkdirSync(
|
|
1292
|
-
//
|
|
1311
|
+
fs.mkdirSync(schedulerPaths.prdsRoot(), { recursive: true });
|
|
1312
|
+
fs.mkdirSync(schedulerPaths.runsDir(), { recursive: true });
|
|
1313
|
+
// Re-seed the guide whenever the bundled template's version stamp differs.
|
|
1293
1314
|
try {
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1315
|
+
const authoringDest = path.join(schedulerPaths.scheduledPlansRoot(), 'PRD_AUTHORING.md');
|
|
1316
|
+
seedAuthoringGuide({
|
|
1317
|
+
src: PRD_AUTHORING_TEMPLATE,
|
|
1318
|
+
dest: authoringDest,
|
|
1319
|
+
write: (abs, text) => config.writeTextAtomic(abs, text, { writer: 'scheduler' }),
|
|
1320
|
+
}).catch(() => { /* non-fatal, same as below */ });
|
|
1297
1321
|
} catch { /* non-fatal: the guide is a convenience, not load-bearing for a run */ }
|
|
1298
1322
|
}
|
|
1299
1323
|
|
|
@@ -1321,8 +1345,9 @@ function ensureDirs() {
|
|
|
1321
1345
|
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
1322
1346
|
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
1323
1347
|
*/
|
|
1324
|
-
async function consolidateAllFlatPrds(cwds) {
|
|
1348
|
+
async function consolidateAllFlatPrds(cwds, skipCwds) {
|
|
1325
1349
|
for (const cwd of cwds) {
|
|
1350
|
+
if (skipCwds?.has(cwd)) continue; // torn shard: its PRDs are not ours to touch this pass
|
|
1326
1351
|
try {
|
|
1327
1352
|
const c = await consolidateFlatPrds(cwd);
|
|
1328
1353
|
if (c.moved > 0) {
|
|
@@ -1354,7 +1379,7 @@ async function consolidateAllFlatPrds(cwds) {
|
|
|
1354
1379
|
async function runPrdMigration() {
|
|
1355
1380
|
let result;
|
|
1356
1381
|
try {
|
|
1357
|
-
result = await migratePrds(
|
|
1382
|
+
result = await migratePrds(schedulerPaths.prdsRoot());
|
|
1358
1383
|
} catch (e) {
|
|
1359
1384
|
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
|
|
1360
1385
|
return null;
|
|
@@ -1365,7 +1390,7 @@ async function runPrdMigration() {
|
|
|
1365
1390
|
level: 'warn',
|
|
1366
1391
|
scope: 'scheduler',
|
|
1367
1392
|
message: `PRD migration: ${result.unresolved.length} file(s) left in legacy dir`,
|
|
1368
|
-
meta: { legacyDir:
|
|
1393
|
+
meta: { legacyDir: schedulerPaths.prdsRoot(), unresolved: result.unresolved },
|
|
1369
1394
|
});
|
|
1370
1395
|
for (const u of result.unresolved) {
|
|
1371
1396
|
console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
|
|
@@ -1432,7 +1457,7 @@ const QUEUE_BAK_KEEP = 5;
|
|
|
1432
1457
|
async function sweepQueueBackups() {
|
|
1433
1458
|
let entries;
|
|
1434
1459
|
try {
|
|
1435
|
-
entries = await fsp.readdir(
|
|
1460
|
+
entries = await fsp.readdir(schedulerPaths.scheduledPlansRoot());
|
|
1436
1461
|
} catch {
|
|
1437
1462
|
return;
|
|
1438
1463
|
}
|
|
@@ -1448,7 +1473,7 @@ async function sweepQueueBackups() {
|
|
|
1448
1473
|
let removed = 0;
|
|
1449
1474
|
for (const f of toDelete) {
|
|
1450
1475
|
try {
|
|
1451
|
-
await fsp.unlink(path.join(
|
|
1476
|
+
await fsp.unlink(path.join(schedulerPaths.scheduledPlansRoot(), f));
|
|
1452
1477
|
removed++;
|
|
1453
1478
|
} catch (e) {
|
|
1454
1479
|
console.warn('[scheduler] backup sweep: unlink failed', f, e?.message);
|
|
@@ -1464,20 +1489,24 @@ async function sweepQueueBackups() {
|
|
|
1464
1489
|
// callback that must flush meta.json before resolving) — replacing with async
|
|
1465
1490
|
// would deadlock the exit path.
|
|
1466
1491
|
const config = require('./config.cjs');
|
|
1492
|
+
const { seedAuthoringGuide } = require('./lib/prdAuthoringSeed.cjs');
|
|
1467
1493
|
const atomicWriteJsonSync = (p, data) => config.writeJsonSync(p, data);
|
|
1468
1494
|
|
|
1469
1495
|
// ---------- scheduler-state.json (sidecar) ----------
|
|
1470
1496
|
|
|
1471
1497
|
function loadSchedulerState() {
|
|
1472
1498
|
try {
|
|
1473
|
-
const raw = fs.readFileSync(
|
|
1499
|
+
const raw = fs.readFileSync(schedulerPaths.schedulerStatePath(), 'utf8');
|
|
1474
1500
|
const s = JSON.parse(raw);
|
|
1475
1501
|
if (s.lastObservedReset) cachedNextReset = s.lastObservedReset;
|
|
1502
|
+
if (typeof s.lastResetObservedAt === 'number') lastResetObservedAtMs = s.lastResetObservedAt;
|
|
1476
1503
|
if (typeof s.consecutiveFailures === 'number') consecutiveFailures = s.consecutiveFailures;
|
|
1477
1504
|
if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
|
|
1478
1505
|
if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
|
|
1479
1506
|
if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
|
|
1480
1507
|
if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
|
|
1508
|
+
if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
|
|
1509
|
+
if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
|
|
1481
1510
|
} catch { /* first boot or corrupt — start fresh */ }
|
|
1482
1511
|
}
|
|
1483
1512
|
|
|
@@ -1487,10 +1516,14 @@ function persistSchedulerState() {
|
|
|
1487
1516
|
// require threading awaits through pause/resume bookkeeping for negligible
|
|
1488
1517
|
// benefit — the file is well under one page.
|
|
1489
1518
|
try {
|
|
1490
|
-
config.writeJsonSync(
|
|
1519
|
+
config.writeJsonSync(schedulerPaths.schedulerStatePath(), {
|
|
1491
1520
|
version: 1,
|
|
1492
1521
|
lastObservedReset: cachedNextReset,
|
|
1493
|
-
|
|
1522
|
+
// Only stamped at the moment a FRESH reset was actually observed (see
|
|
1523
|
+
// recordObservedReset) — never Date.now() on every persist call, which
|
|
1524
|
+
// used to make a stale cachedNextReset look freshly-confirmed on every
|
|
1525
|
+
// tick even when nothing new had been read.
|
|
1526
|
+
lastResetObservedAt: lastResetObservedAtMs,
|
|
1494
1527
|
lastPollAt,
|
|
1495
1528
|
consecutiveFailures,
|
|
1496
1529
|
backoffMs,
|
|
@@ -1498,6 +1531,13 @@ function persistSchedulerState() {
|
|
|
1498
1531
|
pausedSince: null,
|
|
1499
1532
|
pauseClearedManuallyAt,
|
|
1500
1533
|
failureStreakWarned,
|
|
1534
|
+
failureStreakWarnedAt,
|
|
1535
|
+
lastEscalationAt: lastEscalationAtMs,
|
|
1536
|
+
// Circuit fields are read fresh from the live shared breaker each
|
|
1537
|
+
// persist — health.cjs (a separate `npm run health` process) reads
|
|
1538
|
+
// THESE persisted values, since it never holds the in-memory circuit.
|
|
1539
|
+
usageCircuitState: billing.usageCircuit.state(),
|
|
1540
|
+
usageCircuitOpenedAt: billing.usageCircuit.openedAt(),
|
|
1501
1541
|
});
|
|
1502
1542
|
} catch (e) {
|
|
1503
1543
|
console.warn('[scheduler] failed to persist scheduler state', e?.message);
|
|
@@ -1510,18 +1550,279 @@ function appendHeartbeat(entry) {
|
|
|
1510
1550
|
try {
|
|
1511
1551
|
const line = JSON.stringify(entry) + '\n';
|
|
1512
1552
|
let size = 0;
|
|
1513
|
-
try { size = fs.statSync(
|
|
1553
|
+
try { size = fs.statSync(schedulerPaths.heartbeatPath()).size; } catch { /* new file */ }
|
|
1514
1554
|
if (size >= HEARTBEAT_MAX_BYTES) {
|
|
1515
|
-
const rotated =
|
|
1555
|
+
const rotated = schedulerPaths.heartbeatPath() + '.1';
|
|
1516
1556
|
try { fs.unlinkSync(rotated); } catch { /* */ }
|
|
1517
|
-
try { fs.renameSync(
|
|
1557
|
+
try { fs.renameSync(schedulerPaths.heartbeatPath(), rotated); } catch { /* */ }
|
|
1518
1558
|
}
|
|
1519
|
-
fs.appendFileSync(
|
|
1559
|
+
fs.appendFileSync(schedulerPaths.heartbeatPath(), line);
|
|
1520
1560
|
} catch (e) {
|
|
1521
1561
|
console.warn('[scheduler] heartbeat write failed', e?.message);
|
|
1522
1562
|
}
|
|
1523
1563
|
}
|
|
1524
1564
|
|
|
1565
|
+
// Build identity stamped on every heartbeat line — memoized at boot
|
|
1566
|
+
// (SCHEDULER_BUILD_IDENTITY), so a tick costs no git or fs work.
|
|
1567
|
+
function heartbeatBuild() {
|
|
1568
|
+
const { version, codeSha, builtAt } = SCHEDULER_BUILD_IDENTITY;
|
|
1569
|
+
return { version, codeSha, builtAt };
|
|
1570
|
+
}
|
|
1571
|
+
|
|
1572
|
+
// ---------- upgrade drain driver (lib/upgradeDrain.cjs) ----------
|
|
1573
|
+
|
|
1574
|
+
// Set by index.cjs: performs the actual app teardown + relaunch + exit.
|
|
1575
|
+
let restartHandler = null;
|
|
1576
|
+
function setRestartHandler(fn) { restartHandler = typeof fn === 'function' ? fn : null; }
|
|
1577
|
+
let drainDriving = false;
|
|
1578
|
+
|
|
1579
|
+
function drainSnapshot(jobs) {
|
|
1580
|
+
const running = new Set();
|
|
1581
|
+
let investigating = 0;
|
|
1582
|
+
for (const j of jobs ?? []) {
|
|
1583
|
+
if (j?.status === 'running') running.add(j.slug);
|
|
1584
|
+
else if (j?.status === 'investigating') investigating++;
|
|
1585
|
+
}
|
|
1586
|
+
for (const slug of runningSet) running.add(slug);
|
|
1587
|
+
// Deferred investigations are not busy: while draining they never spawn.
|
|
1588
|
+
return { running: running.size, investigating: Math.max(investigating, runtimeState.investigationCount()) };
|
|
1589
|
+
}
|
|
1590
|
+
|
|
1591
|
+
/**
|
|
1592
|
+
* Restart is triggered automatically ONLY when the installed build-info.json
|
|
1593
|
+
* differs from the running codeSha (an install/update already happened) —
|
|
1594
|
+
* never by polling npm. SM_AUTO_UPGRADE_RESTART=0 disables it.
|
|
1595
|
+
*/
|
|
1596
|
+
function maybeAutoRequestRestart() {
|
|
1597
|
+
if (process.env.SM_AUTO_UPGRADE_RESTART === '0' || process.env.SM_DEV === '1') return null;
|
|
1598
|
+
const installed = readInstalledBuildInfo();
|
|
1599
|
+
const installedSha = typeof installed?.gitShortSha === 'string' ? installed.gitShortSha : null;
|
|
1600
|
+
if (!upgradeDrain.installedBuildDiffers({ running: SCHEDULER_CODE_SHA, installed: installedSha })) return null;
|
|
1601
|
+
return upgradeDrain.requestRestart({ reason: `installed build ${installedSha} differs from running ${SCHEDULER_CODE_SHA}`, requestedBy: 'auto-upgrade' });
|
|
1602
|
+
}
|
|
1603
|
+
|
|
1604
|
+
async function driveUpgradeDrain(state) {
|
|
1605
|
+
if (drainDriving) return;
|
|
1606
|
+
drainDriving = true;
|
|
1607
|
+
try {
|
|
1608
|
+
let request = upgradeDrain.readRestartRequest();
|
|
1609
|
+
if (!request && !state.drain?.active) request = maybeAutoRequestRestart();
|
|
1610
|
+
const { action, reason } = upgradeDrain.evaluateDrain({
|
|
1611
|
+
request,
|
|
1612
|
+
queueSnapshot: drainSnapshot(state.jobs),
|
|
1613
|
+
drainState: state.drain,
|
|
1614
|
+
now: Date.now(),
|
|
1615
|
+
});
|
|
1616
|
+
if (action === 'none' || action === 'wait') { drainActive = Boolean(state.drain?.active); return; }
|
|
1617
|
+
if (action === 'pause') {
|
|
1618
|
+
await mutate((s) => { s.drain = { active: true, since: new Date().toISOString(), requestedAt: request.requestedAt }; });
|
|
1619
|
+
drainActive = true;
|
|
1620
|
+
appendAuditEvent('upgrade_drain_started', { reason: request.reason, requestedBy: request.requestedBy });
|
|
1621
|
+
await broadcast({ flush: true });
|
|
1622
|
+
return;
|
|
1623
|
+
}
|
|
1624
|
+
if (action === 'abort') {
|
|
1625
|
+
upgradeDrain.retireRestartRequest();
|
|
1626
|
+
await mutate((s) => { s.drain = null; });
|
|
1627
|
+
drainActive = false;
|
|
1628
|
+
appendAuditEvent('upgrade_drain_aborted', { reason });
|
|
1629
|
+
await broadcast({ flush: true });
|
|
1630
|
+
runDueJobs().catch(() => {});
|
|
1631
|
+
return;
|
|
1632
|
+
}
|
|
1633
|
+
// 'restart': the FINAL zero-busy check runs inside a mutate, immediately
|
|
1634
|
+
// before exit — the snapshot above may be stale by now.
|
|
1635
|
+
let go = false;
|
|
1636
|
+
await mutate((s) => {
|
|
1637
|
+
const snap = drainSnapshot(s.jobs);
|
|
1638
|
+
if (!s.drain?.active || snap.running + snap.investigating > 0) return;
|
|
1639
|
+
upgradeDrain.stampDrainCompleted();
|
|
1640
|
+
go = true;
|
|
1641
|
+
});
|
|
1642
|
+
if (!go) return;
|
|
1643
|
+
appendAuditEvent('upgrade_drain_restart', { reason: request.reason, requestedBy: request.requestedBy });
|
|
1644
|
+
try {
|
|
1645
|
+
if (!restartHandler) throw new Error('no restart handler registered');
|
|
1646
|
+
upgradeDrain.markRestarting();
|
|
1647
|
+
await restartHandler(request);
|
|
1648
|
+
// Only reached when the handler did NOT exit the process (dev-server
|
|
1649
|
+
// in-place reboot): the restart is done, so retire the drain here.
|
|
1650
|
+
upgradeDrain.clearRestartingMarker();
|
|
1651
|
+
upgradeDrain.retireRestartRequest();
|
|
1652
|
+
await mutate((s) => { s.drain = null; });
|
|
1653
|
+
drainActive = false;
|
|
1654
|
+
} catch (e) {
|
|
1655
|
+
// Never strand the queue drained-and-paused: fall back to an abort.
|
|
1656
|
+
console.error('[scheduler] drain restart failed — aborting drain:', e?.message ?? e);
|
|
1657
|
+
upgradeDrain.clearRestartingMarker();
|
|
1658
|
+
upgradeDrain.retireRestartRequest();
|
|
1659
|
+
await mutate((s) => { s.drain = null; });
|
|
1660
|
+
drainActive = false;
|
|
1661
|
+
runDueJobs().catch(() => {});
|
|
1662
|
+
}
|
|
1663
|
+
} finally {
|
|
1664
|
+
drainDriving = false;
|
|
1665
|
+
}
|
|
1666
|
+
}
|
|
1667
|
+
|
|
1668
|
+
/** Boot: clear a leftover drain whose request is complete (the restart happened) or gone. */
|
|
1669
|
+
async function clearStaleDrainAtBoot(boot) {
|
|
1670
|
+
const request = upgradeDrain.readRestartRequest();
|
|
1671
|
+
const action = upgradeDrain.bootDrainAction({ drainState: boot.drain, request });
|
|
1672
|
+
upgradeDrain.clearRestartingMarker();
|
|
1673
|
+
if (request?.drainCompletedAt) upgradeDrain.retireRestartRequest();
|
|
1674
|
+
if (action === 'clear') await mutate((s) => { s.drain = null; });
|
|
1675
|
+
drainActive = action === 'keep';
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
/**
|
|
1679
|
+
* heartbeatTick(deps?) — one 60 s heartbeat interval body. Each subsystem
|
|
1680
|
+
* (queue read + starvation watchdog, stall detector, heartbeat write) runs in
|
|
1681
|
+
* its own try/catch so one throw can't silently skip the others. Any failure
|
|
1682
|
+
* makes the written line `degraded: true` + `errors`; watchdogHelpers'
|
|
1683
|
+
* heartbeatFresh() and health.cjs's readFreshHeartbeat() treat such a line as
|
|
1684
|
+
* NOT fresh, so a throw never disarms the external watchdog or fakes
|
|
1685
|
+
* utilization health never read.
|
|
1686
|
+
*/
|
|
1687
|
+
function heartbeatTick(deps = {}) {
|
|
1688
|
+
const readQueue = deps.readQueueSync ?? readQueueSync;
|
|
1689
|
+
const errors = [];
|
|
1690
|
+
const guard = (subsystem, fn) => {
|
|
1691
|
+
try {
|
|
1692
|
+
return fn();
|
|
1693
|
+
} catch (e) {
|
|
1694
|
+
errors.push({ subsystem, error: e?.message ?? String(e) });
|
|
1695
|
+
console.error(`[scheduler] heartbeat subsystem "${subsystem}" failed`, e);
|
|
1696
|
+
return undefined;
|
|
1697
|
+
}
|
|
1698
|
+
};
|
|
1699
|
+
|
|
1700
|
+
const s = guard('queue-read-starvation-watchdog', () => {
|
|
1701
|
+
const q = readQueue();
|
|
1702
|
+
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
1703
|
+
// running, something must drive it. This is the only driver that does
|
|
1704
|
+
// not depend on the billing poll loop, a pause timer, or a completing
|
|
1705
|
+
// job to schedule the next tick — every one of which has failed at
|
|
1706
|
+
// least once. See classifyQueueStarvation.
|
|
1707
|
+
if (!q.unreadable) {
|
|
1708
|
+
runQueueStarvationWatchdog(q).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
1709
|
+
// Restart-request drain state machine rides this same 60 s interval —
|
|
1710
|
+
// no new driver.
|
|
1711
|
+
driveUpgradeDrain(q).catch((e) => console.error('[scheduler] upgrade drain error', e));
|
|
1712
|
+
}
|
|
1713
|
+
return q;
|
|
1714
|
+
});
|
|
1715
|
+
|
|
1716
|
+
let stall;
|
|
1717
|
+
if (s) {
|
|
1718
|
+
stall = guard('stall-detector', () => {
|
|
1719
|
+
const summary = computeStallSummary(s);
|
|
1720
|
+
// Per-project alerting (see computeStallSummary's header): a project
|
|
1721
|
+
// stalled while others are busy must still fire, and one project
|
|
1722
|
+
// recovering must not clear or suppress another's still-open episode —
|
|
1723
|
+
// that is exactly what a single module-level stallSince/stallToasted
|
|
1724
|
+
// flag masked before (the burrow-vs-others incident this PRD fixes).
|
|
1725
|
+
const now = Date.now();
|
|
1726
|
+
const stalledCwds = Object.keys(summary.byProject).filter((cwd) => summary.byProject[cwd].stalled);
|
|
1727
|
+
for (const cwd of [...stallSince.keys()]) {
|
|
1728
|
+
if (!stalledCwds.includes(cwd)) {
|
|
1729
|
+
stallSince.delete(cwd);
|
|
1730
|
+
stallToasted.delete(cwd);
|
|
1731
|
+
}
|
|
1732
|
+
}
|
|
1733
|
+
const toAlert = [];
|
|
1734
|
+
for (const cwd of stalledCwds) {
|
|
1735
|
+
if (!stallSince.has(cwd)) stallSince.set(cwd, now);
|
|
1736
|
+
if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
|
|
1737
|
+
stallToasted.set(cwd, true);
|
|
1738
|
+
toAlert.push(cwd);
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
1741
|
+
if (toAlert.length > 0) {
|
|
1742
|
+
console.error(
|
|
1743
|
+
`[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
|
|
1744
|
+
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
1745
|
+
summary.byProject,
|
|
1746
|
+
);
|
|
1747
|
+
appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: summary.total, byProject: summary.byProject });
|
|
1748
|
+
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
1749
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
1750
|
+
message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
|
|
1751
|
+
projects: toAlert,
|
|
1752
|
+
total: summary.total,
|
|
1753
|
+
byProject: summary.byProject,
|
|
1754
|
+
});
|
|
1755
|
+
}
|
|
1756
|
+
}
|
|
1757
|
+
return summary;
|
|
1758
|
+
});
|
|
1759
|
+
}
|
|
1760
|
+
|
|
1761
|
+
let entry = null;
|
|
1762
|
+
if (s && stall && errors.length === 0) {
|
|
1763
|
+
entry = guard('heartbeat-write', () => {
|
|
1764
|
+
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
1765
|
+
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
1766
|
+
// failed }` literal silently minted a NEW key for any other value, which
|
|
1767
|
+
// is how a heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
1768
|
+
// visibility instead of the alarm it should have been. Any row whose
|
|
1769
|
+
// status isn't in JOB_STATUSES routes into `unknown`, never a
|
|
1770
|
+
// freshly-minted key.
|
|
1771
|
+
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
1772
|
+
counts.unknown = 0;
|
|
1773
|
+
for (const j of s.jobs) {
|
|
1774
|
+
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
1775
|
+
counts[j.status] += 1;
|
|
1776
|
+
} else {
|
|
1777
|
+
counts.unknown += 1;
|
|
1778
|
+
}
|
|
1779
|
+
}
|
|
1780
|
+
// Logical-liveness signal for the external watchdog (see watchdogHelpers
|
|
1781
|
+
// evaluateDispatchLiveness): computed once per heartbeat from the same
|
|
1782
|
+
// queue snapshot. pendingDispatchable = pending rows minus those
|
|
1783
|
+
// terminally blocked behind a failed/skipped dependency.
|
|
1784
|
+
const blockedPending = computeBlockedChains(s.jobs).reduce((n, c) => n + c.blocked, 0);
|
|
1785
|
+
const dispatch = {
|
|
1786
|
+
lastDispatchAttemptAt: s.lastDispatchAttemptAt ?? null,
|
|
1787
|
+
lastRunAt: s.lastRunAt ?? null,
|
|
1788
|
+
lastTickReason: lastTick?.reason ?? null,
|
|
1789
|
+
pendingDispatchable: Math.max(0, counts.pending - blockedPending),
|
|
1790
|
+
runningCount: counts.running,
|
|
1791
|
+
paused: Boolean(s.paused),
|
|
1792
|
+
drain: s.drain?.active ? { since: s.drain.since ?? null, requestedAt: s.drain.requestedAt ?? null } : null,
|
|
1793
|
+
};
|
|
1794
|
+
return {
|
|
1795
|
+
ts: Date.now(),
|
|
1796
|
+
pid: process.pid,
|
|
1797
|
+
build: heartbeatBuild(),
|
|
1798
|
+
counts,
|
|
1799
|
+
dispatch,
|
|
1800
|
+
stall: { stalled: stall.stalled, total: stall.total },
|
|
1801
|
+
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
1802
|
+
quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
|
|
1803
|
+
nextReset: cachedNextReset,
|
|
1804
|
+
utilization: cachedUtilization,
|
|
1805
|
+
consecutiveFailures,
|
|
1806
|
+
// State/consecutiveFailures/degraded-budget snapshot of the shared
|
|
1807
|
+
// usage-meter breaker, so a human reading only the heartbeat log can
|
|
1808
|
+
// see the meter's own health apart from the queue's.
|
|
1809
|
+
usageMeter: {
|
|
1810
|
+
state: billing.usageCircuit.state(),
|
|
1811
|
+
consecutiveFailures,
|
|
1812
|
+
degradedBudget: computeDegradedBudget(),
|
|
1813
|
+
},
|
|
1814
|
+
};
|
|
1815
|
+
});
|
|
1816
|
+
}
|
|
1817
|
+
if (!entry || errors.length > 0) {
|
|
1818
|
+
// Deliberately carries no utilization/counts: this line says "I ran but
|
|
1819
|
+
// could not read state", and consumers must not mistake it for a fresh read.
|
|
1820
|
+
entry = { ts: Date.now(), pid: process.pid, build: heartbeatBuild(), degraded: true, errors };
|
|
1821
|
+
}
|
|
1822
|
+
appendHeartbeat(entry);
|
|
1823
|
+
return entry;
|
|
1824
|
+
}
|
|
1825
|
+
|
|
1525
1826
|
/**
|
|
1526
1827
|
* computeStallSummary(state) → { stalled, total, running, pending, byProject }
|
|
1527
1828
|
*
|
|
@@ -1564,7 +1865,12 @@ function computeStallSummary(state) {
|
|
|
1564
1865
|
byProject[key].invalid = (byProject[key].invalid || 0) + 1;
|
|
1565
1866
|
}
|
|
1566
1867
|
const total = jobs.length + invalidJobs.length;
|
|
1567
|
-
|
|
1868
|
+
// A drained queue (every row completed/skipped) is idle, not stalled — a
|
|
1869
|
+
// stall needs at least one parked problem row (failed/needs_review/
|
|
1870
|
+
// quarantined/invalid) that is waiting on someone.
|
|
1871
|
+
const isProblemStatus = (st) => st !== 'completed' && st !== 'skipped';
|
|
1872
|
+
const stalled = total > 0 && running === 0 && pending === 0 && !state?.paused
|
|
1873
|
+
&& (invalidJobs.length > 0 || jobs.some((j) => isProblemStatus(j.status)));
|
|
1568
1874
|
for (const key of Object.keys(byProject)) {
|
|
1569
1875
|
const counts = byProject[key];
|
|
1570
1876
|
const projRunning = counts.running || 0;
|
|
@@ -1572,7 +1878,9 @@ function computeStallSummary(state) {
|
|
|
1572
1878
|
const projTotal = Object.keys(counts)
|
|
1573
1879
|
.filter((k) => k !== 'stalled')
|
|
1574
1880
|
.reduce((sum, k) => sum + counts[k], 0);
|
|
1575
|
-
|
|
1881
|
+
const projProblem = Object.keys(counts)
|
|
1882
|
+
.some((k) => k !== 'stalled' && k !== 'completed' && k !== 'skipped' && counts[k] > 0);
|
|
1883
|
+
counts.stalled = projTotal > 0 && projRunning === 0 && projPending === 0 && !state?.paused && projProblem;
|
|
1576
1884
|
}
|
|
1577
1885
|
return { stalled, total, running, pending, byProject };
|
|
1578
1886
|
}
|
|
@@ -2015,6 +2323,22 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
|
2015
2323
|
// the .bak-* snapshots.
|
|
2016
2324
|
const quarantinedPaths = new Set();
|
|
2017
2325
|
function flagUnreadable(state) {
|
|
2326
|
+
// Per-shard quarantine (queueStore.unreadableCwds): snapshot each torn shard
|
|
2327
|
+
// once and name it, but never halt — other projects keep dispatching.
|
|
2328
|
+
for (const u of state.unreadableCwds ?? []) {
|
|
2329
|
+
if (!quarantinedPaths.has(u.file)) {
|
|
2330
|
+
quarantinedPaths.add(u.file);
|
|
2331
|
+
try {
|
|
2332
|
+
fs.copyFileSync(u.file, `${u.file}.corrupt-${Date.now()}`);
|
|
2333
|
+
} catch { /* best-effort: the read already failed, the copy may too */ }
|
|
2334
|
+
console.error(`[scheduler] project queue shard quarantined (${u.cwd}): ${u.error}`);
|
|
2335
|
+
logs.writeLine({
|
|
2336
|
+
level: 'error', scope: 'scheduler',
|
|
2337
|
+
message: `project queue shard unreadable — ${u.cwd} is quarantined, other projects keep dispatching`,
|
|
2338
|
+
meta: { cwd: u.cwd, path: u.file, error: u.error },
|
|
2339
|
+
});
|
|
2340
|
+
}
|
|
2341
|
+
}
|
|
2018
2342
|
if (!state.unreadable) return state;
|
|
2019
2343
|
if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
|
|
2020
2344
|
quarantinedPaths.add(state.unreadablePath);
|
|
@@ -2073,8 +2397,35 @@ async function writeQueue(state) {
|
|
|
2073
2397
|
// preceding mutate threw, so the chain never deadlocks.
|
|
2074
2398
|
let mutateTail = Promise.resolve();
|
|
2075
2399
|
|
|
2400
|
+
// Observe-only watchdog: a mutate body over MUTATE_WATCHDOG_MS is logged and
|
|
2401
|
+
// audited once per episode (latched until a mutate completes). mutateTail is
|
|
2402
|
+
// deliberately NEVER reset — it is what enforces the single-writer law, and
|
|
2403
|
+
// abandoning a live writer would let two read-modify-writes interleave.
|
|
2404
|
+
const MUTATE_WATCHDOG_MS = 60_000;
|
|
2405
|
+
let mutateWedgeLatched = false;
|
|
2406
|
+
|
|
2076
2407
|
function mutate(fn) {
|
|
2077
2408
|
const next = mutateTail.then(async () => {
|
|
2409
|
+
const wedgeTimer = setTimeout(() => {
|
|
2410
|
+
if (mutateWedgeLatched) return;
|
|
2411
|
+
mutateWedgeLatched = true;
|
|
2412
|
+
console.warn(`[scheduler] MUTATE WEDGED: a queue mutation has run > ${MUTATE_WATCHDOG_MS}ms`);
|
|
2413
|
+
appendAuditEvent('mutate_wedged', { budgetMs: MUTATE_WATCHDOG_MS });
|
|
2414
|
+
}, MUTATE_WATCHDOG_MS);
|
|
2415
|
+
if (typeof wedgeTimer.unref === 'function') wedgeTimer.unref();
|
|
2416
|
+
try {
|
|
2417
|
+
return await mutateBody(fn);
|
|
2418
|
+
} finally {
|
|
2419
|
+
clearTimeout(wedgeTimer);
|
|
2420
|
+
mutateWedgeLatched = false;
|
|
2421
|
+
}
|
|
2422
|
+
});
|
|
2423
|
+
mutateTail = next.catch(() => {}); // keep chain alive on errors
|
|
2424
|
+
return next;
|
|
2425
|
+
}
|
|
2426
|
+
|
|
2427
|
+
async function mutateBody(fn) {
|
|
2428
|
+
{
|
|
2078
2429
|
const state = await readQueue();
|
|
2079
2430
|
// Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
|
|
2080
2431
|
// state would compute its result from a queue that isn't there, and
|
|
@@ -2120,9 +2471,7 @@ function mutate(fn) {
|
|
|
2120
2471
|
}
|
|
2121
2472
|
await writeQueue(state);
|
|
2122
2473
|
return ret;
|
|
2123
|
-
}
|
|
2124
|
-
mutateTail = next.catch(() => {}); // keep chain alive on errors
|
|
2125
|
-
return next;
|
|
2474
|
+
}
|
|
2126
2475
|
}
|
|
2127
2476
|
|
|
2128
2477
|
// ---------- PRD parsing ----------
|
|
@@ -2142,11 +2491,23 @@ const parsePrd = prdParser.parsePrd;
|
|
|
2142
2491
|
// one — acceptable: PRD counts per project are bounded (hundreds, not
|
|
2143
2492
|
// millions), and correctness across multiple project dirs matters more than
|
|
2144
2493
|
// preserving the single-dir cache's steady-state zero-read fast path.
|
|
2145
|
-
async function listPrdFiles() {
|
|
2494
|
+
async function listPrdFiles(skipCwds) {
|
|
2146
2495
|
ensureDirs();
|
|
2147
2496
|
const dirs = candidatePrdsDirs();
|
|
2148
2497
|
const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
|
|
2149
|
-
|
|
2498
|
+
let files = perDir.flat();
|
|
2499
|
+
// A quarantined project has no job rows this pass; scanning its PRDs would
|
|
2500
|
+
// mint fresh `pending` rows for work that may already be running.
|
|
2501
|
+
if (skipCwds && skipCwds.size > 0) {
|
|
2502
|
+
const prefixes = [...skipCwds].map((c) => c + path.sep);
|
|
2503
|
+
files = files.filter((f) => !prefixes.some((p) => f.startsWith(p)));
|
|
2504
|
+
}
|
|
2505
|
+
return { files: files.sort(), dirCount: dirs.length };
|
|
2506
|
+
}
|
|
2507
|
+
|
|
2508
|
+
/** Set of cwds whose shard is quarantined in this merged read. */
|
|
2509
|
+
function quarantinedCwdSet(state) {
|
|
2510
|
+
return new Set((state?.unreadableCwds ?? []).map((u) => u.cwd));
|
|
2150
2511
|
}
|
|
2151
2512
|
|
|
2152
2513
|
/**
|
|
@@ -2186,7 +2547,22 @@ async function allocateParallelGroup(cwd) {
|
|
|
2186
2547
|
* Safety:
|
|
2187
2548
|
* - PID-recycling: between app death and this call, another process may have
|
|
2188
2549
|
* reused the PID. We read /proc/<pid>/cmdline (Linux) or `ps -p` (macOS)
|
|
2189
|
-
* and only SIGTERM if the cmdline
|
|
2550
|
+
* and only SIGTERM if the cmdline matches /\bclaude\b/. Since procName
|
|
2551
|
+
* aliasing, cmdline[0] is the alias path (`.../procnames/sm-claude-job`) or
|
|
2552
|
+
* the smArgv0 label (`sm-claude-job:<slug>`), NOT the claude bin path —
|
|
2553
|
+
* both still contain the word `claude`. The macOS `ps -p <pid> -o command=`
|
|
2554
|
+
* branch has the same exposure and the same guarantee (ps shows argv0).
|
|
2555
|
+
* Migration: cmdline is fixed at exec, and no claude procIdentity is
|
|
2556
|
+
* persisted (job.runtime carries none), so a pre-aliasing process recorded
|
|
2557
|
+
* and compared after upgrade still compares equal to itself — no
|
|
2558
|
+
* tolerance needed; unaliased legacy cmdlines also match /\bclaude\b/.
|
|
2559
|
+
* - recordedIdentity (optional): when the caller has a COMPLETE prior
|
|
2560
|
+
* procIdentity for this pid (startTicks + cmdline), it is used only as a
|
|
2561
|
+
* VETO — if it provably differs from the pid's live identity right now,
|
|
2562
|
+
* the pid was recycled and the kill is refused ('mismatch') even before
|
|
2563
|
+
* the cmdline heuristic below runs. No recorded identity (today's only
|
|
2564
|
+
* case — job.runtime carries no identity yet) falls through unchanged to
|
|
2565
|
+
* the existing /\bclaude\b/ + `ps -p` heuristics.
|
|
2190
2566
|
* - Detached process group: jobs are spawned with detached:true so we kill
|
|
2191
2567
|
* -pid (the group). If the group leader is already gone, that fails
|
|
2192
2568
|
* silently and we fall back to single-pid kill.
|
|
@@ -2194,12 +2570,17 @@ async function allocateParallelGroup(cwd) {
|
|
|
2194
2570
|
* scheduled via setTimeout to clean up any process ignoring SIGTERM.
|
|
2195
2571
|
*
|
|
2196
2572
|
* Returns: 'killed' (cmdline matched + signal sent), 'gone' (pid not alive),
|
|
2197
|
-
* 'mismatch' (pid alive but cmdline doesn't look like claude
|
|
2573
|
+
* 'mismatch' (pid alive but cmdline doesn't look like claude, or a
|
|
2574
|
+
* complete recorded identity proves the pid was recycled),
|
|
2198
2575
|
* 'unknown' (couldn't read cmdline — leave the pid alone).
|
|
2199
2576
|
*/
|
|
2200
|
-
function killOrphanClaudePid(pid) {
|
|
2577
|
+
function killOrphanClaudePid(pid, recordedIdentity = null) {
|
|
2201
2578
|
if (!pid || typeof pid !== 'number' || pid <= 1) return 'gone';
|
|
2202
2579
|
try { process.kill(pid, 0); } catch { return 'gone'; }
|
|
2580
|
+
if (recordedIdentity && recordedIdentity.complete
|
|
2581
|
+
&& isDifferentProcess(recordedIdentity, procIdentityOf(pid))) {
|
|
2582
|
+
return 'mismatch';
|
|
2583
|
+
}
|
|
2203
2584
|
let cmdline = '';
|
|
2204
2585
|
try {
|
|
2205
2586
|
cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
|
|
@@ -2307,25 +2688,48 @@ async function reconcile(state) {
|
|
|
2307
2688
|
// has no queue row yet, so it is never "live" and gets archived here
|
|
2308
2689
|
// instead of ever reaching the onDisk scan below.
|
|
2309
2690
|
let phaseStartMs = Date.now();
|
|
2310
|
-
|
|
2691
|
+
const skipCwds = quarantinedCwdSet(state);
|
|
2692
|
+
await consolidateAllFlatPrds(allProjectCwds(), skipCwds);
|
|
2311
2693
|
phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
|
|
2312
2694
|
|
|
2313
2695
|
phaseStartMs = Date.now();
|
|
2314
|
-
const { files, dirCount } = await listPrdFiles();
|
|
2696
|
+
const { files, dirCount } = await listPrdFiles(skipCwds);
|
|
2315
2697
|
phaseMs.prdDirResolve = Date.now() - phaseStartMs;
|
|
2316
2698
|
|
|
2317
2699
|
phaseStartMs = Date.now();
|
|
2318
2700
|
const onDisk = new Map();
|
|
2701
|
+
// Slugs derive from title text with no cwd salt, so two different projects
|
|
2702
|
+
// can legitimately queue an identically-slugged PRD — onDisk alone can
|
|
2703
|
+
// only hold ONE parsed PRD per slug (last-file-wins), which would silently
|
|
2704
|
+
// hand an EXISTING row the wrong project's PRD (or none at all) when two
|
|
2705
|
+
// projects collide on a slug. This side index lets the two existing-row
|
|
2706
|
+
// lookups below (job refresh + invalid-row repair) disambiguate by the
|
|
2707
|
+
// row's own cwd first; the fresh-discovery loop further down still reads
|
|
2708
|
+
// the bare `onDisk` (unscoped) since a same-slug NEW-PRD collision across
|
|
2709
|
+
// two projects is a rarer edge this reconcile pass doesn't yet resolve.
|
|
2710
|
+
const onDiskByCwd = new Map();
|
|
2319
2711
|
for (const f of files) {
|
|
2320
2712
|
try {
|
|
2321
2713
|
// Per-file await: parsing is mtime-cached so steady-state hits zero
|
|
2322
2714
|
// disk reads; on cold cache the awaits keep the main thread responsive.
|
|
2323
2715
|
const p = await parsePrd(f);
|
|
2324
2716
|
onDisk.set(p.slug, p);
|
|
2717
|
+
if (p.cwd) onDiskByCwd.set(`${p.slug}::${p.cwd}`, p);
|
|
2325
2718
|
} catch (e) {
|
|
2326
2719
|
console.warn('[scheduler] failed to parse', f, e?.message);
|
|
2327
2720
|
}
|
|
2328
2721
|
}
|
|
2722
|
+
// resolvePrdForJob(slug, cwd) — cwd-scoped PRD lookup for an EXISTING
|
|
2723
|
+
// queue row, falling back to the unscoped onDisk entry when this exact
|
|
2724
|
+
// (slug, cwd) pair has no PRD (e.g. cwd is null/stale) — same behavior as
|
|
2725
|
+
// a bare onDisk.get() for every slug that isn't cross-project-colliding.
|
|
2726
|
+
function resolvePrdForJob(slug, cwd) {
|
|
2727
|
+
if (cwd) {
|
|
2728
|
+
const scoped = onDiskByCwd.get(`${slug}::${cwd}`);
|
|
2729
|
+
if (scoped) return scoped;
|
|
2730
|
+
}
|
|
2731
|
+
return onDisk.get(slug);
|
|
2732
|
+
}
|
|
2329
2733
|
phaseMs.parseLoop = Date.now() - phaseStartMs;
|
|
2330
2734
|
|
|
2331
2735
|
const next = [];
|
|
@@ -2345,7 +2749,7 @@ async function reconcile(state) {
|
|
|
2345
2749
|
// historyTerminalBySlug() below and backfilled before being dropped.
|
|
2346
2750
|
const terminalDroppedNeedingHistoryCheck = [];
|
|
2347
2751
|
for (const job of state.jobs) {
|
|
2348
|
-
const p =
|
|
2752
|
+
const p = resolvePrdForJob(job.slug, job.cwd);
|
|
2349
2753
|
if (!p) {
|
|
2350
2754
|
// A terminal job whose .md is gone was archived on purpose — dropping
|
|
2351
2755
|
// its row is the intended end of the auto-archive flow, PROVIDED it's
|
|
@@ -2379,10 +2783,24 @@ async function reconcile(state) {
|
|
|
2379
2783
|
continue;
|
|
2380
2784
|
}
|
|
2381
2785
|
seen.add(job.slug);
|
|
2786
|
+
// p.cwd REFINES the row's existing cwd; it never erases one. A PRD
|
|
2787
|
+
// file with no `cwd:` frontmatter parses p.cwd as undefined — falling
|
|
2788
|
+
// through to a bare `cwd: p.cwd` here nulled the row's real cwd,
|
|
2789
|
+
// which queueStore.writeSplit then buckets into
|
|
2790
|
+
// schedulerBatch.js's DEFAULT_PROJECT_CWD, silently relocating the
|
|
2791
|
+
// row into the WRONG project's queue.json shard and emptying the
|
|
2792
|
+
// owning project's shard underneath it (2026-09 data-loss incident).
|
|
2793
|
+
// Resolved ONCE into a local so originSessionId's fallback below
|
|
2794
|
+
// resolves against the SAME cwd this row actually gets, not the raw
|
|
2795
|
+
// (possibly undefined) p.cwd — resolveOriginSessionId(undefined, ...)
|
|
2796
|
+
// returns null unconditionally, which silently dropped the origin link
|
|
2797
|
+
// for every PRD with no `cwd:` frontmatter even though a good cwd was
|
|
2798
|
+
// available one line below.
|
|
2799
|
+
const refreshedCwd = p.cwd ?? job.cwd ?? null;
|
|
2382
2800
|
const updatedJob = {
|
|
2383
2801
|
...job,
|
|
2384
2802
|
title: p.title,
|
|
2385
|
-
cwd:
|
|
2803
|
+
cwd: refreshedCwd,
|
|
2386
2804
|
parallelGroup: p.parallelGroup,
|
|
2387
2805
|
estimateMinutes: p.estimateMinutes,
|
|
2388
2806
|
sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
|
|
@@ -2396,7 +2814,7 @@ async function reconcile(state) {
|
|
|
2396
2814
|
quietMachine: p.quietMachine === true,
|
|
2397
2815
|
budgetExempt: p.budgetExempt === true,
|
|
2398
2816
|
originSessionId: job.originSessionId
|
|
2399
|
-
?? resolveOriginSessionId(
|
|
2817
|
+
?? resolveOriginSessionId(refreshedCwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
2400
2818
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2401
2819
|
agentType: p.agentType ?? job.agentType ?? null,
|
|
2402
2820
|
};
|
|
@@ -2479,7 +2897,7 @@ async function reconcile(state) {
|
|
|
2479
2897
|
for (const inv of invalidJobs) {
|
|
2480
2898
|
if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
|
|
2481
2899
|
const oldStatus = inv.row?.status;
|
|
2482
|
-
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir:
|
|
2900
|
+
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: schedulerPaths.runsDir() });
|
|
2483
2901
|
if (hist) {
|
|
2484
2902
|
// Never resurrect: this slug already has a durable terminal record
|
|
2485
2903
|
// elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
|
|
@@ -2492,17 +2910,22 @@ async function reconcile(state) {
|
|
|
2492
2910
|
});
|
|
2493
2911
|
continue;
|
|
2494
2912
|
}
|
|
2495
|
-
const p =
|
|
2913
|
+
const p = resolvePrdForJob(inv.slug, inv.row?.cwd);
|
|
2496
2914
|
if (!p) {
|
|
2497
2915
|
// PRD file also gone with no terminal record anywhere — nothing to
|
|
2498
2916
|
// repair against. queueStore already logged the quarantine once.
|
|
2499
2917
|
continue;
|
|
2500
2918
|
}
|
|
2919
|
+
// Same cwd-refines-not-erases rule as the normal refresh path above, and
|
|
2920
|
+
// same reason for resolving it once into a local: originSessionId's
|
|
2921
|
+
// fallback must resolve against the cwd this row actually gets, not the
|
|
2922
|
+
// raw (possibly undefined) p.cwd.
|
|
2923
|
+
const repairedCwd = p.cwd ?? inv.row?.cwd ?? null;
|
|
2501
2924
|
const job = {
|
|
2502
2925
|
...inv.row,
|
|
2503
2926
|
slug: inv.slug,
|
|
2504
2927
|
title: p.title,
|
|
2505
|
-
cwd:
|
|
2928
|
+
cwd: repairedCwd,
|
|
2506
2929
|
parallelGroup: p.parallelGroup,
|
|
2507
2930
|
estimateMinutes: p.estimateMinutes,
|
|
2508
2931
|
sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
|
|
@@ -2512,7 +2935,7 @@ async function reconcile(state) {
|
|
|
2512
2935
|
disposition: p.disposition ?? null,
|
|
2513
2936
|
quietMachine: p.quietMachine === true,
|
|
2514
2937
|
budgetExempt: p.budgetExempt === true,
|
|
2515
|
-
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(
|
|
2938
|
+
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
|
|
2516
2939
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2517
2940
|
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
2518
2941
|
};
|
|
@@ -2616,17 +3039,27 @@ async function reconcile(state) {
|
|
|
2616
3039
|
// guard above inert. Fall back to reading the slug's own newest run
|
|
2617
3040
|
// sidecars straight off disk — same "don't resurrect an already-terminal
|
|
2618
3041
|
// slug" intent, independent of history.jsonl's existence.
|
|
2619
|
-
const fallback = latestTerminalOutcomeForSlug(slug, { runsDir:
|
|
3042
|
+
const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: schedulerPaths.runsDir() });
|
|
2620
3043
|
if (fallback) {
|
|
2621
3044
|
if (fallback.status === 'completed') {
|
|
2622
3045
|
historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
|
|
2623
3046
|
}
|
|
2624
3047
|
continue;
|
|
2625
3048
|
}
|
|
3049
|
+
// No prior row exists to fall back to (this is a fresh discovery), so a
|
|
3050
|
+
// PRD file with no `cwd:` frontmatter falls back to the project root it
|
|
3051
|
+
// was actually found under (derived from its own file path) rather than
|
|
3052
|
+
// nulling out to schedulerBatch.js's DEFAULT_PROJECT_CWD. Resolved once
|
|
3053
|
+
// so originSessionId (below) resolves against this SAME cwd — passing
|
|
3054
|
+
// the raw p.cwd there instead would resolve against `undefined` for
|
|
3055
|
+
// exactly the no-frontmatter case this fallback exists to handle, since
|
|
3056
|
+
// resolveOriginSessionId(cwd, ...) returns null unconditionally when
|
|
3057
|
+
// `cwd` is falsy.
|
|
3058
|
+
const discoveredCwd = p.cwd ?? deriveProjectCwdFromPrdPath(p.path) ?? null;
|
|
2626
3059
|
const entry = {
|
|
2627
3060
|
slug,
|
|
2628
3061
|
title: p.title,
|
|
2629
|
-
cwd:
|
|
3062
|
+
cwd: discoveredCwd,
|
|
2630
3063
|
parallelGroup: p.parallelGroup,
|
|
2631
3064
|
estimateMinutes: p.estimateMinutes,
|
|
2632
3065
|
sourcePromptId: p.sourcePromptId,
|
|
@@ -2636,7 +3069,7 @@ async function reconcile(state) {
|
|
|
2636
3069
|
disposition: p.disposition ?? null,
|
|
2637
3070
|
quietMachine: p.quietMachine === true,
|
|
2638
3071
|
budgetExempt: p.budgetExempt === true,
|
|
2639
|
-
originSessionId: resolveOriginSessionId(
|
|
3072
|
+
originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
|
|
2640
3073
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2641
3074
|
agentType: p.agentType ?? null,
|
|
2642
3075
|
status: 'pending',
|
|
@@ -2787,14 +3220,73 @@ async function reconcile(state) {
|
|
|
2787
3220
|
// ---------- next-reset detection ----------
|
|
2788
3221
|
|
|
2789
3222
|
let cachedNextReset = null; // bare ISO string or null
|
|
2790
|
-
let cachedUtilization = null; //
|
|
3223
|
+
let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
|
|
3224
|
+
// ms timestamp of the last FRESH reset observation (see recordObservedReset)
|
|
3225
|
+
// — distinct from Date.now(), so persistSchedulerState never re-stamps a
|
|
3226
|
+
// stale cachedNextReset as "just observed" on every poll cycle.
|
|
3227
|
+
let lastResetObservedAtMs = null;
|
|
3228
|
+
// Full usage payload (`{ five_hour, limits?, ... }`) from the last SUCCESSFUL
|
|
3229
|
+
// poll — the input degradedBudget() carries forward while the meter is down.
|
|
3230
|
+
// Never itself defaults to 0; absent (null) reads as "no signal yet" and
|
|
3231
|
+
// degradedBudget() treats that conservatively (100% / capped concurrency).
|
|
3232
|
+
let lastGoodUsagePayload = null;
|
|
3233
|
+
// Non-null while the meter is degraded (circuit open, or a poll otherwise
|
|
3234
|
+
// failed to return 'ok') — narrows tickQueue's freeSlots as a picker-side
|
|
3235
|
+
// hold instead of a second slot pool (see tickQueue's freeSlots computation).
|
|
3236
|
+
// Cleared to null the moment a poll succeeds or the meter is inapplicable.
|
|
3237
|
+
let degradedConcurrencyCapValue = null;
|
|
3238
|
+
// Deliberately a named constant, not a bare zero literal assigned straight
|
|
3239
|
+
// into cachedUtilization: a genuine "no consumer meter to poll" (enterprise
|
|
3240
|
+
// auth) is categorically different from "the meter is down and we don't
|
|
3241
|
+
// know" — the latter must never read as 0%/full-speed-ahead.
|
|
3242
|
+
const NO_METER_UTILIZATION = 0;
|
|
3243
|
+
|
|
3244
|
+
/**
|
|
3245
|
+
* Records a freshly-observed reset, stamping lastResetObservedAtMs only when
|
|
3246
|
+
* there actually WAS a reset to observe (never on every poll regardless of
|
|
3247
|
+
* payload content — see persistSchedulerState's header).
|
|
3248
|
+
*/
|
|
3249
|
+
function recordObservedReset(resetIso) {
|
|
3250
|
+
if (resetIso) {
|
|
3251
|
+
cachedNextReset = resetIso;
|
|
3252
|
+
lastResetObservedAtMs = Date.now();
|
|
3253
|
+
} else {
|
|
3254
|
+
cachedNextReset = null;
|
|
3255
|
+
}
|
|
3256
|
+
}
|
|
3257
|
+
|
|
3258
|
+
/** Pure: this poll/executor cycle's conservative budget while the meter is degraded. */
|
|
3259
|
+
function computeDegradedBudget() {
|
|
3260
|
+
return degradedBudget(lastGoodUsagePayload, {
|
|
3261
|
+
observed429: lastFailureKind === 'meter_rate_limited',
|
|
3262
|
+
resetsAt: cachedNextReset,
|
|
3263
|
+
now: Date.now(),
|
|
3264
|
+
configuredCap: sessionSlots.snapshot().total,
|
|
3265
|
+
});
|
|
3266
|
+
}
|
|
3267
|
+
|
|
3268
|
+
/**
|
|
3269
|
+
* Computes AND applies this cycle's degraded budget to cachedUtilization /
|
|
3270
|
+
* degradedConcurrencyCapValue in one call — every "meter down" branch in
|
|
3271
|
+
* pollLoop/runQueueStarvationWatchdog needs the exact same
|
|
3272
|
+
* compute-then-assign-both-fields pair, so it lives once here rather than
|
|
3273
|
+
* being copy-pasted at each call site (where a future change to how the
|
|
3274
|
+
* budget applies would otherwise have to be made N times).
|
|
3275
|
+
*/
|
|
3276
|
+
function applyDegradedBudget() {
|
|
3277
|
+
const budget = computeDegradedBudget();
|
|
3278
|
+
cachedUtilization = budget.utilization;
|
|
3279
|
+
degradedConcurrencyCapValue = budget.concurrencyCap;
|
|
3280
|
+
return budget;
|
|
3281
|
+
}
|
|
2791
3282
|
|
|
2792
3283
|
/** Fetches latest usage from billing API. Throws on any error — callers handle it. */
|
|
2793
3284
|
async function refreshNextReset() {
|
|
2794
3285
|
const r = await billing.fetchUsage();
|
|
2795
3286
|
if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
|
|
2796
|
-
|
|
2797
|
-
|
|
3287
|
+
const window = bindingWindow(r.data?.usage);
|
|
3288
|
+
recordObservedReset(window.resets_at ?? null);
|
|
3289
|
+
cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
|
|
2798
3290
|
return cachedNextReset;
|
|
2799
3291
|
}
|
|
2800
3292
|
|
|
@@ -2805,15 +3297,38 @@ function getNextResetCached() {
|
|
|
2805
3297
|
/**
|
|
2806
3298
|
* Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
|
|
2807
3299
|
* Prefers the BINDING window read off the run's own log — refreshNextReset()
|
|
2808
|
-
* only ever reports
|
|
2809
|
-
*
|
|
2810
|
-
*
|
|
2811
|
-
* to the billing-endpoint-derived reset only when the log yields nothing
|
|
2812
|
-
|
|
2813
|
-
|
|
3300
|
+
* only ever reports the binding window at POLL time, which can be a
|
|
3301
|
+
* different clock than what actually 429'd this run (five_hour can read 0%
|
|
3302
|
+
* utilization at the very same moment a seven_day window binds). Falls back
|
|
3303
|
+
* to the billing-endpoint-derived reset only when the log yields nothing —
|
|
3304
|
+
* and rejects EITHER source when it has already passed relative to `nowMs`
|
|
3305
|
+
* (usageCircuit.isResetFresh): a stale reset must never arm a resume timer
|
|
3306
|
+
* that already elapsed (that's the 30-second-nap bug computeEffectiveResumeAt
|
|
3307
|
+
* below also guards against), so a stale value here returns null and lets
|
|
3308
|
+
* the caller's 30-minute fallback take over instead.
|
|
3309
|
+
*/
|
|
3310
|
+
function resolveRateLimitPauseReset(logPath, billingResetIso, nowMs = Date.now()) {
|
|
2814
3311
|
const logReset = resolveBindingRateLimitReset(logPath);
|
|
2815
|
-
if (logReset != null)
|
|
2816
|
-
|
|
3312
|
+
if (logReset != null) {
|
|
3313
|
+
const iso = new Date(logReset * 1000).toISOString();
|
|
3314
|
+
if (isResetFresh(iso, nowMs)) return iso;
|
|
3315
|
+
}
|
|
3316
|
+
if (billingResetIso && isResetFresh(billingResetIso, nowMs)) return billingResetIso;
|
|
3317
|
+
return null;
|
|
3318
|
+
}
|
|
3319
|
+
|
|
3320
|
+
/**
|
|
3321
|
+
* Shared by both rate-limit pause sites (spawnJob's rateLimited branch and
|
|
3322
|
+
* reapDeadRunningJobs): the billing-endpoint-derived fallback reset used
|
|
3323
|
+
* when the run's own log yields no binding window. While the shared
|
|
3324
|
+
* usageCircuit is OPEN, skip calling refreshNextReset() — it would just be
|
|
3325
|
+
* another request against the endpoint the breaker just decided is down —
|
|
3326
|
+
* and fall back to whatever was last cached instead (resolveRateLimitPauseReset
|
|
3327
|
+
* itself still rejects that cached value if it has since gone stale).
|
|
3328
|
+
*/
|
|
3329
|
+
async function billingResetForPause() {
|
|
3330
|
+
if (billing.usageCircuit.state() === 'open') return cachedNextReset;
|
|
3331
|
+
return refreshNextReset().catch(() => cachedNextReset);
|
|
2817
3332
|
}
|
|
2818
3333
|
|
|
2819
3334
|
// ---------- health / poll state ----------
|
|
@@ -2828,10 +3343,27 @@ let firstFailureAt = null;
|
|
|
2828
3343
|
let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
|
|
2829
3344
|
let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
|
|
2830
3345
|
let pauseClearedManuallyAt = null;
|
|
3346
|
+
// In-memory only (no new persisted field): when clearPause last actually lifted a pause — a
|
|
3347
|
+
// legitimate restart point of the dispatch-idleness clock (dispatchIdleMs).
|
|
3348
|
+
let lastPauseClearedAt = null;
|
|
2831
3349
|
// PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
|
|
2832
3350
|
// not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
|
|
2833
3351
|
// Reset alongside consecutiveFailures everywhere that resets to 0.
|
|
2834
3352
|
let failureStreakWarned = false;
|
|
3353
|
+
// ms timestamp the initial WARN fired this streak — anchors the periodic
|
|
3354
|
+
// escalation cadence below. Reset to null alongside failureStreakWarned.
|
|
3355
|
+
let failureStreakWarnedAt = null;
|
|
3356
|
+
// ms timestamp of the last periodic escalation (audit event + opsErrorLog
|
|
3357
|
+
// line) this streak. Reset to null alongside failureStreakWarned so a LATER
|
|
3358
|
+
// streak re-arms both the initial WARN and the escalation cadence.
|
|
3359
|
+
let lastEscalationAtMs = null;
|
|
3360
|
+
|
|
3361
|
+
/** The failure-streak-WARN trio must always reset together — one helper, not 3 copies. */
|
|
3362
|
+
function resetFailureStreak() {
|
|
3363
|
+
failureStreakWarned = false;
|
|
3364
|
+
failureStreakWarnedAt = null;
|
|
3365
|
+
lastEscalationAtMs = null;
|
|
3366
|
+
}
|
|
2835
3367
|
// Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
|
|
2836
3368
|
// branch and the 'meter_rate_limited' branch below share this cap — a single
|
|
2837
3369
|
// constant so the two never drift to different ceilings).
|
|
@@ -2840,6 +3372,11 @@ const BACKOFF_MAX_MS = 480_000; // 8 minutes
|
|
|
2840
3372
|
// jitter and becomes worth a human's attention. health.cjs imports this so the
|
|
2841
3373
|
// WARN and the `npm run health` non-GREEN trip at the exact same count.
|
|
2842
3374
|
const FAILURE_STREAK_WARN_THRESHOLD = 5;
|
|
3375
|
+
// How often a PERSISTING failure streak re-escalates (audit event +
|
|
3376
|
+
// opsErrorLog line) after the initial WARN, and the health.cjs YELLOW->RED
|
|
3377
|
+
// threshold for how long the usageCircuit has been open — one constant so
|
|
3378
|
+
// the log cadence and the health-color flip never drift apart.
|
|
3379
|
+
const FAILURE_STREAK_ESCALATION_MS = 30 * 60_000; // 30 minutes
|
|
2843
3380
|
|
|
2844
3381
|
/** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
|
|
2845
3382
|
function nextBackoffMs(prevBackoffMs) {
|
|
@@ -2856,19 +3393,59 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
|
|
|
2856
3393
|
return consecutiveFailures >= threshold && !alreadyWarned;
|
|
2857
3394
|
}
|
|
2858
3395
|
|
|
2859
|
-
/**
|
|
3396
|
+
/**
|
|
3397
|
+
* Pure: does a PERSISTING failure streak warrant another escalation (audit
|
|
3398
|
+
* event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
|
|
3399
|
+
* relevant once the streak has already crossed `warnThreshold` (the initial
|
|
3400
|
+
* WARN); `lastEscalatedAtMs` null means no escalation has fired yet this
|
|
3401
|
+
* streak, so the first one is due immediately. Re-arms automatically once a
|
|
3402
|
+
* streak clears (the caller resets `lastEscalatedAtMs` to null alongside
|
|
3403
|
+
* `failureStreakWarned`), so a later independent streak escalates again.
|
|
3404
|
+
*/
|
|
3405
|
+
function shouldEscalateFailureStreak(consecutiveFailures, lastEscalatedAtMs, nowMs, thresholdMs = FAILURE_STREAK_ESCALATION_MS, warnThreshold = FAILURE_STREAK_WARN_THRESHOLD) {
|
|
3406
|
+
if (consecutiveFailures < warnThreshold) return false;
|
|
3407
|
+
if (!lastEscalatedAtMs) return true;
|
|
3408
|
+
return nowMs - lastEscalatedAtMs >= thresholdMs;
|
|
3409
|
+
}
|
|
3410
|
+
|
|
3411
|
+
/**
|
|
3412
|
+
* Emits the one-time opsErrorLog WARN the moment a failure streak crosses
|
|
3413
|
+
* the threshold, then — while that streak PERSISTS — re-escalates (audit
|
|
3414
|
+
* event + another opsErrorLog line) every FAILURE_STREAK_ESCALATION_MS so a
|
|
3415
|
+
* human watching only the audit log still sees a live incident, not just the
|
|
3416
|
+
* single opening WARN from hours ago.
|
|
3417
|
+
*/
|
|
2860
3418
|
function warnFailureStreakIfNeeded() {
|
|
2861
|
-
|
|
2862
|
-
failureStreakWarned
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
|
|
2866
|
-
|
|
2867
|
-
|
|
2868
|
-
|
|
2869
|
-
|
|
2870
|
-
|
|
2871
|
-
|
|
3419
|
+
const nowMs = Date.now();
|
|
3420
|
+
if (shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) {
|
|
3421
|
+
failureStreakWarned = true;
|
|
3422
|
+
failureStreakWarnedAt = nowMs;
|
|
3423
|
+
lastEscalationAtMs = nowMs;
|
|
3424
|
+
try {
|
|
3425
|
+
appendError({
|
|
3426
|
+
cwd: DEFAULT_PROJECT_CWD,
|
|
3427
|
+
scope: 'scheduler',
|
|
3428
|
+
level: 'warn',
|
|
3429
|
+
message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
|
|
3430
|
+
meta: { consecutiveFailures, backoffMs, lastFailureKind },
|
|
3431
|
+
});
|
|
3432
|
+
} catch { /* durable logging must never break the poll loop */ }
|
|
3433
|
+
return;
|
|
3434
|
+
}
|
|
3435
|
+
if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
|
|
3436
|
+
lastEscalationAtMs = nowMs;
|
|
3437
|
+
const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
|
|
3438
|
+
try {
|
|
3439
|
+
appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
|
|
3440
|
+
appendError({
|
|
3441
|
+
cwd: DEFAULT_PROJECT_CWD,
|
|
3442
|
+
scope: 'scheduler',
|
|
3443
|
+
level: 'warn',
|
|
3444
|
+
message: `usage/rate-limit poller streak still failing after ${persistedMinutes}m (${consecutiveFailures} consecutive, kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
|
|
3445
|
+
meta: { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes },
|
|
3446
|
+
});
|
|
3447
|
+
} catch { /* durable logging must never break the poll loop */ }
|
|
3448
|
+
}
|
|
2872
3449
|
}
|
|
2873
3450
|
// PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
|
|
2874
3451
|
// See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
|
|
@@ -2882,6 +3459,7 @@ let resumeTimer = null;
|
|
|
2882
3459
|
let pollLoopTimer = null;
|
|
2883
3460
|
let rescheduleInterval = null;
|
|
2884
3461
|
let heartbeatInterval = null;
|
|
3462
|
+
let dispatchLoopHandle = null;
|
|
2885
3463
|
// Stall-detector state (computeStallSummary), read/written only inside the
|
|
2886
3464
|
// heartbeat interval below. Keyed per-project cwd (never a single value) —
|
|
2887
3465
|
// a single module-level flag would let one busy project's activity clear or
|
|
@@ -2905,12 +3483,13 @@ const runningSet = new Set();
|
|
|
2905
3483
|
// N concurrent Opus processes — the >3-concurrent class that OOM-killed Electron.
|
|
2906
3484
|
// Over-cap requests are QUEUED (not dropped) and drained as slots free, so a failed
|
|
2907
3485
|
// PRD that never reaches 'needs_review' still eventually gets its fix-plan authored.
|
|
2908
|
-
let investigationsInFlight = 0;
|
|
2909
3486
|
const MAX_CONCURRENT_INVESTIGATIONS = 1;
|
|
2910
3487
|
const deferredInvestigations = new Map(); // fixable-job slug -> { failedJob, runDir }
|
|
3488
|
+
// Mirror of machine `drain.active`, kept in memory so spawn sites need no queue read.
|
|
3489
|
+
let drainActive = false;
|
|
2911
3490
|
|
|
2912
3491
|
function drainDeferredInvestigation() {
|
|
2913
|
-
if (
|
|
3492
|
+
if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) return;
|
|
2914
3493
|
const next = deferredInvestigations.entries().next();
|
|
2915
3494
|
if (next.done) return;
|
|
2916
3495
|
const [slug, ctx] = next.value;
|
|
@@ -3001,6 +3580,7 @@ function buildScheduleStatePayload(state) {
|
|
|
3001
3580
|
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
3002
3581
|
nextReset: getNextResetCached(),
|
|
3003
3582
|
paused: state.paused,
|
|
3583
|
+
drain: state.drain ?? null,
|
|
3004
3584
|
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
3005
3585
|
// now and why, plus any degraded-mode env in force. Empty objects when healthy.
|
|
3006
3586
|
launchBlocks: state.launchBlocks ?? {},
|
|
@@ -3170,13 +3750,17 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
|
|
|
3170
3750
|
/**
|
|
3171
3751
|
* Pure: decides the resumeAt actually armed for a pause. 'network' and
|
|
3172
3752
|
* 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
|
|
3173
|
-
* explicit resumeAt is supplied
|
|
3174
|
-
*
|
|
3175
|
-
*
|
|
3176
|
-
*
|
|
3753
|
+
* explicit resumeAt is supplied, OR when the supplied resumeAtIso has
|
|
3754
|
+
* already passed (usageCircuit.isResetFresh) — a stale reset used to produce
|
|
3755
|
+
* `Math.max(30_000, <negative>)` downstream in computeResumeDelay, a
|
|
3756
|
+
* 30-SECOND nap instead of a real pause, spinning the queue right back into
|
|
3757
|
+
* the same still-active rate limit. The live failure mode is the billing
|
|
3758
|
+
* usage endpoint itself 429ing while the log yields no fresh binding window
|
|
3759
|
+
* either, which used to leave an indefinite pause with no resume timer at
|
|
3760
|
+
* all (a queue that never comes back on its own) — this covers both.
|
|
3177
3761
|
*/
|
|
3178
3762
|
function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
|
|
3179
|
-
if (resumeAtIso) return resumeAtIso;
|
|
3763
|
+
if (resumeAtIso && isResetFresh(resumeAtIso, nowMs)) return resumeAtIso;
|
|
3180
3764
|
if (reason === 'network' || reason === 'rate_limit') {
|
|
3181
3765
|
return new Date(nowMs + 30 * 60_000).toISOString();
|
|
3182
3766
|
}
|
|
@@ -3196,11 +3780,14 @@ function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
|
|
|
3196
3780
|
|
|
3197
3781
|
async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
3198
3782
|
const { observedAt = null, force = false } = opts;
|
|
3783
|
+
const isManual = reason === 'manual';
|
|
3199
3784
|
// Honor manual-override cooldown: if the user cleared a pause within the
|
|
3200
3785
|
// last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
|
|
3201
3786
|
// backed by a fresh observation (a run that started after the clear) or is
|
|
3202
3787
|
// forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
|
|
3203
|
-
|
|
3788
|
+
// A user-initiated 'manual' pause is never an auto-detection, so the cooldown
|
|
3789
|
+
// (which exists to ignore STALE auto-detections) does not apply to it.
|
|
3790
|
+
if (!isManual && isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
|
|
3204
3791
|
console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
|
|
3205
3792
|
return;
|
|
3206
3793
|
}
|
|
@@ -3210,15 +3797,24 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
|
3210
3797
|
console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
|
|
3211
3798
|
}
|
|
3212
3799
|
|
|
3213
|
-
|
|
3800
|
+
// 'manual' never arms a resume timer: only schedule:resume / run-now clears it.
|
|
3801
|
+
const effectiveResumeAt = isManual ? null : computeEffectiveResumeAt(reason, resumeAtIso);
|
|
3214
3802
|
|
|
3215
|
-
await mutate((s) => {
|
|
3803
|
+
const kept = await mutate((s) => {
|
|
3804
|
+
// A user pause outranks every auto-pause (rate_limit/auth/network): the
|
|
3805
|
+
// auto path must not overwrite it, or its resume timer would auto-clear it.
|
|
3806
|
+
if (!isManual && s.paused && s.paused.reason === 'manual') return true;
|
|
3216
3807
|
if (s.paused && s.paused.reason === reason) {
|
|
3217
3808
|
if (effectiveResumeAt) s.paused.resumeAt = effectiveResumeAt;
|
|
3218
3809
|
} else {
|
|
3219
3810
|
s.paused = { reason, since: new Date().toISOString(), resumeAt: effectiveResumeAt || null };
|
|
3220
3811
|
}
|
|
3812
|
+
return false;
|
|
3221
3813
|
});
|
|
3814
|
+
if (kept) {
|
|
3815
|
+
console.log(`[scheduler] setPaused(${reason}) ignored: a manual pause is in force`);
|
|
3816
|
+
return;
|
|
3817
|
+
}
|
|
3222
3818
|
await broadcast({ flush: true });
|
|
3223
3819
|
cancelToken.cancelled = true;
|
|
3224
3820
|
if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
|
|
@@ -3256,6 +3852,7 @@ async function clearPause(source) {
|
|
|
3256
3852
|
});
|
|
3257
3853
|
// Un-cancel the tick guard on every recovery path, not just runDueJobs().
|
|
3258
3854
|
applyPauseCleared(wasPaused, cancelToken);
|
|
3855
|
+
if (wasPaused) lastPauseClearedAt = Date.now();
|
|
3259
3856
|
// Track manual clears for the auto-pause cooldown.
|
|
3260
3857
|
if (source === 'manual' || source === 'run-now') {
|
|
3261
3858
|
pauseClearedManuallyAt = Date.now();
|
|
@@ -3268,7 +3865,7 @@ async function clearPause(source) {
|
|
|
3268
3865
|
firstFailureAt = null;
|
|
3269
3866
|
firstNon429FailureAt = null;
|
|
3270
3867
|
lastFailureKind = null;
|
|
3271
|
-
|
|
3868
|
+
resetFailureStreak();
|
|
3272
3869
|
persistSchedulerState();
|
|
3273
3870
|
}
|
|
3274
3871
|
if (wasPaused) await broadcast({ flush: true });
|
|
@@ -3351,37 +3948,65 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
3351
3948
|
return true;
|
|
3352
3949
|
}
|
|
3353
3950
|
|
|
3354
|
-
// Grace period between a boot orphan's SIGTERM and reading its log to
|
|
3355
|
-
// classify the outcome — matches killOrphanClaudePid's own internal 5s
|
|
3356
|
-
// SIGKILL follow-up delay, plus a small margin so classification always runs
|
|
3357
|
-
// after that SIGKILL has had a chance to land.
|
|
3358
|
-
const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
|
|
3359
|
-
|
|
3360
3951
|
/**
|
|
3361
|
-
* partitionBootOrphans(jobs,
|
|
3952
|
+
* partitionBootOrphans(jobs, liveness) → { immediate: string[], adopted: string[] }
|
|
3362
3953
|
*
|
|
3363
|
-
* Pure decision split for boot reconciliation
|
|
3364
|
-
*
|
|
3365
|
-
*
|
|
3366
|
-
*
|
|
3367
|
-
*
|
|
3368
|
-
*
|
|
3369
|
-
*
|
|
3370
|
-
*
|
|
3371
|
-
|
|
3372
|
-
|
|
3954
|
+
* Pure decision split for boot reconciliation of 'running' rows. A row PROVEN
|
|
3955
|
+
* ALIVE is `adopted`: left `running`, never signalled — the steady-state
|
|
3956
|
+
* reaper (reapDeadRunningJobs) finishes it on exit, exactly as it does for any
|
|
3957
|
+
* pidless-recovered row. Only rows proven dead or exited are `immediate` and go
|
|
3958
|
+
* through applyOrphanOutcome. `liveness` is the same injected set
|
|
3959
|
+
* selectReapableJobs takes (plus readRecord/runsDir/identityOf):
|
|
3960
|
+
* { pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
|
|
3961
|
+
* readRecord(runDir, slug) }.
|
|
3962
|
+
*
|
|
3963
|
+
* classifyAdoption runs FIRST when the row has a supervisor record: 'adopt' and
|
|
3964
|
+
* 'over-budget' are alive (budget re-arm across restart is a separate PRD, so
|
|
3965
|
+
* an over-budget row is spared, not killed); 'exited' / 'dead' / 'foreign-pid'
|
|
3966
|
+
* are not. A row with no record falls to the reaper's own ladder: recorded
|
|
3967
|
+
* pid alive, then fresh log, log-pid alive, /proc cwd scan.
|
|
3968
|
+
* Complexity: O(jobs) plus one /proc probe per running row.
|
|
3969
|
+
*/
|
|
3970
|
+
function partitionBootOrphans(jobs, {
|
|
3971
|
+
pidAlive = claudePidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
|
|
3972
|
+
readRecord = supervisorRecord.readSupervisorRecord, runsDir = null, identityOf = procIdentityOf,
|
|
3973
|
+
now = Date.now(),
|
|
3974
|
+
} = {}) {
|
|
3373
3975
|
const immediate = [];
|
|
3374
|
-
const
|
|
3976
|
+
const adopted = [];
|
|
3375
3977
|
for (const j of jobs) {
|
|
3376
3978
|
if (j.status !== 'running') continue;
|
|
3377
|
-
|
|
3378
|
-
|
|
3379
|
-
|
|
3380
|
-
|
|
3381
|
-
|
|
3382
|
-
|
|
3979
|
+
if (isBootRowAlive(j, {
|
|
3980
|
+
pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
|
|
3981
|
+
})) adopted.push(j.slug);
|
|
3982
|
+
else immediate.push(j.slug);
|
|
3983
|
+
}
|
|
3984
|
+
return { immediate, adopted };
|
|
3985
|
+
}
|
|
3986
|
+
|
|
3987
|
+
function isBootRowAlive(j, {
|
|
3988
|
+
pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
|
|
3989
|
+
}) {
|
|
3990
|
+
const logMtimeMs = typeof getLogMtimeMs === 'function' ? getLogMtimeMs(j) : null;
|
|
3991
|
+
const runDir = j.runId ? path.join(runsDir || schedulerPaths.runsDir(), j.runId) : null;
|
|
3992
|
+
const record = runDir && typeof readRecord === 'function' ? readRecord(runDir, j.slug) : null;
|
|
3993
|
+
if (record && record.pid) {
|
|
3994
|
+
const alive = !!pidAlive(record.pid);
|
|
3995
|
+
const verdict = supervisorRecord.classifyAdoption(record, {
|
|
3996
|
+
identity: alive ? identityOf(record.pid) : null,
|
|
3997
|
+
pidAlive: alive,
|
|
3998
|
+
logMtimeMs,
|
|
3999
|
+
exitMarker: supervisorRecord.hasExitMarker(runDir, record),
|
|
4000
|
+
now,
|
|
4001
|
+
});
|
|
4002
|
+
return verdict === 'adopt' || verdict === 'over-budget';
|
|
3383
4003
|
}
|
|
3384
|
-
|
|
4004
|
+
const pid = j.runtime?.pid;
|
|
4005
|
+
if (pid && pidAlive(pid)) return true;
|
|
4006
|
+
if (typeof logFreshWindowMs === 'number' && Number.isFinite(logMtimeMs) && now - logMtimeMs <= logFreshWindowMs) return true;
|
|
4007
|
+
const logPid = typeof getLogPid === 'function' ? getLogPid(j) : null;
|
|
4008
|
+
if (logPid && pidAlive(logPid)) return true;
|
|
4009
|
+
return !!(typeof findLiveProcess === 'function' && findLiveProcess(j));
|
|
3385
4010
|
}
|
|
3386
4011
|
|
|
3387
4012
|
/**
|
|
@@ -3647,7 +4272,7 @@ async function notifyOriginatingTab(job, {
|
|
|
3647
4272
|
const epicIdForTranscript = prd?.sourcePromptId || prd?.sourceTabId || job.epicId || null;
|
|
3648
4273
|
if (epicIdForTranscript && job.cwd) {
|
|
3649
4274
|
try {
|
|
3650
|
-
const logPath = job.runId ? path.join(
|
|
4275
|
+
const logPath = job.runId ? path.join(schedulerPaths.runsDir(), job.runId, `${job.slug}.log`) : null;
|
|
3651
4276
|
const resultText = readResultFromLog(logPath);
|
|
3652
4277
|
await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
|
|
3653
4278
|
role: 'assistant',
|
|
@@ -4378,6 +5003,39 @@ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
|
|
|
4378
5003
|
}
|
|
4379
5004
|
}
|
|
4380
5005
|
|
|
5006
|
+
/**
|
|
5007
|
+
* True when `sha` is a non-empty commit that is an ancestor of (or equal to)
|
|
5008
|
+
* HEAD in the repo at `cwd` (`git merge-base --is-ancestor`). Bounded, never
|
|
5009
|
+
* throws: an empty sha, an unknown sha, or any git failure is `false`, so the
|
|
5010
|
+
* caller's safe default is "cannot prove the work survived".
|
|
5011
|
+
*/
|
|
5012
|
+
async function landedCommitIsAncestorOfHead(cwd, sha) {
|
|
5013
|
+
if (!sha || typeof sha !== 'string' || !cwd) return false;
|
|
5014
|
+
try {
|
|
5015
|
+
await execGitAt(resolveProjectRoot(cwd), ['merge-base', '--is-ancestor', sha, 'HEAD'], { timeout: 10_000 });
|
|
5016
|
+
return true;
|
|
5017
|
+
} catch {
|
|
5018
|
+
return false;
|
|
5019
|
+
}
|
|
5020
|
+
}
|
|
5021
|
+
|
|
5022
|
+
/**
|
|
5023
|
+
* Pure predicate: a needs_review row parked as shared_tree_reverted that
|
|
5024
|
+
* carries a landedCommit — the only shape reverifyNeedsReview can re-check
|
|
5025
|
+
* against ground truth (landedCommitIsAncestorOfHead). Deliberately
|
|
5026
|
+
* independent of autoFixAttempted: 1229-fo-03 was parked with
|
|
5027
|
+
* autoFixAttempted:true and no autoFixOutcome (isStrandedAutoFixPark shape),
|
|
5028
|
+
* yet isStrandedAutoFixPark could not release it — that ladder only resolves
|
|
5029
|
+
* once job.looksDone is set, and reverifyNeedsReview computes looksDone only
|
|
5030
|
+
* for isRescanCandidate / isGuardParkedWithoutAutoFix rows, neither of which
|
|
5031
|
+
* a shared_tree_reverted + autoFixAttempted row is.
|
|
5032
|
+
*/
|
|
5033
|
+
function isStaleSharedTreeRevertedPark(job) {
|
|
5034
|
+
return !!job && job.status === 'needs_review'
|
|
5035
|
+
&& job.verifierVerdict === 'shared_tree_reverted'
|
|
5036
|
+
&& typeof job.landedCommit === 'string' && job.landedCommit.length > 0;
|
|
5037
|
+
}
|
|
5038
|
+
|
|
4381
5039
|
/**
|
|
4382
5040
|
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
4383
5041
|
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
@@ -4557,7 +5215,7 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
|
|
|
4557
5215
|
// create it — `recursive: true` makes that race safe.
|
|
4558
5216
|
function pickRunDir() {
|
|
4559
5217
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
4560
|
-
const dir = path.join(
|
|
5218
|
+
const dir = path.join(schedulerPaths.runsDir(), ts);
|
|
4561
5219
|
return { runId: ts, dir };
|
|
4562
5220
|
}
|
|
4563
5221
|
|
|
@@ -4617,7 +5275,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4617
5275
|
// Sync write: this is an early-exit error path inside an async function,
|
|
4618
5276
|
// so we could await, but using the sync variant keeps the error path
|
|
4619
5277
|
// ordering identical to the spawn-failed branch below (also sync).
|
|
4620
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0 });
|
|
5278
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
|
|
4621
5279
|
return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
|
|
4622
5280
|
}
|
|
4623
5281
|
|
|
@@ -4778,7 +5436,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4778
5436
|
if (!promptCheck.ok) {
|
|
4779
5437
|
safeLog(`[scheduler] ${promptCheck.error}\n`);
|
|
4780
5438
|
closeFd();
|
|
4781
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0 });
|
|
5439
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
|
|
4782
5440
|
return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
|
|
4783
5441
|
}
|
|
4784
5442
|
|
|
@@ -4972,13 +5630,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4972
5630
|
|
|
4973
5631
|
// ---------- spawn ----------
|
|
4974
5632
|
|
|
5633
|
+
// Distinct comm (`sm-claude-job`) + slug-labelled argv0 for System Monitor.
|
|
5634
|
+
// Both keep the word `claude`, which the /\bclaude\b/ reaper gates need.
|
|
5635
|
+
const jobSpawn = claudeSpawnTarget('job', job.slug, claudeBin);
|
|
5636
|
+
|
|
4975
5637
|
const { child } = withChildAndLog({
|
|
4976
5638
|
fd,
|
|
4977
5639
|
logPath,
|
|
4978
5640
|
safeLog,
|
|
4979
5641
|
closeFd,
|
|
4980
5642
|
spawn: {
|
|
4981
|
-
command:
|
|
5643
|
+
command: jobSpawn.command,
|
|
4982
5644
|
// Resume mode passes `--resume <sessionId>` (reconnect to the SAME
|
|
4983
5645
|
// session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
|
|
4984
5646
|
// never both, see buildClaudeSpawnArgs.
|
|
@@ -4992,6 +5654,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4992
5654
|
options: {
|
|
4993
5655
|
cwd: spawnCwd,
|
|
4994
5656
|
env: childEnv,
|
|
5657
|
+
...(jobSpawn.argv0 ? { argv0: jobSpawn.argv0 } : {}),
|
|
4995
5658
|
// detached:true puts the child in its own process group so we can kill
|
|
4996
5659
|
// the entire descendant tree (including any stray background bashes the
|
|
4997
5660
|
// agent spawned) with `process.kill(-pid)`. Without this, child.kill()
|
|
@@ -5017,7 +5680,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5017
5680
|
sl(`\n[scheduler] ${errMsg}\n`);
|
|
5018
5681
|
// Sync write: inside a Promise executor callback; must flush meta
|
|
5019
5682
|
// before resolve() so the spawnJob mutate() that follows sees it.
|
|
5020
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
5683
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, ...SCHEDULER_META_IDENTITY, originSessionId, contextDigestApplied });
|
|
5021
5684
|
resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
|
|
5022
5685
|
return;
|
|
5023
5686
|
}
|
|
@@ -5079,7 +5742,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5079
5742
|
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
5080
5743
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
5081
5744
|
killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
5082
|
-
|
|
5745
|
+
...SCHEDULER_META_IDENTITY,
|
|
5083
5746
|
originSessionId, contextDigestApplied,
|
|
5084
5747
|
});
|
|
5085
5748
|
resolve({
|
|
@@ -5091,6 +5754,22 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5091
5754
|
|
|
5092
5755
|
if (child) {
|
|
5093
5756
|
safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
|
|
5757
|
+
// The one synchronous, authoritative dispatch record (see
|
|
5758
|
+
// jobSupervisorRecord.cjs). detached:true → setsid → pgid === pid, so
|
|
5759
|
+
// no process.getpgid. A write failure must never fail the dispatch.
|
|
5760
|
+
try {
|
|
5761
|
+
supervisorRecord.writeSupervisorRecord({
|
|
5762
|
+
runDir, slug: job.slug, cwd, runId: path.basename(runDir), pid: child.pid, pgid: child.pid,
|
|
5763
|
+
identity: procIdentityOf(child.pid), execCwd: spawnCwd,
|
|
5764
|
+
worktreeDir: execCwd || null, worktreeBranch: execCwd ? `sm-job/${job.slug}` : null,
|
|
5765
|
+
sessionId, startedAt, budgetMs: budgetExempt ? null : jobBudgetMs, maxDurationMs: null,
|
|
5766
|
+
idleKillMs: IDLE_OUTPUT_KILL_MS, schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
|
|
5767
|
+
});
|
|
5768
|
+
} catch (e) {
|
|
5769
|
+
const message = e?.message ?? String(e);
|
|
5770
|
+
console.error(`[scheduler] FAILED to write supervisor record for ${job.slug} pid=${child.pid}: ${message}`);
|
|
5771
|
+
appendAuditEvent('supervisor_record_write_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
|
|
5772
|
+
}
|
|
5094
5773
|
// Make this job the OOM killer's preferred victim over Electron.
|
|
5095
5774
|
biasJobOomScore(child.pid);
|
|
5096
5775
|
// Persist runtime.pid with one retry — still fire-and-forget (must
|
|
@@ -5380,7 +6059,8 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5380
6059
|
return { deferred: false };
|
|
5381
6060
|
}
|
|
5382
6061
|
}
|
|
5383
|
-
|
|
6062
|
+
// A drain (lib/upgradeDrain.cjs) admits no NEW work: queue instead of spawning.
|
|
6063
|
+
if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) {
|
|
5384
6064
|
// Queue for retry when a slot frees rather than dropping — otherwise a failed
|
|
5385
6065
|
// job (never 'needs_review', so reverifyNeedsReview won't retry it) would
|
|
5386
6066
|
// silently never get an auto-authored fix-plan.
|
|
@@ -5394,12 +6074,12 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5394
6074
|
// both pass the cap check. Released in onExit, on any pre-spawn early return, or
|
|
5395
6075
|
// on a synchronous throw (try/catch below) — and releasing hands the slot to a
|
|
5396
6076
|
// queued investigation so none are stranded.
|
|
5397
|
-
|
|
6077
|
+
runtimeState.reserveInvestigation(failedJob.slug);
|
|
5398
6078
|
let slotReleased = false;
|
|
5399
6079
|
const releaseSlot = () => {
|
|
5400
6080
|
if (slotReleased) return;
|
|
5401
6081
|
slotReleased = true;
|
|
5402
|
-
|
|
6082
|
+
runtimeState.releaseInvestigation(failedJob.slug);
|
|
5403
6083
|
drainDeferredInvestigation();
|
|
5404
6084
|
};
|
|
5405
6085
|
try {
|
|
@@ -5504,13 +6184,14 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5504
6184
|
};
|
|
5505
6185
|
|
|
5506
6186
|
// Phase 2: spawn with lifecycle managed by withChildAndLog.
|
|
6187
|
+
const probeSpawn = claudeSpawnTarget('aux', 'investigate', claudeBin);
|
|
5507
6188
|
const { child } = withChildAndLog({
|
|
5508
6189
|
fd,
|
|
5509
6190
|
logPath: investigationLogPath,
|
|
5510
6191
|
safeLog,
|
|
5511
6192
|
closeFd,
|
|
5512
6193
|
spawn: {
|
|
5513
|
-
command:
|
|
6194
|
+
command: probeSpawn.command,
|
|
5514
6195
|
args: [
|
|
5515
6196
|
'-p', prompt,
|
|
5516
6197
|
'--model', 'opus',
|
|
@@ -5519,7 +6200,7 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5519
6200
|
'--verbose',
|
|
5520
6201
|
'--session-id', sessionId,
|
|
5521
6202
|
],
|
|
5522
|
-
options: { cwd, env: childEnv },
|
|
6203
|
+
options: { cwd, env: childEnv, ...(probeSpawn.argv0 ? { argv0: probeSpawn.argv0 } : {}) },
|
|
5523
6204
|
},
|
|
5524
6205
|
watchdogs: [deadmanWatchdog],
|
|
5525
6206
|
onExit({ exitCode, error, spawnFailed, safeLog: sl }) {
|
|
@@ -5621,6 +6302,20 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5621
6302
|
|
|
5622
6303
|
if (child) {
|
|
5623
6304
|
safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
|
|
6305
|
+
try {
|
|
6306
|
+
supervisorRecord.writeSupervisorRecord({
|
|
6307
|
+
runDir, slug: `${failedJob.slug}.investigation`, kind: 'investigation', cwd: failedJob.cwd ?? null,
|
|
6308
|
+
runId: path.basename(runDir), pid: child.pid, pgid: child.pid, identity: procIdentityOf(child.pid),
|
|
6309
|
+
execCwd: failedJob.cwd ?? null, worktreeDir: null, worktreeBranch: null, sessionId: null,
|
|
6310
|
+
startedAt: Date.now(), budgetMs: null, maxDurationMs: MAX_INVESTIGATION_DURATION_MS, idleKillMs: null,
|
|
6311
|
+
schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
|
|
6312
|
+
});
|
|
6313
|
+
} catch (e) {
|
|
6314
|
+
const message = e?.message ?? String(e);
|
|
6315
|
+
console.error(`[scheduler] FAILED to write supervisor record for investigation ${failedJob.slug} pid=${child.pid}: ${message}`);
|
|
6316
|
+
appendAuditEvent('supervisor_record_write_failed', { slug: failedJob.slug, cwd: failedJob.cwd, pid: child.pid, error: message, kind: 'investigation' });
|
|
6317
|
+
}
|
|
6318
|
+
runtimeState.stampInvestigationPid(failedJob.slug, child.pid);
|
|
5624
6319
|
// Recorded so findStrandedInvestigations (a post-restart maintenance
|
|
5625
6320
|
// sweep — the live process has no other way to know a probe is still
|
|
5626
6321
|
// running) can tell a live probe apart from one whose owning process is
|
|
@@ -5722,9 +6417,22 @@ async function computeDepHistorySatisfaction(state) {
|
|
|
5722
6417
|
for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
|
|
5723
6418
|
for (const dir of listArchivedPrdDirs(cwd)) {
|
|
5724
6419
|
let entries;
|
|
5725
|
-
try { entries = await fsp.readdir(dir); } catch { continue; }
|
|
5726
|
-
for (const
|
|
5727
|
-
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
6420
|
+
try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; }
|
|
6421
|
+
for (const ent of entries) {
|
|
6422
|
+
if (ent.isFile() && ent.name.endsWith('.md')) { satisfied.add(ent.name.slice(0, -3)); continue; }
|
|
6423
|
+
// Option (b) of PRD 1286: a manual archive (queueOps.archiveOne, the
|
|
6424
|
+
// schedule:archive-prd route and scheduler_archive_prd MCP tool) files the PRD
|
|
6425
|
+
// under prds-archived/<ISO-ts>/<slug>.md — one level DEEPER than the auto-archive
|
|
6426
|
+
// layout. Reading only the top level made every manually-archived slug invisible
|
|
6427
|
+
// here, so once its row aged out its dependents held forever as 'unresolved'.
|
|
6428
|
+
// A dep whose PRD file was archived is satisfied; a dep with NO file, row or
|
|
6429
|
+
// history record (never ran, or a typo) still holds — nothing is dropped.
|
|
6430
|
+
if (!ent.isDirectory()) continue;
|
|
6431
|
+
let inner;
|
|
6432
|
+
try { inner = await fsp.readdir(path.join(dir, ent.name)); } catch { continue; }
|
|
6433
|
+
for (const name of inner) {
|
|
6434
|
+
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
6435
|
+
}
|
|
5728
6436
|
}
|
|
5729
6437
|
}
|
|
5730
6438
|
} catch (e) {
|
|
@@ -5821,7 +6529,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5821
6529
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
5822
6530
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
5823
6531
|
// leaves the job pending; the next tick retries when a slot frees up.
|
|
5824
|
-
const slotToken = sessionSlots.acquire(`scheduler:${job.slug}
|
|
6532
|
+
const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`, { claimedAt: Date.now() });
|
|
5825
6533
|
if (!slotToken) {
|
|
5826
6534
|
console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
|
|
5827
6535
|
return;
|
|
@@ -5947,7 +6655,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5947
6655
|
// targets a specific prior session on purpose) and for anything not
|
|
5948
6656
|
// currently 'pending' (e.g. a needs_review->running recovery row).
|
|
5949
6657
|
if (!resumeTarget && s.jobs[idx].status === 'pending') {
|
|
5950
|
-
const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir:
|
|
6658
|
+
const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: schedulerPaths.runsDir() });
|
|
5951
6659
|
const reconcileDecision = evaluateDispatchSidecarReconcile({
|
|
5952
6660
|
rowStatus: s.jobs[idx].status,
|
|
5953
6661
|
rowRunId: s.jobs[idx].runId ?? null,
|
|
@@ -5956,7 +6664,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5956
6664
|
outcome,
|
|
5957
6665
|
});
|
|
5958
6666
|
if (reconcileDecision.skip) {
|
|
5959
|
-
const sidecar = readRunOutcomeSidecars(path.join(
|
|
6667
|
+
const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), reconcileDecision.runId), job.slug);
|
|
5960
6668
|
transitionJob(s.jobs[idx], 'completed', {
|
|
5961
6669
|
reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
|
|
5962
6670
|
source: 'spawnJob:dispatch-sidecar-reconcile',
|
|
@@ -5986,7 +6694,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5986
6694
|
// pass_no_commit_prior_run_verified exemption can fire on this
|
|
5987
6695
|
// run if it turns out to be another no-op re-verification.
|
|
5988
6696
|
if (!s.jobs[idx].landedCommit && outcome?.runId) {
|
|
5989
|
-
const sidecar = readRunOutcomeSidecars(path.join(
|
|
6697
|
+
const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), outcome.runId), job.slug);
|
|
5990
6698
|
if (sidecar.outcome?.landedCommit) {
|
|
5991
6699
|
s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
|
|
5992
6700
|
}
|
|
@@ -6169,6 +6877,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6169
6877
|
const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
|
|
6170
6878
|
try {
|
|
6171
6879
|
res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
|
|
6880
|
+
sessionSlots.stampPid(slotToken, pid);
|
|
6172
6881
|
await mutate((s) => {
|
|
6173
6882
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
6174
6883
|
if (idx >= 0) {
|
|
@@ -6312,8 +7021,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6312
7021
|
}
|
|
6313
7022
|
|
|
6314
7023
|
if (res.rateLimited) {
|
|
7024
|
+
// The executor itself observed a 429 — open the shared circuit even if
|
|
7025
|
+
// the billing poller has been reporting 'ok' all along (AC3: a
|
|
7026
|
+
// different window can 429 the executor than the one binding the
|
|
7027
|
+
// poller's own reads).
|
|
7028
|
+
billing.usageCircuit.recordFailure('executor_429');
|
|
6315
7029
|
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
6316
|
-
const billingResetIso = await
|
|
7030
|
+
const billingResetIso = await billingResetForPause();
|
|
6317
7031
|
const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
|
|
6318
7032
|
const observedAt = dispatchStartedAtMs;
|
|
6319
7033
|
const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
|
|
@@ -6401,6 +7115,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6401
7115
|
allJobs: stateForDeps.jobs,
|
|
6402
7116
|
committedDuringRun,
|
|
6403
7117
|
priorLandedCommit,
|
|
7118
|
+
jobLandedCommitThisRun,
|
|
7119
|
+
exitCode: res.exitCode,
|
|
6404
7120
|
}).catch((e) => ({
|
|
6405
7121
|
verdict: 'verify_unavailable',
|
|
6406
7122
|
reason: `verifier threw: ${e?.message ?? String(e)}`,
|
|
@@ -6506,6 +7222,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6506
7222
|
dirtyBaseline: guardBaselineEntries,
|
|
6507
7223
|
headBefore: guardHeadBefore,
|
|
6508
7224
|
slug: job.slug,
|
|
7225
|
+
landedCommit: jobLandedCommitThisRun,
|
|
6509
7226
|
});
|
|
6510
7227
|
// A restored stash alone isn't silence — it's logged loudly above and
|
|
6511
7228
|
// surfaced on the job row below — but a path that's still missing
|
|
@@ -7251,36 +7968,122 @@ async function spawnResumeRecovery(job, resumeTarget) {
|
|
|
7251
7968
|
// is synchronous and spawnJob is fire-and-forget.
|
|
7252
7969
|
let tickTail = Promise.resolve();
|
|
7253
7970
|
|
|
7971
|
+
// Tick watchdog. The tick BODY (not enqueue-to-settle) is bounded: a body
|
|
7972
|
+
// that never settles (hung reconcile / git walk) is declared wedged, the
|
|
7973
|
+
// chain is reset, and `tickGeneration` is bumped so the abandoned body —
|
|
7974
|
+
// which may resume much later — fails its generation re-check after every
|
|
7975
|
+
// await and returns without spawning or mutating. mutateTail is NOT touched.
|
|
7976
|
+
let tickGeneration = 0;
|
|
7977
|
+
let tickWedgeLatched = false;
|
|
7978
|
+
function tickWatchdogMs() {
|
|
7979
|
+
const raw = process.env.SM_TICK_WATCHDOG_MS;
|
|
7980
|
+
if (raw === undefined || raw === '') return 120_000;
|
|
7981
|
+
const n = Number(raw);
|
|
7982
|
+
return Number.isFinite(n) && n >= 0 ? n : 120_000;
|
|
7983
|
+
}
|
|
7984
|
+
|
|
7254
7985
|
// `bypassLoadGate` is set only by the explicit human run-now / force-tick
|
|
7255
7986
|
// paths (via runDueJobs): the human is asking, so the CPU-load gate yields
|
|
7256
7987
|
// and logs that it did. Every automatic caller leaves it false.
|
|
7988
|
+
/**
|
|
7989
|
+
* Fail-CLOSED expiry of runtime reservations (slot tokens, quiet-machine
|
|
7990
|
+
* lease, investigation set, worktree accounting) whose owner is provably
|
|
7991
|
+
* dead. `now` is the pass start: reservations claimed after it are never
|
|
7992
|
+
* expired. Accounting only — never signals or removes anything.
|
|
7993
|
+
*/
|
|
7994
|
+
function runReservationExpiryPass(jobs, now = Date.now()) {
|
|
7995
|
+
try {
|
|
7996
|
+
const liveSlugs = new Set(runningSet);
|
|
7997
|
+
const terminal = new Set();
|
|
7998
|
+
for (const j of jobs || []) {
|
|
7999
|
+
if (j.status === 'running' || j.status === 'investigating') liveSlugs.add(j.slug);
|
|
8000
|
+
else if (j.status === 'completed' || j.status === 'failed' || j.status === 'skipped') terminal.add(j.slug);
|
|
8001
|
+
}
|
|
8002
|
+
const pidAlive = (pid) => claudePidAlive(pid);
|
|
8003
|
+
sessionSlots.expireDead({ liveSlugs, pidAlive, now });
|
|
8004
|
+
quietMachineLease.expireDead({ liveSlugs, now });
|
|
8005
|
+
runtimeState.expireDeadInvestigations({ liveSlugs, pidAlive, now });
|
|
8006
|
+
gitWorktree.expireDeadWorktreeRegistrations({
|
|
8007
|
+
isTerminalBranch: (branch) => {
|
|
8008
|
+
const key = gitWorktree.keyFromBranch('job', branch);
|
|
8009
|
+
return !!key && !liveSlugs.has(key) && terminal.has(key);
|
|
8010
|
+
},
|
|
8011
|
+
});
|
|
8012
|
+
} catch (e) {
|
|
8013
|
+
console.warn('[scheduler] reservation expiry pass failed', e?.message);
|
|
8014
|
+
}
|
|
8015
|
+
}
|
|
8016
|
+
|
|
7257
8017
|
function tickQueue({ bypassLoadGate = false } = {}) {
|
|
7258
|
-
const
|
|
8018
|
+
const budgetMs = tickWatchdogMs();
|
|
8019
|
+
const next = tickTail.then(() => {
|
|
8020
|
+
const gen = tickGeneration;
|
|
8021
|
+
return withTimeout(() => tickBody(gen, { bypassLoadGate }), budgetMs, () => {
|
|
8022
|
+
tickGeneration++; // fence the abandoned body
|
|
8023
|
+
console.warn(`[scheduler] TICK WEDGED: tick body exceeded ${budgetMs}ms — resetting tick chain`);
|
|
8024
|
+
if (!tickWedgeLatched) {
|
|
8025
|
+
tickWedgeLatched = true;
|
|
8026
|
+
appendAuditEvent('tick_wedged', { budgetMs });
|
|
8027
|
+
}
|
|
8028
|
+
// CAS: only reset if nothing has queued behind this wedged link.
|
|
8029
|
+
if (tickTail === tail) tickTail = Promise.resolve();
|
|
8030
|
+
return recordTick({ fired: false, reason: 'wedged' }, { detail: `tick body exceeded ${budgetMs}ms` });
|
|
8031
|
+
}).then((r) => {
|
|
8032
|
+
if (r?.reason !== 'wedged') tickWedgeLatched = false;
|
|
8033
|
+
return r;
|
|
8034
|
+
});
|
|
8035
|
+
});
|
|
8036
|
+
const tail = next.catch(() => {});
|
|
8037
|
+
tickTail = tail;
|
|
8038
|
+
return next;
|
|
8039
|
+
}
|
|
8040
|
+
|
|
8041
|
+
// The stale sentinel a fenced body returns: never recorded, never acted on.
|
|
8042
|
+
const STALE_TICK = Object.freeze({ fired: false, reason: 'stale-generation' });
|
|
8043
|
+
|
|
8044
|
+
async function tickBody(gen, { bypassLoadGate }) {
|
|
8045
|
+
const tickStartedAt = Date.now();
|
|
8046
|
+
{
|
|
8047
|
+
const stale = () => gen !== tickGeneration;
|
|
7259
8048
|
const state = await readQueue();
|
|
8049
|
+
if (stale()) return STALE_TICK;
|
|
7260
8050
|
// Never reconcile against an unreadable queue: reconcile() would see zero
|
|
7261
8051
|
// job rows for every PRD on disk and resurrect the lot as 'pending'.
|
|
7262
8052
|
if (state.unreadable) {
|
|
7263
8053
|
console.error('[scheduler] tickQueue skipped: queue.json unreadable');
|
|
7264
8054
|
return { fired: false, reason: 'unreadable' };
|
|
7265
8055
|
}
|
|
8056
|
+
// Supervision of an adopted executor is independent of dispatch: it runs
|
|
8057
|
+
// even while paused, before any early return below.
|
|
8058
|
+
await superviseAdoptedRunsPass(state.jobs);
|
|
8059
|
+
if (stale()) return STALE_TICK;
|
|
7266
8060
|
if (state.paused) {
|
|
7267
8061
|
console.log('[scheduler] tickQueue skipped: paused');
|
|
7268
8062
|
return recordTick({ fired: false, reason: 'paused' }, { detail: 'scheduler paused' });
|
|
7269
8063
|
}
|
|
8064
|
+
// Upgrade drain (lib/upgradeDrain.cjs): a separate field from `paused`, so a
|
|
8065
|
+
// rate-limit pause can't overwrite it. Nothing new dispatches; running and
|
|
8066
|
+
// investigating rows finish.
|
|
8067
|
+
if (state.drain?.active) {
|
|
8068
|
+
return recordTick({ fired: false, reason: 'draining' }, { detail: 'draining for restart' });
|
|
8069
|
+
}
|
|
7270
8070
|
if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
|
|
7271
8071
|
|
|
7272
8072
|
// Stamped here — the moment tickQueue actually reaches the picker,
|
|
7273
8073
|
// regardless of whether this pass ends in a launch — so
|
|
7274
8074
|
// classifyQueueStarvation can tell "the engine keeps evaluating the
|
|
7275
8075
|
// queue" apart from "nothing has invoked tickQueue in a long time".
|
|
7276
|
-
// Distinct from `lastRunAt` below
|
|
7277
|
-
//
|
|
8076
|
+
// Distinct from `lastRunAt` below (a batch actually launched): this one
|
|
8077
|
+
// means only "the loop is alive" (heartbeat/health) and must NEVER feed
|
|
8078
|
+
// the idle clock — see dispatchIdleMs.
|
|
7278
8079
|
await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
|
|
8080
|
+
if (stale()) return STALE_TICK;
|
|
7279
8081
|
|
|
7280
8082
|
// The retired-flat-dir sweep now lives inside reconcile() itself (see its
|
|
7281
8083
|
// own comment) so every caller of reconcile — not just this tick — gets
|
|
7282
8084
|
// the guarantee.
|
|
7283
8085
|
await reconcile(state);
|
|
8086
|
+
if (stale()) return STALE_TICK;
|
|
7284
8087
|
// Reclaim any job-kind worktree whose owning row already resolved
|
|
7285
8088
|
// (completed/failed/skipped) without the run ever reaching
|
|
7286
8089
|
// cleanupWorktree — a leaked checkout that would otherwise sit until the
|
|
@@ -7297,9 +8100,19 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
7297
8100
|
// to also carry a private `concurrencyCap` of 3 — the exact per-consumer
|
|
7298
8101
|
// cap that sessionSlots.cjs was written to replace — which silently
|
|
7299
8102
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
7300
|
-
|
|
8103
|
+
// While the usage meter is degraded (degradedConcurrencyCapValue set by
|
|
8104
|
+
// pollLoop — circuit open, or a poll otherwise failed), a picker-side
|
|
8105
|
+
// hold narrows this SAME freeSlots figure instead of standing up a
|
|
8106
|
+
// second pool: the row count admitted this tick simply can't exceed the
|
|
8107
|
+
// degraded cap minus what's already running.
|
|
8108
|
+
runReservationExpiryPass(state.jobs, tickStartedAt);
|
|
8109
|
+
const freeSlots = degradedConcurrencyCapValue != null
|
|
8110
|
+
? Math.max(0, Math.min(sessionSlots.available(), degradedConcurrencyCapValue - runningSet.size))
|
|
8111
|
+
: sessionSlots.available();
|
|
7301
8112
|
const heldSlugs = await computeLaunchHolds(state);
|
|
8113
|
+
if (stale()) return STALE_TICK;
|
|
7302
8114
|
const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
|
|
8115
|
+
if (stale()) return STALE_TICK;
|
|
7303
8116
|
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
7304
8117
|
leaseHeld: quietMachineLease.isHeld(),
|
|
7305
8118
|
machineInUse: sessionSlots.inUse(),
|
|
@@ -7399,18 +8212,18 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
7399
8212
|
}
|
|
7400
8213
|
|
|
7401
8214
|
await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
|
|
8215
|
+
if (stale()) return STALE_TICK;
|
|
7402
8216
|
await broadcast();
|
|
8217
|
+
if (stale()) return STALE_TICK;
|
|
7403
8218
|
|
|
7404
8219
|
const { runId, dir: runDir } = pickRunDir();
|
|
7405
8220
|
for (const job of gatedBatch) {
|
|
7406
|
-
if (cancelToken.cancelled) break;
|
|
8221
|
+
if (cancelToken.cancelled || stale()) break;
|
|
7407
8222
|
// spawnJob is fire-and-forget; it calls tickQueue() on completion.
|
|
7408
8223
|
spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
|
|
7409
8224
|
}
|
|
7410
8225
|
return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
|
|
7411
|
-
}
|
|
7412
|
-
tickTail = next.catch(() => {});
|
|
7413
|
-
return next;
|
|
8226
|
+
}
|
|
7414
8227
|
}
|
|
7415
8228
|
|
|
7416
8229
|
// Translates a tickQueue()/runDueJobs() outcome descriptor into a renderer-facing
|
|
@@ -7486,6 +8299,40 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
7486
8299
|
* second scheduler. */
|
|
7487
8300
|
const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
7488
8301
|
|
|
8302
|
+
/**
|
|
8303
|
+
* dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now }) → ms
|
|
8304
|
+
*
|
|
8305
|
+
* Pure. The ONE dispatch-idleness clock: `now - max(lastRunAt, lastPauseClearedAt,
|
|
8306
|
+
* schedulerBootedAt)`. `lastRunAt` has exactly one writer (tickQueue, right
|
|
8307
|
+
* before the spawn loop) so it already means "a batch actually launched";
|
|
8308
|
+
* a pause clear and a scheduler boot are the other two moments the queue
|
|
8309
|
+
* legitimately (re)starts. Deliberately NOT `lastDispatchAttemptAt`, which
|
|
8310
|
+
* tickQueue stamps before every gate — a queue that ticks every 30 s and
|
|
8311
|
+
* launches nothing (leaked slot, stuck hold) would refresh that stamp
|
|
8312
|
+
* forever and never look idle (the structural repeat of the f18e161 bug
|
|
8313
|
+
* where lastRunAt was refreshed every poll). No finite input → Infinity.
|
|
8314
|
+
*/
|
|
8315
|
+
function dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now } = {}) {
|
|
8316
|
+
const stamps = [lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs].filter(Number.isFinite);
|
|
8317
|
+
return stamps.length ? now - Math.max(...stamps) : Infinity;
|
|
8318
|
+
}
|
|
8319
|
+
|
|
8320
|
+
/**
|
|
8321
|
+
* launchBlockedSlugs(jobs, launchBlocks) → Set<slug>
|
|
8322
|
+
*
|
|
8323
|
+
* Pure, sync (health.cjs runs as a cold process). Pending rows whose persona
|
|
8324
|
+
* has ANY launch-breaker entry — a superset of computeLaunchHolds, which
|
|
8325
|
+
* additionally lets one half-open probe row through per persona.
|
|
8326
|
+
*/
|
|
8327
|
+
function launchBlockedSlugs(jobs, launchBlocks) {
|
|
8328
|
+
const out = new Set();
|
|
8329
|
+
if (!launchBlocks || !Object.keys(launchBlocks).length) return out;
|
|
8330
|
+
for (const j of Array.isArray(jobs) ? jobs : []) {
|
|
8331
|
+
if (j && j.status === 'pending' && launchBlocks[launchFailure.launchBlockKeyFor(j)]) out.add(j.slug);
|
|
8332
|
+
}
|
|
8333
|
+
return out;
|
|
8334
|
+
}
|
|
8335
|
+
|
|
7489
8336
|
/**
|
|
7490
8337
|
* classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
|
|
7491
8338
|
* → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
|
|
@@ -7513,22 +8360,28 @@ const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
|
7513
8360
|
* Returns null when the queue is healthy (work running, nothing pending,
|
|
7514
8361
|
* paused on purpose, or simply not idle long enough yet).
|
|
7515
8362
|
*/
|
|
7516
|
-
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
8363
|
+
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
7517
8364
|
if (paused) return null; // paused is a DECISION, not a stall
|
|
7518
8365
|
if (runningCount > 0) return null; // work is flowing
|
|
7519
8366
|
const rows = Array.isArray(jobs) ? jobs : [];
|
|
7520
8367
|
const pending = rows.filter((j) => j && j.status === 'pending');
|
|
7521
8368
|
if (pending.length === 0) return null; // nothing to run — not a stall
|
|
7522
8369
|
|
|
7523
|
-
const idleMs =
|
|
8370
|
+
const idleMs = dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
|
|
7524
8371
|
if (idleMs < thresholdMs) return null; // give the normal path its chance first
|
|
7525
8372
|
|
|
7526
8373
|
// Which pending rows could actually dispatch? Anything NOT named by a
|
|
7527
|
-
// blocked chain
|
|
7528
|
-
//
|
|
8374
|
+
// blocked chain (computeBlockedChains walks dependsOn with the picker's own
|
|
8375
|
+
// resolution, so the two can never disagree) and NOT held by an open launch
|
|
8376
|
+
// breaker / the quietMachine lease (`heldSlugs` — a separate input, never
|
|
8377
|
+
// folded into the dependsOn walker). Held rows can't launch no matter how
|
|
8378
|
+
// often we tick, so they read as 'blocked' (needs a human), not 'starved'.
|
|
7529
8379
|
const blockedChains = computeBlockedChains(rows);
|
|
7530
|
-
const
|
|
7531
|
-
const
|
|
8380
|
+
const held = heldSlugs?.has ? heldSlugs : new Set(heldSlugs ?? []);
|
|
8381
|
+
const open = held.size > 0 ? rows.filter((j) => !(j.status === 'pending' && held.has(j.slug))) : rows;
|
|
8382
|
+
const openBlocked = held.size > 0 ? computeBlockedChains(open) : blockedChains;
|
|
8383
|
+
const openPending = open.filter((j) => j.status === 'pending').length;
|
|
8384
|
+
const dispatchable = openPending - openBlocked.reduce((n, c) => n + c.blocked, 0);
|
|
7532
8385
|
|
|
7533
8386
|
return {
|
|
7534
8387
|
kind: dispatchable > 0 ? 'starved' : 'blocked',
|
|
@@ -7551,14 +8404,14 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
|
|
|
7551
8404
|
* projects sat starved/blocked for hours, and the watchdog never fired once
|
|
7552
8405
|
* because "work is flowing" was true somewhere else. Partitioning by cwd
|
|
7553
8406
|
* (the same grouping computeBlockedChains already does) fixes DETECTION only
|
|
7554
|
-
* — the idle clock (
|
|
7555
|
-
* `
|
|
8407
|
+
* — the idle clock (see dispatchIdleMs) stays machine-wide, since
|
|
8408
|
+
* `lastRunAt` is machine-level state, and only one tick is ever
|
|
7556
8409
|
* forced per watchdog pass regardless of how many cwds are starved.
|
|
7557
8410
|
*
|
|
7558
8411
|
* Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
|
|
7559
8412
|
* project has a verdict.
|
|
7560
8413
|
*/
|
|
7561
|
-
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
8414
|
+
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
7562
8415
|
if (paused) return [];
|
|
7563
8416
|
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7564
8417
|
const byCwd = new Map();
|
|
@@ -7580,6 +8433,9 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7580
8433
|
paused: false,
|
|
7581
8434
|
runningCount: projRunningCount,
|
|
7582
8435
|
lastRunAtMs,
|
|
8436
|
+
lastPauseClearedAtMs,
|
|
8437
|
+
schedulerBootedAtMs,
|
|
8438
|
+
heldSlugs,
|
|
7583
8439
|
now,
|
|
7584
8440
|
thresholdMs,
|
|
7585
8441
|
});
|
|
@@ -7590,7 +8446,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7590
8446
|
|
|
7591
8447
|
/**
|
|
7592
8448
|
* classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
|
|
7593
|
-
* totalSlots,
|
|
8449
|
+
* totalSlots, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd, thresholdMs })
|
|
7594
8450
|
* → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
|
|
7595
8451
|
*
|
|
7596
8452
|
* Single source of truth for the Scheduler page's queue-health header: the
|
|
@@ -7601,7 +8457,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7601
8457
|
*
|
|
7602
8458
|
* Reuses classifyQueueStarvation for the blocked/stalled read so the header
|
|
7603
8459
|
* can never disagree with runQueueStarvationWatchdog's own decision to force
|
|
7604
|
-
* a tick: both are handed the same
|
|
8460
|
+
* a tick: both are handed the same launch-keyed idle clock (dispatchIdleMs) and
|
|
7605
8461
|
* the same computeBlockedChains walk under the hood. Called here with
|
|
7606
8462
|
* `thresholdMs: 0` first (a live header must say "blocked" the instant every
|
|
7607
8463
|
* pending row is dependency-stuck, not wait out the watchdog's own 10-minute
|
|
@@ -7638,7 +8494,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7638
8494
|
*/
|
|
7639
8495
|
function classifyQueueHealth({
|
|
7640
8496
|
jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
|
|
7641
|
-
|
|
8497
|
+
lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
|
|
7642
8498
|
} = {}) {
|
|
7643
8499
|
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7644
8500
|
const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
|
|
@@ -7688,19 +8544,22 @@ function classifyQueueHealth({
|
|
|
7688
8544
|
// over the same rows), so `base` already carries them.
|
|
7689
8545
|
const immediate = classifyQueueStarvation({
|
|
7690
8546
|
jobs: projectJobs, paused: false, runningCount: 0,
|
|
7691
|
-
lastRunAtMs
|
|
8547
|
+
lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, thresholdMs: 0,
|
|
7692
8548
|
});
|
|
7693
8549
|
// pending.length is already > 0 above, so `immediate` can only be null when
|
|
7694
|
-
//
|
|
7695
|
-
//
|
|
8550
|
+
// the clock stamp is itself in the future (clock skew) — fall back to
|
|
8551
|
+
// computing idleMs the same way rather than asserting a kind we can't
|
|
7696
8552
|
// back up with a real number.
|
|
7697
8553
|
const idleMs = immediate ? immediate.idleMs
|
|
7698
|
-
: (
|
|
8554
|
+
: dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
|
|
7699
8555
|
if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
|
|
7700
8556
|
const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
|
|
7701
8557
|
return { ...base, kind, idleMs };
|
|
7702
8558
|
}
|
|
7703
8559
|
|
|
8560
|
+
// Per-cwd latch for runQueueStarvationWatchdog: cwd → { kind, at, running }.
|
|
8561
|
+
const starvationLatch = new Map();
|
|
8562
|
+
|
|
7704
8563
|
/**
|
|
7705
8564
|
* The watchdog half: acts on classifyQueueStarvationByProject. Called from
|
|
7706
8565
|
* the heartbeat, which already runs on its own timer independent of the
|
|
@@ -7713,33 +8572,55 @@ function classifyQueueHealth({
|
|
|
7713
8572
|
* the tick itself is machine-wide (it drives whatever the picker finds
|
|
7714
8573
|
* across every project), only the DETECTION is per-project.
|
|
7715
8574
|
*/
|
|
7716
|
-
async function runQueueStarvationWatchdog(state, {
|
|
7717
|
-
|
|
7718
|
-
|
|
7719
|
-
|
|
7720
|
-
//
|
|
8575
|
+
async function runQueueStarvationWatchdog(state, {
|
|
8576
|
+
now = Date.now(), thresholdMs = QUEUE_STARVATION_MS,
|
|
8577
|
+
bootedAtMs = Date.parse(SCHEDULER_BOOTED_AT), pauseClearedAtMs = lastPauseClearedAt,
|
|
8578
|
+
} = {}) {
|
|
8579
|
+
// The idle clock is launch-keyed (dispatchIdleMs): NOT lastDispatchAttemptAt,
|
|
8580
|
+
// which tickQueue stamps before every gate — a queue whose 30 s loop ticks
|
|
8581
|
+
// and launches nothing would refresh it forever and the watchdog would
|
|
8582
|
+
// never fire. Rows held by an open launch breaker or the quietMachine lease
|
|
8583
|
+
// can't launch however often we tick, so they're passed in as `heldSlugs`
|
|
8584
|
+
// and read as 'blocked' (needs a human) rather than a false 'starved'.
|
|
8585
|
+
const heldSlugs = new Set((await computeLaunchHolds(state)).keys());
|
|
8586
|
+
if (quietMachineLease.isHeld()) {
|
|
8587
|
+
for (const j of state?.jobs ?? []) if (j?.status === 'pending' && j.quietMachine === true) heldSlugs.add(j.slug);
|
|
8588
|
+
}
|
|
7721
8589
|
const verdicts = classifyQueueStarvationByProject({
|
|
7722
8590
|
jobs: state?.jobs,
|
|
7723
|
-
paused: state
|
|
8591
|
+
paused: upgradeDrain.effectivePaused(state),
|
|
7724
8592
|
runningSet,
|
|
7725
|
-
lastRunAtMs: Date.parse(state?.
|
|
8593
|
+
lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
|
|
8594
|
+
lastPauseClearedAtMs: pauseClearedAtMs,
|
|
8595
|
+
schedulerBootedAtMs: bootedAtMs,
|
|
8596
|
+
heldSlugs,
|
|
7726
8597
|
now,
|
|
7727
8598
|
thresholdMs,
|
|
7728
8599
|
});
|
|
8600
|
+
// Latch: one episode per (cwd, kind) — re-arms only once QUEUE_STARVATION_MS
|
|
8601
|
+
// has elapsed again or the project's running count changes; forgotten the
|
|
8602
|
+
// moment the cwd stops having a verdict at all.
|
|
8603
|
+
const activeKeys = new Set(verdicts.map((v) => v.cwd));
|
|
8604
|
+
for (const cwd of [...starvationLatch.keys()]) if (!activeKeys.has(cwd)) starvationLatch.delete(cwd);
|
|
7729
8605
|
if (verdicts.length === 0) return null;
|
|
7730
8606
|
|
|
7731
8607
|
let anyStarved = false;
|
|
7732
8608
|
let primary = null;
|
|
7733
8609
|
for (const verdict of verdicts) {
|
|
7734
8610
|
const mins = Math.round(verdict.idleMs / 60_000);
|
|
8611
|
+
const running = (state?.jobs ?? []).filter((j) => j?.cwd === verdict.cwd && (j.status === 'running' || runningSet.has(j.slug))).length;
|
|
8612
|
+
const latched = starvationLatch.get(verdict.cwd);
|
|
8613
|
+
const suppressed = !!latched && latched.kind === verdict.kind && latched.running === running && now - latched.at < QUEUE_STARVATION_MS;
|
|
8614
|
+
if (!primary) primary = verdict;
|
|
8615
|
+
if (suppressed) continue;
|
|
8616
|
+
starvationLatch.set(verdict.cwd, { kind: verdict.kind, at: now, running });
|
|
7735
8617
|
if (verdict.kind === 'blocked') {
|
|
7736
8618
|
console.warn(
|
|
7737
8619
|
`[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
7738
|
-
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
8620
|
+
+ `terminal or parked dependency, an open launch breaker, or the quietMachine lease, so ticking cannot help. Blockers: `
|
|
7739
8621
|
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7740
8622
|
);
|
|
7741
8623
|
appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
7742
|
-
if (!primary) primary = verdict;
|
|
7743
8624
|
continue;
|
|
7744
8625
|
}
|
|
7745
8626
|
|
|
@@ -7756,9 +8637,12 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
|
|
|
7756
8637
|
|
|
7757
8638
|
// A never-populated utilization reading is itself one of the ways the
|
|
7758
8639
|
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
7759
|
-
// returns early on null).
|
|
7760
|
-
//
|
|
7761
|
-
|
|
8640
|
+
// returns early on null). Absence of information, not a green light: fall
|
|
8641
|
+
// back to the same conservative degraded budget the poll loop itself uses
|
|
8642
|
+
// rather than a blind cachedUtilization=0.
|
|
8643
|
+
if (cachedUtilization === null || cachedUtilization === undefined) {
|
|
8644
|
+
applyDegradedBudget();
|
|
8645
|
+
}
|
|
7762
8646
|
// The in-process cancelToken is only ever reset by runDueJobs() (force-tick
|
|
7763
8647
|
// / run-now / resume-timer) — every other path that clears a pause
|
|
7764
8648
|
// (clearPause(), the poll loop's own auto-recovery) leaves it untouched
|
|
@@ -7924,6 +8808,72 @@ async function runBranchSweep(jobs) {
|
|
|
7924
8808
|
* skipped too (spawn may still be mid-flight) — see selectReapableJobs for
|
|
7925
8809
|
* the full predicate. Exported so unit tests can invoke it directly.
|
|
7926
8810
|
*/
|
|
8811
|
+
// Adopted-run supervision (see lib/adoptedRunSupervisor.cjs). In-memory by
|
|
8812
|
+
// design: the supervisors die with this process, and the next boot re-arms.
|
|
8813
|
+
const adoptedSupervisors = new Map();
|
|
8814
|
+
|
|
8815
|
+
/** Pid of a boot-time running row via the same ladder isBootRowAlive uses:
|
|
8816
|
+
* supervisor record → runtime.pid → pid spawned per the run log. */
|
|
8817
|
+
function bootRowPid(j, logPathOf) {
|
|
8818
|
+
const runDir = j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null;
|
|
8819
|
+
const record = runDir ? supervisorRecord.readSupervisorRecord(runDir, j.slug) : null;
|
|
8820
|
+
return record?.pid || j.runtime?.pid || readSpawnedPidFromLog(logPathOf(j)) || null;
|
|
8821
|
+
}
|
|
8822
|
+
|
|
8823
|
+
function signalAdoptedGroup(pgid, signal, pid) {
|
|
8824
|
+
try { process.kill(-pgid, signal); } catch {
|
|
8825
|
+
try { process.kill(pid, signal); } catch { /* already dead */ }
|
|
8826
|
+
}
|
|
8827
|
+
}
|
|
8828
|
+
|
|
8829
|
+
async function superviseAdoptedRunsPass(jobs) {
|
|
8830
|
+
try {
|
|
8831
|
+
const rows = jobs || (await readQueue()).jobs;
|
|
8832
|
+
const runDirOf = (j) => (j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null);
|
|
8833
|
+
return await adoptedRunSupervisor.superviseAdoptedRuns(rows, {
|
|
8834
|
+
registry: adoptedSupervisors,
|
|
8835
|
+
runDir: runDirOf,
|
|
8836
|
+
readRecord: supervisorRecord.readSupervisorRecord,
|
|
8837
|
+
lease: quietMachineLease,
|
|
8838
|
+
markSupervised: async (row) => {
|
|
8839
|
+
await mutate((s) => {
|
|
8840
|
+
const j = s.jobs.find((x) => x.slug === row.slug);
|
|
8841
|
+
if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) j.supervisedAt = new Date().toISOString();
|
|
8842
|
+
});
|
|
8843
|
+
},
|
|
8844
|
+
makeDeps: (row, record) => {
|
|
8845
|
+
const logPath = path.join(runDirOf(row), `${row.slug}.log`);
|
|
8846
|
+
return {
|
|
8847
|
+
logPath,
|
|
8848
|
+
statLogMtimeMs: readLogMtimeMs,
|
|
8849
|
+
pidAlive: claudePidAlive,
|
|
8850
|
+
identityOf: procIdentityOf,
|
|
8851
|
+
isDifferentProcess,
|
|
8852
|
+
killGroup: signalAdoptedGroup,
|
|
8853
|
+
// Stamped BEFORE the signal so reapDeadRunningJobs, which finalizes
|
|
8854
|
+
// the row once the process is gone, always sees why it died.
|
|
8855
|
+
stampKill: async (kind, reason) => {
|
|
8856
|
+
await mutate((s) => {
|
|
8857
|
+
const j = s.jobs.find((x) => x.slug === row.slug);
|
|
8858
|
+
if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) {
|
|
8859
|
+
j.adoptedKill = { watchdog: kind, reason, at: new Date().toISOString() };
|
|
8860
|
+
}
|
|
8861
|
+
});
|
|
8862
|
+
try { fs.appendFileSync(logPath, `\n[scheduler] adopted-run ${kind} watchdog: ${reason}\n`); } catch { /* best-effort */ }
|
|
8863
|
+
},
|
|
8864
|
+
log: (msg) => console.log(`[scheduler] ${row.slug}: ${msg}`),
|
|
8865
|
+
checkIntervalMs: IDLE_CHECK_INTERVAL_MS,
|
|
8866
|
+
sigkillAfterMs: POST_RESULT_KILL_MS,
|
|
8867
|
+
defaultMaxDurationMs: MAX_JOB_DURATION_MS,
|
|
8868
|
+
};
|
|
8869
|
+
},
|
|
8870
|
+
});
|
|
8871
|
+
} catch (e) {
|
|
8872
|
+
console.warn('[scheduler] adopted-run supervision pass failed', e?.message);
|
|
8873
|
+
return [];
|
|
8874
|
+
}
|
|
8875
|
+
}
|
|
8876
|
+
|
|
7927
8877
|
async function reapDeadRunningJobs() {
|
|
7928
8878
|
try {
|
|
7929
8879
|
// Do NOT gate on runningSet: spawnJob()'s finally block unconditionally
|
|
@@ -7932,9 +8882,13 @@ async function reapDeadRunningJobs() {
|
|
|
7932
8882
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
7933
8883
|
// queue.json is the source of truth for which jobs are actually running.
|
|
7934
8884
|
const state = await readQueue();
|
|
8885
|
+
// A quarantined shard's rows never loaded; the filter is defence in depth
|
|
8886
|
+
// so a reap can never terminalize a row of a project we cannot persist.
|
|
8887
|
+
const reapSkip = quarantinedCwdSet(state);
|
|
8888
|
+
if (reapSkip.size > 0) state.jobs = state.jobs.filter((j) => !reapSkip.has(j.cwd));
|
|
7935
8889
|
// Shared by the log-evidence injections below and the reapable-processing
|
|
7936
8890
|
// loop further down — same `j.runId` → run log path formula either way.
|
|
7937
|
-
const logPathForJob = (j) => (j?.runId ? path.join(
|
|
8891
|
+
const logPathForJob = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
|
|
7938
8892
|
const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
|
|
7939
8893
|
pidAlive: claudePidAlive,
|
|
7940
8894
|
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
@@ -8008,8 +8962,12 @@ async function reapDeadRunningJobs() {
|
|
|
8008
8962
|
// the same still-active rate limit — the spin loop this PRD exists to
|
|
8009
8963
|
// stop. Done once, outside mutate(), before finalizing any row below.
|
|
8010
8964
|
if (dead.some((d) => d.outcome === 'rate_limited')) {
|
|
8965
|
+
// Same rationale as spawnJob's own rateLimited branch: a dead-process
|
|
8966
|
+
// reap that classifies as rate-limited is just as much an executor-
|
|
8967
|
+
// observed 429 as a live one, and must open the same shared circuit.
|
|
8968
|
+
billing.usageCircuit.recordFailure('executor_429');
|
|
8011
8969
|
const triggering = dead.find((d) => d.outcome === 'rate_limited');
|
|
8012
|
-
const billingResetIso = await
|
|
8970
|
+
const billingResetIso = await billingResetForPause();
|
|
8013
8971
|
const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
|
|
8014
8972
|
const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
|
|
8015
8973
|
const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
|
|
@@ -8115,9 +9073,10 @@ async function reapDeadRunningJobs() {
|
|
|
8115
9073
|
// this runs the whole dead-job batch concurrently rather than one
|
|
8116
9074
|
// dispatch's git-spawn latency at a time.
|
|
8117
9075
|
await Promise.all(dead.map(async (d) => {
|
|
8118
|
-
if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
|
|
8119
|
-
if (d.pidless && d.failureOverride) return;
|
|
8120
9076
|
const row = state.jobs.find((x) => x.slug === d.slug);
|
|
9077
|
+
const adoptedBudgetKill = row?.adoptedKill?.watchdog === 'budget';
|
|
9078
|
+
if (d.outcome === 'rate_limited' || (d.outcome === 'success' && !adoptedBudgetKill)) return;
|
|
9079
|
+
if (d.pidless && d.failureOverride) return;
|
|
8121
9080
|
if (!row?.landedCommit) return;
|
|
8122
9081
|
const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
8123
9082
|
const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
|
|
@@ -8150,7 +9109,7 @@ async function reapDeadRunningJobs() {
|
|
|
8150
9109
|
const baseSet = new Set(s.jobs[idx].guardBaseline);
|
|
8151
9110
|
deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
8152
9111
|
if (deltaPaths.length) {
|
|
8153
|
-
const salvagePath = path.join(
|
|
9112
|
+
const salvagePath = path.join(schedulerPaths.runsDir(), s.jobs[idx].runId, `${slug}.uncommitted.patch`);
|
|
8154
9113
|
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
|
|
8155
9114
|
if (salvage && salvage.ok) {
|
|
8156
9115
|
s.jobs[idx].salvagePatch = salvagePath;
|
|
@@ -8198,6 +9157,17 @@ async function reapDeadRunningJobs() {
|
|
|
8198
9157
|
if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
|
|
8199
9158
|
notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
|
|
8200
9159
|
}
|
|
9160
|
+
// An adopted run the budget watchdog killed (adoptedKill, stamped
|
|
9161
|
+
// before the signal) parks exactly like a native budget kill: the
|
|
9162
|
+
// shared classifyBudgetKill decides, and it wins over success/failure.
|
|
9163
|
+
const adoptedBudgetKill = rateLimited ? null : classifyBudgetKill({
|
|
9164
|
+
killedByWatchdog: s.jobs[idx].adoptedKill?.watchdog,
|
|
9165
|
+
budgetKillReason: s.jobs[idx].adoptedKill?.reason,
|
|
9166
|
+
}, landedCommitEvidence.get(slug) || null);
|
|
9167
|
+
if (adoptedBudgetKill) {
|
|
9168
|
+
effectiveSuccess = false;
|
|
9169
|
+
notLandedInfo = { verdict: 'budget_exceeded', reason: adoptedBudgetKill.reason };
|
|
9170
|
+
}
|
|
8201
9171
|
|
|
8202
9172
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
8203
9173
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
@@ -8249,6 +9219,7 @@ async function reapDeadRunningJobs() {
|
|
|
8249
9219
|
s.jobs[idx].gateOutcome = gateOutcome;
|
|
8250
9220
|
if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
|
|
8251
9221
|
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
9222
|
+
if (adoptedBudgetKill?.landedCommit) s.jobs[idx].landedCommit = adoptedBudgetKill.landedCommit;
|
|
8252
9223
|
}
|
|
8253
9224
|
// A pidless spawn that never wrote its own '<slug>.log' into the
|
|
8254
9225
|
// batch runId dir it was stamped with must not keep that runId —
|
|
@@ -8263,6 +9234,7 @@ async function reapDeadRunningJobs() {
|
|
|
8263
9234
|
s.jobs[idx].runId = null;
|
|
8264
9235
|
}
|
|
8265
9236
|
delete s.jobs[idx].runtime;
|
|
9237
|
+
delete s.jobs[idx].adoptedKill;
|
|
8266
9238
|
delete s.jobs[idx].dispatchPhase;
|
|
8267
9239
|
delete s.jobs[idx].dispatchPhaseAt;
|
|
8268
9240
|
delete s.jobs[idx].overrun;
|
|
@@ -8327,14 +9299,21 @@ async function pollLoop() {
|
|
|
8327
9299
|
// 404/time-out and eventually pause the queue on 'network' — treat usage as
|
|
8328
9300
|
// wide-open and fire on pending + memory alone. (Blackrock-style machines.)
|
|
8329
9301
|
if (!billing.usageMeterApplicable()) {
|
|
8330
|
-
|
|
9302
|
+
// Close the shared circuit if a PRIOR consumer-auth session left it
|
|
9303
|
+
// open/half_open — this process has stopped polling the meter
|
|
9304
|
+
// entirely, so nothing else will ever call recordSuccess() to clear
|
|
9305
|
+
// it, and health.cjs would otherwise read a stale open circuit as
|
|
9306
|
+
// YELLOW/RED forever even though nothing is actually degraded.
|
|
9307
|
+
if (billing.usageCircuit.state() !== 'closed') billing.usageCircuit.recordSuccess({});
|
|
9308
|
+
cachedUtilization = NO_METER_UTILIZATION;
|
|
9309
|
+
degradedConcurrencyCapValue = null;
|
|
8331
9310
|
consecutiveFailures = 0;
|
|
8332
9311
|
backoffMs = 0;
|
|
8333
9312
|
backoffNextAt = null;
|
|
8334
9313
|
firstFailureAt = null;
|
|
8335
9314
|
firstNon429FailureAt = null;
|
|
8336
9315
|
lastFailureKind = null;
|
|
8337
|
-
|
|
9316
|
+
resetFailureStreak();
|
|
8338
9317
|
lastPollAt = Date.now();
|
|
8339
9318
|
lastPollOk = true;
|
|
8340
9319
|
persistSchedulerState();
|
|
@@ -8350,18 +9329,40 @@ async function pollLoop() {
|
|
|
8350
9329
|
return; // finally re-arms the timer
|
|
8351
9330
|
}
|
|
8352
9331
|
|
|
9332
|
+
// Shared breaker over the meter (AC1): while it is OPEN, no request is
|
|
9333
|
+
// made except the half-open probe below (billing.fetchUsage() is only
|
|
9334
|
+
// ever reached, further down, from the closed/half_open paths). "Meter
|
|
9335
|
+
// down" reads as absence of information, not a green light — the
|
|
9336
|
+
// conservative degraded budget stands in for both the utilization-
|
|
9337
|
+
// threshold gate (maybeLaunchWhenAvailable) and the concurrency cap
|
|
9338
|
+
// (tickQueue's freeSlots), never a blind cachedUtilization=0.
|
|
9339
|
+
if (billing.usageCircuit.state() === 'open') {
|
|
9340
|
+
applyDegradedBudget();
|
|
9341
|
+
lastPollAt = Date.now();
|
|
9342
|
+
lastPollOk = false;
|
|
9343
|
+
warnFailureStreakIfNeeded();
|
|
9344
|
+
persistSchedulerState();
|
|
9345
|
+
const cur = await readQueue();
|
|
9346
|
+
await maybeLaunchWhenAvailable(cur);
|
|
9347
|
+
await broadcast();
|
|
9348
|
+
return;
|
|
9349
|
+
}
|
|
9350
|
+
|
|
8353
9351
|
const r = await billing.fetchUsage();
|
|
8354
9352
|
|
|
8355
9353
|
if (r.kind === 'ok') {
|
|
8356
|
-
|
|
8357
|
-
|
|
9354
|
+
const window = bindingWindow(r.data?.usage);
|
|
9355
|
+
recordObservedReset(window.resets_at ?? null);
|
|
9356
|
+
cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
|
|
9357
|
+
lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
|
|
9358
|
+
degradedConcurrencyCapValue = null;
|
|
8358
9359
|
consecutiveFailures = 0;
|
|
8359
9360
|
backoffMs = 0;
|
|
8360
9361
|
backoffNextAt = null;
|
|
8361
9362
|
firstFailureAt = null;
|
|
8362
9363
|
firstNon429FailureAt = null;
|
|
8363
9364
|
lastFailureKind = null;
|
|
8364
|
-
|
|
9365
|
+
resetFailureStreak();
|
|
8365
9366
|
lastPollAt = Date.now();
|
|
8366
9367
|
lastPollOk = true;
|
|
8367
9368
|
persistSchedulerState();
|
|
@@ -8377,14 +9378,14 @@ async function pollLoop() {
|
|
|
8377
9378
|
await maybeLaunchWhenAvailable(cur);
|
|
8378
9379
|
await broadcast();
|
|
8379
9380
|
} else if (r.kind === 'meter_rate_limited') {
|
|
8380
|
-
// Billing meter is itself being rate-limited
|
|
8381
|
-
//
|
|
8382
|
-
//
|
|
8383
|
-
//
|
|
8384
|
-
//
|
|
8385
|
-
//
|
|
8386
|
-
//
|
|
8387
|
-
//
|
|
9381
|
+
// Billing meter is itself being rate-limited — absence of information,
|
|
9382
|
+
// not a green light. Still back off the POLL cadence itself (same
|
|
9383
|
+
// curve/cap as the transient branch) and persist state every cycle —
|
|
9384
|
+
// without this, a sustained 429 streak hammered the already-rate-
|
|
9385
|
+
// limited endpoint every POLL_INTERVAL_MS forever AND never wrote
|
|
9386
|
+
// lastPollAt/consecutiveFailures back to scheduler-state.json, so the
|
|
9387
|
+
// sidecar froze stale while the loop kept failing silently underneath
|
|
9388
|
+
// it (the 57-consecutive-failure incident).
|
|
8388
9389
|
lastPollAt = Date.now();
|
|
8389
9390
|
lastPollOk = false;
|
|
8390
9391
|
consecutiveFailures++;
|
|
@@ -8392,8 +9393,8 @@ async function pollLoop() {
|
|
|
8392
9393
|
// Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
|
|
8393
9394
|
backoffMs = nextBackoffMs(backoffMs);
|
|
8394
9395
|
backoffNextAt = Date.now() + backoffMs;
|
|
8395
|
-
|
|
8396
|
-
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on
|
|
9396
|
+
applyDegradedBudget();
|
|
9397
|
+
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on degraded budget (util=${cachedUtilization}%, cap=${degradedConcurrencyCapValue}) (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
|
|
8397
9398
|
warnFailureStreakIfNeeded();
|
|
8398
9399
|
persistSchedulerState();
|
|
8399
9400
|
const cur = await readQueue();
|
|
@@ -8433,13 +9434,12 @@ async function pollLoop() {
|
|
|
8433
9434
|
// 'ok' and 'meter_rate_limited' branches used to reach
|
|
8434
9435
|
// maybeLaunchWhenAvailable, so auth/transient failures left ready
|
|
8435
9436
|
// pending work untouched until either the queue-starvation watchdog's
|
|
8436
|
-
// 10-minute safety net fired or the poll itself recovered.
|
|
8437
|
-
//
|
|
8438
|
-
//
|
|
8439
|
-
//
|
|
8440
|
-
//
|
|
8441
|
-
|
|
8442
|
-
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
9437
|
+
// 10-minute safety net fired or the poll itself recovered. Absence of
|
|
9438
|
+
// information, not a green light: fall back to the degraded budget
|
|
9439
|
+
// rather than a blind cachedUtilization=0. maybeLaunchWhenAvailable
|
|
9440
|
+
// itself still honors an 'auth'/'network' pause (state.paused), so
|
|
9441
|
+
// this is a no-op whenever setPaused() above actually engaged one.
|
|
9442
|
+
applyDegradedBudget();
|
|
8443
9443
|
await maybeLaunchWhenAvailable(await readQueue());
|
|
8444
9444
|
await broadcast();
|
|
8445
9445
|
}
|
|
@@ -8458,7 +9458,7 @@ async function pollLoop() {
|
|
|
8458
9458
|
// Same rationale as the auth/transient branch above: the outer catch
|
|
8459
9459
|
// must not be a silent dispatch dead-end either.
|
|
8460
9460
|
try {
|
|
8461
|
-
|
|
9461
|
+
applyDegradedBudget();
|
|
8462
9462
|
await maybeLaunchWhenAvailable(await readQueue());
|
|
8463
9463
|
await broadcast();
|
|
8464
9464
|
} catch { /* best-effort — the poll loop must still re-arm below */ }
|
|
@@ -8521,6 +9521,34 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
8521
9521
|
// anyway). For non-fix-plan jobs the exemption never applies, so rescanning
|
|
8522
9522
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
8523
9523
|
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
9524
|
+
// RESCANNABLE_VERDICTS is a HINT, not a gate: it names the verdicts whose
|
|
9525
|
+
// recovery rung is a transcript re-verification (verifyRun). Every other
|
|
9526
|
+
// needs_review verdict is still a heal candidate (isRescanCandidate) — it just
|
|
9527
|
+
// gets the evidence-only rung (computeLooksDone) instead of a transcript
|
|
9528
|
+
// rescan, because verifyRun cannot see a commit-guard / shared-tree verdict and
|
|
9529
|
+
// would return 'clean' and falsely heal it.
|
|
9530
|
+
|
|
9531
|
+
// The ONLY needs_review verdicts NOT eligible for the periodic heal ladder.
|
|
9532
|
+
// An allow-list here was reopened three times (2026-09-12 x2, 2026-09-18
|
|
9533
|
+
// shared_tree_reverted) because a new park reason was born invisible to
|
|
9534
|
+
// self-healing. Add a verdict here only with a one-line proof that no
|
|
9535
|
+
// re-verification or evidence scan can ever change it.
|
|
9536
|
+
const RESCAN_EXCLUDED_VERDICTS = new Set([
|
|
9537
|
+
// Commit-guard verdict verifyRun never inspects: a rescan returns 'clean' and would heal genuinely unfinished work.
|
|
9538
|
+
'uncommitted_changes',
|
|
9539
|
+
// Its damage IS a commit stranded on an unmerged sm-job branch — a landedCommit restates it; selectMechanicalRecoveryTarget owns the real re-merge.
|
|
9540
|
+
'worktree_integration_failed',
|
|
9541
|
+
// The run overran its own time/cost estimate; no transcript or git evidence can un-overrun it (selectAutoFixTargets excludes it too).
|
|
9542
|
+
'budget_exceeded',
|
|
9543
|
+
]);
|
|
9544
|
+
|
|
9545
|
+
// Per-pass / per-row bounds on the evidence-only rung (the widened candidate
|
|
9546
|
+
// set). Each scan costs one computeLooksDone: a per-cwd-deduped `git fetch`
|
|
9547
|
+
// (<=~20s) + a git log. Unbounded, a backlog of N parked rows would pay N of
|
|
9548
|
+
// those every 10 minutes forever.
|
|
9549
|
+
const REVERIFY_INTERVAL_MS = 10 * 60_000;
|
|
9550
|
+
const EVIDENCE_SCAN_MAX_PER_PASS = 20;
|
|
9551
|
+
const EVIDENCE_SCAN_MIN_INTERVAL_MS = 6 * REVERIFY_INTERVAL_MS;
|
|
8524
9552
|
|
|
8525
9553
|
// Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
|
|
8526
9554
|
// slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
|
|
@@ -8563,7 +9591,7 @@ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
|
|
|
8563
9591
|
* check, no nested loop over user-scaled data. Dir names are ISO timestamps,
|
|
8564
9592
|
* so lexical-descending sort picks the newest match. Exported for tests.
|
|
8565
9593
|
*/
|
|
8566
|
-
function resolveRunId(job, { runsDir =
|
|
9594
|
+
function resolveRunId(job, { runsDir = schedulerPaths.runsDir() } = {}) {
|
|
8567
9595
|
if (!job || job.runId) return job?.runId || null;
|
|
8568
9596
|
if (!job.slug) return null;
|
|
8569
9597
|
let dirs;
|
|
@@ -8665,18 +9693,71 @@ function isGuardParkedWithoutAutoFix(job) {
|
|
|
8665
9693
|
return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
|
|
8666
9694
|
}
|
|
8667
9695
|
|
|
9696
|
+
/**
|
|
9697
|
+
* Pure predicate, no I/O: a needs_review row whose auto-fix investigation
|
|
9698
|
+
* genuinely ran (autoFixAttempted === true) but whose outcome was NEVER
|
|
9699
|
+
* durably stamped at all — and that has nothing left in flight to wait on:
|
|
9700
|
+
* no live/queued fix-plan row at fixSlugFor(job).
|
|
9701
|
+
*
|
|
9702
|
+
* Distinct from isExhaustedAutoFix, which requires autoFixRetries >= 1 to
|
|
9703
|
+
* have already accumulated. spawnInvestigation's onExit handler restores the
|
|
9704
|
+
* job's status from 'investigating' back to needs_review in ONE mutate()
|
|
9705
|
+
* call (scheduler.cjs's spawnInvestigation, source
|
|
9706
|
+
* 'spawnInvestigation:onExit') and stamps autoFixOutcome ('plan' / 'no-plan'
|
|
9707
|
+
* / 'error') in a SEPARATE, later mutate() call — an app restart or process
|
|
9708
|
+
* death between the two leaves autoFixOutcome permanently unset, with
|
|
9709
|
+
* autoFixRetries never incremented either, so isExhaustedAutoFix never fires
|
|
9710
|
+
* and the row falls through every existing resolving door forever, re-scanned
|
|
9711
|
+
* by the periodic reverify pass against the same frozen transcript with no
|
|
9712
|
+
* new outcome to observe.
|
|
9713
|
+
*
|
|
9714
|
+
* Job 1218-fo-01 (2026-09-13, findings filed at
|
|
9715
|
+
* session-manager-operations/reviews/2026-09-13-scheduler-stability-investigation.md,
|
|
9716
|
+
* "post-run adjudication" section) sat exactly in this state: needs_review,
|
|
9717
|
+
* verifierVerdict transcript_errors, autoFixAttempted: true, autoFixOutcome:
|
|
9718
|
+
* undefined, autoFixRetries: undefined, statusHistory ending in
|
|
9719
|
+
* "investigation probe exited — restoring prior status" — with a landed
|
|
9720
|
+
* commit no existing ladder rung would credit.
|
|
9721
|
+
*
|
|
9722
|
+
* Deliberately narrower than "unset, 'error', or 'no-plan'": a row that DID
|
|
9723
|
+
* get a durably-stamped 'error'/'no-plan' outcome with its one bounded retry
|
|
9724
|
+
* still unspent (autoFixRetries < 1) is exactly the row
|
|
9725
|
+
* selectAutoFixTargets's own retryEligible check still owns and will retry
|
|
9726
|
+
* on its own — pulling it into THIS ladder instead would race it away from
|
|
9727
|
+
* that retry (scheduler-needs-review-autoresolve.test.cjs's "a non-exhausted
|
|
9728
|
+
* needs_review row … is left alone" guards exactly this). Only the
|
|
9729
|
+
* outcome-truly-never-stamped case is structurally unrecoverable by any
|
|
9730
|
+
* OTHER existing door, because nothing ever wrote a value selectAutoFixTargets
|
|
9731
|
+
* or isExhaustedAutoFix could act on.
|
|
9732
|
+
* Exported for tests.
|
|
9733
|
+
*/
|
|
9734
|
+
function isStrandedAutoFixPark(job, jobsInProject) {
|
|
9735
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
9736
|
+
if (job.autoFixAttempted !== true) return false;
|
|
9737
|
+
if (job.autoFixOutcome != null) return false;
|
|
9738
|
+
const fixSlug = fixSlugFor(job);
|
|
9739
|
+
const liveOrQueuedChild = (jobsInProject || []).some(
|
|
9740
|
+
(j) => j.slug === fixSlug && j.status !== 'completed' && !DEAD_FIX_CHILD_STATUSES.has(j.status),
|
|
9741
|
+
);
|
|
9742
|
+
return !liveOrQueuedChild;
|
|
9743
|
+
}
|
|
9744
|
+
|
|
8668
9745
|
/**
|
|
8669
9746
|
* Pure predicate, no I/O: is this needs_review row eligible for the bounded
|
|
8670
9747
|
* auto-resolve ladder at all — either because its auto-fix path is genuinely
|
|
8671
|
-
* spent (isExhaustedAutoFix),
|
|
8672
|
-
*
|
|
8673
|
-
*
|
|
8674
|
-
*
|
|
8675
|
-
*
|
|
9748
|
+
* spent (isExhaustedAutoFix), because it was parked by a GUARD verdict that
|
|
9749
|
+
* never entered auto-fix in the first place (isGuardParkedWithoutAutoFix),
|
|
9750
|
+
* or because its auto-fix investigation ran but was stranded before
|
|
9751
|
+
* recording any outcome (isStrandedAutoFixPark). All three classes share ONE
|
|
9752
|
+
* ladder (applyNeedsReviewAutoResolve) rather than a duplicated one — the
|
|
9753
|
+
* ladder itself doesn't care which door a row came through, only whether it
|
|
9754
|
+
* now carries completion evidence (job.looksDone). `jobsInProject` is only
|
|
9755
|
+
* consulted by isStrandedAutoFixPark (to check for a live/queued fix-plan
|
|
9756
|
+
* child) and defaults to empty so existing single-arg callers are unaffected.
|
|
8676
9757
|
* Exported for tests.
|
|
8677
9758
|
*/
|
|
8678
|
-
function isEligibleForNeedsReviewAutoResolve(job) {
|
|
8679
|
-
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
|
|
9759
|
+
function isEligibleForNeedsReviewAutoResolve(job, jobsInProject = []) {
|
|
9760
|
+
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job) || isStrandedAutoFixPark(job, jobsInProject);
|
|
8680
9761
|
}
|
|
8681
9762
|
|
|
8682
9763
|
/**
|
|
@@ -8789,18 +9870,55 @@ function isFailedUnverifiedShaped(job) {
|
|
|
8789
9870
|
if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
|
|
8790
9871
|
const runId = job.runId || resolveRunId(job);
|
|
8791
9872
|
if (!runId) return false;
|
|
8792
|
-
const logPath = path.join(
|
|
9873
|
+
const logPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.log`);
|
|
8793
9874
|
return classifyRunOutcome(logPath) === 'no_result';
|
|
8794
9875
|
}
|
|
8795
9876
|
|
|
8796
9877
|
function isRescanCandidate(job) {
|
|
8797
9878
|
if (!job) return false;
|
|
9879
|
+
// Default-ELIGIBLE: every needs_review row is a heal candidate unless its
|
|
9880
|
+
// verdict is in RESCAN_EXCLUDED_VERDICTS. No runId requirement here — a row
|
|
9881
|
+
// without one still gets the evidence rung and the unresolvable annotation.
|
|
9882
|
+
if (job.status === 'needs_review') return !RESCAN_EXCLUDED_VERDICTS.has(job.verifierVerdict);
|
|
8798
9883
|
if (!(job.runId || resolveRunId(job))) return false;
|
|
8799
|
-
if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
|
|
8800
9884
|
if (job.status === 'failed') return isFailedUnverifiedShaped(job);
|
|
8801
9885
|
return false;
|
|
8802
9886
|
}
|
|
8803
9887
|
|
|
9888
|
+
/**
|
|
9889
|
+
* Which rung a needs_review candidate gets (RESCANNABLE_VERDICTS as a hint):
|
|
9890
|
+
* true = transcript re-verification (needs a run dir to read); false = the
|
|
9891
|
+
* evidence-only rung. I/O only when a rescannable-verdict row lacks a runId.
|
|
9892
|
+
*/
|
|
9893
|
+
function isTranscriptRescannable(job) {
|
|
9894
|
+
return !!job && RESCANNABLE_VERDICTS.has(job.verifierVerdict) && !!(job.runId || resolveRunId(job));
|
|
9895
|
+
}
|
|
9896
|
+
|
|
9897
|
+
/**
|
|
9898
|
+
* Pure, no I/O: the bounded subset of evidence-only needs_review candidates
|
|
9899
|
+
* reverifyNeedsReview scans this pass. Skips rows already carrying looksDone,
|
|
9900
|
+
* rows scanned within EVIDENCE_SCAN_MIN_INTERVAL_MS (evidenceScannedAt), and —
|
|
9901
|
+
* PRD 1136 — rows with a live auto-fix history unless they are an
|
|
9902
|
+
* auto-resolve door (isEligibleForNeedsReviewAutoResolve, which is what
|
|
9903
|
+
* consumes looksDone). Never-scanned rows go first, then least-recently
|
|
9904
|
+
* scanned; capped at EVIDENCE_SCAN_MAX_PER_PASS. O(n log n) in needs_review rows.
|
|
9905
|
+
* Per-pass cost ceiling: EVIDENCE_SCAN_MAX_PER_PASS computeLooksDone calls.
|
|
9906
|
+
*/
|
|
9907
|
+
function selectEvidenceScanTargets(jobs, now = Date.now()) {
|
|
9908
|
+
const due = [];
|
|
9909
|
+
for (const j of jobs ?? []) {
|
|
9910
|
+
if (j.status !== 'needs_review' || !isRescanCandidate(j)) continue;
|
|
9911
|
+
if (isTranscriptRescannable(j)) continue;
|
|
9912
|
+
if (j.looksDone) continue;
|
|
9913
|
+
if (j.autoFixAttempted === true && !isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
|
|
9914
|
+
const last = Date.parse(j.evidenceScannedAt ?? '');
|
|
9915
|
+
if (!Number.isNaN(last) && now - last < EVIDENCE_SCAN_MIN_INTERVAL_MS) continue;
|
|
9916
|
+
due.push({ j, last: Number.isNaN(last) ? 0 : last });
|
|
9917
|
+
}
|
|
9918
|
+
due.sort((a, b) => a.last - b.last);
|
|
9919
|
+
return due.slice(0, EVIDENCE_SCAN_MAX_PER_PASS).map((d) => d.j);
|
|
9920
|
+
}
|
|
9921
|
+
|
|
8804
9922
|
/**
|
|
8805
9923
|
* Cheap-guard for the 10-minute periodic reverify tick. MUST be expressed in
|
|
8806
9924
|
* terms of isRescanCandidate — not a hand-written status test — because the
|
|
@@ -8834,6 +9952,14 @@ function isRescanCandidate(job) {
|
|
|
8834
9952
|
* that function). Same rule as always: never let this guard be narrower than
|
|
8835
9953
|
* the work reverifyNeedsReview actually performs.
|
|
8836
9954
|
*
|
|
9955
|
+
* Reopened a THIRD time 2026-09-18 (shared_tree_reverted parked 1229-fo-03
|
|
9956
|
+
* falsely, 19 of 20 pending rows held): the fix was not another OR-clause but
|
|
9957
|
+
* inverting the default — isRescanCandidate is now default-ELIGIBLE for every
|
|
9958
|
+
* needs_review row (RESCAN_EXCLUDED_VERDICTS names the few exceptions), so a
|
|
9959
|
+
* new park reason can never again be born unhealable. The OR-clauses below
|
|
9960
|
+
* are now redundant for needs_review rows and kept only for their
|
|
9961
|
+
* non-needs_review inputs.
|
|
9962
|
+
*
|
|
8837
9963
|
* Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
|
|
8838
9964
|
* isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
|
|
8839
9965
|
* called with an injected fixSlugExists that always returns false — cheap
|
|
@@ -8845,7 +9971,7 @@ function isRescanCandidate(job) {
|
|
|
8845
9971
|
*/
|
|
8846
9972
|
function shouldRunPeriodicReverify(jobs) {
|
|
8847
9973
|
if (!Array.isArray(jobs)) return false;
|
|
8848
|
-
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
|
|
9974
|
+
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j))) return true;
|
|
8849
9975
|
if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
|
|
8850
9976
|
return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
|
|
8851
9977
|
}
|
|
@@ -9012,11 +10138,12 @@ function needsReviewAutoResolveDisabled() {
|
|
|
9012
10138
|
* [{ slug, cwd, ageMs, attempts }]
|
|
9013
10139
|
*
|
|
9014
10140
|
* Pure selector — no IO. Selects `needs_review` rows eligible for the
|
|
9015
|
-
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve —
|
|
9016
|
-
*
|
|
9017
|
-
* auto-fix
|
|
9018
|
-
* 'needs_review'` is older than
|
|
9019
|
-
* exhaustedResolveAttempts counter has not yet
|
|
10141
|
+
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — auto-fix
|
|
10142
|
+
* genuinely spent, parked by a GUARD verdict that never entered auto-fix at
|
|
10143
|
+
* all, or a stranded auto-fix park with no outcome ever recorded), whose
|
|
10144
|
+
* newest statusHistory entry with `to === 'needs_review'` is older than
|
|
10145
|
+
* `thresholdMs`, and whose exhaustedResolveAttempts counter has not yet
|
|
10146
|
+
* spent its cap.
|
|
9020
10147
|
*
|
|
9021
10148
|
* The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
|
|
9022
10149
|
* CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
|
|
@@ -9029,7 +10156,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
9029
10156
|
const targets = [];
|
|
9030
10157
|
for (const j of jobs ?? []) {
|
|
9031
10158
|
if (j.status !== 'needs_review') continue;
|
|
9032
|
-
if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
|
|
10159
|
+
if (!isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
|
|
9033
10160
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
|
|
9034
10161
|
const history = j.statusHistory || [];
|
|
9035
10162
|
let entry = null;
|
|
@@ -9073,8 +10200,8 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
9073
10200
|
* reason text (and the Queue UI's job.error) name the RIGHT evidence — a
|
|
9074
10201
|
* guard-parked row was never "exhausted auto-fix" and must never claim to be.
|
|
9075
10202
|
*/
|
|
9076
|
-
function applyNeedsReviewAutoResolve(j) {
|
|
9077
|
-
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
|
|
10203
|
+
function applyNeedsReviewAutoResolve(j, jobsInProject = []) {
|
|
10204
|
+
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j, jobsInProject)) return null;
|
|
9078
10205
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
|
|
9079
10206
|
const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
|
|
9080
10207
|
|
|
@@ -9356,6 +10483,55 @@ async function computeLooksDone(job, fetchedCwds) {
|
|
|
9356
10483
|
return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
|
|
9357
10484
|
}
|
|
9358
10485
|
|
|
10486
|
+
/**
|
|
10487
|
+
* Shadow gate (observation only): run a needs_review row's authored gate at
|
|
10488
|
+
* the project's current HEAD and record what it WOULD have decided as
|
|
10489
|
+
* `gateShadow` on the verdicts sidecar and the row. Changes NO status, takes
|
|
10490
|
+
* no slot (not a claude -p run — runGateSequence keeps one shadow gate in
|
|
10491
|
+
* flight machine-wide). Never called from finalize: only the reverify pass.
|
|
10492
|
+
* Returns the recorded gateShadow, or null when nothing was recorded (already
|
|
10493
|
+
* recorded at this HEAD, PRD unreadable, or another shadow gate is running).
|
|
10494
|
+
*/
|
|
10495
|
+
async function runGateShadow(job) {
|
|
10496
|
+
if (!job || !job.slug || !job.cwd) return null;
|
|
10497
|
+
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
10498
|
+
let prdText;
|
|
10499
|
+
try { prdText = fs.readFileSync(prdPath, 'utf8'); } catch { return null; }
|
|
10500
|
+
const head = await gitHead(job.cwd);
|
|
10501
|
+
if (job.gateShadow && job.gateShadow.head === head) return null;
|
|
10502
|
+
const gate = resolveGate(prdText);
|
|
10503
|
+
let outcome;
|
|
10504
|
+
if (gate.source === 'none') outcome = { status: 'unavailable', reason: 'gate-opt-out', results: [] };
|
|
10505
|
+
else if (!gate.sequence.length) outcome = { status: 'unavailable', reason: 'no-parseable-gate', results: [] };
|
|
10506
|
+
else {
|
|
10507
|
+
const r = await runGateSequence(gate.sequence, { cwd: job.cwd });
|
|
10508
|
+
if (r.status === 'busy') return null;
|
|
10509
|
+
outcome = r;
|
|
10510
|
+
}
|
|
10511
|
+
const gateShadow = { ...outcome, head, source: gate.source, ranAt: new Date().toISOString() };
|
|
10512
|
+
const runId = job.runId || resolveRunId(job);
|
|
10513
|
+
if (runId) {
|
|
10514
|
+
const verdictsPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.verdicts.json`);
|
|
10515
|
+
// Read-merge (single-writer law: runVerify owns the sidecar's other keys).
|
|
10516
|
+
// Only merge into an existing run dir — never conjure one.
|
|
10517
|
+
if (fs.existsSync(path.dirname(verdictsPath))) {
|
|
10518
|
+
let existing = {};
|
|
10519
|
+
try { existing = JSON.parse(fs.readFileSync(verdictsPath, 'utf8')) || {}; } catch { /* absent/unparseable → fresh */ }
|
|
10520
|
+
try { atomicWriteJsonSync(verdictsPath, { ...existing, gateShadow }); } catch { /* best-effort */ }
|
|
10521
|
+
}
|
|
10522
|
+
}
|
|
10523
|
+
await mutate((s) => {
|
|
10524
|
+
for (const j of s.jobs) {
|
|
10525
|
+
if (j.slug === job.slug && j.status === 'needs_review') j.gateShadow = gateShadow;
|
|
10526
|
+
}
|
|
10527
|
+
});
|
|
10528
|
+
await broadcast();
|
|
10529
|
+
return gateShadow;
|
|
10530
|
+
}
|
|
10531
|
+
|
|
10532
|
+
// Tail of the last background shadow gate — lets tests (and only tests) await it.
|
|
10533
|
+
let gateShadowPending = null;
|
|
10534
|
+
|
|
9359
10535
|
async function reverifyNeedsReview() {
|
|
9360
10536
|
const snap = await readQueue();
|
|
9361
10537
|
// isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
|
|
@@ -9365,7 +10541,7 @@ async function reverifyNeedsReview() {
|
|
|
9365
10541
|
// guard-verdict auto-resolve gap this PRD closes. Handled in its own
|
|
9366
10542
|
// branch below (no transcript rescan — there is no transcript verdict to
|
|
9367
10543
|
// rescan) rather than through the isRescanCandidate machinery.
|
|
9368
|
-
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
|
|
10544
|
+
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j));
|
|
9369
10545
|
const healed = [];
|
|
9370
10546
|
const leftForReview = [];
|
|
9371
10547
|
const looksDoneUpdates = [];
|
|
@@ -9373,12 +10549,33 @@ async function reverifyNeedsReview() {
|
|
|
9373
10549
|
// the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
|
|
9374
10550
|
// header) rather than re-fetching the same repo once per candidate row.
|
|
9375
10551
|
const fetchedCwds = new Set();
|
|
10552
|
+
const evidenceSlugs = new Set(selectEvidenceScanTargets(snap.jobs).map((j) => j.slug));
|
|
10553
|
+
const evidenceScanned = [];
|
|
9376
10554
|
for (const job of candidates) {
|
|
9377
|
-
if (
|
|
9378
|
-
//
|
|
9379
|
-
//
|
|
9380
|
-
//
|
|
9381
|
-
|
|
10555
|
+
if (isStaleSharedTreeRevertedPark(job)) {
|
|
10556
|
+
// Re-apply the corrected shared-tree check: the row's own landedCommit
|
|
10557
|
+
// (this dispatch's, per resolveLandedCommitEvidence) still being an
|
|
10558
|
+
// ancestor of HEAD means the park was a false positive — heal it.
|
|
10559
|
+
const cwd = job.cwd || DEFAULT_PROJECT_CWD;
|
|
10560
|
+
if (await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)
|
|
10561
|
+
&& await module.exports.landedCommitIsAncestorOfHead(cwd, job.landedCommit)) {
|
|
10562
|
+
healed.push(job.slug);
|
|
10563
|
+
continue;
|
|
10564
|
+
}
|
|
10565
|
+
if (!isRescanCandidate(job) && !isGuardParkedWithoutAutoFix(job)) {
|
|
10566
|
+
leftForReview.push({ slug: job.slug, reason: 'shared_tree_reverted: landed commit not an ancestor of HEAD' });
|
|
10567
|
+
continue;
|
|
10568
|
+
}
|
|
10569
|
+
}
|
|
10570
|
+
if (job.status === 'needs_review' && !isTranscriptRescannable(job)) {
|
|
10571
|
+
// Any needs_review row whose verdict is not a transcript-verifier one
|
|
10572
|
+
// (a guard verdict, a not-yet-invented verdict, a stranded auto-fix
|
|
10573
|
+
// park): only evidence gathering, never a transcript rescan (verifyRun
|
|
10574
|
+
// would call it clean) and never a direct heal —
|
|
10575
|
+
// applyNeedsReviewAutoResolve is the sole place that turns this
|
|
10576
|
+
// annotation into a status change. Bounded by selectEvidenceScanTargets.
|
|
10577
|
+
if (!evidenceSlugs.has(job.slug)) continue;
|
|
10578
|
+
evidenceScanned.push(job.slug);
|
|
9382
10579
|
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
9383
10580
|
if (looksDone) {
|
|
9384
10581
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
@@ -9401,7 +10598,7 @@ async function reverifyNeedsReview() {
|
|
|
9401
10598
|
}
|
|
9402
10599
|
continue;
|
|
9403
10600
|
}
|
|
9404
|
-
const runDir = path.join(
|
|
10601
|
+
const runDir = path.join(schedulerPaths.runsDir(), job.runId || resolveRunId(job));
|
|
9405
10602
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
9406
10603
|
// Derive committedDuringRun from the recorded run window. The live
|
|
9407
10604
|
// commit-guard uses gitHead() (before/after HEAD diff); here the run is
|
|
@@ -9424,6 +10621,13 @@ async function reverifyNeedsReview() {
|
|
|
9424
10621
|
committedDuringRun,
|
|
9425
10622
|
allowPreSentinelHeal: true,
|
|
9426
10623
|
priorLandedCommit,
|
|
10624
|
+
// job.landedCommit is THIS row's own last-run attribution (stamped by
|
|
10625
|
+
// spawnJob's finalize, survives resetJobFields) — the same
|
|
10626
|
+
// ground-truth-outranks-heuristics evidence spawnJob passes live,
|
|
10627
|
+
// just read back post-hoc since there is no in-flight guardHeadBefore/
|
|
10628
|
+
// headAtExit pair to recompute for an already-terminal row.
|
|
10629
|
+
jobLandedCommitThisRun: job.landedCommit ?? null,
|
|
10630
|
+
exitCode: job.exitCode ?? null,
|
|
9427
10631
|
});
|
|
9428
10632
|
} catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
|
|
9429
10633
|
const refusal = healRefusalReason(job, v, committedDuringRun);
|
|
@@ -9454,6 +10658,23 @@ async function reverifyNeedsReview() {
|
|
|
9454
10658
|
}
|
|
9455
10659
|
}
|
|
9456
10660
|
}
|
|
10661
|
+
// Shadow gate (observation only): at most ONE needs_review row per pass,
|
|
10662
|
+
// fired in the background so a 15-minute gate never stalls this pass.
|
|
10663
|
+
if (!gateShadowPending && process.env.SM_GATE_SHADOW_DISABLE !== '1') {
|
|
10664
|
+
const gateTarget = snap.jobs.find((j) => j.status === 'needs_review' && !j.gateShadow);
|
|
10665
|
+
if (gateTarget) {
|
|
10666
|
+
gateShadowPending = runGateShadow(gateTarget)
|
|
10667
|
+
.catch((e) => { console.error('[scheduler] gate shadow error', gateTarget.slug, e); })
|
|
10668
|
+
.finally(() => { gateShadowPending = null; });
|
|
10669
|
+
}
|
|
10670
|
+
}
|
|
10671
|
+
if (evidenceScanned.length) {
|
|
10672
|
+
const scannedSet = new Set(evidenceScanned);
|
|
10673
|
+
const stamp = new Date().toISOString();
|
|
10674
|
+
await mutate((s) => {
|
|
10675
|
+
for (const j of s.jobs) if (scannedSet.has(j.slug) && j.status === 'needs_review') j.evidenceScannedAt = stamp;
|
|
10676
|
+
});
|
|
10677
|
+
}
|
|
9457
10678
|
if (looksDoneUpdates.length) {
|
|
9458
10679
|
const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
|
|
9459
10680
|
await mutate((s) => {
|
|
@@ -9666,7 +10887,7 @@ async function reverifyNeedsReview() {
|
|
|
9666
10887
|
});
|
|
9667
10888
|
for (const job of targets) {
|
|
9668
10889
|
const runId = job.runId || resolveRunId(job);
|
|
9669
|
-
const runDir = path.join(
|
|
10890
|
+
const runDir = path.join(schedulerPaths.runsDir(), runId);
|
|
9670
10891
|
const isRetryAttempt = job.autoFixAttempted === true;
|
|
9671
10892
|
const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
|
|
9672
10893
|
const deadChild = isDeadFixPlanReopen
|
|
@@ -9813,12 +11034,14 @@ function registerScheduleHandlers() {
|
|
|
9813
11034
|
const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
|
|
9814
11035
|
const verdict = classifyQueueHealth({
|
|
9815
11036
|
jobs: state.jobs,
|
|
9816
|
-
paused: state
|
|
11037
|
+
paused: upgradeDrain.effectivePaused(state),
|
|
9817
11038
|
launchBlocks: state.launchBlocks,
|
|
9818
11039
|
runningSet,
|
|
9819
11040
|
freeSlots,
|
|
9820
11041
|
totalSlots: slotSnapshot.total,
|
|
9821
|
-
|
|
11042
|
+
lastRunAtMs: Date.parse(state.lastRunAt ?? ''),
|
|
11043
|
+
lastPauseClearedAtMs: lastPauseClearedAt,
|
|
11044
|
+
schedulerBootedAtMs: Date.parse(SCHEDULER_BOOTED_AT),
|
|
9822
11045
|
now,
|
|
9823
11046
|
cwd,
|
|
9824
11047
|
});
|
|
@@ -9949,6 +11172,11 @@ function registerScheduleHandlers() {
|
|
|
9949
11172
|
return { ok: true };
|
|
9950
11173
|
});
|
|
9951
11174
|
|
|
11175
|
+
ipcMain.handle('schedule:pause', async () => {
|
|
11176
|
+
await setPaused('manual', null);
|
|
11177
|
+
return { ok: true };
|
|
11178
|
+
});
|
|
11179
|
+
|
|
9952
11180
|
ipcMain.handle('schedule:resume', async () => {
|
|
9953
11181
|
await clearPause('manual');
|
|
9954
11182
|
return { ok: true };
|
|
@@ -9980,7 +11208,7 @@ function registerScheduleHandlers() {
|
|
|
9980
11208
|
ipcMain.handle('schedule:clear-queue', async () => {
|
|
9981
11209
|
ensureDirs();
|
|
9982
11210
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
9983
|
-
const archiveDir = path.join(
|
|
11211
|
+
const archiveDir = path.join(schedulerPaths.scheduledPlansRoot(), 'prds-archived', ts);
|
|
9984
11212
|
const state = await readQueue();
|
|
9985
11213
|
const victims = state.jobs.filter((j) => j.status !== 'running');
|
|
9986
11214
|
if (victims.length === 0) {
|
|
@@ -10027,7 +11255,7 @@ function registerScheduleHandlers() {
|
|
|
10027
11255
|
|
|
10028
11256
|
ipcMain.handle('schedule:open-folder', async () => {
|
|
10029
11257
|
const { shell } = require('electron');
|
|
10030
|
-
await shell.openPath(
|
|
11258
|
+
await shell.openPath(schedulerPaths.scheduledPlansRoot());
|
|
10031
11259
|
return { ok: true };
|
|
10032
11260
|
});
|
|
10033
11261
|
|
|
@@ -10045,8 +11273,8 @@ function registerScheduleHandlers() {
|
|
|
10045
11273
|
ipcMain.handle('schedule:read-log', validated(schemas.scheduleReadLog, async ({ slug, runId }) => {
|
|
10046
11274
|
// Defense-in-depth: re-check containment after path.resolve even though
|
|
10047
11275
|
// SLUG_RE / RUN_ID_RE already forbid path separators.
|
|
10048
|
-
const logPath = path.resolve(path.join(
|
|
10049
|
-
if (!logPath.startsWith(
|
|
11276
|
+
const logPath = path.resolve(path.join(schedulerPaths.runsDir(), runId, `${slug}.log`));
|
|
11277
|
+
if (!logPath.startsWith(schedulerPaths.runsDir() + path.sep)) {
|
|
10050
11278
|
return { ok: false, error: 'invalid slug or runId' };
|
|
10051
11279
|
}
|
|
10052
11280
|
try {
|
|
@@ -10063,8 +11291,8 @@ function registerScheduleHandlers() {
|
|
|
10063
11291
|
// template, authored before the user fills in `cwd`) falls back to the
|
|
10064
11292
|
// legacy global dir until it's re-saved with a real cwd and migrated by
|
|
10065
11293
|
// the next reconcile-driven scan.
|
|
10066
|
-
const dir = (await findPrdDir(data.slug)) ??
|
|
10067
|
-
if (dir ===
|
|
11294
|
+
const dir = (await findPrdDir(data.slug)) ?? schedulerPaths.prdsRoot();
|
|
11295
|
+
if (dir === schedulerPaths.prdsRoot()) ensureDirs();
|
|
10068
11296
|
const resolved = safeSlugPathIn(dir, data.slug);
|
|
10069
11297
|
if (!resolved) return { ok: false, error: 'invalid slug' };
|
|
10070
11298
|
try {
|
|
@@ -10096,6 +11324,15 @@ function registerScheduleHandlers() {
|
|
|
10096
11324
|
});
|
|
10097
11325
|
}
|
|
10098
11326
|
|
|
11327
|
+
function stopDispatchLoop() {
|
|
11328
|
+
if (dispatchLoopHandle) { dispatchLoopHandle.stop(); dispatchLoopHandle = null; }
|
|
11329
|
+
}
|
|
11330
|
+
|
|
11331
|
+
/** Shutdown path: stop the timers this module owns. */
|
|
11332
|
+
function stop() {
|
|
11333
|
+
stopDispatchLoop();
|
|
11334
|
+
}
|
|
11335
|
+
|
|
10099
11336
|
async function init() {
|
|
10100
11337
|
ensureDirs();
|
|
10101
11338
|
// Boot phase — reconciliation, migrations, self-heal, first reset probe.
|
|
@@ -10111,6 +11348,9 @@ async function init() {
|
|
|
10111
11348
|
// A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
|
|
10112
11349
|
// batch — advance the queue without waiting for the next 60s poll.
|
|
10113
11350
|
sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
|
|
11351
|
+
// Boot-time expiry pass (process-local state is empty after a restart, so
|
|
11352
|
+
// this is a cheap belt-and-braces run against the freshly read queue).
|
|
11353
|
+
try { runReservationExpiryPass((await readQueue()).jobs); } catch { /* best-effort */ }
|
|
10114
11354
|
// Retire the global queue.json: split its rows into per-project shards
|
|
10115
11355
|
// BEFORE the first read below, so boot reconciliation sees the shards.
|
|
10116
11356
|
try {
|
|
@@ -10131,17 +11371,17 @@ async function init() {
|
|
|
10131
11371
|
// Boot reconciliation: finalize any job that was 'running' when the app died.
|
|
10132
11372
|
// Check the run log first — a job that emitted result/success before the crash
|
|
10133
11373
|
// should be marked 'completed', not 'failed', so it doesn't wedge the queue
|
|
10134
|
-
// via the failure-gate.
|
|
10135
|
-
// it from continuing to write to the project unsupervised (2026-05-21 incident).
|
|
11374
|
+
// via the failure-gate. A still-live executor is spared, not killed.
|
|
10136
11375
|
//
|
|
10137
11376
|
// classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
|
|
10138
11377
|
// Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
|
|
10139
11378
|
// does not stall the event loop or hold the mutateTail chain during startup.
|
|
10140
11379
|
//
|
|
10141
|
-
//
|
|
10142
|
-
//
|
|
11380
|
+
// Rows proven alive are adopted (left running, never killed) — see
|
|
11381
|
+
// partitionBootOrphans. Everything else is proven dead/exited and is safe to
|
|
10143
11382
|
// classify immediately below.
|
|
10144
11383
|
const bootSnap = readQueueSync();
|
|
11384
|
+
const bootLogPath = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
|
|
10145
11385
|
|
|
10146
11386
|
// Worktree boot reconciliation (PRD 994): a job worktree that survives an
|
|
10147
11387
|
// app crash/host reboot must not leak disk or a dangling branch forever —
|
|
@@ -10156,11 +11396,15 @@ async function init() {
|
|
|
10156
11396
|
// itself, proof its run already died. isLive checks the already-read
|
|
10157
11397
|
// bootSnap (no extra queue read) for a live running-row pid, OR a live
|
|
10158
11398
|
// /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
|
|
11399
|
+
// rowPid walks the SAME record → runtime.pid → log-pid ladder
|
|
11400
|
+
// partitionBootOrphans uses, so the sweep and the partition can never
|
|
11401
|
+
// disagree about which executor is alive.
|
|
10159
11402
|
const isLive = buildJobWorktreeIsLive({
|
|
10160
11403
|
bootJobs: bootSnap.jobs,
|
|
10161
11404
|
claudePidAlive,
|
|
10162
11405
|
hasLiveHolder: gitWorktree.hasLiveHolder,
|
|
10163
11406
|
cwdHolders: gitWorktree.listCwdHolders(),
|
|
11407
|
+
rowPid: (j) => bootRowPid(j, bootLogPath),
|
|
10164
11408
|
});
|
|
10165
11409
|
await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
|
|
10166
11410
|
} catch (e) {
|
|
@@ -10180,12 +11424,30 @@ async function init() {
|
|
|
10180
11424
|
console.error('[scheduler] boot epic-worktree reconciliation failed', e?.message);
|
|
10181
11425
|
}
|
|
10182
11426
|
|
|
10183
|
-
const { immediate: immediateSlugs,
|
|
11427
|
+
const { immediate: immediateSlugs, adopted: adoptedSlugs } = partitionBootOrphans(bootSnap.jobs, {
|
|
11428
|
+
pidAlive: claudePidAlive,
|
|
11429
|
+
getLogPid: (j) => readSpawnedPidFromLog(bootLogPath(j)),
|
|
11430
|
+
getLogMtimeMs: (j) => readLogMtimeMs(bootLogPath(j)),
|
|
11431
|
+
logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
|
|
11432
|
+
findLiveProcess: (j) => findLiveProcessForJob(j, {
|
|
11433
|
+
worktreeDir: jobWorktree.worktreeDirFor(j.cwd || DEFAULT_PROJECT_CWD, j.slug),
|
|
11434
|
+
runCwd: j.runtime?.cwd || j.cwd,
|
|
11435
|
+
}),
|
|
11436
|
+
readRecord: supervisorRecord.readSupervisorRecord,
|
|
11437
|
+
});
|
|
10184
11438
|
const bootOutcomes = new Map();
|
|
10185
11439
|
for (const j of bootSnap.jobs) {
|
|
10186
11440
|
if (!immediateSlugs.includes(j.slug)) continue;
|
|
10187
|
-
const logPath = j.runId ? path.join(
|
|
10188
|
-
|
|
11441
|
+
const logPath = j.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null;
|
|
11442
|
+
let outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
11443
|
+
// A row whose run already wrote its exit marker (meta.json) is finalized
|
|
11444
|
+
// from that meta when the log tail alone can't say (killed/torn tail).
|
|
11445
|
+
if (outcome === 'unknown' || outcome === 'no_result') {
|
|
11446
|
+
let meta = null;
|
|
11447
|
+
try { meta = j.runId ? JSON.parse(fs.readFileSync(path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.meta.json`), 'utf8')) : null; } catch { /* no/torn meta — keep the log outcome */ }
|
|
11448
|
+
if (meta && typeof meta.exitCode === 'number') outcome = meta.exitCode === 0 ? 'success' : 'failed';
|
|
11449
|
+
}
|
|
11450
|
+
bootOutcomes.set(j.slug, outcome);
|
|
10189
11451
|
}
|
|
10190
11452
|
// Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
|
|
10191
11453
|
// BEFORE mutate() for the same reason (git spawn work must never run
|
|
@@ -10215,53 +11477,32 @@ async function init() {
|
|
|
10215
11477
|
await archiveCompletedPrd(slug, cwd);
|
|
10216
11478
|
}
|
|
10217
11479
|
|
|
10218
|
-
//
|
|
10219
|
-
//
|
|
10220
|
-
//
|
|
10221
|
-
//
|
|
10222
|
-
|
|
10223
|
-
|
|
10224
|
-
const
|
|
10225
|
-
|
|
10226
|
-
|
|
10227
|
-
|
|
10228
|
-
|
|
10229
|
-
|
|
10230
|
-
|
|
10231
|
-
|
|
10232
|
-
|
|
10233
|
-
|
|
10234
|
-
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
10235
|
-
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
10236
|
-
// Same evidence-before-failure gate as the immediate-orphan path
|
|
10237
|
-
// above, resolved before mutate() for the same reason (git spawn
|
|
10238
|
-
// work must never run inside mutate()'s serialization chain). Uses
|
|
10239
|
-
// the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
|
|
10240
|
-
// race guard below already confirms `cur` is still this same run
|
|
10241
|
-
// (runId === bootRunId) before this evidence is applied.
|
|
10242
|
-
const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
|
|
10243
|
-
? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
|
|
10244
|
-
: null;
|
|
10245
|
-
let deferredCompletedCwd;
|
|
10246
|
-
mutate((state) => {
|
|
10247
|
-
const cur = state.jobs.find((x) => x.slug === slug);
|
|
10248
|
-
// Race guard: bail if the job already resolved, OR if it's already been
|
|
10249
|
-
// re-picked into a NEW run (different runId) within the grace window —
|
|
10250
|
-
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
10251
|
-
// touched by this stale classification.
|
|
10252
|
-
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
10253
|
-
applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
|
|
10254
|
-
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
10255
|
-
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
10256
|
-
}).then(() => {
|
|
10257
|
-
if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
|
|
10258
|
-
}).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
|
|
10259
|
-
}, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
|
|
11480
|
+
// Proven-alive rows stay `running` and are never signalled (the boot worktree
|
|
11481
|
+
// sweep above already spares their checkout). No sessionSlots token is
|
|
11482
|
+
// acquired: pickNextBatch's untrackedRunning correction counts the row
|
|
11483
|
+
// against the pool, and reapDeadRunningJobs finalizes it on exit.
|
|
11484
|
+
if (adoptedSlugs.length) {
|
|
11485
|
+
const adoptedAtBoot = new Date().toISOString();
|
|
11486
|
+
const adoptedRunIds = new Map(bootSnap.jobs.filter((j) => adoptedSlugs.includes(j.slug)).map((j) => [j.slug, j.runId ?? null]));
|
|
11487
|
+
await mutate((state) => {
|
|
11488
|
+
for (const j of state.jobs) {
|
|
11489
|
+
// runId guard: never stamp a DIFFERENT later run of the same slug.
|
|
11490
|
+
if (j.status !== 'running' || !adoptedRunIds.has(j.slug) || (j.runId ?? null) !== adoptedRunIds.get(j.slug)) continue;
|
|
11491
|
+
j.adoptedAtBoot = adoptedAtBoot;
|
|
11492
|
+
delete j.supervisedAt; // a prior process's supervisor died with it
|
|
11493
|
+
console.log(`[scheduler] boot: adopted live executor for ${j.slug} (pid=${j.runtime?.pid ?? 'unknown'}) — left running, no signal`);
|
|
11494
|
+
}
|
|
11495
|
+
});
|
|
10260
11496
|
}
|
|
10261
11497
|
|
|
11498
|
+
// Re-arm budget/idle/deadman + the quietMachine lease for the adopted rows
|
|
11499
|
+
// (a dispatch-loop pass repeats this for any row left without a supervisor).
|
|
11500
|
+
await superviseAdoptedRunsPass();
|
|
11501
|
+
|
|
10262
11502
|
// If we boot up while paused with a resumeAt in the past, clear it. This
|
|
10263
11503
|
// happens when the app was closed across the reset window.
|
|
10264
11504
|
const boot = await readQueue();
|
|
11505
|
+
await clearStaleDrainAtBoot(boot);
|
|
10265
11506
|
if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
|
|
10266
11507
|
await clearPause('boot-elapsed');
|
|
10267
11508
|
} else if (boot.paused && boot.paused.resumeAt) {
|
|
@@ -10483,7 +11724,7 @@ async function init() {
|
|
|
10483
11724
|
}
|
|
10484
11725
|
for (const target of exhaustedNeedsReviewTargets) {
|
|
10485
11726
|
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
10486
|
-
const outcome = applyNeedsReviewAutoResolve(j);
|
|
11727
|
+
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
10487
11728
|
if (outcome) {
|
|
10488
11729
|
console.warn(
|
|
10489
11730
|
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
@@ -10504,7 +11745,7 @@ async function init() {
|
|
|
10504
11745
|
}
|
|
10505
11746
|
}).catch(() => {});
|
|
10506
11747
|
}
|
|
10507
|
-
},
|
|
11748
|
+
}, REVERIFY_INTERVAL_MS);
|
|
10508
11749
|
|
|
10509
11750
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
10510
11751
|
// old fixed-interval pollTimer + initialPollTimeout.
|
|
@@ -10525,87 +11766,18 @@ async function init() {
|
|
|
10525
11766
|
// setInterval callback is sync; readQueueSync stays sync to avoid awaiting
|
|
10526
11767
|
// inside the timer body (and the 60s cadence makes the cost moot).
|
|
10527
11768
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
10528
|
-
heartbeatInterval = setInterval(() =>
|
|
10529
|
-
const s = readQueueSync();
|
|
10530
|
-
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
10531
|
-
// running, something must drive it. This is the only driver that does
|
|
10532
|
-
// not depend on the billing poll loop, a pause timer, or a completing
|
|
10533
|
-
// job to schedule the next tick — every one of which has failed at
|
|
10534
|
-
// least once. See classifyQueueStarvation.
|
|
10535
|
-
if (!s.unreadable) {
|
|
10536
|
-
runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
10537
|
-
}
|
|
10538
|
-
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
10539
|
-
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
10540
|
-
// failed }` literal silently minted a NEW key for any other value
|
|
10541
|
-
// (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
|
|
10542
|
-
// heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
10543
|
-
// visibility instead of the alarm it should have been. Any row whose
|
|
10544
|
-
// status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
|
|
10545
|
-
// this is the last line of defence) routes into `unknown`, never a
|
|
10546
|
-
// freshly-minted key.
|
|
10547
|
-
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
10548
|
-
counts.unknown = 0;
|
|
10549
|
-
for (const j of s.jobs) {
|
|
10550
|
-
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
10551
|
-
counts[j.status] += 1;
|
|
10552
|
-
} else {
|
|
10553
|
-
counts.unknown += 1;
|
|
10554
|
-
}
|
|
10555
|
-
}
|
|
10556
|
-
|
|
10557
|
-
const stall = computeStallSummary(s);
|
|
10558
|
-
// Per-project alerting (see computeStallSummary's header): a project
|
|
10559
|
-
// stalled while others are busy must still fire, and one project
|
|
10560
|
-
// recovering must not clear or suppress another's still-open episode —
|
|
10561
|
-
// that is exactly what a single module-level stallSince/stallToasted
|
|
10562
|
-
// flag masked before (the burrow-vs-others incident this PRD fixes).
|
|
10563
|
-
const now = Date.now();
|
|
10564
|
-
const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
|
|
10565
|
-
for (const cwd of [...stallSince.keys()]) {
|
|
10566
|
-
if (!stalledCwds.includes(cwd)) {
|
|
10567
|
-
stallSince.delete(cwd);
|
|
10568
|
-
stallToasted.delete(cwd);
|
|
10569
|
-
}
|
|
10570
|
-
}
|
|
10571
|
-
const toAlert = [];
|
|
10572
|
-
for (const cwd of stalledCwds) {
|
|
10573
|
-
if (!stallSince.has(cwd)) stallSince.set(cwd, now);
|
|
10574
|
-
if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
|
|
10575
|
-
stallToasted.set(cwd, true);
|
|
10576
|
-
toAlert.push(cwd);
|
|
10577
|
-
}
|
|
10578
|
-
}
|
|
10579
|
-
if (toAlert.length > 0) {
|
|
10580
|
-
console.error(
|
|
10581
|
-
`[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
|
|
10582
|
-
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
10583
|
-
stall.byProject,
|
|
10584
|
-
);
|
|
10585
|
-
appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
|
|
10586
|
-
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
10587
|
-
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
10588
|
-
message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
|
|
10589
|
-
projects: toAlert,
|
|
10590
|
-
total: stall.total,
|
|
10591
|
-
byProject: stall.byProject,
|
|
10592
|
-
});
|
|
10593
|
-
}
|
|
10594
|
-
}
|
|
10595
|
-
|
|
10596
|
-
appendHeartbeat({
|
|
10597
|
-
ts: Date.now(),
|
|
10598
|
-
pid: process.pid,
|
|
10599
|
-
counts,
|
|
10600
|
-
stall: { stalled: stall.stalled, total: stall.total },
|
|
10601
|
-
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
10602
|
-
nextReset: cachedNextReset,
|
|
10603
|
-
utilization: cachedUtilization,
|
|
10604
|
-
consecutiveFailures,
|
|
10605
|
-
});
|
|
10606
|
-
}, 60_000);
|
|
11769
|
+
heartbeatInterval = setInterval(() => heartbeatTick(), 60_000);
|
|
10607
11770
|
if (heartbeatInterval.unref) heartbeatInterval.unref();
|
|
10608
11771
|
|
|
11772
|
+
// Dispatch's own periodic driver: cadence is independent of pollLoop's billing
|
|
11773
|
+
// backoff. A loop tick meeting a cancelled cancelToken returns 'cancelled' from
|
|
11774
|
+
// tickBody; clearing stays with the starvation watchdog's existing force-clear.
|
|
11775
|
+
stopDispatchLoop();
|
|
11776
|
+
dispatchLoopHandle = startDispatchLoop({
|
|
11777
|
+
tick: () => tickQueue(),
|
|
11778
|
+
onError: (e) => console.warn('[scheduler] dispatch loop tick failed', e?.message),
|
|
11779
|
+
});
|
|
11780
|
+
|
|
10609
11781
|
// Wake-from-sleep: immediately re-poll and re-evaluate the queue.
|
|
10610
11782
|
try {
|
|
10611
11783
|
const { powerMonitor } = require('electron');
|
|
@@ -10828,8 +12000,8 @@ const remote = {
|
|
|
10828
12000
|
}
|
|
10829
12001
|
await fsp.mkdir(dir, { recursive: true });
|
|
10830
12002
|
} else {
|
|
10831
|
-
dir = (await findPrdDir(slug)) ??
|
|
10832
|
-
if (dir ===
|
|
12003
|
+
dir = (await findPrdDir(slug)) ?? schedulerPaths.prdsRoot();
|
|
12004
|
+
if (dir === schedulerPaths.prdsRoot()) ensureDirs();
|
|
10833
12005
|
}
|
|
10834
12006
|
|
|
10835
12007
|
// writePrd only ever JOINS an existing Epic now (no mintAuthority
|
|
@@ -10866,6 +12038,19 @@ const remote = {
|
|
|
10866
12038
|
}
|
|
10867
12039
|
},
|
|
10868
12040
|
|
|
12041
|
+
// User-initiated pause/resume — the admin-route/MCP twins of the
|
|
12042
|
+
// schedule:pause / schedule:resume IPC handlers, through the same setPaused /
|
|
12043
|
+
// clearPause. Pause stops NEW dispatch only; running jobs are never touched.
|
|
12044
|
+
async pause() {
|
|
12045
|
+
await setPaused('manual', null);
|
|
12046
|
+
return { ok: true };
|
|
12047
|
+
},
|
|
12048
|
+
|
|
12049
|
+
async resume() {
|
|
12050
|
+
await clearPause('manual');
|
|
12051
|
+
return { ok: true };
|
|
12052
|
+
},
|
|
12053
|
+
|
|
10869
12054
|
async resetJob(slug, opts = {}) {
|
|
10870
12055
|
const resolved = await resolveSlugOrReason(slug, opts.cwd);
|
|
10871
12056
|
if (!resolved.ok) {
|
|
@@ -11193,6 +12378,14 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
11193
12378
|
sendJson(res, 200, jobs);
|
|
11194
12379
|
});
|
|
11195
12380
|
|
|
12381
|
+
adminHttp.registerRoute('POST', '/admin/scheduler/pause', async (req, res) => {
|
|
12382
|
+
sendJson(res, 200, await remoteObj.pause());
|
|
12383
|
+
});
|
|
12384
|
+
|
|
12385
|
+
adminHttp.registerRoute('POST', '/admin/scheduler/resume', async (req, res) => {
|
|
12386
|
+
sendJson(res, 200, await remoteObj.resume());
|
|
12387
|
+
});
|
|
12388
|
+
|
|
11196
12389
|
adminHttp.registerRoute('POST', '/admin/scheduler/reset-job', async (req, res) => {
|
|
11197
12390
|
const raw = await readBody(req);
|
|
11198
12391
|
let parsed;
|
|
@@ -11217,6 +12410,8 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
11217
12410
|
module.exports = {
|
|
11218
12411
|
classifyQueueStarvation,
|
|
11219
12412
|
classifyQueueStarvationByProject,
|
|
12413
|
+
dispatchIdleMs,
|
|
12414
|
+
launchBlockedSlugs,
|
|
11220
12415
|
classifyQueueHealth,
|
|
11221
12416
|
runQueueStarvationWatchdog,
|
|
11222
12417
|
QUEUE_STARVATION_MS,
|
|
@@ -11240,13 +12435,13 @@ module.exports = {
|
|
|
11240
12435
|
registerScheduleHandlers,
|
|
11241
12436
|
attachWindow,
|
|
11242
12437
|
init,
|
|
11243
|
-
ROOT,
|
|
11244
|
-
PRDS_DIR,
|
|
11245
|
-
SCHEDULER_STATE_PATH,
|
|
11246
12438
|
BACKOFF_MAX_MS,
|
|
11247
12439
|
FAILURE_STREAK_WARN_THRESHOLD,
|
|
12440
|
+
FAILURE_STREAK_ESCALATION_MS,
|
|
11248
12441
|
nextBackoffMs,
|
|
11249
12442
|
shouldWarnFailureStreak,
|
|
12443
|
+
shouldEscalateFailureStreak,
|
|
12444
|
+
computeDegradedBudget,
|
|
11250
12445
|
healRefusalReason,
|
|
11251
12446
|
writeQueue,
|
|
11252
12447
|
reconcile,
|
|
@@ -11269,6 +12464,8 @@ module.exports = {
|
|
|
11269
12464
|
memoryLimitedBatchSize,
|
|
11270
12465
|
availableForJobs,
|
|
11271
12466
|
reverifyNeedsReview,
|
|
12467
|
+
runGateShadow,
|
|
12468
|
+
awaitGateShadowIdle: async () => { while (gateShadowPending) await gateShadowPending; },
|
|
11272
12469
|
shouldRunPeriodicReverify,
|
|
11273
12470
|
findStuckFailedJobs,
|
|
11274
12471
|
STUCK_FAILED_ESCALATE_MS,
|
|
@@ -11283,6 +12480,13 @@ module.exports = {
|
|
|
11283
12480
|
NEEDS_REVIEW_RESOLVE_MS,
|
|
11284
12481
|
needsReviewAutoResolveDisabled,
|
|
11285
12482
|
isRescanCandidate,
|
|
12483
|
+
isTranscriptRescannable,
|
|
12484
|
+
selectEvidenceScanTargets,
|
|
12485
|
+
RESCAN_EXCLUDED_VERDICTS,
|
|
12486
|
+
RESCANNABLE_VERDICTS,
|
|
12487
|
+
EVIDENCE_SCAN_MAX_PER_PASS,
|
|
12488
|
+
EVIDENCE_SCAN_MIN_INTERVAL_MS,
|
|
12489
|
+
REVERIFY_INTERVAL_MS,
|
|
11286
12490
|
isFailedUnverifiedShaped,
|
|
11287
12491
|
computeLooksDone,
|
|
11288
12492
|
attributeLandedCommits,
|
|
@@ -11295,6 +12499,9 @@ module.exports = {
|
|
|
11295
12499
|
isExhaustedAutoFix,
|
|
11296
12500
|
GUARD_VERDICT_EVIDENCE_ELIGIBLE,
|
|
11297
12501
|
isGuardParkedWithoutAutoFix,
|
|
12502
|
+
isStrandedAutoFixPark,
|
|
12503
|
+
isStaleSharedTreeRevertedPark,
|
|
12504
|
+
landedCommitIsAncestorOfHead,
|
|
11298
12505
|
isEligibleForNeedsReviewAutoResolve,
|
|
11299
12506
|
isPlanUnqueued,
|
|
11300
12507
|
isFixPlanDead,
|
|
@@ -11334,7 +12541,6 @@ module.exports = {
|
|
|
11334
12541
|
buildScheduleStatePayload,
|
|
11335
12542
|
partitionBootOrphans,
|
|
11336
12543
|
applyOrphanOutcome,
|
|
11337
|
-
BOOT_ORPHAN_KILL_GRACE_MS,
|
|
11338
12544
|
registerAdminRoutes,
|
|
11339
12545
|
notifyOriginatingTab,
|
|
11340
12546
|
notifyNeedsReview,
|
|
@@ -11360,10 +12566,13 @@ module.exports = {
|
|
|
11360
12566
|
SCHEDULER_CODE_SHA,
|
|
11361
12567
|
resetJobFields,
|
|
11362
12568
|
executeJob,
|
|
12569
|
+
killOrphanClaudePid,
|
|
11363
12570
|
prdArchivedSkipResult,
|
|
11364
12571
|
spawnJob,
|
|
11365
12572
|
listPrdsInternal,
|
|
11366
12573
|
computeStallSummary,
|
|
12574
|
+
heartbeatTick,
|
|
12575
|
+
appendHeartbeat,
|
|
11367
12576
|
findStaleQuarantinedJobs,
|
|
11368
12577
|
QUARANTINE_ESCALATE_MS,
|
|
11369
12578
|
selectQuarantineAutoResolveTargets,
|
|
@@ -11402,6 +12611,10 @@ module.exports = {
|
|
|
11402
12611
|
setPaused,
|
|
11403
12612
|
clearPause,
|
|
11404
12613
|
tickQueue,
|
|
12614
|
+
setRestartHandler,
|
|
12615
|
+
driveUpgradeDrain,
|
|
12616
|
+
clearStaleDrainAtBoot,
|
|
12617
|
+
stop,
|
|
11405
12618
|
runDueJobs,
|
|
11406
12619
|
pollLoop,
|
|
11407
12620
|
maybeLaunchWhenAvailable,
|
|
@@ -11410,12 +12623,23 @@ module.exports = {
|
|
|
11410
12623
|
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|
|
11411
12624
|
RAPID_RATE_LIMIT_WINDOW_MS,
|
|
11412
12625
|
MANUAL_PAUSE_COOLDOWN_MS,
|
|
11413
|
-
RUNS_DIR,
|
|
11414
12626
|
pickRunDir,
|
|
11415
12627
|
resolveRateLimitPauseReset,
|
|
12628
|
+
billingResetForPause,
|
|
11416
12629
|
computeEffectiveResumeAt,
|
|
11417
12630
|
computeResumeDelay,
|
|
11418
12631
|
FOREIGN_WIP_BLOCK_STREAK_LIMIT,
|
|
11419
12632
|
validateForeignWipBlockClaim,
|
|
11420
12633
|
requeueForeignWipBlockedJobs,
|
|
11421
12634
|
};
|
|
12635
|
+
|
|
12636
|
+
// Lazy path getters: resolved from SM_SCHEDULER_HOME at each read, never frozen
|
|
12637
|
+
// at require time (see lib/schedulerPaths.cjs).
|
|
12638
|
+
for (const [name, resolve] of [
|
|
12639
|
+
['ROOT', schedulerPaths.scheduledPlansRoot],
|
|
12640
|
+
['PRDS_DIR', schedulerPaths.prdsRoot],
|
|
12641
|
+
['RUNS_DIR', schedulerPaths.runsDir],
|
|
12642
|
+
['SCHEDULER_STATE_PATH', schedulerPaths.schedulerStatePath],
|
|
12643
|
+
]) {
|
|
12644
|
+
Object.defineProperty(module.exports, name, { get: resolve, enumerable: true });
|
|
12645
|
+
}
|