claude-code-session-manager 0.87.0 → 0.88.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -53
- package/dist/assets/{AgentLibrary-DyLWzZDf.js → AgentLibrary-BZ6IY2g1.js} +1 -1
- package/dist/assets/{DataModel--mISIJ6h.js → DataModel-dAoGAODX.js} +1 -1
- package/dist/assets/{History-C2ahUXTg.js → History-CF157HyX.js} +1 -1
- package/dist/assets/{Hooks-BiC6oyR2.js → Hooks-CCPPipBk.js} +1 -1
- package/dist/assets/{HostBilko-BPleEOld.js → HostBilko-Dlmz2m06.js} +1 -1
- package/dist/assets/{Library-Dc8Qst1R.js → Library-BaXcboWX.js} +1 -1
- package/dist/assets/{ListDetail-DIXh-OLX.js → ListDetail-CoUCzG5S.js} +1 -1
- package/dist/assets/{MarkdownEditor-C90bkLXK.js → MarkdownEditor-DTg3JlOT.js} +1 -1
- package/dist/assets/{McpServers-DqcbLOLZ.js → McpServers-DdjY1sA3.js} +1 -1
- package/dist/assets/{Memory-CW62MXlh.js → Memory-BJ3Jlw0C.js} +1 -1
- package/dist/assets/{Panel-Bw1FhRuF.js → Panel-ygRSgc41.js} +1 -1
- package/dist/assets/{Permissions-BcUC-5y8.js → Permissions-m0KAnccF.js} +1 -1
- package/dist/assets/{Plugins-BnKx9flD.js → Plugins-CJRl6B3Q.js} +2 -2
- package/dist/assets/{ProvenanceBadge-Bw5vNVPT.js → ProvenanceBadge-w8cnlVvW.js} +1 -1
- package/dist/assets/{SaveBar-CWr0O_w-.js → SaveBar-BpyPcUIk.js} +1 -1
- package/dist/assets/{Scheduler-DYdLuUqq.js → Scheduler-D2ppxSfy.js} +7 -7
- package/dist/assets/{ScopeSwitcher-CrBLbg8s.js → ScopeSwitcher-EyuIwt-j.js} +1 -1
- package/dist/assets/{Settings-DluB-vN1.js → Settings-BHSGGFEc.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-CHLSseay.js → SkillReferenceGraph-BNaZ2JCF.js} +1 -1
- package/dist/assets/{Skills-gNdo_HNK.js → Skills-Dl3q75QV.js} +1 -1
- package/dist/assets/{SystemPrompt-Cru05-Ia.js → SystemPrompt-Bg6F-Cyp.js} +1 -1
- package/dist/assets/{TagLibrary-DNHY0xou.js → TagLibrary-BSutflCy.js} +1 -1
- package/dist/assets/{TiptapBody-I4lmbCgP.js → TiptapBody-CXih4xrc.js} +1 -1
- package/dist/assets/{Toggle-bWMHjmRh.js → Toggle-2qtUHhA9.js} +1 -1
- package/dist/assets/{index-fc_JjdxL.js → index-B3emy2oI.js} +356 -356
- package/dist/assets/{index-DV3PorRY.css → index-BHTX4OTc.css} +1 -1
- package/dist/assets/{settingsSchema-BfhtZnGD.js → settingsSchema-CdzxaqwM.js} +1 -1
- package/dist/assets/{whisperWorker-Dbia1OpC.js → whisperWorker-C7ZGQwKg.js} +7 -7
- package/dist/index.html +2 -2
- package/dist/vad/ort-wasm-simd-threaded.asyncify.mjs +106 -110
- package/dist/vad/ort-wasm-simd-threaded.asyncify.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.jsep.mjs +98 -98
- package/dist/vad/ort-wasm-simd-threaded.jsep.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.jspi.mjs +99 -102
- package/dist/vad/ort-wasm-simd-threaded.jspi.wasm +0 -0
- package/dist/vad/ort-wasm-simd-threaded.mjs +46 -46
- package/dist/vad/ort-wasm-simd-threaded.wasm +0 -0
- package/package.json +10 -13
- package/scripts/README.md +59 -0
- package/scripts/audit-ops-hygiene.cjs +350 -0
- package/scripts/hooks/guard-destructive-git.cjs +10 -34
- package/scripts/hooks/guard-inline-implementation.cjs +14 -8
- package/scripts/hooks/guard-prd-writes.cjs +7 -36
- package/scripts/hooks/guard-self-schedule.cjs +175 -0
- package/scripts/ops-sweep.cjs +355 -0
- package/scripts/scheduler-mcp-server.cjs +28 -71
- package/src/main/__tests__/bilkoHost-integration.test.cjs +3 -3
- package/src/main/__tests__/chat-cancel-terminal.test.cjs +6 -9
- package/src/main/__tests__/chat-exit-close-race.test.cjs +3 -3
- package/src/main/__tests__/chat-mcp-consent-notice.test.cjs +4 -5
- package/src/main/__tests__/chat-queue.test.cjs +2 -2
- package/src/main/__tests__/chat-stop-signal.test.cjs +2 -2
- package/src/main/__tests__/dep-orphan-archive-health.test.cjs +77 -0
- package/src/main/__tests__/dod-batchkey.test.cjs +2 -2
- package/src/main/__tests__/dod-drain-hook.test.cjs +2 -2
- package/src/main/__tests__/dod-report.test.cjs +2 -2
- package/src/main/__tests__/dod-reverify.test.cjs +2 -2
- package/src/main/__tests__/epicMint.test.cjs +2 -2
- package/src/main/__tests__/exchanges.test.cjs +2 -2
- package/src/main/__tests__/extractJson.test.cjs +2 -2
- package/src/main/__tests__/files-reject-credentials.test.cjs +1 -1
- package/src/main/__tests__/fixtures/1218-fo-01-move-scripts-lib-into-src-main-lib.log +556 -0
- package/src/main/__tests__/health-build-freshness.test.cjs +39 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +15 -1
- package/src/main/__tests__/health-queue-dispatch.test.cjs +58 -7
- package/src/main/__tests__/health-tick-liveness.test.cjs +11 -3
- package/src/main/__tests__/health-usage-poller.test.cjs +70 -23
- package/src/main/__tests__/historyRollup.test.cjs +2 -2
- package/src/main/__tests__/kg-augment.test.cjs +2 -2
- package/src/main/__tests__/mcpStatus.test.cjs +2 -2
- package/src/main/__tests__/memoryAggregate.test.cjs +1 -1
- package/src/main/__tests__/memoryStale.test.cjs +1 -1
- package/src/main/__tests__/opsErrorLogTelemetryTap.test.cjs +25 -1
- package/src/main/__tests__/pollLoop-dispatch-on-failure.test.cjs +33 -3
- package/src/main/__tests__/prd-group-allocator.test.cjs +2 -2
- package/src/main/__tests__/prdAdminRouteParity.test.cjs +2 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +14 -2
- package/src/main/__tests__/prdAuthoringSeed.test.cjs +39 -0
- package/src/main/__tests__/prdLocationsArchived.test.cjs +20 -13
- package/src/main/__tests__/proc-role-env.test.cjs +125 -0
- package/src/main/__tests__/procname-claude-spawn-sites.test.cjs +304 -0
- package/src/main/__tests__/procname-sm-processes.test.cjs +127 -0
- package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +81 -401
- package/src/main/__tests__/projectPages.test.cjs +63 -149
- package/src/main/__tests__/queue-health-verdict.test.cjs +9 -9
- package/src/main/__tests__/queue-starvation-dispatch-driver.test.cjs +117 -20
- package/src/main/__tests__/queueHistory.test.cjs +2 -2
- package/src/main/__tests__/rateLimitPollerStreak.test.cjs +38 -3
- package/src/main/__tests__/runVerify-landed-commit-outranks.test.cjs +181 -0
- package/src/main/__tests__/runVerify.test.cjs +5 -5
- package/src/main/__tests__/scheduleJobStatusDrift.test.cjs +3 -3
- package/src/main/__tests__/scheduleJobTransitions.test.cjs +2 -2
- package/src/main/__tests__/scheduler-adopted-run-supervision.test.cjs +143 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +2 -2
- package/src/main/__tests__/scheduler-autopromote.test.cjs +2 -2
- package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +3 -5
- package/src/main/__tests__/scheduler-boot-orphans.test.cjs +78 -97
- package/src/main/__tests__/scheduler-default-eligible-heal.test.cjs +161 -0
- package/src/main/__tests__/scheduler-dispatch-loop.test.cjs +58 -0
- package/src/main/__tests__/scheduler-epic-digest.test.cjs +3 -5
- package/src/main/__tests__/scheduler-force-tick-outcome.test.cjs +2 -2
- package/src/main/__tests__/scheduler-gate-shadow.test.cjs +119 -0
- package/src/main/__tests__/scheduler-guard-verdict-autoresolve.test.cjs +46 -0
- package/src/main/__tests__/scheduler-heartbeat-payload.test.cjs +80 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +25 -18
- package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +4 -4
- package/src/main/__tests__/scheduler-launch-failure.test.cjs +3 -5
- package/src/main/__tests__/scheduler-looks-done.test.cjs +26 -7
- package/src/main/__tests__/scheduler-manual-pause.test.cjs +118 -0
- package/src/main/__tests__/scheduler-meta-code-sha.test.cjs +26 -3
- package/src/main/__tests__/scheduler-prd-missing-skip.test.cjs +17 -2
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +3 -5
- package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +41 -6
- package/src/main/__tests__/scheduler-rate-limit-pause.test.cjs +62 -6
- package/src/main/__tests__/scheduler-rate-limit-spin-guard.test.cjs +3 -5
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +23 -4
- package/src/main/__tests__/scheduler-reconcile-cwd-preserve.test.cjs +100 -0
- package/src/main/__tests__/scheduler-reconcile-invalid-repair.test.cjs +3 -3
- package/src/main/__tests__/scheduler-reconcile-quarantine.test.cjs +4 -4
- package/src/main/__tests__/scheduler-shard-quarantine.test.cjs +110 -0
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +81 -1
- package/src/main/__tests__/scheduler-starve-escalation.test.cjs +4 -4
- package/src/main/__tests__/scheduler-stranded-autofix-park.test.cjs +245 -0
- package/src/main/__tests__/scheduler-supervisor-record.test.cjs +81 -0
- package/src/main/__tests__/scheduler-tick-cancel-token.test.cjs +2 -2
- package/src/main/__tests__/scheduler-tick-wedge.test.cjs +172 -0
- package/src/main/__tests__/scheduler-transient-failure.test.cjs +2 -2
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +37 -11
- package/src/main/__tests__/scheduler-worktree-exec-cwd.test.cjs +3 -5
- package/src/main/__tests__/usageSingleFlight.test.cjs +157 -0
- package/src/main/__tests__/workTypeLibrary.test.cjs +1 -1
- package/src/main/bilkoHost.cjs +16 -7
- package/src/main/build-info.json +8 -0
- package/src/main/chatRunner.cjs +4 -2
- package/src/main/config.cjs +2 -3
- package/src/main/docEdit.cjs +4 -2
- package/src/main/health.cjs +267 -51
- package/src/main/heapSnapshot.cjs +2 -2
- package/src/main/historyAggregator.cjs +1 -1
- package/src/main/index.cjs +61 -10
- package/src/main/ipcSchemas.cjs +6 -17
- package/src/main/lib/__tests__/auditLog.test.cjs +38 -0
- package/src/main/lib/__tests__/buildIdentity.test.cjs +121 -0
- package/src/main/lib/__tests__/cwdClassify.test.cjs +111 -0
- package/src/main/lib/__tests__/definitionOfDoneSequence.test.cjs +95 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +1 -1
- package/src/main/lib/__tests__/dispatchLoop.test.cjs +63 -0
- package/src/main/lib/__tests__/gateFixtures.json +20 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +8 -1
- package/src/main/lib/__tests__/instanceLock.test.cjs +93 -7
- package/src/main/lib/__tests__/jobSupervisorRecord.test.cjs +78 -0
- package/src/main/lib/__tests__/jobWorktreeBootLive.test.cjs +11 -0
- package/src/main/lib/__tests__/localAdminHttp.test.cjs +1 -1
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +6 -1
- package/src/main/lib/__tests__/procIdentity.test.cjs +119 -0
- package/src/main/lib/__tests__/procName.test.cjs +92 -0
- package/src/main/lib/__tests__/queueStoreMachineStateRecovery.test.cjs +67 -0
- package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +6 -7
- package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +37 -204
- package/src/main/lib/__tests__/schedulerPaths.test.cjs +226 -0
- package/src/main/lib/__tests__/schedulerPathsWorktree.test.cjs +94 -0
- package/src/main/lib/__tests__/schedulerRuntimeState.test.cjs +56 -0
- package/src/main/lib/__tests__/sessionSlots.test.cjs +45 -0
- package/src/main/lib/__tests__/telemetryBacklog.test.cjs +1 -1
- package/src/main/lib/__tests__/upgradeDrain.test.cjs +130 -0
- package/src/main/lib/__tests__/usageCircuit.test.cjs +61 -0
- package/src/main/lib/__tests__/watchdog-helpers.test.cjs +63 -0
- package/src/main/lib/__tests__/watchdog-relaunch.test.cjs +73 -0
- package/src/main/lib/activeIndexRebuild.cjs +1 -3
- package/src/main/lib/activeSessions.cjs +20 -88
- package/src/main/lib/adoptedRunSupervisor.cjs +136 -0
- package/src/main/lib/agentModelResolve.cjs +33 -1
- package/src/main/lib/agentPersonaSchema.cjs +2 -2
- package/src/main/lib/auditLog.cjs +30 -5
- package/src/main/lib/buildIdentity.cjs +113 -0
- package/src/main/lib/classifyPromptTicket.cjs +3 -2
- package/src/main/lib/claudeBin.cjs +37 -2
- package/src/main/lib/cleanEnv.cjs +27 -1
- package/src/main/lib/credentials.cjs +4 -2
- package/src/main/lib/cwdClassify.cjs +185 -0
- package/src/main/lib/definitionOfDone.cjs +296 -52
- package/src/main/lib/dispatchLoop.cjs +40 -0
- package/src/main/lib/effectiveModelInfo.cjs +9 -10
- package/src/main/lib/ephemeralCwd.cjs +7 -28
- package/src/main/lib/gitWorktree.cjs +95 -8
- package/src/main/lib/guardShims.cjs +3 -3
- package/src/main/lib/historyRollup.cjs +6 -7
- package/src/main/lib/instanceLock.cjs +32 -5
- package/src/main/lib/jobSupervisorRecord.cjs +147 -0
- package/src/main/lib/jobWorktreeBootLive.cjs +9 -5
- package/src/main/lib/localAdminHttp.cjs +9 -27
- package/src/main/lib/mcpToolCatalog.cjs +29 -77
- package/src/main/lib/opsOwnership.cjs +27 -19
- package/src/main/lib/prdAuthoringSeed.cjs +36 -0
- package/src/main/lib/prdLocations.cjs +48 -1
- package/src/main/lib/procIdentity.cjs +126 -0
- package/src/main/lib/procName.cjs +98 -0
- package/src/main/lib/projectHomeAdminRoutes.cjs +56 -327
- package/src/main/lib/queueHistory.cjs +8 -7
- package/src/main/lib/queueStore.cjs +57 -37
- package/src/main/lib/quietMachineLease.cjs +19 -2
- package/src/main/lib/reservationExpiry.cjs +30 -0
- package/src/main/lib/runClaudeP.cjs +4 -2
- package/src/main/lib/runLogRetention.cjs +3 -2
- package/src/main/lib/scheduleJobSchema.cjs +2 -2
- package/src/main/lib/scheduleJobTransitions.cjs +29 -2
- package/src/main/lib/schedulerBatch.cjs +27 -2
- package/src/main/lib/schedulerPaths.cjs +175 -0
- package/src/main/lib/schedulerRuntimeState.cjs +59 -0
- package/src/main/lib/sessionSlots.cjs +47 -9
- package/src/main/lib/smProcNames.cjs +51 -0
- package/src/main/lib/upgradeDrain.cjs +188 -0
- package/src/main/lib/usageCircuit.cjs +53 -6
- package/src/main/lib/watchdogHelpers.cjs +70 -37
- package/src/main/lib/withTimeout.cjs +33 -0
- package/src/main/mcpStatus.cjs +4 -2
- package/src/main/pluginInstall.cjs +6 -2
- package/src/main/projectPages.cjs +41 -145
- package/src/main/pty.cjs +8 -1
- package/src/main/queueOps.cjs +10 -10
- package/src/main/runVerify.cjs +69 -3
- package/src/main/scheduler.cjs +1598 -381
- package/src/main/seedAgentPersonas.cjs +1 -1
- package/src/main/seedDevPlugin.cjs +1 -1
- package/src/main/seedSchedulerMcp.cjs +7 -6
- package/src/main/seedStatus.cjs +1 -1
- package/src/main/supervisor.cjs +5 -3
- package/src/main/usage.cjs +126 -62
- package/src/preload/api.d.ts +18 -64
- package/src/preload/index.cjs +2 -4
- package/src/seed/agents/project-home-builder.md +31 -48
- package/screenshots/.gitkeep +0 -0
- package/screenshots/README-screenshots.md +0 -13
- package/src/main/lib/projectPageSummarySchema.cjs +0 -181
- package/src/main/teams.cjs +0 -95
- package/src/main/templates/project-pages-catalog.json +0 -741
- package/src/main/templates/project-pages-default-home.html +0 -123
- package/src/main/templates/project-pages-pipeline.md +0 -417
- package/src/seed/prompts/code-review/ac-coverage-check.md +0 -8
- package/src/seed/prompts/code-review/correctness-only.md +0 -8
- package/src/seed/prompts/code-review/full-spectrum-high.md +0 -8
- package/src/seed/prompts/code-review/hallucination-check.md +0 -8
- package/src/seed/prompts/code-review/public-api-compat.md +0 -8
- package/src/seed/prompts/code-review/readability-naming.md +0 -8
- package/src/seed/prompts/debugging/bug-as-failing-test.md +0 -8
- package/src/seed/prompts/debugging/git-bisect-regression.md +0 -8
- package/src/seed/prompts/debugging/instrument-intermittent-bug.md +0 -8
- package/src/seed/prompts/debugging/localize-pipeline-failure.md +0 -8
- package/src/seed/prompts/debugging/reproduce-then-diagnose.md +0 -8
- package/src/seed/prompts/documentation/adr-from-change.md +0 -8
- package/src/seed/prompts/documentation/module-readme.md +0 -8
- package/src/seed/prompts/documentation/onboarding-plan.md +0 -8
- package/src/seed/prompts/documentation/refresh-claude-md.md +0 -8
- package/src/seed/prompts/documentation/tsdoc-public-exports.md +0 -8
- package/src/seed/prompts/git-pr/conventional-commit.md +0 -8
- package/src/seed/prompts/git-pr/draft-pr-title-body.md +0 -8
- package/src/seed/prompts/git-pr/pre-commit-safety-sweep.md +0 -8
- package/src/seed/prompts/git-pr/release-notes-block.md +0 -8
- package/src/seed/prompts/git-pr/split-large-pr.md +0 -8
- package/src/seed/prompts/performance/bundle-startup-audit.md +0 -8
- package/src/seed/prompts/performance/complexity-audit.md +0 -8
- package/src/seed/prompts/performance/cpu-profile-hot-path.md +0 -8
- package/src/seed/prompts/performance/db-query-plan-review.md +0 -8
- package/src/seed/prompts/performance/memory-leak-hunt.md +0 -8
- package/src/seed/prompts/qa/api-contract-tests.md +0 -8
- package/src/seed/prompts/qa/e2e-critical-path.md +0 -8
- package/src/seed/prompts/qa/failing-test-for-bug.md +0 -8
- package/src/seed/prompts/qa/find-missing-test-coverage.md +0 -8
- package/src/seed/prompts/qa/stabilize-flaky-test.md +0 -8
- package/src/seed/prompts/qa/tdd-red-first.md +0 -8
- package/src/seed/prompts/qa/visual-regression-review.md +0 -8
- package/src/seed/prompts/qa/wcag-axe-scan.md +0 -8
- package/src/seed/prompts/refactoring/dead-code-sweep.md +0 -8
- package/src/seed/prompts/refactoring/extract-duplicated-pattern.md +0 -8
- package/src/seed/prompts/refactoring/modernize-legacy-file.md +0 -8
- package/src/seed/prompts/refactoring/reduce-cyclomatic-complexity.md +0 -8
- package/src/seed/prompts/refactoring/tighten-module-boundaries.md +0 -8
- package/src/seed/prompts/security/authz-audit.md +0 -8
- package/src/seed/prompts/security/crypto-correctness.md +0 -8
- package/src/seed/prompts/security/cwe-top-25-hunt.md +0 -8
- package/src/seed/prompts/security/dependency-audit.md +0 -8
- package/src/seed/prompts/security/ipc-boundary-hardening.md +0 -8
- package/src/seed/prompts/security/owasp-top-10-staged-diff.md +0 -8
- package/src/seed/prompts/security/secret-credential-scan.md +0 -8
- package/web/README.md +0 -41
- package/web/project-pages/logic/dist/logic.cjs +0 -4709
- package/web/project-pages/render.cjs +0 -70
- package/web/project-pages/renderer/dist/renderer.cjs +0 -18900
- package/web/project-pages/validate-summary.cjs +0 -62
package/src/main/scheduler.cjs
CHANGED
|
@@ -47,13 +47,15 @@ const fs = require('node:fs');
|
|
|
47
47
|
const fsp = require('node:fs/promises');
|
|
48
48
|
const path = require('node:path');
|
|
49
49
|
const os = require('node:os');
|
|
50
|
+
const { startDispatchLoop } = require('./lib/dispatchLoop.cjs');
|
|
51
|
+
const schedulerPaths = require('./lib/schedulerPaths.cjs');
|
|
50
52
|
const { randomUUID } = require('node:crypto');
|
|
51
53
|
const { execFile, execFileSync } = require('node:child_process');
|
|
52
54
|
const { ipcMain } = require('electron');
|
|
53
55
|
const billing = require('./usage.cjs');
|
|
54
56
|
const { cleanChildEnv, pathWithUserBins } = require('./lib/cleanEnv.cjs');
|
|
55
57
|
const supervisor = require('./supervisor.cjs');
|
|
56
|
-
const { resolveClaudeBin, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
58
|
+
const { resolveClaudeBin, claudeSpawnTarget, probeClaudeVersion } = require('./lib/claudeBin.cjs');
|
|
57
59
|
const launchFailure = require('./lib/launchFailure.cjs');
|
|
58
60
|
const { appendError } = require('./lib/opsErrorLog.cjs');
|
|
59
61
|
const { readTail } = require('./lib/fileTail.cjs');
|
|
@@ -67,6 +69,7 @@ const { sweepStrandedJobBranches } = require('./lib/branchSweep.cjs');
|
|
|
67
69
|
const { detectRateLimitInLog } = require('./lib/rateLimitDetect.cjs');
|
|
68
70
|
const { stripAppOwnedChurn } = require('./lib/jobDirtFilter.cjs');
|
|
69
71
|
const { resolveBindingRateLimitReset } = require('./lib/rateLimitWindow.cjs');
|
|
72
|
+
const { isResetFresh, bindingWindow, degradedBudget } = require('./lib/usageCircuit.cjs');
|
|
70
73
|
const { computeQueueHealth } = require('./lib/queueHealth.cjs');
|
|
71
74
|
const { createLoadGate, topCpuConsumers } = require('./lib/loadGate.cjs');
|
|
72
75
|
const { openLog, withChildAndLog } = require('./lib/childWithLog.cjs');
|
|
@@ -83,6 +86,7 @@ const { latestTerminalOutcomeForSlug, COMPLETED_EQUIVALENT_VERDICTS } = require(
|
|
|
83
86
|
const { isFixPlanSlug, classifyDiscoveredFixPlan, resolveIsFixPlan } = require('./lib/fixPlanSlug.cjs');
|
|
84
87
|
const { landedSinceRun, landedOnMainSince } = require('./lib/landedSinceRun.cjs');
|
|
85
88
|
const { declaredPathsForPrd } = require('./lib/prdDeclaredPaths.cjs');
|
|
89
|
+
const { identity: procIdentityOf, isDifferentProcess } = require('./lib/procIdentity.cjs');
|
|
86
90
|
const logs = require('./logs.cjs');
|
|
87
91
|
const { schemas, validated, SCHEDULE_SLUG_RE } = require('./ipcSchemas.cjs');
|
|
88
92
|
const { readBody, sendJson } = require('./lib/localAdminHttp.cjs');
|
|
@@ -131,19 +135,21 @@ const { pickForProject, pickNextBatch, findStarvedProjects, DEFAULT_PROJECT_CWD,
|
|
|
131
135
|
const { runDefinitionOfDoneOnDrain } = require('./lib/dodDrainHook.cjs');
|
|
132
136
|
const { writeRcaReport, extractRcaBlock } = require('./lib/rcaReport.cjs');
|
|
133
137
|
const queueHistory = require('./lib/queueHistory.cjs');
|
|
138
|
+
const { resolveGate, runGateSequence } = require('./lib/definitionOfDone.cjs');
|
|
134
139
|
const queueOps = require('./queueOps.cjs');
|
|
135
140
|
// Feedback-auto-PRD sweep — formerly only run by the external scheduler-watchdog
|
|
136
141
|
// while the app was down (PRD 686 moved it in-app so it also runs while alive).
|
|
137
142
|
// Plain Node module, no Electron dependency; queuePath/prdsDir defaults already
|
|
138
143
|
// match ROOT/QUEUE_PATH below since both resolve the same ~/.claude/session-manager
|
|
139
144
|
// home-dir layout.
|
|
140
|
-
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs } = require('./lib/prdLocations.cjs');
|
|
145
|
+
const { resolvePrdsDirs, resolveArchivedPrdsDirs, resolvePrdWriteDir, listEpicPrdDirs, listArchivedPrdDirs, deriveProjectCwdFromPrdPath } = require('./lib/prdLocations.cjs');
|
|
141
146
|
const { ensureEpic, appendPrdCreatedEvent, readActiveIndex } = require('./lib/epicMint.cjs');
|
|
142
147
|
const agentModelResolve = require('./lib/agentModelResolve.cjs');
|
|
143
148
|
const { transitionJob, STATUS_HISTORY_CAP, LEGAL_TRANSITIONS } = require('./lib/scheduleJobTransitions.cjs');
|
|
144
149
|
const { buildContextDigest, composeExecutorPrompt } = require('./lib/epicContextDigest.cjs');
|
|
145
150
|
const { JOB_STATUSES } = require('./lib/scheduleJobSchema.cjs');
|
|
146
151
|
const { appendAuditEvent } = require('./lib/auditLog.cjs');
|
|
152
|
+
const { withTimeout } = require('./lib/withTimeout.cjs');
|
|
147
153
|
|
|
148
154
|
// ---------- origin session resolution (PRD 832) ----------
|
|
149
155
|
// An Epic IS a tagged claude session — job rows carry the originating
|
|
@@ -164,12 +170,15 @@ function resolveOriginSessionId(cwd, epicId) {
|
|
|
164
170
|
}
|
|
165
171
|
const sessionSlots = require('./lib/sessionSlots.cjs');
|
|
166
172
|
const quietMachineLease = require('./lib/quietMachineLease.cjs');
|
|
173
|
+
const runtimeState = require('./lib/schedulerRuntimeState.cjs');
|
|
167
174
|
const jobWorktree = require('./lib/jobWorktree.cjs');
|
|
168
175
|
const gitWorktree = require('./lib/gitWorktree.cjs');
|
|
169
176
|
const { buildJobWorktreeIsLive } = require('./lib/jobWorktreeBootLive.cjs');
|
|
170
177
|
const { buildTerminalOrphanIsLive } = require('./lib/jobWorktreeTerminalOrphanLive.cjs');
|
|
171
178
|
const { reconcileEpicWorktreesOnBoot } = require('./lib/epicWorktreeBoot.cjs');
|
|
172
179
|
const queueStore = require('./lib/queueStore.cjs');
|
|
180
|
+
const supervisorRecord = require('./lib/jobSupervisorRecord.cjs');
|
|
181
|
+
const adoptedRunSupervisor = require('./lib/adoptedRunSupervisor.cjs');
|
|
173
182
|
const { splitFrontmatter, parsePrdFile, serializePrdFile } = require('./lib/prdFrontmatter.cjs');
|
|
174
183
|
const { resolveDepSlug, findNearMatches } = require('./lib/depSlugResolve.cjs');
|
|
175
184
|
const { computeDispositionRewrite } = require('./lib/prdDisposition.cjs');
|
|
@@ -182,20 +191,26 @@ const { allProjectCwds } = require('./lib/activeSessions.cjs');
|
|
|
182
191
|
// an exemption it should have applied landed on disk, and nothing in the
|
|
183
192
|
// run record showed that; this is the fix).
|
|
184
193
|
const SCHEDULER_BOOTED_AT = new Date().toISOString();
|
|
185
|
-
//
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
194
|
+
// A production npx install ships no .git at all, so a runtime `git
|
|
195
|
+
// rev-parse` from __dirname was structurally always null there — every
|
|
196
|
+
// production run-meta sidecar recorded schedulerCodeSha: null. buildIdentity
|
|
197
|
+
// resolves build-info.json (baked at publish time) first, falling back to a
|
|
198
|
+
// non-walking git read only in a dev checkout / job worktree — see
|
|
199
|
+
// src/main/lib/buildIdentity.cjs's header.
|
|
200
|
+
const { resolveBuildIdentity, readInstalledBuildInfo } = require('./lib/buildIdentity.cjs');
|
|
201
|
+
const upgradeDrain = require('./lib/upgradeDrain.cjs');
|
|
202
|
+
const SCHEDULER_BUILD_IDENTITY = resolveBuildIdentity({ bootedAt: SCHEDULER_BOOTED_AT });
|
|
203
|
+
const SCHEDULER_CODE_SHA = SCHEDULER_BUILD_IDENTITY.codeSha;
|
|
204
|
+
// Spread into EVERY metaPath writer below (grep `metaPath` for the full
|
|
205
|
+
// list) — single source so a future field never lands in some sidecars and
|
|
206
|
+
// not others, the exact gap that left 3 of 5 writers silently missing
|
|
207
|
+
// schedulerBootedAt/schedulerCodeSha before this constant existed.
|
|
208
|
+
const SCHEDULER_META_IDENTITY = {
|
|
209
|
+
schedulerBootedAt: SCHEDULER_BOOTED_AT,
|
|
210
|
+
schedulerCodeSha: SCHEDULER_CODE_SHA,
|
|
211
|
+
schedulerVersion: SCHEDULER_BUILD_IDENTITY.version,
|
|
212
|
+
schedulerBuiltAt: SCHEDULER_BUILD_IDENTITY.builtAt,
|
|
213
|
+
};
|
|
199
214
|
|
|
200
215
|
const MAX_INVESTIGATION_DURATION_MS = 30 * 60_000;
|
|
201
216
|
|
|
@@ -590,7 +605,7 @@ function evaluateSharedTreeGuard({ stashBefore, stashAfter, dirtyBefore, dirtyAf
|
|
|
590
605
|
// executor-created stash (never guesses when there are 2+); reports anything
|
|
591
606
|
// it can't safely resolve on the returned object so the caller can surface it
|
|
592
607
|
// on the job row instead of finishing silently green.
|
|
593
|
-
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug }) {
|
|
608
|
+
async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBefore, slug, landedCommit }) {
|
|
594
609
|
try {
|
|
595
610
|
const [stashAfter, headAfter] = await Promise.all([
|
|
596
611
|
module.exports.stashList(cwd),
|
|
@@ -655,7 +670,16 @@ async function checkSharedTreeGuard({ cwd, stashBaseline, dirtyBaseline, headBef
|
|
|
655
670
|
pathsCommittedDuringRun,
|
|
656
671
|
existsAfter,
|
|
657
672
|
});
|
|
658
|
-
|
|
673
|
+
// Ground truth outranks the baseline diff (2026-09-18, 1229-fo-03): the
|
|
674
|
+
// dirty baseline is invalidated by ANY later writer (a human commit that
|
|
675
|
+
// sweeps the same paths), so it can't prove a revert on its own. The
|
|
676
|
+
// job's own landedCommit still being an ancestor of HEAD proves its work
|
|
677
|
+
// was not discarded — anchored to that sha, not to the baseline.
|
|
678
|
+
const workSurvives = reverted.length > 0
|
|
679
|
+
&& await module.exports.landedCommitIsAncestorOfHead(cwd, landedCommit);
|
|
680
|
+
if (workSurvives) {
|
|
681
|
+
console.log(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} baseline path(s) went clean but landed commit ${String(landedCommit).slice(0, 7)} is still an ancestor of HEAD — not a revert`);
|
|
682
|
+
} else if (reverted.length) {
|
|
659
683
|
result.reverted = reverted;
|
|
660
684
|
console.error(`[scheduler] ${slug}: shared-tree guard: ${reverted.length} path(s) reverted in the shared tree with no commit to explain it (${reverted.slice(0, 3).join(', ')})`);
|
|
661
685
|
}
|
|
@@ -895,13 +919,9 @@ function isQueueRowRegression({ statusBefore, statusAfter, historyLenBefore, his
|
|
|
895
919
|
return statusBefore === 'running' && statusAfter === 'pending' && historyLenAfter < historyLenBefore;
|
|
896
920
|
}
|
|
897
921
|
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
const PRDS_ARCHIVE_DIR = path.join(ROOT, 'prds-archived');
|
|
902
|
-
const QUEUE_PATH = path.join(ROOT, 'queue.json');
|
|
903
|
-
const SCHEDULER_STATE_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-state.json');
|
|
904
|
-
const HEARTBEAT_PATH = path.join(os.homedir(), '.claude', 'session-manager', 'scheduler-heartbeat.log');
|
|
922
|
+
// Machine-wide roots resolve lazily via lib/schedulerPaths.cjs (SM_SCHEDULER_HOME
|
|
923
|
+
// override) — never module-scope consts. The ROOT/PRDS_DIR/RUNS_DIR/
|
|
924
|
+
// SCHEDULER_STATE_PATH exports below are lazy getters over the same resolver.
|
|
905
925
|
const HEARTBEAT_MAX_BYTES = 1024 * 1024;
|
|
906
926
|
// DEFAULT_PROJECT_CWD imported from lib/schedulerBatch.cjs (single source of truth).
|
|
907
927
|
|
|
@@ -1020,7 +1040,7 @@ function biasJobOomScore(pid) {
|
|
|
1020
1040
|
* (reconcile, list-prds, lint, rescan).
|
|
1021
1041
|
*/
|
|
1022
1042
|
function candidatePrdsDirs() {
|
|
1023
|
-
return [
|
|
1043
|
+
return [schedulerPaths.prdsRoot(), ...resolvePrdsDirs()];
|
|
1024
1044
|
}
|
|
1025
1045
|
|
|
1026
1046
|
/**
|
|
@@ -1141,7 +1161,7 @@ function prdArchivedSkipResult(job, cwd, sessionId, startedAt, safeLog, closeFd,
|
|
|
1141
1161
|
const finishedAt = Date.now();
|
|
1142
1162
|
config.writeJsonSync(metaPath, {
|
|
1143
1163
|
slug: job.slug, cwd, sessionId, exitCode: 0, skipped: reason,
|
|
1144
|
-
note: msg, startedAt, finishedAt, durationMs: 0,
|
|
1164
|
+
note: msg, startedAt, finishedAt, durationMs: 0, ...SCHEDULER_META_IDENTITY,
|
|
1145
1165
|
});
|
|
1146
1166
|
return { exitCode: 0, durationMs: 0, skipped: reason, note: msg, sessionId };
|
|
1147
1167
|
}
|
|
@@ -1283,17 +1303,21 @@ async function retireCompletedSlugs(slugs) {
|
|
|
1283
1303
|
// Bundled authoring guide seeded into the scheduler dir so the session-manager-dev
|
|
1284
1304
|
// plugin's /develop and /prd skills — which reference this stable `~`-absolute
|
|
1285
1305
|
// path — work on any user's machine, not just the author's.
|
|
1306
|
+
// Line 1 of the template is `<!-- PRD_AUTHORING.md vN -->`: bump vN whenever the
|
|
1307
|
+
// template changes, or existing installs never receive the update.
|
|
1286
1308
|
const PRD_AUTHORING_TEMPLATE = path.join(__dirname, 'templates', 'PRD_AUTHORING.md');
|
|
1287
|
-
const PRD_AUTHORING_DEST = path.join(ROOT, 'PRD_AUTHORING.md');
|
|
1288
1309
|
|
|
1289
1310
|
function ensureDirs() {
|
|
1290
|
-
fs.mkdirSync(
|
|
1291
|
-
fs.mkdirSync(
|
|
1292
|
-
//
|
|
1311
|
+
fs.mkdirSync(schedulerPaths.prdsRoot(), { recursive: true });
|
|
1312
|
+
fs.mkdirSync(schedulerPaths.runsDir(), { recursive: true });
|
|
1313
|
+
// Re-seed the guide whenever the bundled template's version stamp differs.
|
|
1293
1314
|
try {
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1315
|
+
const authoringDest = path.join(schedulerPaths.scheduledPlansRoot(), 'PRD_AUTHORING.md');
|
|
1316
|
+
seedAuthoringGuide({
|
|
1317
|
+
src: PRD_AUTHORING_TEMPLATE,
|
|
1318
|
+
dest: authoringDest,
|
|
1319
|
+
write: (abs, text) => config.writeTextAtomic(abs, text, { writer: 'scheduler' }),
|
|
1320
|
+
}).catch(() => { /* non-fatal, same as below */ });
|
|
1297
1321
|
} catch { /* non-fatal: the guide is a convenience, not load-bearing for a run */ }
|
|
1298
1322
|
}
|
|
1299
1323
|
|
|
@@ -1321,8 +1345,9 @@ function ensureDirs() {
|
|
|
1321
1345
|
* queue row yet at that point, so it is never in LIVE_JOB_STATUSES and this
|
|
1322
1346
|
* sweep archives it before reconcile can ever turn it into a pending job.
|
|
1323
1347
|
*/
|
|
1324
|
-
async function consolidateAllFlatPrds(cwds) {
|
|
1348
|
+
async function consolidateAllFlatPrds(cwds, skipCwds) {
|
|
1325
1349
|
for (const cwd of cwds) {
|
|
1350
|
+
if (skipCwds?.has(cwd)) continue; // torn shard: its PRDs are not ours to touch this pass
|
|
1326
1351
|
try {
|
|
1327
1352
|
const c = await consolidateFlatPrds(cwd);
|
|
1328
1353
|
if (c.moved > 0) {
|
|
@@ -1354,7 +1379,7 @@ async function consolidateAllFlatPrds(cwds) {
|
|
|
1354
1379
|
async function runPrdMigration() {
|
|
1355
1380
|
let result;
|
|
1356
1381
|
try {
|
|
1357
|
-
result = await migratePrds(
|
|
1382
|
+
result = await migratePrds(schedulerPaths.prdsRoot());
|
|
1358
1383
|
} catch (e) {
|
|
1359
1384
|
logs.writeLine({ level: 'error', scope: 'scheduler', message: 'PRD migration failed', meta: { error: e?.message } });
|
|
1360
1385
|
return null;
|
|
@@ -1365,7 +1390,7 @@ async function runPrdMigration() {
|
|
|
1365
1390
|
level: 'warn',
|
|
1366
1391
|
scope: 'scheduler',
|
|
1367
1392
|
message: `PRD migration: ${result.unresolved.length} file(s) left in legacy dir`,
|
|
1368
|
-
meta: { legacyDir:
|
|
1393
|
+
meta: { legacyDir: schedulerPaths.prdsRoot(), unresolved: result.unresolved },
|
|
1369
1394
|
});
|
|
1370
1395
|
for (const u of result.unresolved) {
|
|
1371
1396
|
console.warn(`[scheduler] PRD migration: left ${u.file} in legacy dir (${u.reason})`);
|
|
@@ -1432,7 +1457,7 @@ const QUEUE_BAK_KEEP = 5;
|
|
|
1432
1457
|
async function sweepQueueBackups() {
|
|
1433
1458
|
let entries;
|
|
1434
1459
|
try {
|
|
1435
|
-
entries = await fsp.readdir(
|
|
1460
|
+
entries = await fsp.readdir(schedulerPaths.scheduledPlansRoot());
|
|
1436
1461
|
} catch {
|
|
1437
1462
|
return;
|
|
1438
1463
|
}
|
|
@@ -1448,7 +1473,7 @@ async function sweepQueueBackups() {
|
|
|
1448
1473
|
let removed = 0;
|
|
1449
1474
|
for (const f of toDelete) {
|
|
1450
1475
|
try {
|
|
1451
|
-
await fsp.unlink(path.join(
|
|
1476
|
+
await fsp.unlink(path.join(schedulerPaths.scheduledPlansRoot(), f));
|
|
1452
1477
|
removed++;
|
|
1453
1478
|
} catch (e) {
|
|
1454
1479
|
console.warn('[scheduler] backup sweep: unlink failed', f, e?.message);
|
|
@@ -1464,20 +1489,24 @@ async function sweepQueueBackups() {
|
|
|
1464
1489
|
// callback that must flush meta.json before resolving) — replacing with async
|
|
1465
1490
|
// would deadlock the exit path.
|
|
1466
1491
|
const config = require('./config.cjs');
|
|
1492
|
+
const { seedAuthoringGuide } = require('./lib/prdAuthoringSeed.cjs');
|
|
1467
1493
|
const atomicWriteJsonSync = (p, data) => config.writeJsonSync(p, data);
|
|
1468
1494
|
|
|
1469
1495
|
// ---------- scheduler-state.json (sidecar) ----------
|
|
1470
1496
|
|
|
1471
1497
|
function loadSchedulerState() {
|
|
1472
1498
|
try {
|
|
1473
|
-
const raw = fs.readFileSync(
|
|
1499
|
+
const raw = fs.readFileSync(schedulerPaths.schedulerStatePath(), 'utf8');
|
|
1474
1500
|
const s = JSON.parse(raw);
|
|
1475
1501
|
if (s.lastObservedReset) cachedNextReset = s.lastObservedReset;
|
|
1502
|
+
if (typeof s.lastResetObservedAt === 'number') lastResetObservedAtMs = s.lastResetObservedAt;
|
|
1476
1503
|
if (typeof s.consecutiveFailures === 'number') consecutiveFailures = s.consecutiveFailures;
|
|
1477
1504
|
if (typeof s.backoffMs === 'number') backoffMs = s.backoffMs;
|
|
1478
1505
|
if (typeof s.pauseClearedManuallyAt === 'number') pauseClearedManuallyAt = s.pauseClearedManuallyAt;
|
|
1479
1506
|
if (typeof s.lastPollAt === 'number') lastPollAt = s.lastPollAt;
|
|
1480
1507
|
if (typeof s.failureStreakWarned === 'boolean') failureStreakWarned = s.failureStreakWarned;
|
|
1508
|
+
if (typeof s.failureStreakWarnedAt === 'number') failureStreakWarnedAt = s.failureStreakWarnedAt;
|
|
1509
|
+
if (typeof s.lastEscalationAt === 'number') lastEscalationAtMs = s.lastEscalationAt;
|
|
1481
1510
|
} catch { /* first boot or corrupt — start fresh */ }
|
|
1482
1511
|
}
|
|
1483
1512
|
|
|
@@ -1487,10 +1516,14 @@ function persistSchedulerState() {
|
|
|
1487
1516
|
// require threading awaits through pause/resume bookkeeping for negligible
|
|
1488
1517
|
// benefit — the file is well under one page.
|
|
1489
1518
|
try {
|
|
1490
|
-
config.writeJsonSync(
|
|
1519
|
+
config.writeJsonSync(schedulerPaths.schedulerStatePath(), {
|
|
1491
1520
|
version: 1,
|
|
1492
1521
|
lastObservedReset: cachedNextReset,
|
|
1493
|
-
|
|
1522
|
+
// Only stamped at the moment a FRESH reset was actually observed (see
|
|
1523
|
+
// recordObservedReset) — never Date.now() on every persist call, which
|
|
1524
|
+
// used to make a stale cachedNextReset look freshly-confirmed on every
|
|
1525
|
+
// tick even when nothing new had been read.
|
|
1526
|
+
lastResetObservedAt: lastResetObservedAtMs,
|
|
1494
1527
|
lastPollAt,
|
|
1495
1528
|
consecutiveFailures,
|
|
1496
1529
|
backoffMs,
|
|
@@ -1498,6 +1531,13 @@ function persistSchedulerState() {
|
|
|
1498
1531
|
pausedSince: null,
|
|
1499
1532
|
pauseClearedManuallyAt,
|
|
1500
1533
|
failureStreakWarned,
|
|
1534
|
+
failureStreakWarnedAt,
|
|
1535
|
+
lastEscalationAt: lastEscalationAtMs,
|
|
1536
|
+
// Circuit fields are read fresh from the live shared breaker each
|
|
1537
|
+
// persist — health.cjs (a separate `npm run health` process) reads
|
|
1538
|
+
// THESE persisted values, since it never holds the in-memory circuit.
|
|
1539
|
+
usageCircuitState: billing.usageCircuit.state(),
|
|
1540
|
+
usageCircuitOpenedAt: billing.usageCircuit.openedAt(),
|
|
1501
1541
|
});
|
|
1502
1542
|
} catch (e) {
|
|
1503
1543
|
console.warn('[scheduler] failed to persist scheduler state', e?.message);
|
|
@@ -1510,18 +1550,279 @@ function appendHeartbeat(entry) {
|
|
|
1510
1550
|
try {
|
|
1511
1551
|
const line = JSON.stringify(entry) + '\n';
|
|
1512
1552
|
let size = 0;
|
|
1513
|
-
try { size = fs.statSync(
|
|
1553
|
+
try { size = fs.statSync(schedulerPaths.heartbeatPath()).size; } catch { /* new file */ }
|
|
1514
1554
|
if (size >= HEARTBEAT_MAX_BYTES) {
|
|
1515
|
-
const rotated =
|
|
1555
|
+
const rotated = schedulerPaths.heartbeatPath() + '.1';
|
|
1516
1556
|
try { fs.unlinkSync(rotated); } catch { /* */ }
|
|
1517
|
-
try { fs.renameSync(
|
|
1557
|
+
try { fs.renameSync(schedulerPaths.heartbeatPath(), rotated); } catch { /* */ }
|
|
1518
1558
|
}
|
|
1519
|
-
fs.appendFileSync(
|
|
1559
|
+
fs.appendFileSync(schedulerPaths.heartbeatPath(), line);
|
|
1520
1560
|
} catch (e) {
|
|
1521
1561
|
console.warn('[scheduler] heartbeat write failed', e?.message);
|
|
1522
1562
|
}
|
|
1523
1563
|
}
|
|
1524
1564
|
|
|
1565
|
+
// Build identity stamped on every heartbeat line — memoized at boot
|
|
1566
|
+
// (SCHEDULER_BUILD_IDENTITY), so a tick costs no git or fs work.
|
|
1567
|
+
function heartbeatBuild() {
|
|
1568
|
+
const { version, codeSha, builtAt } = SCHEDULER_BUILD_IDENTITY;
|
|
1569
|
+
return { version, codeSha, builtAt };
|
|
1570
|
+
}
|
|
1571
|
+
|
|
1572
|
+
// ---------- upgrade drain driver (lib/upgradeDrain.cjs) ----------
|
|
1573
|
+
|
|
1574
|
+
// Set by index.cjs: performs the actual app teardown + relaunch + exit.
|
|
1575
|
+
let restartHandler = null;
|
|
1576
|
+
function setRestartHandler(fn) { restartHandler = typeof fn === 'function' ? fn : null; }
|
|
1577
|
+
let drainDriving = false;
|
|
1578
|
+
|
|
1579
|
+
function drainSnapshot(jobs) {
|
|
1580
|
+
const running = new Set();
|
|
1581
|
+
let investigating = 0;
|
|
1582
|
+
for (const j of jobs ?? []) {
|
|
1583
|
+
if (j?.status === 'running') running.add(j.slug);
|
|
1584
|
+
else if (j?.status === 'investigating') investigating++;
|
|
1585
|
+
}
|
|
1586
|
+
for (const slug of runningSet) running.add(slug);
|
|
1587
|
+
// Deferred investigations are not busy: while draining they never spawn.
|
|
1588
|
+
return { running: running.size, investigating: Math.max(investigating, runtimeState.investigationCount()) };
|
|
1589
|
+
}
|
|
1590
|
+
|
|
1591
|
+
/**
|
|
1592
|
+
* Restart is triggered automatically ONLY when the installed build-info.json
|
|
1593
|
+
* differs from the running codeSha (an install/update already happened) —
|
|
1594
|
+
* never by polling npm. SM_AUTO_UPGRADE_RESTART=0 disables it.
|
|
1595
|
+
*/
|
|
1596
|
+
function maybeAutoRequestRestart() {
|
|
1597
|
+
if (process.env.SM_AUTO_UPGRADE_RESTART === '0' || process.env.SM_DEV === '1') return null;
|
|
1598
|
+
const installed = readInstalledBuildInfo();
|
|
1599
|
+
const installedSha = typeof installed?.gitShortSha === 'string' ? installed.gitShortSha : null;
|
|
1600
|
+
if (!upgradeDrain.installedBuildDiffers({ running: SCHEDULER_CODE_SHA, installed: installedSha })) return null;
|
|
1601
|
+
return upgradeDrain.requestRestart({ reason: `installed build ${installedSha} differs from running ${SCHEDULER_CODE_SHA}`, requestedBy: 'auto-upgrade' });
|
|
1602
|
+
}
|
|
1603
|
+
|
|
1604
|
+
async function driveUpgradeDrain(state) {
|
|
1605
|
+
if (drainDriving) return;
|
|
1606
|
+
drainDriving = true;
|
|
1607
|
+
try {
|
|
1608
|
+
let request = upgradeDrain.readRestartRequest();
|
|
1609
|
+
if (!request && !state.drain?.active) request = maybeAutoRequestRestart();
|
|
1610
|
+
const { action, reason } = upgradeDrain.evaluateDrain({
|
|
1611
|
+
request,
|
|
1612
|
+
queueSnapshot: drainSnapshot(state.jobs),
|
|
1613
|
+
drainState: state.drain,
|
|
1614
|
+
now: Date.now(),
|
|
1615
|
+
});
|
|
1616
|
+
if (action === 'none' || action === 'wait') { drainActive = Boolean(state.drain?.active); return; }
|
|
1617
|
+
if (action === 'pause') {
|
|
1618
|
+
await mutate((s) => { s.drain = { active: true, since: new Date().toISOString(), requestedAt: request.requestedAt }; });
|
|
1619
|
+
drainActive = true;
|
|
1620
|
+
appendAuditEvent('upgrade_drain_started', { reason: request.reason, requestedBy: request.requestedBy });
|
|
1621
|
+
await broadcast({ flush: true });
|
|
1622
|
+
return;
|
|
1623
|
+
}
|
|
1624
|
+
if (action === 'abort') {
|
|
1625
|
+
upgradeDrain.retireRestartRequest();
|
|
1626
|
+
await mutate((s) => { s.drain = null; });
|
|
1627
|
+
drainActive = false;
|
|
1628
|
+
appendAuditEvent('upgrade_drain_aborted', { reason });
|
|
1629
|
+
await broadcast({ flush: true });
|
|
1630
|
+
runDueJobs().catch(() => {});
|
|
1631
|
+
return;
|
|
1632
|
+
}
|
|
1633
|
+
// 'restart': the FINAL zero-busy check runs inside a mutate, immediately
|
|
1634
|
+
// before exit — the snapshot above may be stale by now.
|
|
1635
|
+
let go = false;
|
|
1636
|
+
await mutate((s) => {
|
|
1637
|
+
const snap = drainSnapshot(s.jobs);
|
|
1638
|
+
if (!s.drain?.active || snap.running + snap.investigating > 0) return;
|
|
1639
|
+
upgradeDrain.stampDrainCompleted();
|
|
1640
|
+
go = true;
|
|
1641
|
+
});
|
|
1642
|
+
if (!go) return;
|
|
1643
|
+
appendAuditEvent('upgrade_drain_restart', { reason: request.reason, requestedBy: request.requestedBy });
|
|
1644
|
+
try {
|
|
1645
|
+
if (!restartHandler) throw new Error('no restart handler registered');
|
|
1646
|
+
upgradeDrain.markRestarting();
|
|
1647
|
+
await restartHandler(request);
|
|
1648
|
+
// Only reached when the handler did NOT exit the process (dev-server
|
|
1649
|
+
// in-place reboot): the restart is done, so retire the drain here.
|
|
1650
|
+
upgradeDrain.clearRestartingMarker();
|
|
1651
|
+
upgradeDrain.retireRestartRequest();
|
|
1652
|
+
await mutate((s) => { s.drain = null; });
|
|
1653
|
+
drainActive = false;
|
|
1654
|
+
} catch (e) {
|
|
1655
|
+
// Never strand the queue drained-and-paused: fall back to an abort.
|
|
1656
|
+
console.error('[scheduler] drain restart failed — aborting drain:', e?.message ?? e);
|
|
1657
|
+
upgradeDrain.clearRestartingMarker();
|
|
1658
|
+
upgradeDrain.retireRestartRequest();
|
|
1659
|
+
await mutate((s) => { s.drain = null; });
|
|
1660
|
+
drainActive = false;
|
|
1661
|
+
runDueJobs().catch(() => {});
|
|
1662
|
+
}
|
|
1663
|
+
} finally {
|
|
1664
|
+
drainDriving = false;
|
|
1665
|
+
}
|
|
1666
|
+
}
|
|
1667
|
+
|
|
1668
|
+
/** Boot: clear a leftover drain whose request is complete (the restart happened) or gone. */
|
|
1669
|
+
async function clearStaleDrainAtBoot(boot) {
|
|
1670
|
+
const request = upgradeDrain.readRestartRequest();
|
|
1671
|
+
const action = upgradeDrain.bootDrainAction({ drainState: boot.drain, request });
|
|
1672
|
+
upgradeDrain.clearRestartingMarker();
|
|
1673
|
+
if (request?.drainCompletedAt) upgradeDrain.retireRestartRequest();
|
|
1674
|
+
if (action === 'clear') await mutate((s) => { s.drain = null; });
|
|
1675
|
+
drainActive = action === 'keep';
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
/**
|
|
1679
|
+
* heartbeatTick(deps?) — one 60 s heartbeat interval body. Each subsystem
|
|
1680
|
+
* (queue read + starvation watchdog, stall detector, heartbeat write) runs in
|
|
1681
|
+
* its own try/catch so one throw can't silently skip the others. Any failure
|
|
1682
|
+
* makes the written line `degraded: true` + `errors`; watchdogHelpers'
|
|
1683
|
+
* heartbeatFresh() and health.cjs's readFreshHeartbeat() treat such a line as
|
|
1684
|
+
* NOT fresh, so a throw never disarms the external watchdog or fakes
|
|
1685
|
+
* utilization health never read.
|
|
1686
|
+
*/
|
|
1687
|
+
function heartbeatTick(deps = {}) {
|
|
1688
|
+
const readQueue = deps.readQueueSync ?? readQueueSync;
|
|
1689
|
+
const errors = [];
|
|
1690
|
+
const guard = (subsystem, fn) => {
|
|
1691
|
+
try {
|
|
1692
|
+
return fn();
|
|
1693
|
+
} catch (e) {
|
|
1694
|
+
errors.push({ subsystem, error: e?.message ?? String(e) });
|
|
1695
|
+
console.error(`[scheduler] heartbeat subsystem "${subsystem}" failed`, e);
|
|
1696
|
+
return undefined;
|
|
1697
|
+
}
|
|
1698
|
+
};
|
|
1699
|
+
|
|
1700
|
+
const s = guard('queue-read-starvation-watchdog', () => {
|
|
1701
|
+
const q = readQueue();
|
|
1702
|
+
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
1703
|
+
// running, something must drive it. This is the only driver that does
|
|
1704
|
+
// not depend on the billing poll loop, a pause timer, or a completing
|
|
1705
|
+
// job to schedule the next tick — every one of which has failed at
|
|
1706
|
+
// least once. See classifyQueueStarvation.
|
|
1707
|
+
if (!q.unreadable) {
|
|
1708
|
+
runQueueStarvationWatchdog(q).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
1709
|
+
// Restart-request drain state machine rides this same 60 s interval —
|
|
1710
|
+
// no new driver.
|
|
1711
|
+
driveUpgradeDrain(q).catch((e) => console.error('[scheduler] upgrade drain error', e));
|
|
1712
|
+
}
|
|
1713
|
+
return q;
|
|
1714
|
+
});
|
|
1715
|
+
|
|
1716
|
+
let stall;
|
|
1717
|
+
if (s) {
|
|
1718
|
+
stall = guard('stall-detector', () => {
|
|
1719
|
+
const summary = computeStallSummary(s);
|
|
1720
|
+
// Per-project alerting (see computeStallSummary's header): a project
|
|
1721
|
+
// stalled while others are busy must still fire, and one project
|
|
1722
|
+
// recovering must not clear or suppress another's still-open episode —
|
|
1723
|
+
// that is exactly what a single module-level stallSince/stallToasted
|
|
1724
|
+
// flag masked before (the burrow-vs-others incident this PRD fixes).
|
|
1725
|
+
const now = Date.now();
|
|
1726
|
+
const stalledCwds = Object.keys(summary.byProject).filter((cwd) => summary.byProject[cwd].stalled);
|
|
1727
|
+
for (const cwd of [...stallSince.keys()]) {
|
|
1728
|
+
if (!stalledCwds.includes(cwd)) {
|
|
1729
|
+
stallSince.delete(cwd);
|
|
1730
|
+
stallToasted.delete(cwd);
|
|
1731
|
+
}
|
|
1732
|
+
}
|
|
1733
|
+
const toAlert = [];
|
|
1734
|
+
for (const cwd of stalledCwds) {
|
|
1735
|
+
if (!stallSince.has(cwd)) stallSince.set(cwd, now);
|
|
1736
|
+
if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
|
|
1737
|
+
stallToasted.set(cwd, true);
|
|
1738
|
+
toAlert.push(cwd);
|
|
1739
|
+
}
|
|
1740
|
+
}
|
|
1741
|
+
if (toAlert.length > 0) {
|
|
1742
|
+
console.error(
|
|
1743
|
+
`[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
|
|
1744
|
+
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
1745
|
+
summary.byProject,
|
|
1746
|
+
);
|
|
1747
|
+
appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: summary.total, byProject: summary.byProject });
|
|
1748
|
+
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
1749
|
+
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
1750
|
+
message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
|
|
1751
|
+
projects: toAlert,
|
|
1752
|
+
total: summary.total,
|
|
1753
|
+
byProject: summary.byProject,
|
|
1754
|
+
});
|
|
1755
|
+
}
|
|
1756
|
+
}
|
|
1757
|
+
return summary;
|
|
1758
|
+
});
|
|
1759
|
+
}
|
|
1760
|
+
|
|
1761
|
+
let entry = null;
|
|
1762
|
+
if (s && stall && errors.length === 0) {
|
|
1763
|
+
entry = guard('heartbeat-write', () => {
|
|
1764
|
+
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
1765
|
+
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
1766
|
+
// failed }` literal silently minted a NEW key for any other value, which
|
|
1767
|
+
// is how a heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
1768
|
+
// visibility instead of the alarm it should have been. Any row whose
|
|
1769
|
+
// status isn't in JOB_STATUSES routes into `unknown`, never a
|
|
1770
|
+
// freshly-minted key.
|
|
1771
|
+
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
1772
|
+
counts.unknown = 0;
|
|
1773
|
+
for (const j of s.jobs) {
|
|
1774
|
+
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
1775
|
+
counts[j.status] += 1;
|
|
1776
|
+
} else {
|
|
1777
|
+
counts.unknown += 1;
|
|
1778
|
+
}
|
|
1779
|
+
}
|
|
1780
|
+
// Logical-liveness signal for the external watchdog (see watchdogHelpers
|
|
1781
|
+
// evaluateDispatchLiveness): computed once per heartbeat from the same
|
|
1782
|
+
// queue snapshot. pendingDispatchable = pending rows minus those
|
|
1783
|
+
// terminally blocked behind a failed/skipped dependency.
|
|
1784
|
+
const blockedPending = computeBlockedChains(s.jobs).reduce((n, c) => n + c.blocked, 0);
|
|
1785
|
+
const dispatch = {
|
|
1786
|
+
lastDispatchAttemptAt: s.lastDispatchAttemptAt ?? null,
|
|
1787
|
+
lastRunAt: s.lastRunAt ?? null,
|
|
1788
|
+
lastTickReason: lastTick?.reason ?? null,
|
|
1789
|
+
pendingDispatchable: Math.max(0, counts.pending - blockedPending),
|
|
1790
|
+
runningCount: counts.running,
|
|
1791
|
+
paused: Boolean(s.paused),
|
|
1792
|
+
drain: s.drain?.active ? { since: s.drain.since ?? null, requestedAt: s.drain.requestedAt ?? null } : null,
|
|
1793
|
+
};
|
|
1794
|
+
return {
|
|
1795
|
+
ts: Date.now(),
|
|
1796
|
+
pid: process.pid,
|
|
1797
|
+
build: heartbeatBuild(),
|
|
1798
|
+
counts,
|
|
1799
|
+
dispatch,
|
|
1800
|
+
stall: { stalled: stall.stalled, total: stall.total },
|
|
1801
|
+
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
1802
|
+
quarantinedCwds: (s.unreadableCwds ?? []).map((u) => u.cwd),
|
|
1803
|
+
nextReset: cachedNextReset,
|
|
1804
|
+
utilization: cachedUtilization,
|
|
1805
|
+
consecutiveFailures,
|
|
1806
|
+
// State/consecutiveFailures/degraded-budget snapshot of the shared
|
|
1807
|
+
// usage-meter breaker, so a human reading only the heartbeat log can
|
|
1808
|
+
// see the meter's own health apart from the queue's.
|
|
1809
|
+
usageMeter: {
|
|
1810
|
+
state: billing.usageCircuit.state(),
|
|
1811
|
+
consecutiveFailures,
|
|
1812
|
+
degradedBudget: computeDegradedBudget(),
|
|
1813
|
+
},
|
|
1814
|
+
};
|
|
1815
|
+
});
|
|
1816
|
+
}
|
|
1817
|
+
if (!entry || errors.length > 0) {
|
|
1818
|
+
// Deliberately carries no utilization/counts: this line says "I ran but
|
|
1819
|
+
// could not read state", and consumers must not mistake it for a fresh read.
|
|
1820
|
+
entry = { ts: Date.now(), pid: process.pid, build: heartbeatBuild(), degraded: true, errors };
|
|
1821
|
+
}
|
|
1822
|
+
appendHeartbeat(entry);
|
|
1823
|
+
return entry;
|
|
1824
|
+
}
|
|
1825
|
+
|
|
1525
1826
|
/**
|
|
1526
1827
|
* computeStallSummary(state) → { stalled, total, running, pending, byProject }
|
|
1527
1828
|
*
|
|
@@ -2015,6 +2316,22 @@ function findStrandedInvestigations(jobs, now, maxMs, isAlive = claudePidAlive)
|
|
|
2015
2316
|
// the .bak-* snapshots.
|
|
2016
2317
|
const quarantinedPaths = new Set();
|
|
2017
2318
|
function flagUnreadable(state) {
|
|
2319
|
+
// Per-shard quarantine (queueStore.unreadableCwds): snapshot each torn shard
|
|
2320
|
+
// once and name it, but never halt — other projects keep dispatching.
|
|
2321
|
+
for (const u of state.unreadableCwds ?? []) {
|
|
2322
|
+
if (!quarantinedPaths.has(u.file)) {
|
|
2323
|
+
quarantinedPaths.add(u.file);
|
|
2324
|
+
try {
|
|
2325
|
+
fs.copyFileSync(u.file, `${u.file}.corrupt-${Date.now()}`);
|
|
2326
|
+
} catch { /* best-effort: the read already failed, the copy may too */ }
|
|
2327
|
+
console.error(`[scheduler] project queue shard quarantined (${u.cwd}): ${u.error}`);
|
|
2328
|
+
logs.writeLine({
|
|
2329
|
+
level: 'error', scope: 'scheduler',
|
|
2330
|
+
message: `project queue shard unreadable — ${u.cwd} is quarantined, other projects keep dispatching`,
|
|
2331
|
+
meta: { cwd: u.cwd, path: u.file, error: u.error },
|
|
2332
|
+
});
|
|
2333
|
+
}
|
|
2334
|
+
}
|
|
2018
2335
|
if (!state.unreadable) return state;
|
|
2019
2336
|
if (state.unreadablePath && !quarantinedPaths.has(state.unreadablePath)) {
|
|
2020
2337
|
quarantinedPaths.add(state.unreadablePath);
|
|
@@ -2073,8 +2390,35 @@ async function writeQueue(state) {
|
|
|
2073
2390
|
// preceding mutate threw, so the chain never deadlocks.
|
|
2074
2391
|
let mutateTail = Promise.resolve();
|
|
2075
2392
|
|
|
2393
|
+
// Observe-only watchdog: a mutate body over MUTATE_WATCHDOG_MS is logged and
|
|
2394
|
+
// audited once per episode (latched until a mutate completes). mutateTail is
|
|
2395
|
+
// deliberately NEVER reset — it is what enforces the single-writer law, and
|
|
2396
|
+
// abandoning a live writer would let two read-modify-writes interleave.
|
|
2397
|
+
const MUTATE_WATCHDOG_MS = 60_000;
|
|
2398
|
+
let mutateWedgeLatched = false;
|
|
2399
|
+
|
|
2076
2400
|
function mutate(fn) {
|
|
2077
2401
|
const next = mutateTail.then(async () => {
|
|
2402
|
+
const wedgeTimer = setTimeout(() => {
|
|
2403
|
+
if (mutateWedgeLatched) return;
|
|
2404
|
+
mutateWedgeLatched = true;
|
|
2405
|
+
console.warn(`[scheduler] MUTATE WEDGED: a queue mutation has run > ${MUTATE_WATCHDOG_MS}ms`);
|
|
2406
|
+
appendAuditEvent('mutate_wedged', { budgetMs: MUTATE_WATCHDOG_MS });
|
|
2407
|
+
}, MUTATE_WATCHDOG_MS);
|
|
2408
|
+
if (typeof wedgeTimer.unref === 'function') wedgeTimer.unref();
|
|
2409
|
+
try {
|
|
2410
|
+
return await mutateBody(fn);
|
|
2411
|
+
} finally {
|
|
2412
|
+
clearTimeout(wedgeTimer);
|
|
2413
|
+
mutateWedgeLatched = false;
|
|
2414
|
+
}
|
|
2415
|
+
});
|
|
2416
|
+
mutateTail = next.catch(() => {}); // keep chain alive on errors
|
|
2417
|
+
return next;
|
|
2418
|
+
}
|
|
2419
|
+
|
|
2420
|
+
async function mutateBody(fn) {
|
|
2421
|
+
{
|
|
2078
2422
|
const state = await readQueue();
|
|
2079
2423
|
// Bail BEFORE fn runs: a mutator handed an unreadable (therefore empty)
|
|
2080
2424
|
// state would compute its result from a queue that isn't there, and
|
|
@@ -2120,9 +2464,7 @@ function mutate(fn) {
|
|
|
2120
2464
|
}
|
|
2121
2465
|
await writeQueue(state);
|
|
2122
2466
|
return ret;
|
|
2123
|
-
}
|
|
2124
|
-
mutateTail = next.catch(() => {}); // keep chain alive on errors
|
|
2125
|
-
return next;
|
|
2467
|
+
}
|
|
2126
2468
|
}
|
|
2127
2469
|
|
|
2128
2470
|
// ---------- PRD parsing ----------
|
|
@@ -2142,11 +2484,23 @@ const parsePrd = prdParser.parsePrd;
|
|
|
2142
2484
|
// one — acceptable: PRD counts per project are bounded (hundreds, not
|
|
2143
2485
|
// millions), and correctness across multiple project dirs matters more than
|
|
2144
2486
|
// preserving the single-dir cache's steady-state zero-read fast path.
|
|
2145
|
-
async function listPrdFiles() {
|
|
2487
|
+
async function listPrdFiles(skipCwds) {
|
|
2146
2488
|
ensureDirs();
|
|
2147
2489
|
const dirs = candidatePrdsDirs();
|
|
2148
2490
|
const perDir = await Promise.all(dirs.map((dir) => prdParser.listPrdFiles(dir)));
|
|
2149
|
-
|
|
2491
|
+
let files = perDir.flat();
|
|
2492
|
+
// A quarantined project has no job rows this pass; scanning its PRDs would
|
|
2493
|
+
// mint fresh `pending` rows for work that may already be running.
|
|
2494
|
+
if (skipCwds && skipCwds.size > 0) {
|
|
2495
|
+
const prefixes = [...skipCwds].map((c) => c + path.sep);
|
|
2496
|
+
files = files.filter((f) => !prefixes.some((p) => f.startsWith(p)));
|
|
2497
|
+
}
|
|
2498
|
+
return { files: files.sort(), dirCount: dirs.length };
|
|
2499
|
+
}
|
|
2500
|
+
|
|
2501
|
+
/** Set of cwds whose shard is quarantined in this merged read. */
|
|
2502
|
+
function quarantinedCwdSet(state) {
|
|
2503
|
+
return new Set((state?.unreadableCwds ?? []).map((u) => u.cwd));
|
|
2150
2504
|
}
|
|
2151
2505
|
|
|
2152
2506
|
/**
|
|
@@ -2186,7 +2540,22 @@ async function allocateParallelGroup(cwd) {
|
|
|
2186
2540
|
* Safety:
|
|
2187
2541
|
* - PID-recycling: between app death and this call, another process may have
|
|
2188
2542
|
* reused the PID. We read /proc/<pid>/cmdline (Linux) or `ps -p` (macOS)
|
|
2189
|
-
* and only SIGTERM if the cmdline
|
|
2543
|
+
* and only SIGTERM if the cmdline matches /\bclaude\b/. Since procName
|
|
2544
|
+
* aliasing, cmdline[0] is the alias path (`.../procnames/sm-claude-job`) or
|
|
2545
|
+
* the smArgv0 label (`sm-claude-job:<slug>`), NOT the claude bin path —
|
|
2546
|
+
* both still contain the word `claude`. The macOS `ps -p <pid> -o command=`
|
|
2547
|
+
* branch has the same exposure and the same guarantee (ps shows argv0).
|
|
2548
|
+
* Migration: cmdline is fixed at exec, and no claude procIdentity is
|
|
2549
|
+
* persisted (job.runtime carries none), so a pre-aliasing process recorded
|
|
2550
|
+
* and compared after upgrade still compares equal to itself — no
|
|
2551
|
+
* tolerance needed; unaliased legacy cmdlines also match /\bclaude\b/.
|
|
2552
|
+
* - recordedIdentity (optional): when the caller has a COMPLETE prior
|
|
2553
|
+
* procIdentity for this pid (startTicks + cmdline), it is used only as a
|
|
2554
|
+
* VETO — if it provably differs from the pid's live identity right now,
|
|
2555
|
+
* the pid was recycled and the kill is refused ('mismatch') even before
|
|
2556
|
+
* the cmdline heuristic below runs. No recorded identity (today's only
|
|
2557
|
+
* case — job.runtime carries no identity yet) falls through unchanged to
|
|
2558
|
+
* the existing /\bclaude\b/ + `ps -p` heuristics.
|
|
2190
2559
|
* - Detached process group: jobs are spawned with detached:true so we kill
|
|
2191
2560
|
* -pid (the group). If the group leader is already gone, that fails
|
|
2192
2561
|
* silently and we fall back to single-pid kill.
|
|
@@ -2194,12 +2563,17 @@ async function allocateParallelGroup(cwd) {
|
|
|
2194
2563
|
* scheduled via setTimeout to clean up any process ignoring SIGTERM.
|
|
2195
2564
|
*
|
|
2196
2565
|
* Returns: 'killed' (cmdline matched + signal sent), 'gone' (pid not alive),
|
|
2197
|
-
* 'mismatch' (pid alive but cmdline doesn't look like claude
|
|
2566
|
+
* 'mismatch' (pid alive but cmdline doesn't look like claude, or a
|
|
2567
|
+
* complete recorded identity proves the pid was recycled),
|
|
2198
2568
|
* 'unknown' (couldn't read cmdline — leave the pid alone).
|
|
2199
2569
|
*/
|
|
2200
|
-
function killOrphanClaudePid(pid) {
|
|
2570
|
+
function killOrphanClaudePid(pid, recordedIdentity = null) {
|
|
2201
2571
|
if (!pid || typeof pid !== 'number' || pid <= 1) return 'gone';
|
|
2202
2572
|
try { process.kill(pid, 0); } catch { return 'gone'; }
|
|
2573
|
+
if (recordedIdentity && recordedIdentity.complete
|
|
2574
|
+
&& isDifferentProcess(recordedIdentity, procIdentityOf(pid))) {
|
|
2575
|
+
return 'mismatch';
|
|
2576
|
+
}
|
|
2203
2577
|
let cmdline = '';
|
|
2204
2578
|
try {
|
|
2205
2579
|
cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8').replace(/\0/g, ' ');
|
|
@@ -2307,25 +2681,48 @@ async function reconcile(state) {
|
|
|
2307
2681
|
// has no queue row yet, so it is never "live" and gets archived here
|
|
2308
2682
|
// instead of ever reaching the onDisk scan below.
|
|
2309
2683
|
let phaseStartMs = Date.now();
|
|
2310
|
-
|
|
2684
|
+
const skipCwds = quarantinedCwdSet(state);
|
|
2685
|
+
await consolidateAllFlatPrds(allProjectCwds(), skipCwds);
|
|
2311
2686
|
phaseMs.flatPrdSweep = Date.now() - phaseStartMs;
|
|
2312
2687
|
|
|
2313
2688
|
phaseStartMs = Date.now();
|
|
2314
|
-
const { files, dirCount } = await listPrdFiles();
|
|
2689
|
+
const { files, dirCount } = await listPrdFiles(skipCwds);
|
|
2315
2690
|
phaseMs.prdDirResolve = Date.now() - phaseStartMs;
|
|
2316
2691
|
|
|
2317
2692
|
phaseStartMs = Date.now();
|
|
2318
2693
|
const onDisk = new Map();
|
|
2694
|
+
// Slugs derive from title text with no cwd salt, so two different projects
|
|
2695
|
+
// can legitimately queue an identically-slugged PRD — onDisk alone can
|
|
2696
|
+
// only hold ONE parsed PRD per slug (last-file-wins), which would silently
|
|
2697
|
+
// hand an EXISTING row the wrong project's PRD (or none at all) when two
|
|
2698
|
+
// projects collide on a slug. This side index lets the two existing-row
|
|
2699
|
+
// lookups below (job refresh + invalid-row repair) disambiguate by the
|
|
2700
|
+
// row's own cwd first; the fresh-discovery loop further down still reads
|
|
2701
|
+
// the bare `onDisk` (unscoped) since a same-slug NEW-PRD collision across
|
|
2702
|
+
// two projects is a rarer edge this reconcile pass doesn't yet resolve.
|
|
2703
|
+
const onDiskByCwd = new Map();
|
|
2319
2704
|
for (const f of files) {
|
|
2320
2705
|
try {
|
|
2321
2706
|
// Per-file await: parsing is mtime-cached so steady-state hits zero
|
|
2322
2707
|
// disk reads; on cold cache the awaits keep the main thread responsive.
|
|
2323
2708
|
const p = await parsePrd(f);
|
|
2324
2709
|
onDisk.set(p.slug, p);
|
|
2710
|
+
if (p.cwd) onDiskByCwd.set(`${p.slug}::${p.cwd}`, p);
|
|
2325
2711
|
} catch (e) {
|
|
2326
2712
|
console.warn('[scheduler] failed to parse', f, e?.message);
|
|
2327
2713
|
}
|
|
2328
2714
|
}
|
|
2715
|
+
// resolvePrdForJob(slug, cwd) — cwd-scoped PRD lookup for an EXISTING
|
|
2716
|
+
// queue row, falling back to the unscoped onDisk entry when this exact
|
|
2717
|
+
// (slug, cwd) pair has no PRD (e.g. cwd is null/stale) — same behavior as
|
|
2718
|
+
// a bare onDisk.get() for every slug that isn't cross-project-colliding.
|
|
2719
|
+
function resolvePrdForJob(slug, cwd) {
|
|
2720
|
+
if (cwd) {
|
|
2721
|
+
const scoped = onDiskByCwd.get(`${slug}::${cwd}`);
|
|
2722
|
+
if (scoped) return scoped;
|
|
2723
|
+
}
|
|
2724
|
+
return onDisk.get(slug);
|
|
2725
|
+
}
|
|
2329
2726
|
phaseMs.parseLoop = Date.now() - phaseStartMs;
|
|
2330
2727
|
|
|
2331
2728
|
const next = [];
|
|
@@ -2345,7 +2742,7 @@ async function reconcile(state) {
|
|
|
2345
2742
|
// historyTerminalBySlug() below and backfilled before being dropped.
|
|
2346
2743
|
const terminalDroppedNeedingHistoryCheck = [];
|
|
2347
2744
|
for (const job of state.jobs) {
|
|
2348
|
-
const p =
|
|
2745
|
+
const p = resolvePrdForJob(job.slug, job.cwd);
|
|
2349
2746
|
if (!p) {
|
|
2350
2747
|
// A terminal job whose .md is gone was archived on purpose — dropping
|
|
2351
2748
|
// its row is the intended end of the auto-archive flow, PROVIDED it's
|
|
@@ -2379,10 +2776,24 @@ async function reconcile(state) {
|
|
|
2379
2776
|
continue;
|
|
2380
2777
|
}
|
|
2381
2778
|
seen.add(job.slug);
|
|
2779
|
+
// p.cwd REFINES the row's existing cwd; it never erases one. A PRD
|
|
2780
|
+
// file with no `cwd:` frontmatter parses p.cwd as undefined — falling
|
|
2781
|
+
// through to a bare `cwd: p.cwd` here nulled the row's real cwd,
|
|
2782
|
+
// which queueStore.writeSplit then buckets into
|
|
2783
|
+
// schedulerBatch.js's DEFAULT_PROJECT_CWD, silently relocating the
|
|
2784
|
+
// row into the WRONG project's queue.json shard and emptying the
|
|
2785
|
+
// owning project's shard underneath it (2026-09 data-loss incident).
|
|
2786
|
+
// Resolved ONCE into a local so originSessionId's fallback below
|
|
2787
|
+
// resolves against the SAME cwd this row actually gets, not the raw
|
|
2788
|
+
// (possibly undefined) p.cwd — resolveOriginSessionId(undefined, ...)
|
|
2789
|
+
// returns null unconditionally, which silently dropped the origin link
|
|
2790
|
+
// for every PRD with no `cwd:` frontmatter even though a good cwd was
|
|
2791
|
+
// available one line below.
|
|
2792
|
+
const refreshedCwd = p.cwd ?? job.cwd ?? null;
|
|
2382
2793
|
const updatedJob = {
|
|
2383
2794
|
...job,
|
|
2384
2795
|
title: p.title,
|
|
2385
|
-
cwd:
|
|
2796
|
+
cwd: refreshedCwd,
|
|
2386
2797
|
parallelGroup: p.parallelGroup,
|
|
2387
2798
|
estimateMinutes: p.estimateMinutes,
|
|
2388
2799
|
sourcePromptId: reconcileSourcePromptId(job, p.sourcePromptId),
|
|
@@ -2396,7 +2807,7 @@ async function reconcile(state) {
|
|
|
2396
2807
|
quietMachine: p.quietMachine === true,
|
|
2397
2808
|
budgetExempt: p.budgetExempt === true,
|
|
2398
2809
|
originSessionId: job.originSessionId
|
|
2399
|
-
?? resolveOriginSessionId(
|
|
2810
|
+
?? resolveOriginSessionId(refreshedCwd, p.epicId ?? reconcileSourcePromptId(job, p.sourcePromptId)),
|
|
2400
2811
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2401
2812
|
agentType: p.agentType ?? job.agentType ?? null,
|
|
2402
2813
|
};
|
|
@@ -2479,7 +2890,7 @@ async function reconcile(state) {
|
|
|
2479
2890
|
for (const inv of invalidJobs) {
|
|
2480
2891
|
if (seen.has(inv.slug)) continue; // a valid row for this slug already exists
|
|
2481
2892
|
const oldStatus = inv.row?.status;
|
|
2482
|
-
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir:
|
|
2893
|
+
const hist = historyBySlug.get(inv.slug) ?? latestTerminalOutcomeForSlug(inv.slug, { runsDir: schedulerPaths.runsDir() });
|
|
2483
2894
|
if (hist) {
|
|
2484
2895
|
// Never resurrect: this slug already has a durable terminal record
|
|
2485
2896
|
// elsewhere (history.jsonl or a run sidecar) — repairing its corrupted
|
|
@@ -2492,17 +2903,22 @@ async function reconcile(state) {
|
|
|
2492
2903
|
});
|
|
2493
2904
|
continue;
|
|
2494
2905
|
}
|
|
2495
|
-
const p =
|
|
2906
|
+
const p = resolvePrdForJob(inv.slug, inv.row?.cwd);
|
|
2496
2907
|
if (!p) {
|
|
2497
2908
|
// PRD file also gone with no terminal record anywhere — nothing to
|
|
2498
2909
|
// repair against. queueStore already logged the quarantine once.
|
|
2499
2910
|
continue;
|
|
2500
2911
|
}
|
|
2912
|
+
// Same cwd-refines-not-erases rule as the normal refresh path above, and
|
|
2913
|
+
// same reason for resolving it once into a local: originSessionId's
|
|
2914
|
+
// fallback must resolve against the cwd this row actually gets, not the
|
|
2915
|
+
// raw (possibly undefined) p.cwd.
|
|
2916
|
+
const repairedCwd = p.cwd ?? inv.row?.cwd ?? null;
|
|
2501
2917
|
const job = {
|
|
2502
2918
|
...inv.row,
|
|
2503
2919
|
slug: inv.slug,
|
|
2504
2920
|
title: p.title,
|
|
2505
|
-
cwd:
|
|
2921
|
+
cwd: repairedCwd,
|
|
2506
2922
|
parallelGroup: p.parallelGroup,
|
|
2507
2923
|
estimateMinutes: p.estimateMinutes,
|
|
2508
2924
|
sourcePromptId: p.sourcePromptId ?? inv.row?.sourcePromptId ?? null,
|
|
@@ -2512,7 +2928,7 @@ async function reconcile(state) {
|
|
|
2512
2928
|
disposition: p.disposition ?? null,
|
|
2513
2929
|
quietMachine: p.quietMachine === true,
|
|
2514
2930
|
budgetExempt: p.budgetExempt === true,
|
|
2515
|
-
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(
|
|
2931
|
+
originSessionId: inv.row?.originSessionId ?? resolveOriginSessionId(repairedCwd, p.epicId ?? p.sourcePromptId),
|
|
2516
2932
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2517
2933
|
agentType: p.agentType ?? inv.row?.agentType ?? null,
|
|
2518
2934
|
};
|
|
@@ -2616,17 +3032,27 @@ async function reconcile(state) {
|
|
|
2616
3032
|
// guard above inert. Fall back to reading the slug's own newest run
|
|
2617
3033
|
// sidecars straight off disk — same "don't resurrect an already-terminal
|
|
2618
3034
|
// slug" intent, independent of history.jsonl's existence.
|
|
2619
|
-
const fallback = latestTerminalOutcomeForSlug(slug, { runsDir:
|
|
3035
|
+
const fallback = latestTerminalOutcomeForSlug(slug, { runsDir: schedulerPaths.runsDir() });
|
|
2620
3036
|
if (fallback) {
|
|
2621
3037
|
if (fallback.status === 'completed') {
|
|
2622
3038
|
historyArchiveCandidates.push({ slug, status: fallback.status, finishedAt: fallback.finishedAt });
|
|
2623
3039
|
}
|
|
2624
3040
|
continue;
|
|
2625
3041
|
}
|
|
3042
|
+
// No prior row exists to fall back to (this is a fresh discovery), so a
|
|
3043
|
+
// PRD file with no `cwd:` frontmatter falls back to the project root it
|
|
3044
|
+
// was actually found under (derived from its own file path) rather than
|
|
3045
|
+
// nulling out to schedulerBatch.js's DEFAULT_PROJECT_CWD. Resolved once
|
|
3046
|
+
// so originSessionId (below) resolves against this SAME cwd — passing
|
|
3047
|
+
// the raw p.cwd there instead would resolve against `undefined` for
|
|
3048
|
+
// exactly the no-frontmatter case this fallback exists to handle, since
|
|
3049
|
+
// resolveOriginSessionId(cwd, ...) returns null unconditionally when
|
|
3050
|
+
// `cwd` is falsy.
|
|
3051
|
+
const discoveredCwd = p.cwd ?? deriveProjectCwdFromPrdPath(p.path) ?? null;
|
|
2626
3052
|
const entry = {
|
|
2627
3053
|
slug,
|
|
2628
3054
|
title: p.title,
|
|
2629
|
-
cwd:
|
|
3055
|
+
cwd: discoveredCwd,
|
|
2630
3056
|
parallelGroup: p.parallelGroup,
|
|
2631
3057
|
estimateMinutes: p.estimateMinutes,
|
|
2632
3058
|
sourcePromptId: p.sourcePromptId,
|
|
@@ -2636,7 +3062,7 @@ async function reconcile(state) {
|
|
|
2636
3062
|
disposition: p.disposition ?? null,
|
|
2637
3063
|
quietMachine: p.quietMachine === true,
|
|
2638
3064
|
budgetExempt: p.budgetExempt === true,
|
|
2639
|
-
originSessionId: resolveOriginSessionId(
|
|
3065
|
+
originSessionId: resolveOriginSessionId(discoveredCwd, p.epicId ?? p.sourcePromptId),
|
|
2640
3066
|
bodyPreview: p.body.split('\n').slice(0, 6).join('\n'),
|
|
2641
3067
|
agentType: p.agentType ?? null,
|
|
2642
3068
|
status: 'pending',
|
|
@@ -2787,14 +3213,73 @@ async function reconcile(state) {
|
|
|
2787
3213
|
// ---------- next-reset detection ----------
|
|
2788
3214
|
|
|
2789
3215
|
let cachedNextReset = null; // bare ISO string or null
|
|
2790
|
-
let cachedUtilization = null; //
|
|
3216
|
+
let cachedUtilization = null; // binding-window utilization %, 0–100, or null if unknown
|
|
3217
|
+
// ms timestamp of the last FRESH reset observation (see recordObservedReset)
|
|
3218
|
+
// — distinct from Date.now(), so persistSchedulerState never re-stamps a
|
|
3219
|
+
// stale cachedNextReset as "just observed" on every poll cycle.
|
|
3220
|
+
let lastResetObservedAtMs = null;
|
|
3221
|
+
// Full usage payload (`{ five_hour, limits?, ... }`) from the last SUCCESSFUL
|
|
3222
|
+
// poll — the input degradedBudget() carries forward while the meter is down.
|
|
3223
|
+
// Never itself defaults to 0; absent (null) reads as "no signal yet" and
|
|
3224
|
+
// degradedBudget() treats that conservatively (100% / capped concurrency).
|
|
3225
|
+
let lastGoodUsagePayload = null;
|
|
3226
|
+
// Non-null while the meter is degraded (circuit open, or a poll otherwise
|
|
3227
|
+
// failed to return 'ok') — narrows tickQueue's freeSlots as a picker-side
|
|
3228
|
+
// hold instead of a second slot pool (see tickQueue's freeSlots computation).
|
|
3229
|
+
// Cleared to null the moment a poll succeeds or the meter is inapplicable.
|
|
3230
|
+
let degradedConcurrencyCapValue = null;
|
|
3231
|
+
// Deliberately a named constant, not a bare zero literal assigned straight
|
|
3232
|
+
// into cachedUtilization: a genuine "no consumer meter to poll" (enterprise
|
|
3233
|
+
// auth) is categorically different from "the meter is down and we don't
|
|
3234
|
+
// know" — the latter must never read as 0%/full-speed-ahead.
|
|
3235
|
+
const NO_METER_UTILIZATION = 0;
|
|
3236
|
+
|
|
3237
|
+
/**
|
|
3238
|
+
* Records a freshly-observed reset, stamping lastResetObservedAtMs only when
|
|
3239
|
+
* there actually WAS a reset to observe (never on every poll regardless of
|
|
3240
|
+
* payload content — see persistSchedulerState's header).
|
|
3241
|
+
*/
|
|
3242
|
+
function recordObservedReset(resetIso) {
|
|
3243
|
+
if (resetIso) {
|
|
3244
|
+
cachedNextReset = resetIso;
|
|
3245
|
+
lastResetObservedAtMs = Date.now();
|
|
3246
|
+
} else {
|
|
3247
|
+
cachedNextReset = null;
|
|
3248
|
+
}
|
|
3249
|
+
}
|
|
3250
|
+
|
|
3251
|
+
/** Pure: this poll/executor cycle's conservative budget while the meter is degraded. */
|
|
3252
|
+
function computeDegradedBudget() {
|
|
3253
|
+
return degradedBudget(lastGoodUsagePayload, {
|
|
3254
|
+
observed429: lastFailureKind === 'meter_rate_limited',
|
|
3255
|
+
resetsAt: cachedNextReset,
|
|
3256
|
+
now: Date.now(),
|
|
3257
|
+
configuredCap: sessionSlots.snapshot().total,
|
|
3258
|
+
});
|
|
3259
|
+
}
|
|
3260
|
+
|
|
3261
|
+
/**
|
|
3262
|
+
* Computes AND applies this cycle's degraded budget to cachedUtilization /
|
|
3263
|
+
* degradedConcurrencyCapValue in one call — every "meter down" branch in
|
|
3264
|
+
* pollLoop/runQueueStarvationWatchdog needs the exact same
|
|
3265
|
+
* compute-then-assign-both-fields pair, so it lives once here rather than
|
|
3266
|
+
* being copy-pasted at each call site (where a future change to how the
|
|
3267
|
+
* budget applies would otherwise have to be made N times).
|
|
3268
|
+
*/
|
|
3269
|
+
function applyDegradedBudget() {
|
|
3270
|
+
const budget = computeDegradedBudget();
|
|
3271
|
+
cachedUtilization = budget.utilization;
|
|
3272
|
+
degradedConcurrencyCapValue = budget.concurrencyCap;
|
|
3273
|
+
return budget;
|
|
3274
|
+
}
|
|
2791
3275
|
|
|
2792
3276
|
/** Fetches latest usage from billing API. Throws on any error — callers handle it. */
|
|
2793
3277
|
async function refreshNextReset() {
|
|
2794
3278
|
const r = await billing.fetchUsage();
|
|
2795
3279
|
if (r.kind !== 'ok') throw new Error(`usage fetch failed (${r.kind}): ${r.message ?? ''}`);
|
|
2796
|
-
|
|
2797
|
-
|
|
3280
|
+
const window = bindingWindow(r.data?.usage);
|
|
3281
|
+
recordObservedReset(window.resets_at ?? null);
|
|
3282
|
+
cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
|
|
2798
3283
|
return cachedNextReset;
|
|
2799
3284
|
}
|
|
2800
3285
|
|
|
@@ -2805,15 +3290,38 @@ function getNextResetCached() {
|
|
|
2805
3290
|
/**
|
|
2806
3291
|
* Pure: picks the reset to pause against for a rate-limited run (PRD 1118).
|
|
2807
3292
|
* Prefers the BINDING window read off the run's own log — refreshNextReset()
|
|
2808
|
-
* only ever reports
|
|
2809
|
-
*
|
|
2810
|
-
*
|
|
2811
|
-
* to the billing-endpoint-derived reset only when the log yields nothing
|
|
2812
|
-
|
|
2813
|
-
|
|
3293
|
+
* only ever reports the binding window at POLL time, which can be a
|
|
3294
|
+
* different clock than what actually 429'd this run (five_hour can read 0%
|
|
3295
|
+
* utilization at the very same moment a seven_day window binds). Falls back
|
|
3296
|
+
* to the billing-endpoint-derived reset only when the log yields nothing —
|
|
3297
|
+
* and rejects EITHER source when it has already passed relative to `nowMs`
|
|
3298
|
+
* (usageCircuit.isResetFresh): a stale reset must never arm a resume timer
|
|
3299
|
+
* that already elapsed (that's the 30-second-nap bug computeEffectiveResumeAt
|
|
3300
|
+
* below also guards against), so a stale value here returns null and lets
|
|
3301
|
+
* the caller's 30-minute fallback take over instead.
|
|
3302
|
+
*/
|
|
3303
|
+
function resolveRateLimitPauseReset(logPath, billingResetIso, nowMs = Date.now()) {
|
|
2814
3304
|
const logReset = resolveBindingRateLimitReset(logPath);
|
|
2815
|
-
if (logReset != null)
|
|
2816
|
-
|
|
3305
|
+
if (logReset != null) {
|
|
3306
|
+
const iso = new Date(logReset * 1000).toISOString();
|
|
3307
|
+
if (isResetFresh(iso, nowMs)) return iso;
|
|
3308
|
+
}
|
|
3309
|
+
if (billingResetIso && isResetFresh(billingResetIso, nowMs)) return billingResetIso;
|
|
3310
|
+
return null;
|
|
3311
|
+
}
|
|
3312
|
+
|
|
3313
|
+
/**
|
|
3314
|
+
* Shared by both rate-limit pause sites (spawnJob's rateLimited branch and
|
|
3315
|
+
* reapDeadRunningJobs): the billing-endpoint-derived fallback reset used
|
|
3316
|
+
* when the run's own log yields no binding window. While the shared
|
|
3317
|
+
* usageCircuit is OPEN, skip calling refreshNextReset() — it would just be
|
|
3318
|
+
* another request against the endpoint the breaker just decided is down —
|
|
3319
|
+
* and fall back to whatever was last cached instead (resolveRateLimitPauseReset
|
|
3320
|
+
* itself still rejects that cached value if it has since gone stale).
|
|
3321
|
+
*/
|
|
3322
|
+
async function billingResetForPause() {
|
|
3323
|
+
if (billing.usageCircuit.state() === 'open') return cachedNextReset;
|
|
3324
|
+
return refreshNextReset().catch(() => cachedNextReset);
|
|
2817
3325
|
}
|
|
2818
3326
|
|
|
2819
3327
|
// ---------- health / poll state ----------
|
|
@@ -2828,10 +3336,27 @@ let firstFailureAt = null;
|
|
|
2828
3336
|
let firstNon429FailureAt = null; // tracks only transient/config failures; 429s don't count toward network-pause threshold
|
|
2829
3337
|
let lastFailureKind = null; // 'transient' | 'meter_rate_limited' | 'auth' | null
|
|
2830
3338
|
let pauseClearedManuallyAt = null;
|
|
3339
|
+
// In-memory only (no new persisted field): when clearPause last actually lifted a pause — a
|
|
3340
|
+
// legitimate restart point of the dispatch-idleness clock (dispatchIdleMs).
|
|
3341
|
+
let lastPauseClearedAt = null;
|
|
2831
3342
|
// PRD: the usage-poller silent-failure-streak WARN is emitted once per streak,
|
|
2832
3343
|
// not once per failure (57 failures must produce ONE opsErrorLog line, not 57).
|
|
2833
3344
|
// Reset alongside consecutiveFailures everywhere that resets to 0.
|
|
2834
3345
|
let failureStreakWarned = false;
|
|
3346
|
+
// ms timestamp the initial WARN fired this streak — anchors the periodic
|
|
3347
|
+
// escalation cadence below. Reset to null alongside failureStreakWarned.
|
|
3348
|
+
let failureStreakWarnedAt = null;
|
|
3349
|
+
// ms timestamp of the last periodic escalation (audit event + opsErrorLog
|
|
3350
|
+
// line) this streak. Reset to null alongside failureStreakWarned so a LATER
|
|
3351
|
+
// streak re-arms both the initial WARN and the escalation cadence.
|
|
3352
|
+
let lastEscalationAtMs = null;
|
|
3353
|
+
|
|
3354
|
+
/** The failure-streak-WARN trio must always reset together — one helper, not 3 copies. */
|
|
3355
|
+
function resetFailureStreak() {
|
|
3356
|
+
failureStreakWarned = false;
|
|
3357
|
+
failureStreakWarnedAt = null;
|
|
3358
|
+
lastEscalationAtMs = null;
|
|
3359
|
+
}
|
|
2835
3360
|
// Ceiling on pollLoop's exponential poll backoff (both the 'transient'/'config'
|
|
2836
3361
|
// branch and the 'meter_rate_limited' branch below share this cap — a single
|
|
2837
3362
|
// constant so the two never drift to different ceilings).
|
|
@@ -2840,6 +3365,11 @@ const BACKOFF_MAX_MS = 480_000; // 8 minutes
|
|
|
2840
3365
|
// jitter and becomes worth a human's attention. health.cjs imports this so the
|
|
2841
3366
|
// WARN and the `npm run health` non-GREEN trip at the exact same count.
|
|
2842
3367
|
const FAILURE_STREAK_WARN_THRESHOLD = 5;
|
|
3368
|
+
// How often a PERSISTING failure streak re-escalates (audit event +
|
|
3369
|
+
// opsErrorLog line) after the initial WARN, and the health.cjs YELLOW->RED
|
|
3370
|
+
// threshold for how long the usageCircuit has been open — one constant so
|
|
3371
|
+
// the log cadence and the health-color flip never drift apart.
|
|
3372
|
+
const FAILURE_STREAK_ESCALATION_MS = 30 * 60_000; // 30 minutes
|
|
2843
3373
|
|
|
2844
3374
|
/** Pure: exponential backoff with a cap, shared by every pollLoop failure branch. Exported for unit testing. */
|
|
2845
3375
|
function nextBackoffMs(prevBackoffMs) {
|
|
@@ -2856,19 +3386,59 @@ function shouldWarnFailureStreak(consecutiveFailures, alreadyWarned, threshold =
|
|
|
2856
3386
|
return consecutiveFailures >= threshold && !alreadyWarned;
|
|
2857
3387
|
}
|
|
2858
3388
|
|
|
2859
|
-
/**
|
|
3389
|
+
/**
|
|
3390
|
+
* Pure: does a PERSISTING failure streak warrant another escalation (audit
|
|
3391
|
+
* event + opsErrorLog line) at `nowMs`? Exported for unit testing. Only
|
|
3392
|
+
* relevant once the streak has already crossed `warnThreshold` (the initial
|
|
3393
|
+
* WARN); `lastEscalatedAtMs` null means no escalation has fired yet this
|
|
3394
|
+
* streak, so the first one is due immediately. Re-arms automatically once a
|
|
3395
|
+
* streak clears (the caller resets `lastEscalatedAtMs` to null alongside
|
|
3396
|
+
* `failureStreakWarned`), so a later independent streak escalates again.
|
|
3397
|
+
*/
|
|
3398
|
+
function shouldEscalateFailureStreak(consecutiveFailures, lastEscalatedAtMs, nowMs, thresholdMs = FAILURE_STREAK_ESCALATION_MS, warnThreshold = FAILURE_STREAK_WARN_THRESHOLD) {
|
|
3399
|
+
if (consecutiveFailures < warnThreshold) return false;
|
|
3400
|
+
if (!lastEscalatedAtMs) return true;
|
|
3401
|
+
return nowMs - lastEscalatedAtMs >= thresholdMs;
|
|
3402
|
+
}
|
|
3403
|
+
|
|
3404
|
+
/**
|
|
3405
|
+
* Emits the one-time opsErrorLog WARN the moment a failure streak crosses
|
|
3406
|
+
* the threshold, then — while that streak PERSISTS — re-escalates (audit
|
|
3407
|
+
* event + another opsErrorLog line) every FAILURE_STREAK_ESCALATION_MS so a
|
|
3408
|
+
* human watching only the audit log still sees a live incident, not just the
|
|
3409
|
+
* single opening WARN from hours ago.
|
|
3410
|
+
*/
|
|
2860
3411
|
function warnFailureStreakIfNeeded() {
|
|
2861
|
-
|
|
2862
|
-
failureStreakWarned
|
|
2863
|
-
|
|
2864
|
-
|
|
2865
|
-
|
|
2866
|
-
|
|
2867
|
-
|
|
2868
|
-
|
|
2869
|
-
|
|
2870
|
-
|
|
2871
|
-
|
|
3412
|
+
const nowMs = Date.now();
|
|
3413
|
+
if (shouldWarnFailureStreak(consecutiveFailures, failureStreakWarned)) {
|
|
3414
|
+
failureStreakWarned = true;
|
|
3415
|
+
failureStreakWarnedAt = nowMs;
|
|
3416
|
+
lastEscalationAtMs = nowMs;
|
|
3417
|
+
try {
|
|
3418
|
+
appendError({
|
|
3419
|
+
cwd: DEFAULT_PROJECT_CWD,
|
|
3420
|
+
scope: 'scheduler',
|
|
3421
|
+
level: 'warn',
|
|
3422
|
+
message: `usage/rate-limit poller has failed ${consecutiveFailures} consecutive times (kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
|
|
3423
|
+
meta: { consecutiveFailures, backoffMs, lastFailureKind },
|
|
3424
|
+
});
|
|
3425
|
+
} catch { /* durable logging must never break the poll loop */ }
|
|
3426
|
+
return;
|
|
3427
|
+
}
|
|
3428
|
+
if (failureStreakWarned && shouldEscalateFailureStreak(consecutiveFailures, lastEscalationAtMs, nowMs)) {
|
|
3429
|
+
lastEscalationAtMs = nowMs;
|
|
3430
|
+
const persistedMinutes = failureStreakWarnedAt ? Math.round((nowMs - failureStreakWarnedAt) / 60_000) : null;
|
|
3431
|
+
try {
|
|
3432
|
+
appendAuditEvent('usage_poller_failure_streak_persists', { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes });
|
|
3433
|
+
appendError({
|
|
3434
|
+
cwd: DEFAULT_PROJECT_CWD,
|
|
3435
|
+
scope: 'scheduler',
|
|
3436
|
+
level: 'warn',
|
|
3437
|
+
message: `usage/rate-limit poller streak still failing after ${persistedMinutes}m (${consecutiveFailures} consecutive, kind=${lastFailureKind}, backoffMs=${backoffMs}) — see ${schedulerPaths.schedulerStatePath()}`,
|
|
3438
|
+
meta: { consecutiveFailures, backoffMs, lastFailureKind, persistedMinutes },
|
|
3439
|
+
});
|
|
3440
|
+
} catch { /* durable logging must never break the poll loop */ }
|
|
3441
|
+
}
|
|
2872
3442
|
}
|
|
2873
3443
|
// PRD 1119: consecutive-rapid-rate-limit hard-pause tracking, keyed per slug.
|
|
2874
3444
|
// See isCooldownSuppressed/nextRapidRateLimitCount below for the pure rules.
|
|
@@ -2882,6 +3452,7 @@ let resumeTimer = null;
|
|
|
2882
3452
|
let pollLoopTimer = null;
|
|
2883
3453
|
let rescheduleInterval = null;
|
|
2884
3454
|
let heartbeatInterval = null;
|
|
3455
|
+
let dispatchLoopHandle = null;
|
|
2885
3456
|
// Stall-detector state (computeStallSummary), read/written only inside the
|
|
2886
3457
|
// heartbeat interval below. Keyed per-project cwd (never a single value) —
|
|
2887
3458
|
// a single module-level flag would let one busy project's activity clear or
|
|
@@ -2905,12 +3476,13 @@ const runningSet = new Set();
|
|
|
2905
3476
|
// N concurrent Opus processes — the >3-concurrent class that OOM-killed Electron.
|
|
2906
3477
|
// Over-cap requests are QUEUED (not dropped) and drained as slots free, so a failed
|
|
2907
3478
|
// PRD that never reaches 'needs_review' still eventually gets its fix-plan authored.
|
|
2908
|
-
let investigationsInFlight = 0;
|
|
2909
3479
|
const MAX_CONCURRENT_INVESTIGATIONS = 1;
|
|
2910
3480
|
const deferredInvestigations = new Map(); // fixable-job slug -> { failedJob, runDir }
|
|
3481
|
+
// Mirror of machine `drain.active`, kept in memory so spawn sites need no queue read.
|
|
3482
|
+
let drainActive = false;
|
|
2911
3483
|
|
|
2912
3484
|
function drainDeferredInvestigation() {
|
|
2913
|
-
if (
|
|
3485
|
+
if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) return;
|
|
2914
3486
|
const next = deferredInvestigations.entries().next();
|
|
2915
3487
|
if (next.done) return;
|
|
2916
3488
|
const [slug, ctx] = next.value;
|
|
@@ -3001,6 +3573,7 @@ function buildScheduleStatePayload(state) {
|
|
|
3001
3573
|
lastDispatchAttemptAt: state.lastDispatchAttemptAt ?? null,
|
|
3002
3574
|
nextReset: getNextResetCached(),
|
|
3003
3575
|
paused: state.paused,
|
|
3576
|
+
drain: state.drain ?? null,
|
|
3004
3577
|
// Launch circuit breaker (issue #11): which personas cannot launch right
|
|
3005
3578
|
// now and why, plus any degraded-mode env in force. Empty objects when healthy.
|
|
3006
3579
|
launchBlocks: state.launchBlocks ?? {},
|
|
@@ -3170,13 +3743,17 @@ function nextRapidRateLimitCount(prevCount, { rateLimited, durationMs }) {
|
|
|
3170
3743
|
/**
|
|
3171
3744
|
* Pure: decides the resumeAt actually armed for a pause. 'network' and
|
|
3172
3745
|
* 'rate_limit' (PRD 1118) both get a bounded 30-minute fallback when no
|
|
3173
|
-
* explicit resumeAt is supplied
|
|
3174
|
-
*
|
|
3175
|
-
*
|
|
3176
|
-
*
|
|
3746
|
+
* explicit resumeAt is supplied, OR when the supplied resumeAtIso has
|
|
3747
|
+
* already passed (usageCircuit.isResetFresh) — a stale reset used to produce
|
|
3748
|
+
* `Math.max(30_000, <negative>)` downstream in computeResumeDelay, a
|
|
3749
|
+
* 30-SECOND nap instead of a real pause, spinning the queue right back into
|
|
3750
|
+
* the same still-active rate limit. The live failure mode is the billing
|
|
3751
|
+
* usage endpoint itself 429ing while the log yields no fresh binding window
|
|
3752
|
+
* either, which used to leave an indefinite pause with no resume timer at
|
|
3753
|
+
* all (a queue that never comes back on its own) — this covers both.
|
|
3177
3754
|
*/
|
|
3178
3755
|
function computeEffectiveResumeAt(reason, resumeAtIso, nowMs = Date.now()) {
|
|
3179
|
-
if (resumeAtIso) return resumeAtIso;
|
|
3756
|
+
if (resumeAtIso && isResetFresh(resumeAtIso, nowMs)) return resumeAtIso;
|
|
3180
3757
|
if (reason === 'network' || reason === 'rate_limit') {
|
|
3181
3758
|
return new Date(nowMs + 30 * 60_000).toISOString();
|
|
3182
3759
|
}
|
|
@@ -3196,11 +3773,14 @@ function computeResumeDelay(effectiveResumeAtIso, nowMs = Date.now()) {
|
|
|
3196
3773
|
|
|
3197
3774
|
async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
3198
3775
|
const { observedAt = null, force = false } = opts;
|
|
3776
|
+
const isManual = reason === 'manual';
|
|
3199
3777
|
// Honor manual-override cooldown: if the user cleared a pause within the
|
|
3200
3778
|
// last 5 minutes, suppress auto-pause re-engagement UNLESS this pause is
|
|
3201
3779
|
// backed by a fresh observation (a run that started after the clear) or is
|
|
3202
3780
|
// forced (the rapid-repeat hard pause, which the cooldown cannot suppress).
|
|
3203
|
-
|
|
3781
|
+
// A user-initiated 'manual' pause is never an auto-detection, so the cooldown
|
|
3782
|
+
// (which exists to ignore STALE auto-detections) does not apply to it.
|
|
3783
|
+
if (!isManual && isCooldownSuppressed({ pauseClearedManuallyAt, now: Date.now(), observedAt, force })) {
|
|
3204
3784
|
console.log(`[scheduler] setPaused(${reason}) suppressed by manual override cooldown`);
|
|
3205
3785
|
return;
|
|
3206
3786
|
}
|
|
@@ -3210,15 +3790,24 @@ async function setPaused(reason, resumeAtIso, opts = {}) {
|
|
|
3210
3790
|
console.log(`[scheduler] setPaused(${reason}) engaging despite manual override cooldown — triggering run started after the manual clear`);
|
|
3211
3791
|
}
|
|
3212
3792
|
|
|
3213
|
-
|
|
3793
|
+
// 'manual' never arms a resume timer: only schedule:resume / run-now clears it.
|
|
3794
|
+
const effectiveResumeAt = isManual ? null : computeEffectiveResumeAt(reason, resumeAtIso);
|
|
3214
3795
|
|
|
3215
|
-
await mutate((s) => {
|
|
3796
|
+
const kept = await mutate((s) => {
|
|
3797
|
+
// A user pause outranks every auto-pause (rate_limit/auth/network): the
|
|
3798
|
+
// auto path must not overwrite it, or its resume timer would auto-clear it.
|
|
3799
|
+
if (!isManual && s.paused && s.paused.reason === 'manual') return true;
|
|
3216
3800
|
if (s.paused && s.paused.reason === reason) {
|
|
3217
3801
|
if (effectiveResumeAt) s.paused.resumeAt = effectiveResumeAt;
|
|
3218
3802
|
} else {
|
|
3219
3803
|
s.paused = { reason, since: new Date().toISOString(), resumeAt: effectiveResumeAt || null };
|
|
3220
3804
|
}
|
|
3805
|
+
return false;
|
|
3221
3806
|
});
|
|
3807
|
+
if (kept) {
|
|
3808
|
+
console.log(`[scheduler] setPaused(${reason}) ignored: a manual pause is in force`);
|
|
3809
|
+
return;
|
|
3810
|
+
}
|
|
3222
3811
|
await broadcast({ flush: true });
|
|
3223
3812
|
cancelToken.cancelled = true;
|
|
3224
3813
|
if (resumeTimer) { clearTimeout(resumeTimer); resumeTimer = null; }
|
|
@@ -3256,6 +3845,7 @@ async function clearPause(source) {
|
|
|
3256
3845
|
});
|
|
3257
3846
|
// Un-cancel the tick guard on every recovery path, not just runDueJobs().
|
|
3258
3847
|
applyPauseCleared(wasPaused, cancelToken);
|
|
3848
|
+
if (wasPaused) lastPauseClearedAt = Date.now();
|
|
3259
3849
|
// Track manual clears for the auto-pause cooldown.
|
|
3260
3850
|
if (source === 'manual' || source === 'run-now') {
|
|
3261
3851
|
pauseClearedManuallyAt = Date.now();
|
|
@@ -3268,7 +3858,7 @@ async function clearPause(source) {
|
|
|
3268
3858
|
firstFailureAt = null;
|
|
3269
3859
|
firstNon429FailureAt = null;
|
|
3270
3860
|
lastFailureKind = null;
|
|
3271
|
-
|
|
3861
|
+
resetFailureStreak();
|
|
3272
3862
|
persistSchedulerState();
|
|
3273
3863
|
}
|
|
3274
3864
|
if (wasPaused) await broadcast({ flush: true });
|
|
@@ -3351,37 +3941,65 @@ function resetJobFields(job, errorMsg, opts = {}) {
|
|
|
3351
3941
|
return true;
|
|
3352
3942
|
}
|
|
3353
3943
|
|
|
3354
|
-
// Grace period between a boot orphan's SIGTERM and reading its log to
|
|
3355
|
-
// classify the outcome — matches killOrphanClaudePid's own internal 5s
|
|
3356
|
-
// SIGKILL follow-up delay, plus a small margin so classification always runs
|
|
3357
|
-
// after that SIGKILL has had a chance to land.
|
|
3358
|
-
const BOOT_ORPHAN_KILL_GRACE_MS = 6000;
|
|
3359
|
-
|
|
3360
3944
|
/**
|
|
3361
|
-
* partitionBootOrphans(jobs,
|
|
3945
|
+
* partitionBootOrphans(jobs, liveness) → { immediate: string[], adopted: string[] }
|
|
3362
3946
|
*
|
|
3363
|
-
* Pure decision split for boot reconciliation
|
|
3364
|
-
*
|
|
3365
|
-
*
|
|
3366
|
-
*
|
|
3367
|
-
*
|
|
3368
|
-
*
|
|
3369
|
-
*
|
|
3370
|
-
*
|
|
3371
|
-
|
|
3372
|
-
|
|
3947
|
+
* Pure decision split for boot reconciliation of 'running' rows. A row PROVEN
|
|
3948
|
+
* ALIVE is `adopted`: left `running`, never signalled — the steady-state
|
|
3949
|
+
* reaper (reapDeadRunningJobs) finishes it on exit, exactly as it does for any
|
|
3950
|
+
* pidless-recovered row. Only rows proven dead or exited are `immediate` and go
|
|
3951
|
+
* through applyOrphanOutcome. `liveness` is the same injected set
|
|
3952
|
+
* selectReapableJobs takes (plus readRecord/runsDir/identityOf):
|
|
3953
|
+
* { pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
|
|
3954
|
+
* readRecord(runDir, slug) }.
|
|
3955
|
+
*
|
|
3956
|
+
* classifyAdoption runs FIRST when the row has a supervisor record: 'adopt' and
|
|
3957
|
+
* 'over-budget' are alive (budget re-arm across restart is a separate PRD, so
|
|
3958
|
+
* an over-budget row is spared, not killed); 'exited' / 'dead' / 'foreign-pid'
|
|
3959
|
+
* are not. A row with no record falls to the reaper's own ladder: recorded
|
|
3960
|
+
* pid alive, then fresh log, log-pid alive, /proc cwd scan.
|
|
3961
|
+
* Complexity: O(jobs) plus one /proc probe per running row.
|
|
3962
|
+
*/
|
|
3963
|
+
function partitionBootOrphans(jobs, {
|
|
3964
|
+
pidAlive = claudePidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess,
|
|
3965
|
+
readRecord = supervisorRecord.readSupervisorRecord, runsDir = null, identityOf = procIdentityOf,
|
|
3966
|
+
now = Date.now(),
|
|
3967
|
+
} = {}) {
|
|
3373
3968
|
const immediate = [];
|
|
3374
|
-
const
|
|
3969
|
+
const adopted = [];
|
|
3375
3970
|
for (const j of jobs) {
|
|
3376
3971
|
if (j.status !== 'running') continue;
|
|
3377
|
-
|
|
3378
|
-
|
|
3379
|
-
|
|
3380
|
-
|
|
3381
|
-
|
|
3382
|
-
|
|
3972
|
+
if (isBootRowAlive(j, {
|
|
3973
|
+
pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
|
|
3974
|
+
})) adopted.push(j.slug);
|
|
3975
|
+
else immediate.push(j.slug);
|
|
3976
|
+
}
|
|
3977
|
+
return { immediate, adopted };
|
|
3978
|
+
}
|
|
3979
|
+
|
|
3980
|
+
function isBootRowAlive(j, {
|
|
3981
|
+
pidAlive, getLogPid, getLogMtimeMs, logFreshWindowMs, findLiveProcess, readRecord, runsDir, identityOf, now,
|
|
3982
|
+
}) {
|
|
3983
|
+
const logMtimeMs = typeof getLogMtimeMs === 'function' ? getLogMtimeMs(j) : null;
|
|
3984
|
+
const runDir = j.runId ? path.join(runsDir || schedulerPaths.runsDir(), j.runId) : null;
|
|
3985
|
+
const record = runDir && typeof readRecord === 'function' ? readRecord(runDir, j.slug) : null;
|
|
3986
|
+
if (record && record.pid) {
|
|
3987
|
+
const alive = !!pidAlive(record.pid);
|
|
3988
|
+
const verdict = supervisorRecord.classifyAdoption(record, {
|
|
3989
|
+
identity: alive ? identityOf(record.pid) : null,
|
|
3990
|
+
pidAlive: alive,
|
|
3991
|
+
logMtimeMs,
|
|
3992
|
+
exitMarker: supervisorRecord.hasExitMarker(runDir, record),
|
|
3993
|
+
now,
|
|
3994
|
+
});
|
|
3995
|
+
return verdict === 'adopt' || verdict === 'over-budget';
|
|
3383
3996
|
}
|
|
3384
|
-
|
|
3997
|
+
const pid = j.runtime?.pid;
|
|
3998
|
+
if (pid && pidAlive(pid)) return true;
|
|
3999
|
+
if (typeof logFreshWindowMs === 'number' && Number.isFinite(logMtimeMs) && now - logMtimeMs <= logFreshWindowMs) return true;
|
|
4000
|
+
const logPid = typeof getLogPid === 'function' ? getLogPid(j) : null;
|
|
4001
|
+
if (logPid && pidAlive(logPid)) return true;
|
|
4002
|
+
return !!(typeof findLiveProcess === 'function' && findLiveProcess(j));
|
|
3385
4003
|
}
|
|
3386
4004
|
|
|
3387
4005
|
/**
|
|
@@ -3647,7 +4265,7 @@ async function notifyOriginatingTab(job, {
|
|
|
3647
4265
|
const epicIdForTranscript = prd?.sourcePromptId || prd?.sourceTabId || job.epicId || null;
|
|
3648
4266
|
if (epicIdForTranscript && job.cwd) {
|
|
3649
4267
|
try {
|
|
3650
|
-
const logPath = job.runId ? path.join(
|
|
4268
|
+
const logPath = job.runId ? path.join(schedulerPaths.runsDir(), job.runId, `${job.slug}.log`) : null;
|
|
3651
4269
|
const resultText = readResultFromLog(logPath);
|
|
3652
4270
|
await appendTranscriptTurn(job.cwd, epicIdForTranscript, {
|
|
3653
4271
|
role: 'assistant',
|
|
@@ -4378,6 +4996,39 @@ async function resolveLandedCommitEvidence(cwd, sha, sinceIso) {
|
|
|
4378
4996
|
}
|
|
4379
4997
|
}
|
|
4380
4998
|
|
|
4999
|
+
/**
|
|
5000
|
+
* True when `sha` is a non-empty commit that is an ancestor of (or equal to)
|
|
5001
|
+
* HEAD in the repo at `cwd` (`git merge-base --is-ancestor`). Bounded, never
|
|
5002
|
+
* throws: an empty sha, an unknown sha, or any git failure is `false`, so the
|
|
5003
|
+
* caller's safe default is "cannot prove the work survived".
|
|
5004
|
+
*/
|
|
5005
|
+
async function landedCommitIsAncestorOfHead(cwd, sha) {
|
|
5006
|
+
if (!sha || typeof sha !== 'string' || !cwd) return false;
|
|
5007
|
+
try {
|
|
5008
|
+
await execGitAt(resolveProjectRoot(cwd), ['merge-base', '--is-ancestor', sha, 'HEAD'], { timeout: 10_000 });
|
|
5009
|
+
return true;
|
|
5010
|
+
} catch {
|
|
5011
|
+
return false;
|
|
5012
|
+
}
|
|
5013
|
+
}
|
|
5014
|
+
|
|
5015
|
+
/**
|
|
5016
|
+
* Pure predicate: a needs_review row parked as shared_tree_reverted that
|
|
5017
|
+
* carries a landedCommit — the only shape reverifyNeedsReview can re-check
|
|
5018
|
+
* against ground truth (landedCommitIsAncestorOfHead). Deliberately
|
|
5019
|
+
* independent of autoFixAttempted: 1229-fo-03 was parked with
|
|
5020
|
+
* autoFixAttempted:true and no autoFixOutcome (isStrandedAutoFixPark shape),
|
|
5021
|
+
* yet isStrandedAutoFixPark could not release it — that ladder only resolves
|
|
5022
|
+
* once job.looksDone is set, and reverifyNeedsReview computes looksDone only
|
|
5023
|
+
* for isRescanCandidate / isGuardParkedWithoutAutoFix rows, neither of which
|
|
5024
|
+
* a shared_tree_reverted + autoFixAttempted row is.
|
|
5025
|
+
*/
|
|
5026
|
+
function isStaleSharedTreeRevertedPark(job) {
|
|
5027
|
+
return !!job && job.status === 'needs_review'
|
|
5028
|
+
&& job.verifierVerdict === 'shared_tree_reverted'
|
|
5029
|
+
&& typeof job.landedCommit === 'string' && job.landedCommit.length > 0;
|
|
5030
|
+
}
|
|
5031
|
+
|
|
4381
5032
|
/**
|
|
4382
5033
|
* Commit exactly `paths` (must already be dirty on disk) onto a dedicated
|
|
4383
5034
|
* `sm-salvage/<slug>` ref, built from `headBefore` (or current HEAD when
|
|
@@ -4557,7 +5208,7 @@ function buildClaudeSpawnArgs({ prompt, model, sessionId, resume, systemPrompt }
|
|
|
4557
5208
|
// create it — `recursive: true` makes that race safe.
|
|
4558
5209
|
function pickRunDir() {
|
|
4559
5210
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
4560
|
-
const dir = path.join(
|
|
5211
|
+
const dir = path.join(schedulerPaths.runsDir(), ts);
|
|
4561
5212
|
return { runId: ts, dir };
|
|
4562
5213
|
}
|
|
4563
5214
|
|
|
@@ -4617,7 +5268,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4617
5268
|
// Sync write: this is an early-exit error path inside an async function,
|
|
4618
5269
|
// so we could await, but using the sync variant keeps the error path
|
|
4619
5270
|
// ordering identical to the spawn-failed branch below (also sync).
|
|
4620
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0 });
|
|
5271
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
|
|
4621
5272
|
return { exitCode: -1, durationMs: 0, error: errMsg, sessionId };
|
|
4622
5273
|
}
|
|
4623
5274
|
|
|
@@ -4778,7 +5429,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4778
5429
|
if (!promptCheck.ok) {
|
|
4779
5430
|
safeLog(`[scheduler] ${promptCheck.error}\n`);
|
|
4780
5431
|
closeFd();
|
|
4781
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0 });
|
|
5432
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: promptCheck.error, startedAt, finishedAt: Date.now(), durationMs: 0, ...SCHEDULER_META_IDENTITY });
|
|
4782
5433
|
return { exitCode: -1, durationMs: 0, error: promptCheck.error, sessionId };
|
|
4783
5434
|
}
|
|
4784
5435
|
|
|
@@ -4972,13 +5623,17 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4972
5623
|
|
|
4973
5624
|
// ---------- spawn ----------
|
|
4974
5625
|
|
|
5626
|
+
// Distinct comm (`sm-claude-job`) + slug-labelled argv0 for System Monitor.
|
|
5627
|
+
// Both keep the word `claude`, which the /\bclaude\b/ reaper gates need.
|
|
5628
|
+
const jobSpawn = claudeSpawnTarget('job', job.slug, claudeBin);
|
|
5629
|
+
|
|
4975
5630
|
const { child } = withChildAndLog({
|
|
4976
5631
|
fd,
|
|
4977
5632
|
logPath,
|
|
4978
5633
|
safeLog,
|
|
4979
5634
|
closeFd,
|
|
4980
5635
|
spawn: {
|
|
4981
|
-
command:
|
|
5636
|
+
command: jobSpawn.command,
|
|
4982
5637
|
// Resume mode passes `--resume <sessionId>` (reconnect to the SAME
|
|
4983
5638
|
// session) INSTEAD of `--session-id <sessionId>` (mint a new one) —
|
|
4984
5639
|
// never both, see buildClaudeSpawnArgs.
|
|
@@ -4992,6 +5647,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
4992
5647
|
options: {
|
|
4993
5648
|
cwd: spawnCwd,
|
|
4994
5649
|
env: childEnv,
|
|
5650
|
+
...(jobSpawn.argv0 ? { argv0: jobSpawn.argv0 } : {}),
|
|
4995
5651
|
// detached:true puts the child in its own process group so we can kill
|
|
4996
5652
|
// the entire descendant tree (including any stray background bashes the
|
|
4997
5653
|
// agent spawned) with `process.kill(-pid)`. Without this, child.kill()
|
|
@@ -5017,7 +5673,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5017
5673
|
sl(`\n[scheduler] ${errMsg}\n`);
|
|
5018
5674
|
// Sync write: inside a Promise executor callback; must flush meta
|
|
5019
5675
|
// before resolve() so the spawnJob mutate() that follows sees it.
|
|
5020
|
-
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
5676
|
+
config.writeJsonSync(metaPath, { slug: job.slug, cwd, sessionId, exitCode: -1, error: errMsg, startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked, ...SCHEDULER_META_IDENTITY, originSessionId, contextDigestApplied });
|
|
5021
5677
|
resolve({ exitCode: -1, durationMs, error: errMsg, leakedDescendants: leaked, sessionId });
|
|
5022
5678
|
return;
|
|
5023
5679
|
}
|
|
@@ -5079,7 +5735,7 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5079
5735
|
startedAt, finishedAt: Date.now(), durationMs, leakedDescendants: leaked,
|
|
5080
5736
|
agentResultSubtype, mappedFromSignal: mappedToSuccess ? signal || `code=${exitCode}` : null,
|
|
5081
5737
|
killedByWatchdog: effectiveKilledByWatchdog, budgetKillReason,
|
|
5082
|
-
|
|
5738
|
+
...SCHEDULER_META_IDENTITY,
|
|
5083
5739
|
originSessionId, contextDigestApplied,
|
|
5084
5740
|
});
|
|
5085
5741
|
resolve({
|
|
@@ -5091,6 +5747,22 @@ async function executeJob(job, runDir, defaultCwd, onPid, execCwd, resumeTarget
|
|
|
5091
5747
|
|
|
5092
5748
|
if (child) {
|
|
5093
5749
|
safeLog(`[scheduler] spawned pid=${child.pid} sessionId=${sessionId} (process group)\n\n`);
|
|
5750
|
+
// The one synchronous, authoritative dispatch record (see
|
|
5751
|
+
// jobSupervisorRecord.cjs). detached:true → setsid → pgid === pid, so
|
|
5752
|
+
// no process.getpgid. A write failure must never fail the dispatch.
|
|
5753
|
+
try {
|
|
5754
|
+
supervisorRecord.writeSupervisorRecord({
|
|
5755
|
+
runDir, slug: job.slug, cwd, runId: path.basename(runDir), pid: child.pid, pgid: child.pid,
|
|
5756
|
+
identity: procIdentityOf(child.pid), execCwd: spawnCwd,
|
|
5757
|
+
worktreeDir: execCwd || null, worktreeBranch: execCwd ? `sm-job/${job.slug}` : null,
|
|
5758
|
+
sessionId, startedAt, budgetMs: budgetExempt ? null : jobBudgetMs, maxDurationMs: null,
|
|
5759
|
+
idleKillMs: IDLE_OUTPUT_KILL_MS, schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
|
|
5760
|
+
});
|
|
5761
|
+
} catch (e) {
|
|
5762
|
+
const message = e?.message ?? String(e);
|
|
5763
|
+
console.error(`[scheduler] FAILED to write supervisor record for ${job.slug} pid=${child.pid}: ${message}`);
|
|
5764
|
+
appendAuditEvent('supervisor_record_write_failed', { slug: job.slug, cwd, pid: child.pid, error: message });
|
|
5765
|
+
}
|
|
5094
5766
|
// Make this job the OOM killer's preferred victim over Electron.
|
|
5095
5767
|
biasJobOomScore(child.pid);
|
|
5096
5768
|
// Persist runtime.pid with one retry — still fire-and-forget (must
|
|
@@ -5380,7 +6052,8 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5380
6052
|
return { deferred: false };
|
|
5381
6053
|
}
|
|
5382
6054
|
}
|
|
5383
|
-
|
|
6055
|
+
// A drain (lib/upgradeDrain.cjs) admits no NEW work: queue instead of spawning.
|
|
6056
|
+
if (drainActive || runtimeState.investigationCount() >= MAX_CONCURRENT_INVESTIGATIONS) {
|
|
5384
6057
|
// Queue for retry when a slot frees rather than dropping — otherwise a failed
|
|
5385
6058
|
// job (never 'needs_review', so reverifyNeedsReview won't retry it) would
|
|
5386
6059
|
// silently never get an auto-authored fix-plan.
|
|
@@ -5394,12 +6067,12 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5394
6067
|
// both pass the cap check. Released in onExit, on any pre-spawn early return, or
|
|
5395
6068
|
// on a synchronous throw (try/catch below) — and releasing hands the slot to a
|
|
5396
6069
|
// queued investigation so none are stranded.
|
|
5397
|
-
|
|
6070
|
+
runtimeState.reserveInvestigation(failedJob.slug);
|
|
5398
6071
|
let slotReleased = false;
|
|
5399
6072
|
const releaseSlot = () => {
|
|
5400
6073
|
if (slotReleased) return;
|
|
5401
6074
|
slotReleased = true;
|
|
5402
|
-
|
|
6075
|
+
runtimeState.releaseInvestigation(failedJob.slug);
|
|
5403
6076
|
drainDeferredInvestigation();
|
|
5404
6077
|
};
|
|
5405
6078
|
try {
|
|
@@ -5504,13 +6177,14 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5504
6177
|
};
|
|
5505
6178
|
|
|
5506
6179
|
// Phase 2: spawn with lifecycle managed by withChildAndLog.
|
|
6180
|
+
const probeSpawn = claudeSpawnTarget('aux', 'investigate', claudeBin);
|
|
5507
6181
|
const { child } = withChildAndLog({
|
|
5508
6182
|
fd,
|
|
5509
6183
|
logPath: investigationLogPath,
|
|
5510
6184
|
safeLog,
|
|
5511
6185
|
closeFd,
|
|
5512
6186
|
spawn: {
|
|
5513
|
-
command:
|
|
6187
|
+
command: probeSpawn.command,
|
|
5514
6188
|
args: [
|
|
5515
6189
|
'-p', prompt,
|
|
5516
6190
|
'--model', 'opus',
|
|
@@ -5519,7 +6193,7 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5519
6193
|
'--verbose',
|
|
5520
6194
|
'--session-id', sessionId,
|
|
5521
6195
|
],
|
|
5522
|
-
options: { cwd, env: childEnv },
|
|
6196
|
+
options: { cwd, env: childEnv, ...(probeSpawn.argv0 ? { argv0: probeSpawn.argv0 } : {}) },
|
|
5523
6197
|
},
|
|
5524
6198
|
watchdogs: [deadmanWatchdog],
|
|
5525
6199
|
onExit({ exitCode, error, spawnFailed, safeLog: sl }) {
|
|
@@ -5621,6 +6295,20 @@ async function spawnInvestigation(failedJob, runDir, { deadChild = null } = {})
|
|
|
5621
6295
|
|
|
5622
6296
|
if (child) {
|
|
5623
6297
|
safeLog(`[scheduler] investigation pid=${child.pid}\n\n`);
|
|
6298
|
+
try {
|
|
6299
|
+
supervisorRecord.writeSupervisorRecord({
|
|
6300
|
+
runDir, slug: `${failedJob.slug}.investigation`, kind: 'investigation', cwd: failedJob.cwd ?? null,
|
|
6301
|
+
runId: path.basename(runDir), pid: child.pid, pgid: child.pid, identity: procIdentityOf(child.pid),
|
|
6302
|
+
execCwd: failedJob.cwd ?? null, worktreeDir: null, worktreeBranch: null, sessionId: null,
|
|
6303
|
+
startedAt: Date.now(), budgetMs: null, maxDurationMs: MAX_INVESTIGATION_DURATION_MS, idleKillMs: null,
|
|
6304
|
+
schedulerPid: process.pid, codeSha: SCHEDULER_CODE_SHA,
|
|
6305
|
+
});
|
|
6306
|
+
} catch (e) {
|
|
6307
|
+
const message = e?.message ?? String(e);
|
|
6308
|
+
console.error(`[scheduler] FAILED to write supervisor record for investigation ${failedJob.slug} pid=${child.pid}: ${message}`);
|
|
6309
|
+
appendAuditEvent('supervisor_record_write_failed', { slug: failedJob.slug, cwd: failedJob.cwd, pid: child.pid, error: message, kind: 'investigation' });
|
|
6310
|
+
}
|
|
6311
|
+
runtimeState.stampInvestigationPid(failedJob.slug, child.pid);
|
|
5624
6312
|
// Recorded so findStrandedInvestigations (a post-restart maintenance
|
|
5625
6313
|
// sweep — the live process has no other way to know a probe is still
|
|
5626
6314
|
// running) can tell a live probe apart from one whose owning process is
|
|
@@ -5722,9 +6410,22 @@ async function computeDepHistorySatisfaction(state) {
|
|
|
5722
6410
|
for (const slug of await queueHistory.completedSlugsForCwd(cwd)) satisfied.add(slug);
|
|
5723
6411
|
for (const dir of listArchivedPrdDirs(cwd)) {
|
|
5724
6412
|
let entries;
|
|
5725
|
-
try { entries = await fsp.readdir(dir); } catch { continue; }
|
|
5726
|
-
for (const
|
|
5727
|
-
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
6413
|
+
try { entries = await fsp.readdir(dir, { withFileTypes: true }); } catch { continue; }
|
|
6414
|
+
for (const ent of entries) {
|
|
6415
|
+
if (ent.isFile() && ent.name.endsWith('.md')) { satisfied.add(ent.name.slice(0, -3)); continue; }
|
|
6416
|
+
// Option (b) of PRD 1286: a manual archive (queueOps.archiveOne, the
|
|
6417
|
+
// schedule:archive-prd route and scheduler_archive_prd MCP tool) files the PRD
|
|
6418
|
+
// under prds-archived/<ISO-ts>/<slug>.md — one level DEEPER than the auto-archive
|
|
6419
|
+
// layout. Reading only the top level made every manually-archived slug invisible
|
|
6420
|
+
// here, so once its row aged out its dependents held forever as 'unresolved'.
|
|
6421
|
+
// A dep whose PRD file was archived is satisfied; a dep with NO file, row or
|
|
6422
|
+
// history record (never ran, or a typo) still holds — nothing is dropped.
|
|
6423
|
+
if (!ent.isDirectory()) continue;
|
|
6424
|
+
let inner;
|
|
6425
|
+
try { inner = await fsp.readdir(path.join(dir, ent.name)); } catch { continue; }
|
|
6426
|
+
for (const name of inner) {
|
|
6427
|
+
if (name.endsWith('.md')) satisfied.add(name.slice(0, -3));
|
|
6428
|
+
}
|
|
5728
6429
|
}
|
|
5729
6430
|
}
|
|
5730
6431
|
} catch (e) {
|
|
@@ -5821,7 +6522,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5821
6522
|
// Session-Manager owns the machine-wide `claude -p` pool (sessionSlots.cjs)
|
|
5822
6523
|
// — the scheduler REQUESTS capacity, it doesn't own a private cap. A miss
|
|
5823
6524
|
// leaves the job pending; the next tick retries when a slot frees up.
|
|
5824
|
-
const slotToken = sessionSlots.acquire(`scheduler:${job.slug}
|
|
6525
|
+
const slotToken = sessionSlots.acquire(`scheduler:${job.slug}`, { claimedAt: Date.now() });
|
|
5825
6526
|
if (!slotToken) {
|
|
5826
6527
|
console.log(`[scheduler] no session slot free for ${job.slug} — deferring (${JSON.stringify(sessionSlots.snapshot().holders.map((h) => h.owner))})`);
|
|
5827
6528
|
return;
|
|
@@ -5947,7 +6648,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5947
6648
|
// targets a specific prior session on purpose) and for anything not
|
|
5948
6649
|
// currently 'pending' (e.g. a needs_review->running recovery row).
|
|
5949
6650
|
if (!resumeTarget && s.jobs[idx].status === 'pending') {
|
|
5950
|
-
const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir:
|
|
6651
|
+
const outcome = latestTerminalOutcomeForSlug(job.slug, { runsDir: schedulerPaths.runsDir() });
|
|
5951
6652
|
const reconcileDecision = evaluateDispatchSidecarReconcile({
|
|
5952
6653
|
rowStatus: s.jobs[idx].status,
|
|
5953
6654
|
rowRunId: s.jobs[idx].runId ?? null,
|
|
@@ -5956,7 +6657,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5956
6657
|
outcome,
|
|
5957
6658
|
});
|
|
5958
6659
|
if (reconcileDecision.skip) {
|
|
5959
|
-
const sidecar = readRunOutcomeSidecars(path.join(
|
|
6660
|
+
const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), reconcileDecision.runId), job.slug);
|
|
5960
6661
|
transitionJob(s.jobs[idx], 'completed', {
|
|
5961
6662
|
reason: `prior run ${reconcileDecision.runId} already completed this slug (sidecar-reconciled)`,
|
|
5962
6663
|
source: 'spawnJob:dispatch-sidecar-reconcile',
|
|
@@ -5986,7 +6687,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
5986
6687
|
// pass_no_commit_prior_run_verified exemption can fire on this
|
|
5987
6688
|
// run if it turns out to be another no-op re-verification.
|
|
5988
6689
|
if (!s.jobs[idx].landedCommit && outcome?.runId) {
|
|
5989
|
-
const sidecar = readRunOutcomeSidecars(path.join(
|
|
6690
|
+
const sidecar = readRunOutcomeSidecars(path.join(schedulerPaths.runsDir(), outcome.runId), job.slug);
|
|
5990
6691
|
if (sidecar.outcome?.landedCommit) {
|
|
5991
6692
|
s.jobs[idx].landedCommit = sidecar.outcome.landedCommit;
|
|
5992
6693
|
}
|
|
@@ -6169,6 +6870,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6169
6870
|
const foreignWip = worktree.ok ? { carriedPaths } : { preRunDirtyPaths };
|
|
6170
6871
|
try {
|
|
6171
6872
|
res = await executeJob(job, runDir, defaultCwd, async (pid, sessionId, cwd) => {
|
|
6873
|
+
sessionSlots.stampPid(slotToken, pid);
|
|
6172
6874
|
await mutate((s) => {
|
|
6173
6875
|
const idx = s.jobs.findIndex((x) => x.slug === job.slug);
|
|
6174
6876
|
if (idx >= 0) {
|
|
@@ -6312,8 +7014,13 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6312
7014
|
}
|
|
6313
7015
|
|
|
6314
7016
|
if (res.rateLimited) {
|
|
7017
|
+
// The executor itself observed a 429 — open the shared circuit even if
|
|
7018
|
+
// the billing poller has been reporting 'ok' all along (AC3: a
|
|
7019
|
+
// different window can 429 the executor than the one binding the
|
|
7020
|
+
// poller's own reads).
|
|
7021
|
+
billing.usageCircuit.recordFailure('executor_429');
|
|
6315
7022
|
const logPath = path.join(runDir, `${job.slug}.log`);
|
|
6316
|
-
const billingResetIso = await
|
|
7023
|
+
const billingResetIso = await billingResetForPause();
|
|
6317
7024
|
const resetIso = resolveRateLimitPauseReset(logPath, billingResetIso);
|
|
6318
7025
|
const observedAt = dispatchStartedAtMs;
|
|
6319
7026
|
const prevCount = consecutiveRapidRateLimitsBySlug.get(job.slug) || 0;
|
|
@@ -6401,6 +7108,8 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6401
7108
|
allJobs: stateForDeps.jobs,
|
|
6402
7109
|
committedDuringRun,
|
|
6403
7110
|
priorLandedCommit,
|
|
7111
|
+
jobLandedCommitThisRun,
|
|
7112
|
+
exitCode: res.exitCode,
|
|
6404
7113
|
}).catch((e) => ({
|
|
6405
7114
|
verdict: 'verify_unavailable',
|
|
6406
7115
|
reason: `verifier threw: ${e?.message ?? String(e)}`,
|
|
@@ -6506,6 +7215,7 @@ async function spawnJob(job, runId, runDir, defaultCwd, resumeTarget = null) {
|
|
|
6506
7215
|
dirtyBaseline: guardBaselineEntries,
|
|
6507
7216
|
headBefore: guardHeadBefore,
|
|
6508
7217
|
slug: job.slug,
|
|
7218
|
+
landedCommit: jobLandedCommitThisRun,
|
|
6509
7219
|
});
|
|
6510
7220
|
// A restored stash alone isn't silence — it's logged loudly above and
|
|
6511
7221
|
// surfaced on the job row below — but a path that's still missing
|
|
@@ -7251,36 +7961,122 @@ async function spawnResumeRecovery(job, resumeTarget) {
|
|
|
7251
7961
|
// is synchronous and spawnJob is fire-and-forget.
|
|
7252
7962
|
let tickTail = Promise.resolve();
|
|
7253
7963
|
|
|
7964
|
+
// Tick watchdog. The tick BODY (not enqueue-to-settle) is bounded: a body
|
|
7965
|
+
// that never settles (hung reconcile / git walk) is declared wedged, the
|
|
7966
|
+
// chain is reset, and `tickGeneration` is bumped so the abandoned body —
|
|
7967
|
+
// which may resume much later — fails its generation re-check after every
|
|
7968
|
+
// await and returns without spawning or mutating. mutateTail is NOT touched.
|
|
7969
|
+
let tickGeneration = 0;
|
|
7970
|
+
let tickWedgeLatched = false;
|
|
7971
|
+
function tickWatchdogMs() {
|
|
7972
|
+
const raw = process.env.SM_TICK_WATCHDOG_MS;
|
|
7973
|
+
if (raw === undefined || raw === '') return 120_000;
|
|
7974
|
+
const n = Number(raw);
|
|
7975
|
+
return Number.isFinite(n) && n >= 0 ? n : 120_000;
|
|
7976
|
+
}
|
|
7977
|
+
|
|
7254
7978
|
// `bypassLoadGate` is set only by the explicit human run-now / force-tick
|
|
7255
7979
|
// paths (via runDueJobs): the human is asking, so the CPU-load gate yields
|
|
7256
7980
|
// and logs that it did. Every automatic caller leaves it false.
|
|
7981
|
+
/**
|
|
7982
|
+
* Fail-CLOSED expiry of runtime reservations (slot tokens, quiet-machine
|
|
7983
|
+
* lease, investigation set, worktree accounting) whose owner is provably
|
|
7984
|
+
* dead. `now` is the pass start: reservations claimed after it are never
|
|
7985
|
+
* expired. Accounting only — never signals or removes anything.
|
|
7986
|
+
*/
|
|
7987
|
+
function runReservationExpiryPass(jobs, now = Date.now()) {
|
|
7988
|
+
try {
|
|
7989
|
+
const liveSlugs = new Set(runningSet);
|
|
7990
|
+
const terminal = new Set();
|
|
7991
|
+
for (const j of jobs || []) {
|
|
7992
|
+
if (j.status === 'running' || j.status === 'investigating') liveSlugs.add(j.slug);
|
|
7993
|
+
else if (j.status === 'completed' || j.status === 'failed' || j.status === 'skipped') terminal.add(j.slug);
|
|
7994
|
+
}
|
|
7995
|
+
const pidAlive = (pid) => claudePidAlive(pid);
|
|
7996
|
+
sessionSlots.expireDead({ liveSlugs, pidAlive, now });
|
|
7997
|
+
quietMachineLease.expireDead({ liveSlugs, now });
|
|
7998
|
+
runtimeState.expireDeadInvestigations({ liveSlugs, pidAlive, now });
|
|
7999
|
+
gitWorktree.expireDeadWorktreeRegistrations({
|
|
8000
|
+
isTerminalBranch: (branch) => {
|
|
8001
|
+
const key = gitWorktree.keyFromBranch('job', branch);
|
|
8002
|
+
return !!key && !liveSlugs.has(key) && terminal.has(key);
|
|
8003
|
+
},
|
|
8004
|
+
});
|
|
8005
|
+
} catch (e) {
|
|
8006
|
+
console.warn('[scheduler] reservation expiry pass failed', e?.message);
|
|
8007
|
+
}
|
|
8008
|
+
}
|
|
8009
|
+
|
|
7257
8010
|
function tickQueue({ bypassLoadGate = false } = {}) {
|
|
7258
|
-
const
|
|
8011
|
+
const budgetMs = tickWatchdogMs();
|
|
8012
|
+
const next = tickTail.then(() => {
|
|
8013
|
+
const gen = tickGeneration;
|
|
8014
|
+
return withTimeout(() => tickBody(gen, { bypassLoadGate }), budgetMs, () => {
|
|
8015
|
+
tickGeneration++; // fence the abandoned body
|
|
8016
|
+
console.warn(`[scheduler] TICK WEDGED: tick body exceeded ${budgetMs}ms — resetting tick chain`);
|
|
8017
|
+
if (!tickWedgeLatched) {
|
|
8018
|
+
tickWedgeLatched = true;
|
|
8019
|
+
appendAuditEvent('tick_wedged', { budgetMs });
|
|
8020
|
+
}
|
|
8021
|
+
// CAS: only reset if nothing has queued behind this wedged link.
|
|
8022
|
+
if (tickTail === tail) tickTail = Promise.resolve();
|
|
8023
|
+
return recordTick({ fired: false, reason: 'wedged' }, { detail: `tick body exceeded ${budgetMs}ms` });
|
|
8024
|
+
}).then((r) => {
|
|
8025
|
+
if (r?.reason !== 'wedged') tickWedgeLatched = false;
|
|
8026
|
+
return r;
|
|
8027
|
+
});
|
|
8028
|
+
});
|
|
8029
|
+
const tail = next.catch(() => {});
|
|
8030
|
+
tickTail = tail;
|
|
8031
|
+
return next;
|
|
8032
|
+
}
|
|
8033
|
+
|
|
8034
|
+
// The stale sentinel a fenced body returns: never recorded, never acted on.
|
|
8035
|
+
const STALE_TICK = Object.freeze({ fired: false, reason: 'stale-generation' });
|
|
8036
|
+
|
|
8037
|
+
async function tickBody(gen, { bypassLoadGate }) {
|
|
8038
|
+
const tickStartedAt = Date.now();
|
|
8039
|
+
{
|
|
8040
|
+
const stale = () => gen !== tickGeneration;
|
|
7259
8041
|
const state = await readQueue();
|
|
8042
|
+
if (stale()) return STALE_TICK;
|
|
7260
8043
|
// Never reconcile against an unreadable queue: reconcile() would see zero
|
|
7261
8044
|
// job rows for every PRD on disk and resurrect the lot as 'pending'.
|
|
7262
8045
|
if (state.unreadable) {
|
|
7263
8046
|
console.error('[scheduler] tickQueue skipped: queue.json unreadable');
|
|
7264
8047
|
return { fired: false, reason: 'unreadable' };
|
|
7265
8048
|
}
|
|
8049
|
+
// Supervision of an adopted executor is independent of dispatch: it runs
|
|
8050
|
+
// even while paused, before any early return below.
|
|
8051
|
+
await superviseAdoptedRunsPass(state.jobs);
|
|
8052
|
+
if (stale()) return STALE_TICK;
|
|
7266
8053
|
if (state.paused) {
|
|
7267
8054
|
console.log('[scheduler] tickQueue skipped: paused');
|
|
7268
8055
|
return recordTick({ fired: false, reason: 'paused' }, { detail: 'scheduler paused' });
|
|
7269
8056
|
}
|
|
8057
|
+
// Upgrade drain (lib/upgradeDrain.cjs): a separate field from `paused`, so a
|
|
8058
|
+
// rate-limit pause can't overwrite it. Nothing new dispatches; running and
|
|
8059
|
+
// investigating rows finish.
|
|
8060
|
+
if (state.drain?.active) {
|
|
8061
|
+
return recordTick({ fired: false, reason: 'draining' }, { detail: 'draining for restart' });
|
|
8062
|
+
}
|
|
7270
8063
|
if (cancelToken.cancelled) return { fired: false, reason: 'cancelled' };
|
|
7271
8064
|
|
|
7272
8065
|
// Stamped here — the moment tickQueue actually reaches the picker,
|
|
7273
8066
|
// regardless of whether this pass ends in a launch — so
|
|
7274
8067
|
// classifyQueueStarvation can tell "the engine keeps evaluating the
|
|
7275
8068
|
// queue" apart from "nothing has invoked tickQueue in a long time".
|
|
7276
|
-
// Distinct from `lastRunAt` below
|
|
7277
|
-
//
|
|
8069
|
+
// Distinct from `lastRunAt` below (a batch actually launched): this one
|
|
8070
|
+
// means only "the loop is alive" (heartbeat/health) and must NEVER feed
|
|
8071
|
+
// the idle clock — see dispatchIdleMs.
|
|
7278
8072
|
await mutate((s) => { s.lastDispatchAttemptAt = new Date().toISOString(); });
|
|
8073
|
+
if (stale()) return STALE_TICK;
|
|
7279
8074
|
|
|
7280
8075
|
// The retired-flat-dir sweep now lives inside reconcile() itself (see its
|
|
7281
8076
|
// own comment) so every caller of reconcile — not just this tick — gets
|
|
7282
8077
|
// the guarantee.
|
|
7283
8078
|
await reconcile(state);
|
|
8079
|
+
if (stale()) return STALE_TICK;
|
|
7284
8080
|
// Reclaim any job-kind worktree whose owning row already resolved
|
|
7285
8081
|
// (completed/failed/skipped) without the run ever reaching
|
|
7286
8082
|
// cleanupWorktree — a leaked checkout that would otherwise sit until the
|
|
@@ -7297,9 +8093,19 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
7297
8093
|
// to also carry a private `concurrencyCap` of 3 — the exact per-consumer
|
|
7298
8094
|
// cap that sessionSlots.cjs was written to replace — which silently
|
|
7299
8095
|
// ceilinged the queue at 3 while the pool the user configured said 5.
|
|
7300
|
-
|
|
8096
|
+
// While the usage meter is degraded (degradedConcurrencyCapValue set by
|
|
8097
|
+
// pollLoop — circuit open, or a poll otherwise failed), a picker-side
|
|
8098
|
+
// hold narrows this SAME freeSlots figure instead of standing up a
|
|
8099
|
+
// second pool: the row count admitted this tick simply can't exceed the
|
|
8100
|
+
// degraded cap minus what's already running.
|
|
8101
|
+
runReservationExpiryPass(state.jobs, tickStartedAt);
|
|
8102
|
+
const freeSlots = degradedConcurrencyCapValue != null
|
|
8103
|
+
? Math.max(0, Math.min(sessionSlots.available(), degradedConcurrencyCapValue - runningSet.size))
|
|
8104
|
+
: sessionSlots.available();
|
|
7301
8105
|
const heldSlugs = await computeLaunchHolds(state);
|
|
8106
|
+
if (stale()) return STALE_TICK;
|
|
7302
8107
|
const satisfiedSlugsByCwd = await computeDepHistorySatisfaction(state);
|
|
8108
|
+
if (stale()) return STALE_TICK;
|
|
7303
8109
|
const { batch, reason: holdReason, holds } = pickNextBatch(state.jobs, runningSet, freeSlots, {
|
|
7304
8110
|
leaseHeld: quietMachineLease.isHeld(),
|
|
7305
8111
|
machineInUse: sessionSlots.inUse(),
|
|
@@ -7399,18 +8205,18 @@ function tickQueue({ bypassLoadGate = false } = {}) {
|
|
|
7399
8205
|
}
|
|
7400
8206
|
|
|
7401
8207
|
await mutate((s) => { s.lastRunAt = new Date().toISOString(); });
|
|
8208
|
+
if (stale()) return STALE_TICK;
|
|
7402
8209
|
await broadcast();
|
|
8210
|
+
if (stale()) return STALE_TICK;
|
|
7403
8211
|
|
|
7404
8212
|
const { runId, dir: runDir } = pickRunDir();
|
|
7405
8213
|
for (const job of gatedBatch) {
|
|
7406
|
-
if (cancelToken.cancelled) break;
|
|
8214
|
+
if (cancelToken.cancelled || stale()) break;
|
|
7407
8215
|
// spawnJob is fire-and-forget; it calls tickQueue() on completion.
|
|
7408
8216
|
spawnJob(job, runId, runDir, state.config.defaultCwd).catch(() => {});
|
|
7409
8217
|
}
|
|
7410
8218
|
return recordTick({ fired: true, count: gatedBatch.length, group: gatedBatch[0]?.parallelGroup }, { holds });
|
|
7411
|
-
}
|
|
7412
|
-
tickTail = next.catch(() => {});
|
|
7413
|
-
return next;
|
|
8219
|
+
}
|
|
7414
8220
|
}
|
|
7415
8221
|
|
|
7416
8222
|
// Translates a tickQueue()/runDueJobs() outcome descriptor into a renderer-facing
|
|
@@ -7486,6 +8292,40 @@ async function maybeLaunchWhenAvailable(state) {
|
|
|
7486
8292
|
* second scheduler. */
|
|
7487
8293
|
const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
7488
8294
|
|
|
8295
|
+
/**
|
|
8296
|
+
* dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now }) → ms
|
|
8297
|
+
*
|
|
8298
|
+
* Pure. The ONE dispatch-idleness clock: `now - max(lastRunAt, lastPauseClearedAt,
|
|
8299
|
+
* schedulerBootedAt)`. `lastRunAt` has exactly one writer (tickQueue, right
|
|
8300
|
+
* before the spawn loop) so it already means "a batch actually launched";
|
|
8301
|
+
* a pause clear and a scheduler boot are the other two moments the queue
|
|
8302
|
+
* legitimately (re)starts. Deliberately NOT `lastDispatchAttemptAt`, which
|
|
8303
|
+
* tickQueue stamps before every gate — a queue that ticks every 30 s and
|
|
8304
|
+
* launches nothing (leaked slot, stuck hold) would refresh that stamp
|
|
8305
|
+
* forever and never look idle (the structural repeat of the f18e161 bug
|
|
8306
|
+
* where lastRunAt was refreshed every poll). No finite input → Infinity.
|
|
8307
|
+
*/
|
|
8308
|
+
function dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now } = {}) {
|
|
8309
|
+
const stamps = [lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs].filter(Number.isFinite);
|
|
8310
|
+
return stamps.length ? now - Math.max(...stamps) : Infinity;
|
|
8311
|
+
}
|
|
8312
|
+
|
|
8313
|
+
/**
|
|
8314
|
+
* launchBlockedSlugs(jobs, launchBlocks) → Set<slug>
|
|
8315
|
+
*
|
|
8316
|
+
* Pure, sync (health.cjs runs as a cold process). Pending rows whose persona
|
|
8317
|
+
* has ANY launch-breaker entry — a superset of computeLaunchHolds, which
|
|
8318
|
+
* additionally lets one half-open probe row through per persona.
|
|
8319
|
+
*/
|
|
8320
|
+
function launchBlockedSlugs(jobs, launchBlocks) {
|
|
8321
|
+
const out = new Set();
|
|
8322
|
+
if (!launchBlocks || !Object.keys(launchBlocks).length) return out;
|
|
8323
|
+
for (const j of Array.isArray(jobs) ? jobs : []) {
|
|
8324
|
+
if (j && j.status === 'pending' && launchBlocks[launchFailure.launchBlockKeyFor(j)]) out.add(j.slug);
|
|
8325
|
+
}
|
|
8326
|
+
return out;
|
|
8327
|
+
}
|
|
8328
|
+
|
|
7489
8329
|
/**
|
|
7490
8330
|
* classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs })
|
|
7491
8331
|
* → null | { kind: 'starved' | 'blocked', pending, dispatchable, blockedChains, idleMs }
|
|
@@ -7513,22 +8353,28 @@ const QUEUE_STARVATION_MS = 10 * 60_000;
|
|
|
7513
8353
|
* Returns null when the queue is healthy (work running, nothing pending,
|
|
7514
8354
|
* paused on purpose, or simply not idle long enough yet).
|
|
7515
8355
|
*/
|
|
7516
|
-
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
8356
|
+
function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
7517
8357
|
if (paused) return null; // paused is a DECISION, not a stall
|
|
7518
8358
|
if (runningCount > 0) return null; // work is flowing
|
|
7519
8359
|
const rows = Array.isArray(jobs) ? jobs : [];
|
|
7520
8360
|
const pending = rows.filter((j) => j && j.status === 'pending');
|
|
7521
8361
|
if (pending.length === 0) return null; // nothing to run — not a stall
|
|
7522
8362
|
|
|
7523
|
-
const idleMs =
|
|
8363
|
+
const idleMs = dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
|
|
7524
8364
|
if (idleMs < thresholdMs) return null; // give the normal path its chance first
|
|
7525
8365
|
|
|
7526
8366
|
// Which pending rows could actually dispatch? Anything NOT named by a
|
|
7527
|
-
// blocked chain
|
|
7528
|
-
//
|
|
8367
|
+
// blocked chain (computeBlockedChains walks dependsOn with the picker's own
|
|
8368
|
+
// resolution, so the two can never disagree) and NOT held by an open launch
|
|
8369
|
+
// breaker / the quietMachine lease (`heldSlugs` — a separate input, never
|
|
8370
|
+
// folded into the dependsOn walker). Held rows can't launch no matter how
|
|
8371
|
+
// often we tick, so they read as 'blocked' (needs a human), not 'starved'.
|
|
7529
8372
|
const blockedChains = computeBlockedChains(rows);
|
|
7530
|
-
const
|
|
7531
|
-
const
|
|
8373
|
+
const held = heldSlugs?.has ? heldSlugs : new Set(heldSlugs ?? []);
|
|
8374
|
+
const open = held.size > 0 ? rows.filter((j) => !(j.status === 'pending' && held.has(j.slug))) : rows;
|
|
8375
|
+
const openBlocked = held.size > 0 ? computeBlockedChains(open) : blockedChains;
|
|
8376
|
+
const openPending = open.filter((j) => j.status === 'pending').length;
|
|
8377
|
+
const dispatchable = openPending - openBlocked.reduce((n, c) => n + c.blocked, 0);
|
|
7532
8378
|
|
|
7533
8379
|
return {
|
|
7534
8380
|
kind: dispatchable > 0 ? 'starved' : 'blocked',
|
|
@@ -7551,14 +8397,14 @@ function classifyQueueStarvation({ jobs, paused, runningCount, lastRunAtMs, now,
|
|
|
7551
8397
|
* projects sat starved/blocked for hours, and the watchdog never fired once
|
|
7552
8398
|
* because "work is flowing" was true somewhere else. Partitioning by cwd
|
|
7553
8399
|
* (the same grouping computeBlockedChains already does) fixes DETECTION only
|
|
7554
|
-
* — the idle clock (
|
|
7555
|
-
* `
|
|
8400
|
+
* — the idle clock (see dispatchIdleMs) stays machine-wide, since
|
|
8401
|
+
* `lastRunAt` is machine-level state, and only one tick is ever
|
|
7556
8402
|
* forced per watchdog pass regardless of how many cwds are starved.
|
|
7557
8403
|
*
|
|
7558
8404
|
* Pure, no IO. Returns [] when paused (a DECISION, not a stall) or when no
|
|
7559
8405
|
* project has a verdict.
|
|
7560
8406
|
*/
|
|
7561
|
-
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
8407
|
+
function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlugs, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, heldSlugs, now, thresholdMs = QUEUE_STARVATION_MS } = {}) {
|
|
7562
8408
|
if (paused) return [];
|
|
7563
8409
|
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7564
8410
|
const byCwd = new Map();
|
|
@@ -7580,6 +8426,9 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7580
8426
|
paused: false,
|
|
7581
8427
|
runningCount: projRunningCount,
|
|
7582
8428
|
lastRunAtMs,
|
|
8429
|
+
lastPauseClearedAtMs,
|
|
8430
|
+
schedulerBootedAtMs,
|
|
8431
|
+
heldSlugs,
|
|
7583
8432
|
now,
|
|
7584
8433
|
thresholdMs,
|
|
7585
8434
|
});
|
|
@@ -7590,7 +8439,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7590
8439
|
|
|
7591
8440
|
/**
|
|
7592
8441
|
* classifyQueueHealth({ jobs, paused, launchBlocks, runningSet, freeSlots,
|
|
7593
|
-
* totalSlots,
|
|
8442
|
+
* totalSlots, lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd, thresholdMs })
|
|
7594
8443
|
* → { kind, cwd, pending, dispatchable, blockedChains, needsReviewCount, runningCount, ... }
|
|
7595
8444
|
*
|
|
7596
8445
|
* Single source of truth for the Scheduler page's queue-health header: the
|
|
@@ -7601,7 +8450,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7601
8450
|
*
|
|
7602
8451
|
* Reuses classifyQueueStarvation for the blocked/stalled read so the header
|
|
7603
8452
|
* can never disagree with runQueueStarvationWatchdog's own decision to force
|
|
7604
|
-
* a tick: both are handed the same
|
|
8453
|
+
* a tick: both are handed the same launch-keyed idle clock (dispatchIdleMs) and
|
|
7605
8454
|
* the same computeBlockedChains walk under the hood. Called here with
|
|
7606
8455
|
* `thresholdMs: 0` first (a live header must say "blocked" the instant every
|
|
7607
8456
|
* pending row is dependency-stuck, not wait out the watchdog's own 10-minute
|
|
@@ -7638,7 +8487,7 @@ function classifyQueueStarvationByProject({ jobs, paused, runningSet: runningSlu
|
|
|
7638
8487
|
*/
|
|
7639
8488
|
function classifyQueueHealth({
|
|
7640
8489
|
jobs, paused, launchBlocks, runningSet: runningSlugs, freeSlots, totalSlots,
|
|
7641
|
-
|
|
8490
|
+
lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, cwd = null, thresholdMs = QUEUE_STARVATION_MS,
|
|
7642
8491
|
} = {}) {
|
|
7643
8492
|
const rows = (Array.isArray(jobs) ? jobs : []).filter(Boolean);
|
|
7644
8493
|
const projectJobs = cwd ? rows.filter((j) => j.cwd === cwd) : rows;
|
|
@@ -7688,19 +8537,22 @@ function classifyQueueHealth({
|
|
|
7688
8537
|
// over the same rows), so `base` already carries them.
|
|
7689
8538
|
const immediate = classifyQueueStarvation({
|
|
7690
8539
|
jobs: projectJobs, paused: false, runningCount: 0,
|
|
7691
|
-
lastRunAtMs
|
|
8540
|
+
lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now, thresholdMs: 0,
|
|
7692
8541
|
});
|
|
7693
8542
|
// pending.length is already > 0 above, so `immediate` can only be null when
|
|
7694
|
-
//
|
|
7695
|
-
//
|
|
8543
|
+
// the clock stamp is itself in the future (clock skew) — fall back to
|
|
8544
|
+
// computing idleMs the same way rather than asserting a kind we can't
|
|
7696
8545
|
// back up with a real number.
|
|
7697
8546
|
const idleMs = immediate ? immediate.idleMs
|
|
7698
|
-
: (
|
|
8547
|
+
: dispatchIdleMs({ lastRunAtMs, lastPauseClearedAtMs, schedulerBootedAtMs, now });
|
|
7699
8548
|
if (dispatchable === 0) return { ...base, kind: 'blocked', idleMs };
|
|
7700
8549
|
const kind = idleMs >= thresholdMs ? 'stalled' : 'running';
|
|
7701
8550
|
return { ...base, kind, idleMs };
|
|
7702
8551
|
}
|
|
7703
8552
|
|
|
8553
|
+
// Per-cwd latch for runQueueStarvationWatchdog: cwd → { kind, at, running }.
|
|
8554
|
+
const starvationLatch = new Map();
|
|
8555
|
+
|
|
7704
8556
|
/**
|
|
7705
8557
|
* The watchdog half: acts on classifyQueueStarvationByProject. Called from
|
|
7706
8558
|
* the heartbeat, which already runs on its own timer independent of the
|
|
@@ -7713,33 +8565,55 @@ function classifyQueueHealth({
|
|
|
7713
8565
|
* the tick itself is machine-wide (it drives whatever the picker finds
|
|
7714
8566
|
* across every project), only the DETECTION is per-project.
|
|
7715
8567
|
*/
|
|
7716
|
-
async function runQueueStarvationWatchdog(state, {
|
|
7717
|
-
|
|
7718
|
-
|
|
7719
|
-
|
|
7720
|
-
//
|
|
8568
|
+
async function runQueueStarvationWatchdog(state, {
|
|
8569
|
+
now = Date.now(), thresholdMs = QUEUE_STARVATION_MS,
|
|
8570
|
+
bootedAtMs = Date.parse(SCHEDULER_BOOTED_AT), pauseClearedAtMs = lastPauseClearedAt,
|
|
8571
|
+
} = {}) {
|
|
8572
|
+
// The idle clock is launch-keyed (dispatchIdleMs): NOT lastDispatchAttemptAt,
|
|
8573
|
+
// which tickQueue stamps before every gate — a queue whose 30 s loop ticks
|
|
8574
|
+
// and launches nothing would refresh it forever and the watchdog would
|
|
8575
|
+
// never fire. Rows held by an open launch breaker or the quietMachine lease
|
|
8576
|
+
// can't launch however often we tick, so they're passed in as `heldSlugs`
|
|
8577
|
+
// and read as 'blocked' (needs a human) rather than a false 'starved'.
|
|
8578
|
+
const heldSlugs = new Set((await computeLaunchHolds(state)).keys());
|
|
8579
|
+
if (quietMachineLease.isHeld()) {
|
|
8580
|
+
for (const j of state?.jobs ?? []) if (j?.status === 'pending' && j.quietMachine === true) heldSlugs.add(j.slug);
|
|
8581
|
+
}
|
|
7721
8582
|
const verdicts = classifyQueueStarvationByProject({
|
|
7722
8583
|
jobs: state?.jobs,
|
|
7723
|
-
paused: state
|
|
8584
|
+
paused: upgradeDrain.effectivePaused(state),
|
|
7724
8585
|
runningSet,
|
|
7725
|
-
lastRunAtMs: Date.parse(state?.
|
|
8586
|
+
lastRunAtMs: Date.parse(state?.lastRunAt ?? ''),
|
|
8587
|
+
lastPauseClearedAtMs: pauseClearedAtMs,
|
|
8588
|
+
schedulerBootedAtMs: bootedAtMs,
|
|
8589
|
+
heldSlugs,
|
|
7726
8590
|
now,
|
|
7727
8591
|
thresholdMs,
|
|
7728
8592
|
});
|
|
8593
|
+
// Latch: one episode per (cwd, kind) — re-arms only once QUEUE_STARVATION_MS
|
|
8594
|
+
// has elapsed again or the project's running count changes; forgotten the
|
|
8595
|
+
// moment the cwd stops having a verdict at all.
|
|
8596
|
+
const activeKeys = new Set(verdicts.map((v) => v.cwd));
|
|
8597
|
+
for (const cwd of [...starvationLatch.keys()]) if (!activeKeys.has(cwd)) starvationLatch.delete(cwd);
|
|
7729
8598
|
if (verdicts.length === 0) return null;
|
|
7730
8599
|
|
|
7731
8600
|
let anyStarved = false;
|
|
7732
8601
|
let primary = null;
|
|
7733
8602
|
for (const verdict of verdicts) {
|
|
7734
8603
|
const mins = Math.round(verdict.idleMs / 60_000);
|
|
8604
|
+
const running = (state?.jobs ?? []).filter((j) => j?.cwd === verdict.cwd && (j.status === 'running' || runningSet.has(j.slug))).length;
|
|
8605
|
+
const latched = starvationLatch.get(verdict.cwd);
|
|
8606
|
+
const suppressed = !!latched && latched.kind === verdict.kind && latched.running === running && now - latched.at < QUEUE_STARVATION_MS;
|
|
8607
|
+
if (!primary) primary = verdict;
|
|
8608
|
+
if (suppressed) continue;
|
|
8609
|
+
starvationLatch.set(verdict.cwd, { kind: verdict.kind, at: now, running });
|
|
7735
8610
|
if (verdict.kind === 'blocked') {
|
|
7736
8611
|
console.warn(
|
|
7737
8612
|
`[scheduler] QUEUE BLOCKED (${verdict.cwd}): ${verdict.pending} pending job(s), 0 running, idle ${mins}m — every ready row is behind a `
|
|
7738
|
-
+ `terminal or parked dependency, so ticking cannot help. Blockers: `
|
|
8613
|
+
+ `terminal or parked dependency, an open launch breaker, or the quietMachine lease, so ticking cannot help. Blockers: `
|
|
7739
8614
|
+ verdict.blockedChains.map((c) => `${c.cwd} [${c.blockedBy.join(', ')}]`).join(' · '),
|
|
7740
8615
|
);
|
|
7741
8616
|
appendAuditEvent('queue_blocked_stall', { cwd: verdict.cwd, pending: verdict.pending, idleMs: verdict.idleMs, chains: verdict.blockedChains });
|
|
7742
|
-
if (!primary) primary = verdict;
|
|
7743
8617
|
continue;
|
|
7744
8618
|
}
|
|
7745
8619
|
|
|
@@ -7756,9 +8630,12 @@ async function runQueueStarvationWatchdog(state, { now = Date.now(), thresholdMs
|
|
|
7756
8630
|
|
|
7757
8631
|
// A never-populated utilization reading is itself one of the ways the
|
|
7758
8632
|
// when-available path silently never fires (maybeLaunchWhenAvailable
|
|
7759
|
-
// returns early on null).
|
|
7760
|
-
//
|
|
7761
|
-
|
|
8633
|
+
// returns early on null). Absence of information, not a green light: fall
|
|
8634
|
+
// back to the same conservative degraded budget the poll loop itself uses
|
|
8635
|
+
// rather than a blind cachedUtilization=0.
|
|
8636
|
+
if (cachedUtilization === null || cachedUtilization === undefined) {
|
|
8637
|
+
applyDegradedBudget();
|
|
8638
|
+
}
|
|
7762
8639
|
// The in-process cancelToken is only ever reset by runDueJobs() (force-tick
|
|
7763
8640
|
// / run-now / resume-timer) — every other path that clears a pause
|
|
7764
8641
|
// (clearPause(), the poll loop's own auto-recovery) leaves it untouched
|
|
@@ -7924,6 +8801,72 @@ async function runBranchSweep(jobs) {
|
|
|
7924
8801
|
* skipped too (spawn may still be mid-flight) — see selectReapableJobs for
|
|
7925
8802
|
* the full predicate. Exported so unit tests can invoke it directly.
|
|
7926
8803
|
*/
|
|
8804
|
+
// Adopted-run supervision (see lib/adoptedRunSupervisor.cjs). In-memory by
|
|
8805
|
+
// design: the supervisors die with this process, and the next boot re-arms.
|
|
8806
|
+
const adoptedSupervisors = new Map();
|
|
8807
|
+
|
|
8808
|
+
/** Pid of a boot-time running row via the same ladder isBootRowAlive uses:
|
|
8809
|
+
* supervisor record → runtime.pid → pid spawned per the run log. */
|
|
8810
|
+
function bootRowPid(j, logPathOf) {
|
|
8811
|
+
const runDir = j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null;
|
|
8812
|
+
const record = runDir ? supervisorRecord.readSupervisorRecord(runDir, j.slug) : null;
|
|
8813
|
+
return record?.pid || j.runtime?.pid || readSpawnedPidFromLog(logPathOf(j)) || null;
|
|
8814
|
+
}
|
|
8815
|
+
|
|
8816
|
+
function signalAdoptedGroup(pgid, signal, pid) {
|
|
8817
|
+
try { process.kill(-pgid, signal); } catch {
|
|
8818
|
+
try { process.kill(pid, signal); } catch { /* already dead */ }
|
|
8819
|
+
}
|
|
8820
|
+
}
|
|
8821
|
+
|
|
8822
|
+
async function superviseAdoptedRunsPass(jobs) {
|
|
8823
|
+
try {
|
|
8824
|
+
const rows = jobs || (await readQueue()).jobs;
|
|
8825
|
+
const runDirOf = (j) => (j.runId ? path.join(schedulerPaths.runsDir(), j.runId) : null);
|
|
8826
|
+
return await adoptedRunSupervisor.superviseAdoptedRuns(rows, {
|
|
8827
|
+
registry: adoptedSupervisors,
|
|
8828
|
+
runDir: runDirOf,
|
|
8829
|
+
readRecord: supervisorRecord.readSupervisorRecord,
|
|
8830
|
+
lease: quietMachineLease,
|
|
8831
|
+
markSupervised: async (row) => {
|
|
8832
|
+
await mutate((s) => {
|
|
8833
|
+
const j = s.jobs.find((x) => x.slug === row.slug);
|
|
8834
|
+
if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) j.supervisedAt = new Date().toISOString();
|
|
8835
|
+
});
|
|
8836
|
+
},
|
|
8837
|
+
makeDeps: (row, record) => {
|
|
8838
|
+
const logPath = path.join(runDirOf(row), `${row.slug}.log`);
|
|
8839
|
+
return {
|
|
8840
|
+
logPath,
|
|
8841
|
+
statLogMtimeMs: readLogMtimeMs,
|
|
8842
|
+
pidAlive: claudePidAlive,
|
|
8843
|
+
identityOf: procIdentityOf,
|
|
8844
|
+
isDifferentProcess,
|
|
8845
|
+
killGroup: signalAdoptedGroup,
|
|
8846
|
+
// Stamped BEFORE the signal so reapDeadRunningJobs, which finalizes
|
|
8847
|
+
// the row once the process is gone, always sees why it died.
|
|
8848
|
+
stampKill: async (kind, reason) => {
|
|
8849
|
+
await mutate((s) => {
|
|
8850
|
+
const j = s.jobs.find((x) => x.slug === row.slug);
|
|
8851
|
+
if (j && j.status === 'running' && (j.runId ?? null) === (row.runId ?? null)) {
|
|
8852
|
+
j.adoptedKill = { watchdog: kind, reason, at: new Date().toISOString() };
|
|
8853
|
+
}
|
|
8854
|
+
});
|
|
8855
|
+
try { fs.appendFileSync(logPath, `\n[scheduler] adopted-run ${kind} watchdog: ${reason}\n`); } catch { /* best-effort */ }
|
|
8856
|
+
},
|
|
8857
|
+
log: (msg) => console.log(`[scheduler] ${row.slug}: ${msg}`),
|
|
8858
|
+
checkIntervalMs: IDLE_CHECK_INTERVAL_MS,
|
|
8859
|
+
sigkillAfterMs: POST_RESULT_KILL_MS,
|
|
8860
|
+
defaultMaxDurationMs: MAX_JOB_DURATION_MS,
|
|
8861
|
+
};
|
|
8862
|
+
},
|
|
8863
|
+
});
|
|
8864
|
+
} catch (e) {
|
|
8865
|
+
console.warn('[scheduler] adopted-run supervision pass failed', e?.message);
|
|
8866
|
+
return [];
|
|
8867
|
+
}
|
|
8868
|
+
}
|
|
8869
|
+
|
|
7927
8870
|
async function reapDeadRunningJobs() {
|
|
7928
8871
|
try {
|
|
7929
8872
|
// Do NOT gate on runningSet: spawnJob()'s finally block unconditionally
|
|
@@ -7932,9 +8875,13 @@ async function reapDeadRunningJobs() {
|
|
|
7932
8875
|
// status:"running" with no slug left in runningSet to trigger reconciliation.
|
|
7933
8876
|
// queue.json is the source of truth for which jobs are actually running.
|
|
7934
8877
|
const state = await readQueue();
|
|
8878
|
+
// A quarantined shard's rows never loaded; the filter is defence in depth
|
|
8879
|
+
// so a reap can never terminalize a row of a project we cannot persist.
|
|
8880
|
+
const reapSkip = quarantinedCwdSet(state);
|
|
8881
|
+
if (reapSkip.size > 0) state.jobs = state.jobs.filter((j) => !reapSkip.has(j.cwd));
|
|
7935
8882
|
// Shared by the log-evidence injections below and the reapable-processing
|
|
7936
8883
|
// loop further down — same `j.runId` → run log path formula either way.
|
|
7937
|
-
const logPathForJob = (j) => (j?.runId ? path.join(
|
|
8884
|
+
const logPathForJob = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
|
|
7938
8885
|
const { reapable, warnings, recovered } = selectReapableJobs(state.jobs, Date.now(), {
|
|
7939
8886
|
pidAlive: claudePidAlive,
|
|
7940
8887
|
grace: PIDLESS_SPAWN_GRACE_MS,
|
|
@@ -8008,8 +8955,12 @@ async function reapDeadRunningJobs() {
|
|
|
8008
8955
|
// the same still-active rate limit — the spin loop this PRD exists to
|
|
8009
8956
|
// stop. Done once, outside mutate(), before finalizing any row below.
|
|
8010
8957
|
if (dead.some((d) => d.outcome === 'rate_limited')) {
|
|
8958
|
+
// Same rationale as spawnJob's own rateLimited branch: a dead-process
|
|
8959
|
+
// reap that classifies as rate-limited is just as much an executor-
|
|
8960
|
+
// observed 429 as a live one, and must open the same shared circuit.
|
|
8961
|
+
billing.usageCircuit.recordFailure('executor_429');
|
|
8011
8962
|
const triggering = dead.find((d) => d.outcome === 'rate_limited');
|
|
8012
|
-
const billingResetIso = await
|
|
8963
|
+
const billingResetIso = await billingResetForPause();
|
|
8013
8964
|
const resetIso = resolveRateLimitPauseReset(triggering.logPath, billingResetIso);
|
|
8014
8965
|
const triggeringRow = triggering ? state.jobs.find((x) => x.slug === triggering.slug) : null;
|
|
8015
8966
|
const observedAtMs = triggeringRow?.startedAt ? Date.parse(triggeringRow.startedAt) : null;
|
|
@@ -8115,9 +9066,10 @@ async function reapDeadRunningJobs() {
|
|
|
8115
9066
|
// this runs the whole dead-job batch concurrently rather than one
|
|
8116
9067
|
// dispatch's git-spawn latency at a time.
|
|
8117
9068
|
await Promise.all(dead.map(async (d) => {
|
|
8118
|
-
if (d.outcome === 'rate_limited' || d.outcome === 'success') return;
|
|
8119
|
-
if (d.pidless && d.failureOverride) return;
|
|
8120
9069
|
const row = state.jobs.find((x) => x.slug === d.slug);
|
|
9070
|
+
const adoptedBudgetKill = row?.adoptedKill?.watchdog === 'budget';
|
|
9071
|
+
if (d.outcome === 'rate_limited' || (d.outcome === 'success' && !adoptedBudgetKill)) return;
|
|
9072
|
+
if (d.pidless && d.failureOverride) return;
|
|
8121
9073
|
if (!row?.landedCommit) return;
|
|
8122
9074
|
const rowCwd = row.cwd || state.config?.defaultCwd || DEFAULT_PROJECT_CWD;
|
|
8123
9075
|
const resolved = await resolveLandedCommitEvidence(rowCwd, row.landedCommit, row.startedAt);
|
|
@@ -8150,7 +9102,7 @@ async function reapDeadRunningJobs() {
|
|
|
8150
9102
|
const baseSet = new Set(s.jobs[idx].guardBaseline);
|
|
8151
9103
|
deltaPaths = after.filter((p) => !baseSet.has(p));
|
|
8152
9104
|
if (deltaPaths.length) {
|
|
8153
|
-
const salvagePath = path.join(
|
|
9105
|
+
const salvagePath = path.join(schedulerPaths.runsDir(), s.jobs[idx].runId, `${slug}.uncommitted.patch`);
|
|
8154
9106
|
const salvage = await jobWorktree.salvageJobDirtyDelta({ cwd: rowCwd, paths: deltaPaths, outFile: salvagePath });
|
|
8155
9107
|
if (salvage && salvage.ok) {
|
|
8156
9108
|
s.jobs[idx].salvagePatch = salvagePath;
|
|
@@ -8198,6 +9150,17 @@ async function reapDeadRunningJobs() {
|
|
|
8198
9150
|
if (pidless && !effectiveSuccess && !rateLimited && !notLandedInfo && failureOverride) {
|
|
8199
9151
|
notLandedInfo = { verdict: failureOverride.verdict, reason: failureOverride.reason };
|
|
8200
9152
|
}
|
|
9153
|
+
// An adopted run the budget watchdog killed (adoptedKill, stamped
|
|
9154
|
+
// before the signal) parks exactly like a native budget kill: the
|
|
9155
|
+
// shared classifyBudgetKill decides, and it wins over success/failure.
|
|
9156
|
+
const adoptedBudgetKill = rateLimited ? null : classifyBudgetKill({
|
|
9157
|
+
killedByWatchdog: s.jobs[idx].adoptedKill?.watchdog,
|
|
9158
|
+
budgetKillReason: s.jobs[idx].adoptedKill?.reason,
|
|
9159
|
+
}, landedCommitEvidence.get(slug) || null);
|
|
9160
|
+
if (adoptedBudgetKill) {
|
|
9161
|
+
effectiveSuccess = false;
|
|
9162
|
+
notLandedInfo = { verdict: 'budget_exceeded', reason: adoptedBudgetKill.reason };
|
|
9163
|
+
}
|
|
8201
9164
|
|
|
8202
9165
|
const leftoverSuffix = deltaPaths && deltaPaths.length
|
|
8203
9166
|
? ` — left ${deltaPaths.length} files uncommitted`
|
|
@@ -8249,6 +9212,7 @@ async function reapDeadRunningJobs() {
|
|
|
8249
9212
|
s.jobs[idx].gateOutcome = gateOutcome;
|
|
8250
9213
|
if (confirmedLandedCommit) s.jobs[idx].landedCommit = confirmedLandedCommit;
|
|
8251
9214
|
if (landedCommit) s.jobs[idx].landedCommit = landedCommit;
|
|
9215
|
+
if (adoptedBudgetKill?.landedCommit) s.jobs[idx].landedCommit = adoptedBudgetKill.landedCommit;
|
|
8252
9216
|
}
|
|
8253
9217
|
// A pidless spawn that never wrote its own '<slug>.log' into the
|
|
8254
9218
|
// batch runId dir it was stamped with must not keep that runId —
|
|
@@ -8263,6 +9227,7 @@ async function reapDeadRunningJobs() {
|
|
|
8263
9227
|
s.jobs[idx].runId = null;
|
|
8264
9228
|
}
|
|
8265
9229
|
delete s.jobs[idx].runtime;
|
|
9230
|
+
delete s.jobs[idx].adoptedKill;
|
|
8266
9231
|
delete s.jobs[idx].dispatchPhase;
|
|
8267
9232
|
delete s.jobs[idx].dispatchPhaseAt;
|
|
8268
9233
|
delete s.jobs[idx].overrun;
|
|
@@ -8327,14 +9292,21 @@ async function pollLoop() {
|
|
|
8327
9292
|
// 404/time-out and eventually pause the queue on 'network' — treat usage as
|
|
8328
9293
|
// wide-open and fire on pending + memory alone. (Blackrock-style machines.)
|
|
8329
9294
|
if (!billing.usageMeterApplicable()) {
|
|
8330
|
-
|
|
9295
|
+
// Close the shared circuit if a PRIOR consumer-auth session left it
|
|
9296
|
+
// open/half_open — this process has stopped polling the meter
|
|
9297
|
+
// entirely, so nothing else will ever call recordSuccess() to clear
|
|
9298
|
+
// it, and health.cjs would otherwise read a stale open circuit as
|
|
9299
|
+
// YELLOW/RED forever even though nothing is actually degraded.
|
|
9300
|
+
if (billing.usageCircuit.state() !== 'closed') billing.usageCircuit.recordSuccess({});
|
|
9301
|
+
cachedUtilization = NO_METER_UTILIZATION;
|
|
9302
|
+
degradedConcurrencyCapValue = null;
|
|
8331
9303
|
consecutiveFailures = 0;
|
|
8332
9304
|
backoffMs = 0;
|
|
8333
9305
|
backoffNextAt = null;
|
|
8334
9306
|
firstFailureAt = null;
|
|
8335
9307
|
firstNon429FailureAt = null;
|
|
8336
9308
|
lastFailureKind = null;
|
|
8337
|
-
|
|
9309
|
+
resetFailureStreak();
|
|
8338
9310
|
lastPollAt = Date.now();
|
|
8339
9311
|
lastPollOk = true;
|
|
8340
9312
|
persistSchedulerState();
|
|
@@ -8350,18 +9322,40 @@ async function pollLoop() {
|
|
|
8350
9322
|
return; // finally re-arms the timer
|
|
8351
9323
|
}
|
|
8352
9324
|
|
|
9325
|
+
// Shared breaker over the meter (AC1): while it is OPEN, no request is
|
|
9326
|
+
// made except the half-open probe below (billing.fetchUsage() is only
|
|
9327
|
+
// ever reached, further down, from the closed/half_open paths). "Meter
|
|
9328
|
+
// down" reads as absence of information, not a green light — the
|
|
9329
|
+
// conservative degraded budget stands in for both the utilization-
|
|
9330
|
+
// threshold gate (maybeLaunchWhenAvailable) and the concurrency cap
|
|
9331
|
+
// (tickQueue's freeSlots), never a blind cachedUtilization=0.
|
|
9332
|
+
if (billing.usageCircuit.state() === 'open') {
|
|
9333
|
+
applyDegradedBudget();
|
|
9334
|
+
lastPollAt = Date.now();
|
|
9335
|
+
lastPollOk = false;
|
|
9336
|
+
warnFailureStreakIfNeeded();
|
|
9337
|
+
persistSchedulerState();
|
|
9338
|
+
const cur = await readQueue();
|
|
9339
|
+
await maybeLaunchWhenAvailable(cur);
|
|
9340
|
+
await broadcast();
|
|
9341
|
+
return;
|
|
9342
|
+
}
|
|
9343
|
+
|
|
8353
9344
|
const r = await billing.fetchUsage();
|
|
8354
9345
|
|
|
8355
9346
|
if (r.kind === 'ok') {
|
|
8356
|
-
|
|
8357
|
-
|
|
9347
|
+
const window = bindingWindow(r.data?.usage);
|
|
9348
|
+
recordObservedReset(window.resets_at ?? null);
|
|
9349
|
+
cachedUtilization = Number.isFinite(window.utilization) ? window.utilization : cachedUtilization;
|
|
9350
|
+
lastGoodUsagePayload = r.data?.usage ?? lastGoodUsagePayload;
|
|
9351
|
+
degradedConcurrencyCapValue = null;
|
|
8358
9352
|
consecutiveFailures = 0;
|
|
8359
9353
|
backoffMs = 0;
|
|
8360
9354
|
backoffNextAt = null;
|
|
8361
9355
|
firstFailureAt = null;
|
|
8362
9356
|
firstNon429FailureAt = null;
|
|
8363
9357
|
lastFailureKind = null;
|
|
8364
|
-
|
|
9358
|
+
resetFailureStreak();
|
|
8365
9359
|
lastPollAt = Date.now();
|
|
8366
9360
|
lastPollOk = true;
|
|
8367
9361
|
persistSchedulerState();
|
|
@@ -8377,14 +9371,14 @@ async function pollLoop() {
|
|
|
8377
9371
|
await maybeLaunchWhenAvailable(cur);
|
|
8378
9372
|
await broadcast();
|
|
8379
9373
|
} else if (r.kind === 'meter_rate_limited') {
|
|
8380
|
-
// Billing meter is itself being rate-limited
|
|
8381
|
-
//
|
|
8382
|
-
//
|
|
8383
|
-
//
|
|
8384
|
-
//
|
|
8385
|
-
//
|
|
8386
|
-
//
|
|
8387
|
-
//
|
|
9374
|
+
// Billing meter is itself being rate-limited — absence of information,
|
|
9375
|
+
// not a green light. Still back off the POLL cadence itself (same
|
|
9376
|
+
// curve/cap as the transient branch) and persist state every cycle —
|
|
9377
|
+
// without this, a sustained 429 streak hammered the already-rate-
|
|
9378
|
+
// limited endpoint every POLL_INTERVAL_MS forever AND never wrote
|
|
9379
|
+
// lastPollAt/consecutiveFailures back to scheduler-state.json, so the
|
|
9380
|
+
// sidecar froze stale while the loop kept failing silently underneath
|
|
9381
|
+
// it (the 57-consecutive-failure incident).
|
|
8388
9382
|
lastPollAt = Date.now();
|
|
8389
9383
|
lastPollOk = false;
|
|
8390
9384
|
consecutiveFailures++;
|
|
@@ -8392,8 +9386,8 @@ async function pollLoop() {
|
|
|
8392
9386
|
// Don't update firstNon429FailureAt — 429s don't count toward the 30-min network-pause threshold.
|
|
8393
9387
|
backoffMs = nextBackoffMs(backoffMs);
|
|
8394
9388
|
backoffNextAt = Date.now() + backoffMs;
|
|
8395
|
-
|
|
8396
|
-
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on
|
|
9389
|
+
applyDegradedBudget();
|
|
9390
|
+
console.log(`[scheduler] billing meter rate-limited (HTTP 429) — firing on degraded budget (util=${cachedUtilization}%, cap=${degradedConcurrencyCapValue}) (failure #${consecutiveFailures}); retry in ${backoffMs / 1000}s`);
|
|
8397
9391
|
warnFailureStreakIfNeeded();
|
|
8398
9392
|
persistSchedulerState();
|
|
8399
9393
|
const cur = await readQueue();
|
|
@@ -8433,13 +9427,12 @@ async function pollLoop() {
|
|
|
8433
9427
|
// 'ok' and 'meter_rate_limited' branches used to reach
|
|
8434
9428
|
// maybeLaunchWhenAvailable, so auth/transient failures left ready
|
|
8435
9429
|
// pending work untouched until either the queue-starvation watchdog's
|
|
8436
|
-
// 10-minute safety net fired or the poll itself recovered.
|
|
8437
|
-
//
|
|
8438
|
-
//
|
|
8439
|
-
//
|
|
8440
|
-
//
|
|
8441
|
-
|
|
8442
|
-
if (cachedUtilization === null || cachedUtilization === undefined) cachedUtilization = 0;
|
|
9430
|
+
// 10-minute safety net fired or the poll itself recovered. Absence of
|
|
9431
|
+
// information, not a green light: fall back to the degraded budget
|
|
9432
|
+
// rather than a blind cachedUtilization=0. maybeLaunchWhenAvailable
|
|
9433
|
+
// itself still honors an 'auth'/'network' pause (state.paused), so
|
|
9434
|
+
// this is a no-op whenever setPaused() above actually engaged one.
|
|
9435
|
+
applyDegradedBudget();
|
|
8443
9436
|
await maybeLaunchWhenAvailable(await readQueue());
|
|
8444
9437
|
await broadcast();
|
|
8445
9438
|
}
|
|
@@ -8458,7 +9451,7 @@ async function pollLoop() {
|
|
|
8458
9451
|
// Same rationale as the auth/transient branch above: the outer catch
|
|
8459
9452
|
// must not be a silent dispatch dead-end either.
|
|
8460
9453
|
try {
|
|
8461
|
-
|
|
9454
|
+
applyDegradedBudget();
|
|
8462
9455
|
await maybeLaunchWhenAvailable(await readQueue());
|
|
8463
9456
|
await broadcast();
|
|
8464
9457
|
} catch { /* best-effort — the poll loop must still re-arm below */ }
|
|
@@ -8521,6 +9514,34 @@ function selectHistoryJobs(jobs, limit, historyEntries = []) {
|
|
|
8521
9514
|
// anyway). For non-fix-plan jobs the exemption never applies, so rescanning
|
|
8522
9515
|
// their pass_no_commit verdict is a harmless no-op (same facts, same verdict).
|
|
8523
9516
|
const RESCANNABLE_VERDICTS = new Set(['transcript_errors', 'verify_unavailable', 'no_verdict_sentinel', 'abandoned_background_task', 'pass_no_commit', 'pass_no_commit_already_shipped']);
|
|
9517
|
+
// RESCANNABLE_VERDICTS is a HINT, not a gate: it names the verdicts whose
|
|
9518
|
+
// recovery rung is a transcript re-verification (verifyRun). Every other
|
|
9519
|
+
// needs_review verdict is still a heal candidate (isRescanCandidate) — it just
|
|
9520
|
+
// gets the evidence-only rung (computeLooksDone) instead of a transcript
|
|
9521
|
+
// rescan, because verifyRun cannot see a commit-guard / shared-tree verdict and
|
|
9522
|
+
// would return 'clean' and falsely heal it.
|
|
9523
|
+
|
|
9524
|
+
// The ONLY needs_review verdicts NOT eligible for the periodic heal ladder.
|
|
9525
|
+
// An allow-list here was reopened three times (2026-09-12 x2, 2026-09-18
|
|
9526
|
+
// shared_tree_reverted) because a new park reason was born invisible to
|
|
9527
|
+
// self-healing. Add a verdict here only with a one-line proof that no
|
|
9528
|
+
// re-verification or evidence scan can ever change it.
|
|
9529
|
+
const RESCAN_EXCLUDED_VERDICTS = new Set([
|
|
9530
|
+
// Commit-guard verdict verifyRun never inspects: a rescan returns 'clean' and would heal genuinely unfinished work.
|
|
9531
|
+
'uncommitted_changes',
|
|
9532
|
+
// Its damage IS a commit stranded on an unmerged sm-job branch — a landedCommit restates it; selectMechanicalRecoveryTarget owns the real re-merge.
|
|
9533
|
+
'worktree_integration_failed',
|
|
9534
|
+
// The run overran its own time/cost estimate; no transcript or git evidence can un-overrun it (selectAutoFixTargets excludes it too).
|
|
9535
|
+
'budget_exceeded',
|
|
9536
|
+
]);
|
|
9537
|
+
|
|
9538
|
+
// Per-pass / per-row bounds on the evidence-only rung (the widened candidate
|
|
9539
|
+
// set). Each scan costs one computeLooksDone: a per-cwd-deduped `git fetch`
|
|
9540
|
+
// (<=~20s) + a git log. Unbounded, a backlog of N parked rows would pay N of
|
|
9541
|
+
// those every 10 minutes forever.
|
|
9542
|
+
const REVERIFY_INTERVAL_MS = 10 * 60_000;
|
|
9543
|
+
const EVIDENCE_SCAN_MAX_PER_PASS = 20;
|
|
9544
|
+
const EVIDENCE_SCAN_MIN_INTERVAL_MS = 6 * REVERIFY_INTERVAL_MS;
|
|
8524
9545
|
|
|
8525
9546
|
// Bounds fix-plan recursion: cap N permits at most N+1 fix jobs per original
|
|
8526
9547
|
// slug (depth 1 = the original job, depth 2 = its `-fix`, depth 3+ is
|
|
@@ -8563,7 +9584,7 @@ function isFixPlanBeyondDepthCap(slug, investigationDepth, isFixPlan) {
|
|
|
8563
9584
|
* check, no nested loop over user-scaled data. Dir names are ISO timestamps,
|
|
8564
9585
|
* so lexical-descending sort picks the newest match. Exported for tests.
|
|
8565
9586
|
*/
|
|
8566
|
-
function resolveRunId(job, { runsDir =
|
|
9587
|
+
function resolveRunId(job, { runsDir = schedulerPaths.runsDir() } = {}) {
|
|
8567
9588
|
if (!job || job.runId) return job?.runId || null;
|
|
8568
9589
|
if (!job.slug) return null;
|
|
8569
9590
|
let dirs;
|
|
@@ -8665,18 +9686,71 @@ function isGuardParkedWithoutAutoFix(job) {
|
|
|
8665
9686
|
return GUARD_VERDICT_EVIDENCE_ELIGIBLE.has(job.verifierVerdict);
|
|
8666
9687
|
}
|
|
8667
9688
|
|
|
9689
|
+
/**
|
|
9690
|
+
* Pure predicate, no I/O: a needs_review row whose auto-fix investigation
|
|
9691
|
+
* genuinely ran (autoFixAttempted === true) but whose outcome was NEVER
|
|
9692
|
+
* durably stamped at all — and that has nothing left in flight to wait on:
|
|
9693
|
+
* no live/queued fix-plan row at fixSlugFor(job).
|
|
9694
|
+
*
|
|
9695
|
+
* Distinct from isExhaustedAutoFix, which requires autoFixRetries >= 1 to
|
|
9696
|
+
* have already accumulated. spawnInvestigation's onExit handler restores the
|
|
9697
|
+
* job's status from 'investigating' back to needs_review in ONE mutate()
|
|
9698
|
+
* call (scheduler.cjs's spawnInvestigation, source
|
|
9699
|
+
* 'spawnInvestigation:onExit') and stamps autoFixOutcome ('plan' / 'no-plan'
|
|
9700
|
+
* / 'error') in a SEPARATE, later mutate() call — an app restart or process
|
|
9701
|
+
* death between the two leaves autoFixOutcome permanently unset, with
|
|
9702
|
+
* autoFixRetries never incremented either, so isExhaustedAutoFix never fires
|
|
9703
|
+
* and the row falls through every existing resolving door forever, re-scanned
|
|
9704
|
+
* by the periodic reverify pass against the same frozen transcript with no
|
|
9705
|
+
* new outcome to observe.
|
|
9706
|
+
*
|
|
9707
|
+
* Job 1218-fo-01 (2026-09-13, findings filed at
|
|
9708
|
+
* session-manager-operations/reviews/2026-09-13-scheduler-stability-investigation.md,
|
|
9709
|
+
* "post-run adjudication" section) sat exactly in this state: needs_review,
|
|
9710
|
+
* verifierVerdict transcript_errors, autoFixAttempted: true, autoFixOutcome:
|
|
9711
|
+
* undefined, autoFixRetries: undefined, statusHistory ending in
|
|
9712
|
+
* "investigation probe exited — restoring prior status" — with a landed
|
|
9713
|
+
* commit no existing ladder rung would credit.
|
|
9714
|
+
*
|
|
9715
|
+
* Deliberately narrower than "unset, 'error', or 'no-plan'": a row that DID
|
|
9716
|
+
* get a durably-stamped 'error'/'no-plan' outcome with its one bounded retry
|
|
9717
|
+
* still unspent (autoFixRetries < 1) is exactly the row
|
|
9718
|
+
* selectAutoFixTargets's own retryEligible check still owns and will retry
|
|
9719
|
+
* on its own — pulling it into THIS ladder instead would race it away from
|
|
9720
|
+
* that retry (scheduler-needs-review-autoresolve.test.cjs's "a non-exhausted
|
|
9721
|
+
* needs_review row … is left alone" guards exactly this). Only the
|
|
9722
|
+
* outcome-truly-never-stamped case is structurally unrecoverable by any
|
|
9723
|
+
* OTHER existing door, because nothing ever wrote a value selectAutoFixTargets
|
|
9724
|
+
* or isExhaustedAutoFix could act on.
|
|
9725
|
+
* Exported for tests.
|
|
9726
|
+
*/
|
|
9727
|
+
function isStrandedAutoFixPark(job, jobsInProject) {
|
|
9728
|
+
if (!job || job.status !== 'needs_review') return false;
|
|
9729
|
+
if (job.autoFixAttempted !== true) return false;
|
|
9730
|
+
if (job.autoFixOutcome != null) return false;
|
|
9731
|
+
const fixSlug = fixSlugFor(job);
|
|
9732
|
+
const liveOrQueuedChild = (jobsInProject || []).some(
|
|
9733
|
+
(j) => j.slug === fixSlug && j.status !== 'completed' && !DEAD_FIX_CHILD_STATUSES.has(j.status),
|
|
9734
|
+
);
|
|
9735
|
+
return !liveOrQueuedChild;
|
|
9736
|
+
}
|
|
9737
|
+
|
|
8668
9738
|
/**
|
|
8669
9739
|
* Pure predicate, no I/O: is this needs_review row eligible for the bounded
|
|
8670
9740
|
* auto-resolve ladder at all — either because its auto-fix path is genuinely
|
|
8671
|
-
* spent (isExhaustedAutoFix),
|
|
8672
|
-
*
|
|
8673
|
-
*
|
|
8674
|
-
*
|
|
8675
|
-
*
|
|
9741
|
+
* spent (isExhaustedAutoFix), because it was parked by a GUARD verdict that
|
|
9742
|
+
* never entered auto-fix in the first place (isGuardParkedWithoutAutoFix),
|
|
9743
|
+
* or because its auto-fix investigation ran but was stranded before
|
|
9744
|
+
* recording any outcome (isStrandedAutoFixPark). All three classes share ONE
|
|
9745
|
+
* ladder (applyNeedsReviewAutoResolve) rather than a duplicated one — the
|
|
9746
|
+
* ladder itself doesn't care which door a row came through, only whether it
|
|
9747
|
+
* now carries completion evidence (job.looksDone). `jobsInProject` is only
|
|
9748
|
+
* consulted by isStrandedAutoFixPark (to check for a live/queued fix-plan
|
|
9749
|
+
* child) and defaults to empty so existing single-arg callers are unaffected.
|
|
8676
9750
|
* Exported for tests.
|
|
8677
9751
|
*/
|
|
8678
|
-
function isEligibleForNeedsReviewAutoResolve(job) {
|
|
8679
|
-
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job);
|
|
9752
|
+
function isEligibleForNeedsReviewAutoResolve(job, jobsInProject = []) {
|
|
9753
|
+
return isExhaustedAutoFix(job) || isGuardParkedWithoutAutoFix(job) || isStrandedAutoFixPark(job, jobsInProject);
|
|
8680
9754
|
}
|
|
8681
9755
|
|
|
8682
9756
|
/**
|
|
@@ -8789,18 +9863,55 @@ function isFailedUnverifiedShaped(job) {
|
|
|
8789
9863
|
if (job.verifierVerdict && RESCANNABLE_VERDICTS.has(job.verifierVerdict)) return true;
|
|
8790
9864
|
const runId = job.runId || resolveRunId(job);
|
|
8791
9865
|
if (!runId) return false;
|
|
8792
|
-
const logPath = path.join(
|
|
9866
|
+
const logPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.log`);
|
|
8793
9867
|
return classifyRunOutcome(logPath) === 'no_result';
|
|
8794
9868
|
}
|
|
8795
9869
|
|
|
8796
9870
|
function isRescanCandidate(job) {
|
|
8797
9871
|
if (!job) return false;
|
|
9872
|
+
// Default-ELIGIBLE: every needs_review row is a heal candidate unless its
|
|
9873
|
+
// verdict is in RESCAN_EXCLUDED_VERDICTS. No runId requirement here — a row
|
|
9874
|
+
// without one still gets the evidence rung and the unresolvable annotation.
|
|
9875
|
+
if (job.status === 'needs_review') return !RESCAN_EXCLUDED_VERDICTS.has(job.verifierVerdict);
|
|
8798
9876
|
if (!(job.runId || resolveRunId(job))) return false;
|
|
8799
|
-
if (job.status === 'needs_review') return RESCANNABLE_VERDICTS.has(job.verifierVerdict);
|
|
8800
9877
|
if (job.status === 'failed') return isFailedUnverifiedShaped(job);
|
|
8801
9878
|
return false;
|
|
8802
9879
|
}
|
|
8803
9880
|
|
|
9881
|
+
/**
|
|
9882
|
+
* Which rung a needs_review candidate gets (RESCANNABLE_VERDICTS as a hint):
|
|
9883
|
+
* true = transcript re-verification (needs a run dir to read); false = the
|
|
9884
|
+
* evidence-only rung. I/O only when a rescannable-verdict row lacks a runId.
|
|
9885
|
+
*/
|
|
9886
|
+
function isTranscriptRescannable(job) {
|
|
9887
|
+
return !!job && RESCANNABLE_VERDICTS.has(job.verifierVerdict) && !!(job.runId || resolveRunId(job));
|
|
9888
|
+
}
|
|
9889
|
+
|
|
9890
|
+
/**
|
|
9891
|
+
* Pure, no I/O: the bounded subset of evidence-only needs_review candidates
|
|
9892
|
+
* reverifyNeedsReview scans this pass. Skips rows already carrying looksDone,
|
|
9893
|
+
* rows scanned within EVIDENCE_SCAN_MIN_INTERVAL_MS (evidenceScannedAt), and —
|
|
9894
|
+
* PRD 1136 — rows with a live auto-fix history unless they are an
|
|
9895
|
+
* auto-resolve door (isEligibleForNeedsReviewAutoResolve, which is what
|
|
9896
|
+
* consumes looksDone). Never-scanned rows go first, then least-recently
|
|
9897
|
+
* scanned; capped at EVIDENCE_SCAN_MAX_PER_PASS. O(n log n) in needs_review rows.
|
|
9898
|
+
* Per-pass cost ceiling: EVIDENCE_SCAN_MAX_PER_PASS computeLooksDone calls.
|
|
9899
|
+
*/
|
|
9900
|
+
function selectEvidenceScanTargets(jobs, now = Date.now()) {
|
|
9901
|
+
const due = [];
|
|
9902
|
+
for (const j of jobs ?? []) {
|
|
9903
|
+
if (j.status !== 'needs_review' || !isRescanCandidate(j)) continue;
|
|
9904
|
+
if (isTranscriptRescannable(j)) continue;
|
|
9905
|
+
if (j.looksDone) continue;
|
|
9906
|
+
if (j.autoFixAttempted === true && !isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
|
|
9907
|
+
const last = Date.parse(j.evidenceScannedAt ?? '');
|
|
9908
|
+
if (!Number.isNaN(last) && now - last < EVIDENCE_SCAN_MIN_INTERVAL_MS) continue;
|
|
9909
|
+
due.push({ j, last: Number.isNaN(last) ? 0 : last });
|
|
9910
|
+
}
|
|
9911
|
+
due.sort((a, b) => a.last - b.last);
|
|
9912
|
+
return due.slice(0, EVIDENCE_SCAN_MAX_PER_PASS).map((d) => d.j);
|
|
9913
|
+
}
|
|
9914
|
+
|
|
8804
9915
|
/**
|
|
8805
9916
|
* Cheap-guard for the 10-minute periodic reverify tick. MUST be expressed in
|
|
8806
9917
|
* terms of isRescanCandidate — not a hand-written status test — because the
|
|
@@ -8834,6 +9945,14 @@ function isRescanCandidate(job) {
|
|
|
8834
9945
|
* that function). Same rule as always: never let this guard be narrower than
|
|
8835
9946
|
* the work reverifyNeedsReview actually performs.
|
|
8836
9947
|
*
|
|
9948
|
+
* Reopened a THIRD time 2026-09-18 (shared_tree_reverted parked 1229-fo-03
|
|
9949
|
+
* falsely, 19 of 20 pending rows held): the fix was not another OR-clause but
|
|
9950
|
+
* inverting the default — isRescanCandidate is now default-ELIGIBLE for every
|
|
9951
|
+
* needs_review row (RESCAN_EXCLUDED_VERDICTS names the few exceptions), so a
|
|
9952
|
+
* new park reason can never again be born unhealable. The OR-clauses below
|
|
9953
|
+
* are now redundant for needs_review rows and kept only for their
|
|
9954
|
+
* non-needs_review inputs.
|
|
9955
|
+
*
|
|
8837
9956
|
* Cost: selectMechanicalRecoveryTarget/selectResumeRecoveryTarget and
|
|
8838
9957
|
* isGuardParkedWithoutAutoFix are pure (no I/O). selectAutoFixTargets is
|
|
8839
9958
|
* called with an injected fixSlugExists that always returns false — cheap
|
|
@@ -8845,7 +9964,7 @@ function isRescanCandidate(job) {
|
|
|
8845
9964
|
*/
|
|
8846
9965
|
function shouldRunPeriodicReverify(jobs) {
|
|
8847
9966
|
if (!Array.isArray(jobs)) return false;
|
|
8848
|
-
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j))) return true;
|
|
9967
|
+
if (jobs.some((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j))) return true;
|
|
8849
9968
|
if (jobs.some((j) => selectMechanicalRecoveryTarget(j) || selectResumeRecoveryTarget(j))) return true;
|
|
8850
9969
|
return selectAutoFixTargets(jobs, { fixSlugExists: () => false }).length > 0;
|
|
8851
9970
|
}
|
|
@@ -9012,11 +10131,12 @@ function needsReviewAutoResolveDisabled() {
|
|
|
9012
10131
|
* [{ slug, cwd, ageMs, attempts }]
|
|
9013
10132
|
*
|
|
9014
10133
|
* Pure selector — no IO. Selects `needs_review` rows eligible for the
|
|
9015
|
-
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve —
|
|
9016
|
-
*
|
|
9017
|
-
* auto-fix
|
|
9018
|
-
* 'needs_review'` is older than
|
|
9019
|
-
* exhaustedResolveAttempts counter has not yet
|
|
10134
|
+
* bounded auto-resolve ladder (isEligibleForNeedsReviewAutoResolve — auto-fix
|
|
10135
|
+
* genuinely spent, parked by a GUARD verdict that never entered auto-fix at
|
|
10136
|
+
* all, or a stranded auto-fix park with no outcome ever recorded), whose
|
|
10137
|
+
* newest statusHistory entry with `to === 'needs_review'` is older than
|
|
10138
|
+
* `thresholdMs`, and whose exhaustedResolveAttempts counter has not yet
|
|
10139
|
+
* spent its cap.
|
|
9020
10140
|
*
|
|
9021
10141
|
* The inclusion bound is inclusive of the cap itself (`<= CAP`, not `<
|
|
9022
10142
|
* CAP`): NEEDS_REVIEW_RESOLVE_CAP counts REQUEUE attempts already spent, and
|
|
@@ -9029,7 +10149,7 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
9029
10149
|
const targets = [];
|
|
9030
10150
|
for (const j of jobs ?? []) {
|
|
9031
10151
|
if (j.status !== 'needs_review') continue;
|
|
9032
|
-
if (!isEligibleForNeedsReviewAutoResolve(j)) continue;
|
|
10152
|
+
if (!isEligibleForNeedsReviewAutoResolve(j, jobs)) continue;
|
|
9033
10153
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) continue;
|
|
9034
10154
|
const history = j.statusHistory || [];
|
|
9035
10155
|
let entry = null;
|
|
@@ -9073,8 +10193,8 @@ function selectExhaustedNeedsReviewTargets(jobs, now, thresholdMs) {
|
|
|
9073
10193
|
* reason text (and the Queue UI's job.error) name the RIGHT evidence — a
|
|
9074
10194
|
* guard-parked row was never "exhausted auto-fix" and must never claim to be.
|
|
9075
10195
|
*/
|
|
9076
|
-
function applyNeedsReviewAutoResolve(j) {
|
|
9077
|
-
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j)) return null;
|
|
10196
|
+
function applyNeedsReviewAutoResolve(j, jobsInProject = []) {
|
|
10197
|
+
if (!j || j.status !== 'needs_review' || !isEligibleForNeedsReviewAutoResolve(j, jobsInProject)) return null;
|
|
9078
10198
|
if ((j.exhaustedResolveAttempts ?? 0) > NEEDS_REVIEW_RESOLVE_CAP) return null;
|
|
9079
10199
|
const originIsGuardParked = !isExhaustedAutoFix(j) && isGuardParkedWithoutAutoFix(j);
|
|
9080
10200
|
|
|
@@ -9356,6 +10476,55 @@ async function computeLooksDone(job, fetchedCwds) {
|
|
|
9356
10476
|
return { commits: attributed.commits, paths, detectedAt: new Date().toISOString(), rule: attributed.rule };
|
|
9357
10477
|
}
|
|
9358
10478
|
|
|
10479
|
+
/**
|
|
10480
|
+
* Shadow gate (observation only): run a needs_review row's authored gate at
|
|
10481
|
+
* the project's current HEAD and record what it WOULD have decided as
|
|
10482
|
+
* `gateShadow` on the verdicts sidecar and the row. Changes NO status, takes
|
|
10483
|
+
* no slot (not a claude -p run — runGateSequence keeps one shadow gate in
|
|
10484
|
+
* flight machine-wide). Never called from finalize: only the reverify pass.
|
|
10485
|
+
* Returns the recorded gateShadow, or null when nothing was recorded (already
|
|
10486
|
+
* recorded at this HEAD, PRD unreadable, or another shadow gate is running).
|
|
10487
|
+
*/
|
|
10488
|
+
async function runGateShadow(job) {
|
|
10489
|
+
if (!job || !job.slug || !job.cwd) return null;
|
|
10490
|
+
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
10491
|
+
let prdText;
|
|
10492
|
+
try { prdText = fs.readFileSync(prdPath, 'utf8'); } catch { return null; }
|
|
10493
|
+
const head = await gitHead(job.cwd);
|
|
10494
|
+
if (job.gateShadow && job.gateShadow.head === head) return null;
|
|
10495
|
+
const gate = resolveGate(prdText);
|
|
10496
|
+
let outcome;
|
|
10497
|
+
if (gate.source === 'none') outcome = { status: 'unavailable', reason: 'gate-opt-out', results: [] };
|
|
10498
|
+
else if (!gate.sequence.length) outcome = { status: 'unavailable', reason: 'no-parseable-gate', results: [] };
|
|
10499
|
+
else {
|
|
10500
|
+
const r = await runGateSequence(gate.sequence, { cwd: job.cwd });
|
|
10501
|
+
if (r.status === 'busy') return null;
|
|
10502
|
+
outcome = r;
|
|
10503
|
+
}
|
|
10504
|
+
const gateShadow = { ...outcome, head, source: gate.source, ranAt: new Date().toISOString() };
|
|
10505
|
+
const runId = job.runId || resolveRunId(job);
|
|
10506
|
+
if (runId) {
|
|
10507
|
+
const verdictsPath = path.join(schedulerPaths.runsDir(), runId, `${job.slug}.verdicts.json`);
|
|
10508
|
+
// Read-merge (single-writer law: runVerify owns the sidecar's other keys).
|
|
10509
|
+
// Only merge into an existing run dir — never conjure one.
|
|
10510
|
+
if (fs.existsSync(path.dirname(verdictsPath))) {
|
|
10511
|
+
let existing = {};
|
|
10512
|
+
try { existing = JSON.parse(fs.readFileSync(verdictsPath, 'utf8')) || {}; } catch { /* absent/unparseable → fresh */ }
|
|
10513
|
+
try { atomicWriteJsonSync(verdictsPath, { ...existing, gateShadow }); } catch { /* best-effort */ }
|
|
10514
|
+
}
|
|
10515
|
+
}
|
|
10516
|
+
await mutate((s) => {
|
|
10517
|
+
for (const j of s.jobs) {
|
|
10518
|
+
if (j.slug === job.slug && j.status === 'needs_review') j.gateShadow = gateShadow;
|
|
10519
|
+
}
|
|
10520
|
+
});
|
|
10521
|
+
await broadcast();
|
|
10522
|
+
return gateShadow;
|
|
10523
|
+
}
|
|
10524
|
+
|
|
10525
|
+
// Tail of the last background shadow gate — lets tests (and only tests) await it.
|
|
10526
|
+
let gateShadowPending = null;
|
|
10527
|
+
|
|
9359
10528
|
async function reverifyNeedsReview() {
|
|
9360
10529
|
const snap = await readQueue();
|
|
9361
10530
|
// isGuardParkedWithoutAutoFix rows are NOT isRescanCandidate (their
|
|
@@ -9365,7 +10534,7 @@ async function reverifyNeedsReview() {
|
|
|
9365
10534
|
// guard-verdict auto-resolve gap this PRD closes. Handled in its own
|
|
9366
10535
|
// branch below (no transcript rescan — there is no transcript verdict to
|
|
9367
10536
|
// rescan) rather than through the isRescanCandidate machinery.
|
|
9368
|
-
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j));
|
|
10537
|
+
const candidates = snap.jobs.filter((j) => isRescanCandidate(j) || isGuardParkedWithoutAutoFix(j) || isStaleSharedTreeRevertedPark(j));
|
|
9369
10538
|
const healed = [];
|
|
9370
10539
|
const leftForReview = [];
|
|
9371
10540
|
const looksDoneUpdates = [];
|
|
@@ -9373,12 +10542,33 @@ async function reverifyNeedsReview() {
|
|
|
9373
10542
|
// the `git fetch --all --prune` per distinct cwd (see computeLooksDone's
|
|
9374
10543
|
// header) rather than re-fetching the same repo once per candidate row.
|
|
9375
10544
|
const fetchedCwds = new Set();
|
|
10545
|
+
const evidenceSlugs = new Set(selectEvidenceScanTargets(snap.jobs).map((j) => j.slug));
|
|
10546
|
+
const evidenceScanned = [];
|
|
9376
10547
|
for (const job of candidates) {
|
|
9377
|
-
if (
|
|
9378
|
-
//
|
|
9379
|
-
//
|
|
9380
|
-
//
|
|
9381
|
-
|
|
10548
|
+
if (isStaleSharedTreeRevertedPark(job)) {
|
|
10549
|
+
// Re-apply the corrected shared-tree check: the row's own landedCommit
|
|
10550
|
+
// (this dispatch's, per resolveLandedCommitEvidence) still being an
|
|
10551
|
+
// ancestor of HEAD means the park was a false positive — heal it.
|
|
10552
|
+
const cwd = job.cwd || DEFAULT_PROJECT_CWD;
|
|
10553
|
+
if (await resolveLandedCommitEvidence(cwd, job.landedCommit, job.startedAt)
|
|
10554
|
+
&& await module.exports.landedCommitIsAncestorOfHead(cwd, job.landedCommit)) {
|
|
10555
|
+
healed.push(job.slug);
|
|
10556
|
+
continue;
|
|
10557
|
+
}
|
|
10558
|
+
if (!isRescanCandidate(job) && !isGuardParkedWithoutAutoFix(job)) {
|
|
10559
|
+
leftForReview.push({ slug: job.slug, reason: 'shared_tree_reverted: landed commit not an ancestor of HEAD' });
|
|
10560
|
+
continue;
|
|
10561
|
+
}
|
|
10562
|
+
}
|
|
10563
|
+
if (job.status === 'needs_review' && !isTranscriptRescannable(job)) {
|
|
10564
|
+
// Any needs_review row whose verdict is not a transcript-verifier one
|
|
10565
|
+
// (a guard verdict, a not-yet-invented verdict, a stranded auto-fix
|
|
10566
|
+
// park): only evidence gathering, never a transcript rescan (verifyRun
|
|
10567
|
+
// would call it clean) and never a direct heal —
|
|
10568
|
+
// applyNeedsReviewAutoResolve is the sole place that turns this
|
|
10569
|
+
// annotation into a status change. Bounded by selectEvidenceScanTargets.
|
|
10570
|
+
if (!evidenceSlugs.has(job.slug)) continue;
|
|
10571
|
+
evidenceScanned.push(job.slug);
|
|
9382
10572
|
const looksDone = await computeLooksDone(job, fetchedCwds);
|
|
9383
10573
|
if (looksDone) {
|
|
9384
10574
|
looksDoneUpdates.push({ slug: job.slug, cwd: job.cwd, looksDone, fromFailed: false });
|
|
@@ -9401,7 +10591,7 @@ async function reverifyNeedsReview() {
|
|
|
9401
10591
|
}
|
|
9402
10592
|
continue;
|
|
9403
10593
|
}
|
|
9404
|
-
const runDir = path.join(
|
|
10594
|
+
const runDir = path.join(schedulerPaths.runsDir(), job.runId || resolveRunId(job));
|
|
9405
10595
|
const prdPath = (await resolveVerifyPrdPath(job)) ?? archivedPrdPathForJob(job);
|
|
9406
10596
|
// Derive committedDuringRun from the recorded run window. The live
|
|
9407
10597
|
// commit-guard uses gitHead() (before/after HEAD diff); here the run is
|
|
@@ -9424,6 +10614,13 @@ async function reverifyNeedsReview() {
|
|
|
9424
10614
|
committedDuringRun,
|
|
9425
10615
|
allowPreSentinelHeal: true,
|
|
9426
10616
|
priorLandedCommit,
|
|
10617
|
+
// job.landedCommit is THIS row's own last-run attribution (stamped by
|
|
10618
|
+
// spawnJob's finalize, survives resetJobFields) — the same
|
|
10619
|
+
// ground-truth-outranks-heuristics evidence spawnJob passes live,
|
|
10620
|
+
// just read back post-hoc since there is no in-flight guardHeadBefore/
|
|
10621
|
+
// headAtExit pair to recompute for an already-terminal row.
|
|
10622
|
+
jobLandedCommitThisRun: job.landedCommit ?? null,
|
|
10623
|
+
exitCode: job.exitCode ?? null,
|
|
9427
10624
|
});
|
|
9428
10625
|
} catch { leftForReview.push({ slug: job.slug, reason: 'verifyRun threw' }); continue; }
|
|
9429
10626
|
const refusal = healRefusalReason(job, v, committedDuringRun);
|
|
@@ -9454,6 +10651,23 @@ async function reverifyNeedsReview() {
|
|
|
9454
10651
|
}
|
|
9455
10652
|
}
|
|
9456
10653
|
}
|
|
10654
|
+
// Shadow gate (observation only): at most ONE needs_review row per pass,
|
|
10655
|
+
// fired in the background so a 15-minute gate never stalls this pass.
|
|
10656
|
+
if (!gateShadowPending && process.env.SM_GATE_SHADOW_DISABLE !== '1') {
|
|
10657
|
+
const gateTarget = snap.jobs.find((j) => j.status === 'needs_review' && !j.gateShadow);
|
|
10658
|
+
if (gateTarget) {
|
|
10659
|
+
gateShadowPending = runGateShadow(gateTarget)
|
|
10660
|
+
.catch((e) => { console.error('[scheduler] gate shadow error', gateTarget.slug, e); })
|
|
10661
|
+
.finally(() => { gateShadowPending = null; });
|
|
10662
|
+
}
|
|
10663
|
+
}
|
|
10664
|
+
if (evidenceScanned.length) {
|
|
10665
|
+
const scannedSet = new Set(evidenceScanned);
|
|
10666
|
+
const stamp = new Date().toISOString();
|
|
10667
|
+
await mutate((s) => {
|
|
10668
|
+
for (const j of s.jobs) if (scannedSet.has(j.slug) && j.status === 'needs_review') j.evidenceScannedAt = stamp;
|
|
10669
|
+
});
|
|
10670
|
+
}
|
|
9457
10671
|
if (looksDoneUpdates.length) {
|
|
9458
10672
|
const bySlug = new Map(looksDoneUpdates.map((u) => [u.slug, u]));
|
|
9459
10673
|
await mutate((s) => {
|
|
@@ -9666,7 +10880,7 @@ async function reverifyNeedsReview() {
|
|
|
9666
10880
|
});
|
|
9667
10881
|
for (const job of targets) {
|
|
9668
10882
|
const runId = job.runId || resolveRunId(job);
|
|
9669
|
-
const runDir = path.join(
|
|
10883
|
+
const runDir = path.join(schedulerPaths.runsDir(), runId);
|
|
9670
10884
|
const isRetryAttempt = job.autoFixAttempted === true;
|
|
9671
10885
|
const isDeadFixPlanReopen = isFixPlanDead(job, queueForResumeAndAutofix.jobs);
|
|
9672
10886
|
const deadChild = isDeadFixPlanReopen
|
|
@@ -9813,12 +11027,14 @@ function registerScheduleHandlers() {
|
|
|
9813
11027
|
const freeSlots = Math.max(0, slotSnapshot.total - slotSnapshot.inUse);
|
|
9814
11028
|
const verdict = classifyQueueHealth({
|
|
9815
11029
|
jobs: state.jobs,
|
|
9816
|
-
paused: state
|
|
11030
|
+
paused: upgradeDrain.effectivePaused(state),
|
|
9817
11031
|
launchBlocks: state.launchBlocks,
|
|
9818
11032
|
runningSet,
|
|
9819
11033
|
freeSlots,
|
|
9820
11034
|
totalSlots: slotSnapshot.total,
|
|
9821
|
-
|
|
11035
|
+
lastRunAtMs: Date.parse(state.lastRunAt ?? ''),
|
|
11036
|
+
lastPauseClearedAtMs: lastPauseClearedAt,
|
|
11037
|
+
schedulerBootedAtMs: Date.parse(SCHEDULER_BOOTED_AT),
|
|
9822
11038
|
now,
|
|
9823
11039
|
cwd,
|
|
9824
11040
|
});
|
|
@@ -9949,6 +11165,11 @@ function registerScheduleHandlers() {
|
|
|
9949
11165
|
return { ok: true };
|
|
9950
11166
|
});
|
|
9951
11167
|
|
|
11168
|
+
ipcMain.handle('schedule:pause', async () => {
|
|
11169
|
+
await setPaused('manual', null);
|
|
11170
|
+
return { ok: true };
|
|
11171
|
+
});
|
|
11172
|
+
|
|
9952
11173
|
ipcMain.handle('schedule:resume', async () => {
|
|
9953
11174
|
await clearPause('manual');
|
|
9954
11175
|
return { ok: true };
|
|
@@ -9980,7 +11201,7 @@ function registerScheduleHandlers() {
|
|
|
9980
11201
|
ipcMain.handle('schedule:clear-queue', async () => {
|
|
9981
11202
|
ensureDirs();
|
|
9982
11203
|
const ts = new Date().toISOString().replace(/[:.]/g, '-');
|
|
9983
|
-
const archiveDir = path.join(
|
|
11204
|
+
const archiveDir = path.join(schedulerPaths.scheduledPlansRoot(), 'prds-archived', ts);
|
|
9984
11205
|
const state = await readQueue();
|
|
9985
11206
|
const victims = state.jobs.filter((j) => j.status !== 'running');
|
|
9986
11207
|
if (victims.length === 0) {
|
|
@@ -10027,7 +11248,7 @@ function registerScheduleHandlers() {
|
|
|
10027
11248
|
|
|
10028
11249
|
ipcMain.handle('schedule:open-folder', async () => {
|
|
10029
11250
|
const { shell } = require('electron');
|
|
10030
|
-
await shell.openPath(
|
|
11251
|
+
await shell.openPath(schedulerPaths.scheduledPlansRoot());
|
|
10031
11252
|
return { ok: true };
|
|
10032
11253
|
});
|
|
10033
11254
|
|
|
@@ -10045,8 +11266,8 @@ function registerScheduleHandlers() {
|
|
|
10045
11266
|
ipcMain.handle('schedule:read-log', validated(schemas.scheduleReadLog, async ({ slug, runId }) => {
|
|
10046
11267
|
// Defense-in-depth: re-check containment after path.resolve even though
|
|
10047
11268
|
// SLUG_RE / RUN_ID_RE already forbid path separators.
|
|
10048
|
-
const logPath = path.resolve(path.join(
|
|
10049
|
-
if (!logPath.startsWith(
|
|
11269
|
+
const logPath = path.resolve(path.join(schedulerPaths.runsDir(), runId, `${slug}.log`));
|
|
11270
|
+
if (!logPath.startsWith(schedulerPaths.runsDir() + path.sep)) {
|
|
10050
11271
|
return { ok: false, error: 'invalid slug or runId' };
|
|
10051
11272
|
}
|
|
10052
11273
|
try {
|
|
@@ -10063,8 +11284,8 @@ function registerScheduleHandlers() {
|
|
|
10063
11284
|
// template, authored before the user fills in `cwd`) falls back to the
|
|
10064
11285
|
// legacy global dir until it's re-saved with a real cwd and migrated by
|
|
10065
11286
|
// the next reconcile-driven scan.
|
|
10066
|
-
const dir = (await findPrdDir(data.slug)) ??
|
|
10067
|
-
if (dir ===
|
|
11287
|
+
const dir = (await findPrdDir(data.slug)) ?? schedulerPaths.prdsRoot();
|
|
11288
|
+
if (dir === schedulerPaths.prdsRoot()) ensureDirs();
|
|
10068
11289
|
const resolved = safeSlugPathIn(dir, data.slug);
|
|
10069
11290
|
if (!resolved) return { ok: false, error: 'invalid slug' };
|
|
10070
11291
|
try {
|
|
@@ -10096,6 +11317,15 @@ function registerScheduleHandlers() {
|
|
|
10096
11317
|
});
|
|
10097
11318
|
}
|
|
10098
11319
|
|
|
11320
|
+
function stopDispatchLoop() {
|
|
11321
|
+
if (dispatchLoopHandle) { dispatchLoopHandle.stop(); dispatchLoopHandle = null; }
|
|
11322
|
+
}
|
|
11323
|
+
|
|
11324
|
+
/** Shutdown path: stop the timers this module owns. */
|
|
11325
|
+
function stop() {
|
|
11326
|
+
stopDispatchLoop();
|
|
11327
|
+
}
|
|
11328
|
+
|
|
10099
11329
|
async function init() {
|
|
10100
11330
|
ensureDirs();
|
|
10101
11331
|
// Boot phase — reconciliation, migrations, self-heal, first reset probe.
|
|
@@ -10111,6 +11341,9 @@ async function init() {
|
|
|
10111
11341
|
// A slot freed anywhere (e.g. a chat run settled) may unblock a deferred
|
|
10112
11342
|
// batch — advance the queue without waiting for the next 60s poll.
|
|
10113
11343
|
sessionSlots.subscribe(() => { tickQueue().catch(() => {}); });
|
|
11344
|
+
// Boot-time expiry pass (process-local state is empty after a restart, so
|
|
11345
|
+
// this is a cheap belt-and-braces run against the freshly read queue).
|
|
11346
|
+
try { runReservationExpiryPass((await readQueue()).jobs); } catch { /* best-effort */ }
|
|
10114
11347
|
// Retire the global queue.json: split its rows into per-project shards
|
|
10115
11348
|
// BEFORE the first read below, so boot reconciliation sees the shards.
|
|
10116
11349
|
try {
|
|
@@ -10131,17 +11364,17 @@ async function init() {
|
|
|
10131
11364
|
// Boot reconciliation: finalize any job that was 'running' when the app died.
|
|
10132
11365
|
// Check the run log first — a job that emitted result/success before the crash
|
|
10133
11366
|
// should be marked 'completed', not 'failed', so it doesn't wedge the queue
|
|
10134
|
-
// via the failure-gate.
|
|
10135
|
-
// it from continuing to write to the project unsupervised (2026-05-21 incident).
|
|
11367
|
+
// via the failure-gate. A still-live executor is spared, not killed.
|
|
10136
11368
|
//
|
|
10137
11369
|
// classifyRunOutcome calls readTail → fs.readFileSync (up to 64 KB per job).
|
|
10138
11370
|
// Pre-compute all outcomes BEFORE entering the mutate lock so the blocking I/O
|
|
10139
11371
|
// does not stall the event loop or hold the mutateTail chain during startup.
|
|
10140
11372
|
//
|
|
10141
|
-
//
|
|
10142
|
-
//
|
|
11373
|
+
// Rows proven alive are adopted (left running, never killed) — see
|
|
11374
|
+
// partitionBootOrphans. Everything else is proven dead/exited and is safe to
|
|
10143
11375
|
// classify immediately below.
|
|
10144
11376
|
const bootSnap = readQueueSync();
|
|
11377
|
+
const bootLogPath = (j) => (j?.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null);
|
|
10145
11378
|
|
|
10146
11379
|
// Worktree boot reconciliation (PRD 994): a job worktree that survives an
|
|
10147
11380
|
// app crash/host reboot must not leak disk or a dangling branch forever —
|
|
@@ -10156,11 +11389,15 @@ async function init() {
|
|
|
10156
11389
|
// itself, proof its run already died. isLive checks the already-read
|
|
10157
11390
|
// bootSnap (no extra queue read) for a live running-row pid, OR a live
|
|
10158
11391
|
// /proc cwd holder under the checkout itself. See jobWorktreeBootLive.cjs.
|
|
11392
|
+
// rowPid walks the SAME record → runtime.pid → log-pid ladder
|
|
11393
|
+
// partitionBootOrphans uses, so the sweep and the partition can never
|
|
11394
|
+
// disagree about which executor is alive.
|
|
10159
11395
|
const isLive = buildJobWorktreeIsLive({
|
|
10160
11396
|
bootJobs: bootSnap.jobs,
|
|
10161
11397
|
claudePidAlive,
|
|
10162
11398
|
hasLiveHolder: gitWorktree.hasLiveHolder,
|
|
10163
11399
|
cwdHolders: gitWorktree.listCwdHolders(),
|
|
11400
|
+
rowPid: (j) => bootRowPid(j, bootLogPath),
|
|
10164
11401
|
});
|
|
10165
11402
|
await jobWorktree.reconcileWorktreesOnBoot([...worktreeCwds], { isLive });
|
|
10166
11403
|
} catch (e) {
|
|
@@ -10180,12 +11417,30 @@ async function init() {
|
|
|
10180
11417
|
console.error('[scheduler] boot epic-worktree reconciliation failed', e?.message);
|
|
10181
11418
|
}
|
|
10182
11419
|
|
|
10183
|
-
const { immediate: immediateSlugs,
|
|
11420
|
+
const { immediate: immediateSlugs, adopted: adoptedSlugs } = partitionBootOrphans(bootSnap.jobs, {
|
|
11421
|
+
pidAlive: claudePidAlive,
|
|
11422
|
+
getLogPid: (j) => readSpawnedPidFromLog(bootLogPath(j)),
|
|
11423
|
+
getLogMtimeMs: (j) => readLogMtimeMs(bootLogPath(j)),
|
|
11424
|
+
logFreshWindowMs: IDLE_OUTPUT_KILL_MS,
|
|
11425
|
+
findLiveProcess: (j) => findLiveProcessForJob(j, {
|
|
11426
|
+
worktreeDir: jobWorktree.worktreeDirFor(j.cwd || DEFAULT_PROJECT_CWD, j.slug),
|
|
11427
|
+
runCwd: j.runtime?.cwd || j.cwd,
|
|
11428
|
+
}),
|
|
11429
|
+
readRecord: supervisorRecord.readSupervisorRecord,
|
|
11430
|
+
});
|
|
10184
11431
|
const bootOutcomes = new Map();
|
|
10185
11432
|
for (const j of bootSnap.jobs) {
|
|
10186
11433
|
if (!immediateSlugs.includes(j.slug)) continue;
|
|
10187
|
-
const logPath = j.runId ? path.join(
|
|
10188
|
-
|
|
11434
|
+
const logPath = j.runId ? path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.log`) : null;
|
|
11435
|
+
let outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
11436
|
+
// A row whose run already wrote its exit marker (meta.json) is finalized
|
|
11437
|
+
// from that meta when the log tail alone can't say (killed/torn tail).
|
|
11438
|
+
if (outcome === 'unknown' || outcome === 'no_result') {
|
|
11439
|
+
let meta = null;
|
|
11440
|
+
try { meta = j.runId ? JSON.parse(fs.readFileSync(path.join(schedulerPaths.runsDir(), j.runId, `${j.slug}.meta.json`), 'utf8')) : null; } catch { /* no/torn meta — keep the log outcome */ }
|
|
11441
|
+
if (meta && typeof meta.exitCode === 'number') outcome = meta.exitCode === 0 ? 'success' : 'failed';
|
|
11442
|
+
}
|
|
11443
|
+
bootOutcomes.set(j.slug, outcome);
|
|
10189
11444
|
}
|
|
10190
11445
|
// Same evidence-before-failure gate reapDeadRunningJobs applies, resolved
|
|
10191
11446
|
// BEFORE mutate() for the same reason (git spawn work must never run
|
|
@@ -10215,53 +11470,32 @@ async function init() {
|
|
|
10215
11470
|
await archiveCompletedPrd(slug, cwd);
|
|
10216
11471
|
}
|
|
10217
11472
|
|
|
10218
|
-
//
|
|
10219
|
-
//
|
|
10220
|
-
//
|
|
10221
|
-
//
|
|
10222
|
-
|
|
10223
|
-
|
|
10224
|
-
const
|
|
10225
|
-
|
|
10226
|
-
|
|
10227
|
-
|
|
10228
|
-
|
|
10229
|
-
|
|
10230
|
-
|
|
10231
|
-
|
|
10232
|
-
|
|
10233
|
-
|
|
10234
|
-
const logPath = j.runId ? path.join(RUNS_DIR, j.runId, `${j.slug}.log`) : null;
|
|
10235
|
-
const outcome = logPath ? classifyRunOutcome(logPath) : 'unknown';
|
|
10236
|
-
// Same evidence-before-failure gate as the immediate-orphan path
|
|
10237
|
-
// above, resolved before mutate() for the same reason (git spawn
|
|
10238
|
-
// work must never run inside mutate()'s serialization chain). Uses
|
|
10239
|
-
// the captured pre-kill snapshot's landedCommit/cwd/startedAt — the
|
|
10240
|
-
// race guard below already confirms `cur` is still this same run
|
|
10241
|
-
// (runId === bootRunId) before this evidence is applied.
|
|
10242
|
-
const confirmedLandedCommit = (outcome !== 'success' && j.landedCommit)
|
|
10243
|
-
? (await resolveLandedCommitEvidence(j.cwd || DEFAULT_PROJECT_CWD, j.landedCommit, j.startedAt) ? j.landedCommit : null)
|
|
10244
|
-
: null;
|
|
10245
|
-
let deferredCompletedCwd;
|
|
10246
|
-
mutate((state) => {
|
|
10247
|
-
const cur = state.jobs.find((x) => x.slug === slug);
|
|
10248
|
-
// Race guard: bail if the job already resolved, OR if it's already been
|
|
10249
|
-
// re-picked into a NEW run (different runId) within the grace window —
|
|
10250
|
-
// that new run is not the boot orphan we SIGTERM'd and must not be
|
|
10251
|
-
// touched by this stale classification.
|
|
10252
|
-
if (!cur || cur.status !== 'running' || cur.runId !== bootRunId) return;
|
|
10253
|
-
applyOrphanOutcome(cur, outcome, killNote, confirmedLandedCommit);
|
|
10254
|
-
console.log(`[scheduler] boot reconcile (deferred): slug=${slug} outcome=${outcome} → status=${cur.status}`);
|
|
10255
|
-
deferredCompletedCwd = cur.status === 'completed' ? cur.cwd : undefined;
|
|
10256
|
-
}).then(() => {
|
|
10257
|
-
if (deferredCompletedCwd !== undefined) return archiveCompletedPrd(slug, deferredCompletedCwd);
|
|
10258
|
-
}).catch((e) => console.error(`[scheduler] deferred boot reconcile failed for ${slug}:`, e?.message));
|
|
10259
|
-
}, BOOT_ORPHAN_KILL_GRACE_MS).unref?.();
|
|
11473
|
+
// Proven-alive rows stay `running` and are never signalled (the boot worktree
|
|
11474
|
+
// sweep above already spares their checkout). No sessionSlots token is
|
|
11475
|
+
// acquired: pickNextBatch's untrackedRunning correction counts the row
|
|
11476
|
+
// against the pool, and reapDeadRunningJobs finalizes it on exit.
|
|
11477
|
+
if (adoptedSlugs.length) {
|
|
11478
|
+
const adoptedAtBoot = new Date().toISOString();
|
|
11479
|
+
const adoptedRunIds = new Map(bootSnap.jobs.filter((j) => adoptedSlugs.includes(j.slug)).map((j) => [j.slug, j.runId ?? null]));
|
|
11480
|
+
await mutate((state) => {
|
|
11481
|
+
for (const j of state.jobs) {
|
|
11482
|
+
// runId guard: never stamp a DIFFERENT later run of the same slug.
|
|
11483
|
+
if (j.status !== 'running' || !adoptedRunIds.has(j.slug) || (j.runId ?? null) !== adoptedRunIds.get(j.slug)) continue;
|
|
11484
|
+
j.adoptedAtBoot = adoptedAtBoot;
|
|
11485
|
+
delete j.supervisedAt; // a prior process's supervisor died with it
|
|
11486
|
+
console.log(`[scheduler] boot: adopted live executor for ${j.slug} (pid=${j.runtime?.pid ?? 'unknown'}) — left running, no signal`);
|
|
11487
|
+
}
|
|
11488
|
+
});
|
|
10260
11489
|
}
|
|
10261
11490
|
|
|
11491
|
+
// Re-arm budget/idle/deadman + the quietMachine lease for the adopted rows
|
|
11492
|
+
// (a dispatch-loop pass repeats this for any row left without a supervisor).
|
|
11493
|
+
await superviseAdoptedRunsPass();
|
|
11494
|
+
|
|
10262
11495
|
// If we boot up while paused with a resumeAt in the past, clear it. This
|
|
10263
11496
|
// happens when the app was closed across the reset window.
|
|
10264
11497
|
const boot = await readQueue();
|
|
11498
|
+
await clearStaleDrainAtBoot(boot);
|
|
10265
11499
|
if (boot.paused && boot.paused.resumeAt && new Date(boot.paused.resumeAt).getTime() <= Date.now()) {
|
|
10266
11500
|
await clearPause('boot-elapsed');
|
|
10267
11501
|
} else if (boot.paused && boot.paused.resumeAt) {
|
|
@@ -10483,7 +11717,7 @@ async function init() {
|
|
|
10483
11717
|
}
|
|
10484
11718
|
for (const target of exhaustedNeedsReviewTargets) {
|
|
10485
11719
|
const j = ms.jobs.find((x) => x.slug === target.slug);
|
|
10486
|
-
const outcome = applyNeedsReviewAutoResolve(j);
|
|
11720
|
+
const outcome = applyNeedsReviewAutoResolve(j, ms.jobs);
|
|
10487
11721
|
if (outcome) {
|
|
10488
11722
|
console.warn(
|
|
10489
11723
|
`[scheduler] NEEDS_REVIEW AUTO-RESOLVE: project=${j.cwd ?? '(unknown)'} slug=${j.slug} `
|
|
@@ -10504,7 +11738,7 @@ async function init() {
|
|
|
10504
11738
|
}
|
|
10505
11739
|
}).catch(() => {});
|
|
10506
11740
|
}
|
|
10507
|
-
},
|
|
11741
|
+
}, REVERIFY_INTERVAL_MS);
|
|
10508
11742
|
|
|
10509
11743
|
// Self-rescheduling poll loop with exponential backoff. Replaces the
|
|
10510
11744
|
// old fixed-interval pollTimer + initialPollTimeout.
|
|
@@ -10525,87 +11759,18 @@ async function init() {
|
|
|
10525
11759
|
// setInterval callback is sync; readQueueSync stays sync to avoid awaiting
|
|
10526
11760
|
// inside the timer body (and the 60s cadence makes the cost moot).
|
|
10527
11761
|
if (heartbeatInterval) clearInterval(heartbeatInterval);
|
|
10528
|
-
heartbeatInterval = setInterval(() =>
|
|
10529
|
-
const s = readQueueSync();
|
|
10530
|
-
// NEVER-STOP INVARIANT: if a queue holds ready PRDs and nothing is
|
|
10531
|
-
// running, something must drive it. This is the only driver that does
|
|
10532
|
-
// not depend on the billing poll loop, a pause timer, or a completing
|
|
10533
|
-
// job to schedule the next tick — every one of which has failed at
|
|
10534
|
-
// least once. See classifyQueueStarvation.
|
|
10535
|
-
if (!s.unreadable) {
|
|
10536
|
-
runQueueStarvationWatchdog(s).catch((e) => console.error('[scheduler] starvation watchdog error', e));
|
|
10537
|
-
}
|
|
10538
|
-
// Initialise from the real status union (scheduleJobSchema.cjs) rather
|
|
10539
|
-
// than a hand-maintained subset — the old `{ pending, running, completed,
|
|
10540
|
-
// failed }` literal silently minted a NEW key for any other value
|
|
10541
|
-
// (`counts[j.status] = (counts[j.status]||0)+1`), which is exactly how a
|
|
10542
|
-
// heartbeat with a `queued: 2` bucket looked like "normal" 24h
|
|
10543
|
-
// visibility instead of the alarm it should have been. Any row whose
|
|
10544
|
-
// status isn't in JOB_STATUSES (shouldn't happen post-quarantine, but
|
|
10545
|
-
// this is the last line of defence) routes into `unknown`, never a
|
|
10546
|
-
// freshly-minted key.
|
|
10547
|
-
const counts = Object.fromEntries(JOB_STATUSES.map((st) => [st, 0]));
|
|
10548
|
-
counts.unknown = 0;
|
|
10549
|
-
for (const j of s.jobs) {
|
|
10550
|
-
if (Object.prototype.hasOwnProperty.call(counts, j.status) && j.status !== 'unknown') {
|
|
10551
|
-
counts[j.status] += 1;
|
|
10552
|
-
} else {
|
|
10553
|
-
counts.unknown += 1;
|
|
10554
|
-
}
|
|
10555
|
-
}
|
|
10556
|
-
|
|
10557
|
-
const stall = computeStallSummary(s);
|
|
10558
|
-
// Per-project alerting (see computeStallSummary's header): a project
|
|
10559
|
-
// stalled while others are busy must still fire, and one project
|
|
10560
|
-
// recovering must not clear or suppress another's still-open episode —
|
|
10561
|
-
// that is exactly what a single module-level stallSince/stallToasted
|
|
10562
|
-
// flag masked before (the burrow-vs-others incident this PRD fixes).
|
|
10563
|
-
const now = Date.now();
|
|
10564
|
-
const stalledCwds = Object.keys(stall.byProject).filter((cwd) => stall.byProject[cwd].stalled);
|
|
10565
|
-
for (const cwd of [...stallSince.keys()]) {
|
|
10566
|
-
if (!stalledCwds.includes(cwd)) {
|
|
10567
|
-
stallSince.delete(cwd);
|
|
10568
|
-
stallToasted.delete(cwd);
|
|
10569
|
-
}
|
|
10570
|
-
}
|
|
10571
|
-
const toAlert = [];
|
|
10572
|
-
for (const cwd of stalledCwds) {
|
|
10573
|
-
if (!stallSince.has(cwd)) stallSince.set(cwd, now);
|
|
10574
|
-
if (!stallToasted.get(cwd) && now - stallSince.get(cwd) >= POLL_INTERVAL_MS) {
|
|
10575
|
-
stallToasted.set(cwd, true);
|
|
10576
|
-
toAlert.push(cwd);
|
|
10577
|
-
}
|
|
10578
|
-
}
|
|
10579
|
-
if (toAlert.length > 0) {
|
|
10580
|
-
console.error(
|
|
10581
|
-
`[scheduler] STALL DETECTED in project(s): ${toAlert.join(', ')} — 0 running, 0 pending, not paused, `
|
|
10582
|
-
+ `for >= ${Math.round(POLL_INTERVAL_MS / 1000)}s`,
|
|
10583
|
-
stall.byProject,
|
|
10584
|
-
);
|
|
10585
|
-
appendAuditEvent('scheduler_stall_detected', { projects: toAlert, total: stall.total, byProject: stall.byProject });
|
|
10586
|
-
if (mainWindow && !mainWindow.isDestroyed()) {
|
|
10587
|
-
sendIfAlive(mainWindow, 'schedule:stall', {
|
|
10588
|
-
message: `Scheduler stall in ${toAlert.length} project(s): ${toAlert.join(', ')}. Check the Scheduler tab.`,
|
|
10589
|
-
projects: toAlert,
|
|
10590
|
-
total: stall.total,
|
|
10591
|
-
byProject: stall.byProject,
|
|
10592
|
-
});
|
|
10593
|
-
}
|
|
10594
|
-
}
|
|
10595
|
-
|
|
10596
|
-
appendHeartbeat({
|
|
10597
|
-
ts: Date.now(),
|
|
10598
|
-
pid: process.pid,
|
|
10599
|
-
counts,
|
|
10600
|
-
stall: { stalled: stall.stalled, total: stall.total },
|
|
10601
|
-
paused: s.paused ? { reason: s.paused.reason, resumeAt: s.paused.resumeAt } : null,
|
|
10602
|
-
nextReset: cachedNextReset,
|
|
10603
|
-
utilization: cachedUtilization,
|
|
10604
|
-
consecutiveFailures,
|
|
10605
|
-
});
|
|
10606
|
-
}, 60_000);
|
|
11762
|
+
heartbeatInterval = setInterval(() => heartbeatTick(), 60_000);
|
|
10607
11763
|
if (heartbeatInterval.unref) heartbeatInterval.unref();
|
|
10608
11764
|
|
|
11765
|
+
// Dispatch's own periodic driver: cadence is independent of pollLoop's billing
|
|
11766
|
+
// backoff. A loop tick meeting a cancelled cancelToken returns 'cancelled' from
|
|
11767
|
+
// tickBody; clearing stays with the starvation watchdog's existing force-clear.
|
|
11768
|
+
stopDispatchLoop();
|
|
11769
|
+
dispatchLoopHandle = startDispatchLoop({
|
|
11770
|
+
tick: () => tickQueue(),
|
|
11771
|
+
onError: (e) => console.warn('[scheduler] dispatch loop tick failed', e?.message),
|
|
11772
|
+
});
|
|
11773
|
+
|
|
10609
11774
|
// Wake-from-sleep: immediately re-poll and re-evaluate the queue.
|
|
10610
11775
|
try {
|
|
10611
11776
|
const { powerMonitor } = require('electron');
|
|
@@ -10828,8 +11993,8 @@ const remote = {
|
|
|
10828
11993
|
}
|
|
10829
11994
|
await fsp.mkdir(dir, { recursive: true });
|
|
10830
11995
|
} else {
|
|
10831
|
-
dir = (await findPrdDir(slug)) ??
|
|
10832
|
-
if (dir ===
|
|
11996
|
+
dir = (await findPrdDir(slug)) ?? schedulerPaths.prdsRoot();
|
|
11997
|
+
if (dir === schedulerPaths.prdsRoot()) ensureDirs();
|
|
10833
11998
|
}
|
|
10834
11999
|
|
|
10835
12000
|
// writePrd only ever JOINS an existing Epic now (no mintAuthority
|
|
@@ -10866,6 +12031,19 @@ const remote = {
|
|
|
10866
12031
|
}
|
|
10867
12032
|
},
|
|
10868
12033
|
|
|
12034
|
+
// User-initiated pause/resume — the admin-route/MCP twins of the
|
|
12035
|
+
// schedule:pause / schedule:resume IPC handlers, through the same setPaused /
|
|
12036
|
+
// clearPause. Pause stops NEW dispatch only; running jobs are never touched.
|
|
12037
|
+
async pause() {
|
|
12038
|
+
await setPaused('manual', null);
|
|
12039
|
+
return { ok: true };
|
|
12040
|
+
},
|
|
12041
|
+
|
|
12042
|
+
async resume() {
|
|
12043
|
+
await clearPause('manual');
|
|
12044
|
+
return { ok: true };
|
|
12045
|
+
},
|
|
12046
|
+
|
|
10869
12047
|
async resetJob(slug, opts = {}) {
|
|
10870
12048
|
const resolved = await resolveSlugOrReason(slug, opts.cwd);
|
|
10871
12049
|
if (!resolved.ok) {
|
|
@@ -11193,6 +12371,14 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
11193
12371
|
sendJson(res, 200, jobs);
|
|
11194
12372
|
});
|
|
11195
12373
|
|
|
12374
|
+
adminHttp.registerRoute('POST', '/admin/scheduler/pause', async (req, res) => {
|
|
12375
|
+
sendJson(res, 200, await remoteObj.pause());
|
|
12376
|
+
});
|
|
12377
|
+
|
|
12378
|
+
adminHttp.registerRoute('POST', '/admin/scheduler/resume', async (req, res) => {
|
|
12379
|
+
sendJson(res, 200, await remoteObj.resume());
|
|
12380
|
+
});
|
|
12381
|
+
|
|
11196
12382
|
adminHttp.registerRoute('POST', '/admin/scheduler/reset-job', async (req, res) => {
|
|
11197
12383
|
const raw = await readBody(req);
|
|
11198
12384
|
let parsed;
|
|
@@ -11217,6 +12403,8 @@ function registerAdminRoutes(adminHttp, remoteObj = remote) {
|
|
|
11217
12403
|
module.exports = {
|
|
11218
12404
|
classifyQueueStarvation,
|
|
11219
12405
|
classifyQueueStarvationByProject,
|
|
12406
|
+
dispatchIdleMs,
|
|
12407
|
+
launchBlockedSlugs,
|
|
11220
12408
|
classifyQueueHealth,
|
|
11221
12409
|
runQueueStarvationWatchdog,
|
|
11222
12410
|
QUEUE_STARVATION_MS,
|
|
@@ -11240,13 +12428,13 @@ module.exports = {
|
|
|
11240
12428
|
registerScheduleHandlers,
|
|
11241
12429
|
attachWindow,
|
|
11242
12430
|
init,
|
|
11243
|
-
ROOT,
|
|
11244
|
-
PRDS_DIR,
|
|
11245
|
-
SCHEDULER_STATE_PATH,
|
|
11246
12431
|
BACKOFF_MAX_MS,
|
|
11247
12432
|
FAILURE_STREAK_WARN_THRESHOLD,
|
|
12433
|
+
FAILURE_STREAK_ESCALATION_MS,
|
|
11248
12434
|
nextBackoffMs,
|
|
11249
12435
|
shouldWarnFailureStreak,
|
|
12436
|
+
shouldEscalateFailureStreak,
|
|
12437
|
+
computeDegradedBudget,
|
|
11250
12438
|
healRefusalReason,
|
|
11251
12439
|
writeQueue,
|
|
11252
12440
|
reconcile,
|
|
@@ -11269,6 +12457,8 @@ module.exports = {
|
|
|
11269
12457
|
memoryLimitedBatchSize,
|
|
11270
12458
|
availableForJobs,
|
|
11271
12459
|
reverifyNeedsReview,
|
|
12460
|
+
runGateShadow,
|
|
12461
|
+
awaitGateShadowIdle: async () => { while (gateShadowPending) await gateShadowPending; },
|
|
11272
12462
|
shouldRunPeriodicReverify,
|
|
11273
12463
|
findStuckFailedJobs,
|
|
11274
12464
|
STUCK_FAILED_ESCALATE_MS,
|
|
@@ -11283,6 +12473,13 @@ module.exports = {
|
|
|
11283
12473
|
NEEDS_REVIEW_RESOLVE_MS,
|
|
11284
12474
|
needsReviewAutoResolveDisabled,
|
|
11285
12475
|
isRescanCandidate,
|
|
12476
|
+
isTranscriptRescannable,
|
|
12477
|
+
selectEvidenceScanTargets,
|
|
12478
|
+
RESCAN_EXCLUDED_VERDICTS,
|
|
12479
|
+
RESCANNABLE_VERDICTS,
|
|
12480
|
+
EVIDENCE_SCAN_MAX_PER_PASS,
|
|
12481
|
+
EVIDENCE_SCAN_MIN_INTERVAL_MS,
|
|
12482
|
+
REVERIFY_INTERVAL_MS,
|
|
11286
12483
|
isFailedUnverifiedShaped,
|
|
11287
12484
|
computeLooksDone,
|
|
11288
12485
|
attributeLandedCommits,
|
|
@@ -11295,6 +12492,9 @@ module.exports = {
|
|
|
11295
12492
|
isExhaustedAutoFix,
|
|
11296
12493
|
GUARD_VERDICT_EVIDENCE_ELIGIBLE,
|
|
11297
12494
|
isGuardParkedWithoutAutoFix,
|
|
12495
|
+
isStrandedAutoFixPark,
|
|
12496
|
+
isStaleSharedTreeRevertedPark,
|
|
12497
|
+
landedCommitIsAncestorOfHead,
|
|
11298
12498
|
isEligibleForNeedsReviewAutoResolve,
|
|
11299
12499
|
isPlanUnqueued,
|
|
11300
12500
|
isFixPlanDead,
|
|
@@ -11334,7 +12534,6 @@ module.exports = {
|
|
|
11334
12534
|
buildScheduleStatePayload,
|
|
11335
12535
|
partitionBootOrphans,
|
|
11336
12536
|
applyOrphanOutcome,
|
|
11337
|
-
BOOT_ORPHAN_KILL_GRACE_MS,
|
|
11338
12537
|
registerAdminRoutes,
|
|
11339
12538
|
notifyOriginatingTab,
|
|
11340
12539
|
notifyNeedsReview,
|
|
@@ -11360,10 +12559,13 @@ module.exports = {
|
|
|
11360
12559
|
SCHEDULER_CODE_SHA,
|
|
11361
12560
|
resetJobFields,
|
|
11362
12561
|
executeJob,
|
|
12562
|
+
killOrphanClaudePid,
|
|
11363
12563
|
prdArchivedSkipResult,
|
|
11364
12564
|
spawnJob,
|
|
11365
12565
|
listPrdsInternal,
|
|
11366
12566
|
computeStallSummary,
|
|
12567
|
+
heartbeatTick,
|
|
12568
|
+
appendHeartbeat,
|
|
11367
12569
|
findStaleQuarantinedJobs,
|
|
11368
12570
|
QUARANTINE_ESCALATE_MS,
|
|
11369
12571
|
selectQuarantineAutoResolveTargets,
|
|
@@ -11402,6 +12604,10 @@ module.exports = {
|
|
|
11402
12604
|
setPaused,
|
|
11403
12605
|
clearPause,
|
|
11404
12606
|
tickQueue,
|
|
12607
|
+
setRestartHandler,
|
|
12608
|
+
driveUpgradeDrain,
|
|
12609
|
+
clearStaleDrainAtBoot,
|
|
12610
|
+
stop,
|
|
11405
12611
|
runDueJobs,
|
|
11406
12612
|
pollLoop,
|
|
11407
12613
|
maybeLaunchWhenAvailable,
|
|
@@ -11410,12 +12616,23 @@ module.exports = {
|
|
|
11410
12616
|
CONSECUTIVE_RAPID_RATE_LIMIT_THRESHOLD,
|
|
11411
12617
|
RAPID_RATE_LIMIT_WINDOW_MS,
|
|
11412
12618
|
MANUAL_PAUSE_COOLDOWN_MS,
|
|
11413
|
-
RUNS_DIR,
|
|
11414
12619
|
pickRunDir,
|
|
11415
12620
|
resolveRateLimitPauseReset,
|
|
12621
|
+
billingResetForPause,
|
|
11416
12622
|
computeEffectiveResumeAt,
|
|
11417
12623
|
computeResumeDelay,
|
|
11418
12624
|
FOREIGN_WIP_BLOCK_STREAK_LIMIT,
|
|
11419
12625
|
validateForeignWipBlockClaim,
|
|
11420
12626
|
requeueForeignWipBlockedJobs,
|
|
11421
12627
|
};
|
|
12628
|
+
|
|
12629
|
+
// Lazy path getters: resolved from SM_SCHEDULER_HOME at each read, never frozen
|
|
12630
|
+
// at require time (see lib/schedulerPaths.cjs).
|
|
12631
|
+
for (const [name, resolve] of [
|
|
12632
|
+
['ROOT', schedulerPaths.scheduledPlansRoot],
|
|
12633
|
+
['PRDS_DIR', schedulerPaths.prdsRoot],
|
|
12634
|
+
['RUNS_DIR', schedulerPaths.runsDir],
|
|
12635
|
+
['SCHEDULER_STATE_PATH', schedulerPaths.schedulerStatePath],
|
|
12636
|
+
]) {
|
|
12637
|
+
Object.defineProperty(module.exports, name, { get: resolve, enumerable: true });
|
|
12638
|
+
}
|