@yemi33/minions 0.1.2148 → 0.1.2150

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,164 @@
1
+ # Worktree Lifecycle
2
+
3
+ Deep-dive for the four engine pieces that own git-worktree state for dispatch:
4
+ the **pool** (recycling), the **live guard** (don't wipe an agent's work),
5
+ the **quarantine path** (dirty/divergent → quarantine dir + retry), and the
6
+ **Windows file-lock retry** (EPERM/EBUSY footgun). CLAUDE.md → Worktree
7
+ Lifecycle keeps the cross-cutting invariants; the detail lives here.
8
+
9
+ > Source of truth: `engine/worktree-pool.js`, `engine/shared.js#removeWorktree`
10
+ > + `_retryFsOp`, `engine.js` (`_quarantineDirtyWorktree`, `_renameWithRetry`,
11
+ > `_killGitDescendantsForWorktree`, `pruneOrphanWorktrees*`, `gcDispatchWorktreeIfOrphan`),
12
+ > `engine/cleanup.js`. Last verified: 2026-06-09.
13
+
14
+ ## Live-worktree guard (W-mq5rwwss000f30a7)
15
+
16
+ **Invariant:** every code path that wipes, renames, or recycles a worktree
17
+ MUST call `shared.isWorktreePathLive(path, { db?, excludeDispatchId? })`
18
+ first and skip on `true`. Originally added after W-mq5n1zx5000hcfb5 — an
19
+ agent's worktree was wiped 4× consecutively by the reaper while the
20
+ dispatch was still active.
21
+
22
+ - **Backed by SQL** against `dispatches` (status IN ('pending','active')),
23
+ reading `json_extract(data, '$.worktreePath')` with a `data.meta.worktreePath`
24
+ fallback. The pending → active transition persists `item.worktreePath`
25
+ on the dispatch row so the guard has a path to correlate.
26
+ - **Fails OPEN** (returns `true`) when SQLite is unreachable or the query
27
+ throws — better to leak a worktree than nuke an agent's unpushed work.
28
+ - **Wired sites:** `shared.removeWorktree` (accepts `{ excludeDispatchId }`
29
+ forwarded to the guard); pool-return chain in `engine.js`
30
+ (`excludeDispatchId: id` so the dispatch can clean up its own worktree);
31
+ dispatch-end orphan GC (`gcDispatchWorktreeIfOrphan`); `_quarantineDirtyWorktree`
32
+ (returns `{ skipped: true, quarantinedPath: null }` on skip — callers
33
+ MUST honor `skipped` and not set `quarantined: true`); `engine/cleanup.js`
34
+ orphan-dir sweep.
35
+ - **On skip:** drops a deduped operator note at
36
+ `notes/inbox/engine-worktree-skip-live-<basename>-<date>.md`.
37
+
38
+ ## Worktree pool (opt-in)
39
+
40
+ `ENGINE_DEFAULTS.worktreePoolSize > 0` enables `engine/worktree-pool.js` to
41
+ recycle worktree dirs across branches.
42
+
43
+ - **Borrow** in `spawnAgent` only when (a) the new branch doesn't exist on
44
+ origin AND (b) the dispatch is not shared-branch / `useExistingBranch`.
45
+ - **Return** in `onAgentClose` BEFORE `completeDispatch`:
46
+ `git reset --hard HEAD` → `git clean -fd` → `git fetch origin <main>`
47
+ → `git checkout --detach origin/<main>` → mark IDLE.
48
+ - **State** at `engine/worktree-pool.json`; git ops outside any lock.
49
+
50
+ ## Quarantine path (dirty / divergent)
51
+
52
+ When `discoverFromWorkItems` finds a worktree in a `WORKTREE_DIRTY`,
53
+ `WORKTREE_DIVERGENT`, or post-stuck state, `_quarantineDirtyWorktree`
54
+ moves it to `<root>/.quarantined/<basename>-<utc>` and lets the next tick
55
+ recreate a clean worktree. Six layered defenses cover the Windows EBUSY
56
+ race when git status descendants still hold packfile handles
57
+ (W-mq5n1zx5000hcfb5; PR #3156).
58
+
59
+ ### Layer 1b — `--no-optional-locks` on every status probe (load-bearing)
60
+
61
+ `_statusPorcelainCmd()` emits `git --no-optional-locks status …`. Skips
62
+ the `.git/index.lock` acquire around the untracked-cache refresh inside
63
+ `status`. Typical probe duration under AV scanning drops from 6–12s to
64
+ <500ms, which removes the timeout that leaks the child in the first
65
+ place. **Most incidents are closed by 1b alone.**
66
+ Toggle: `ENGINE_DEFAULTS.statusProbeUseNoOptionalLocks` (default `true`).
67
+
68
+ ### Layer 1a — `_renameWithRetry` with jittered backoff
69
+
70
+ 6 attempts × 250 ms base × 2^N exponential + 200 ms random jitter
71
+ (~16 s worst-case). Only retries on `EBUSY|EPERM|EACCES|ENOTEMPTY`;
72
+ rethrows other codes immediately. Toggles:
73
+ `ENGINE_DEFAULTS.quarantineRenameRetryAttempts` (6),
74
+ `quarantineRenameRetryBaseMs` (250).
75
+
76
+ ### Layer 2a — `_killGitDescendantsForWorktree` (Windows-only)
77
+
78
+ PowerShell sweep, 2 s timeout. Shell-out built via single-quoted literal
79
+ (`'` escaped as `''`) so worktree-path interpolation is injection-safe.
80
+ POSIX no-op. Best-effort — failure logged, never throws.
81
+ Toggle: `ENGINE_DEFAULTS.statusProbeKillDescendantsWin32` (default `true`).
82
+
83
+ ### Layer 2b — `git worktree remove --force` fallback
84
+
85
+ Triggers only when 1a exhausts retries; destroys worktree contents (no
86
+ quarantine dir preserved). Toggle:
87
+ `ENGINE_DEFAULTS.quarantineForceRemoveFallback` (default `true`).
88
+
89
+ ### Layer 1c — `FAILURE_CLASS.WORKTREE_QUARANTINE_ENV_BLOCKED`
90
+
91
+ Routed when `cleanResult.quarantineError && !cleanResult.quarantined`.
92
+ Added to `dispatch.js#isRetryableFailureReason` never-retry set so the
93
+ per-agent retry counter doesn't bump (env failure, not the agent's fault).
94
+ The auto-recovery loop in `engine.js#discoverFromWorkItems` recognizes
95
+ the new class via both enum check and regex on legacy `failReason`
96
+ strings, then re-queues under the existing
97
+ `ENGINE_DEFAULTS.quarantineAutoRecoveryMax` (default 2) cap.
98
+
99
+ ### Layer 3a/3b — metrics + inbox alert
100
+
101
+ Counters at
102
+ `metrics._engine.worktreeQuarantineOutcomes.{attempts, success, successAfterRetry, fallbackForceRemove, totalFailure}`.
103
+ Total-failure path writes a structured inbox alert with the exact
104
+ PowerShell/POSIX recovery commands and `git worktree prune` instructions.
105
+
106
+ ### Auto-recovery cap
107
+
108
+ `discoverFromWorkItems` auto-recovers `WORKTREE_DIRTY`/`WORKTREE_DIVERGENT`
109
+ quarantine (#2996) up to `ENGINE_DEFAULTS.quarantineAutoRecoveryMax`
110
+ (default 2, tracked on `_quarantineRecoveryCount`); after the cap,
111
+ `_quarantineRecoveryGaveUp` triggers a warn-once.
112
+
113
+ ## Windows file-lock on removal (footgun #6 detail)
114
+
115
+ `fs.rmSync` against an active git worktree can lose to lingering file
116
+ handles (CC log streams, virus scanners, Explorer previews).
117
+ `shared.removeWorktree` retries via `shared._retryFsOp`
118
+ (`worktreeRemoveRetryAttempts` × exponential `worktreeRemoveRetryBaseMs`,
119
+ codes `EPERM|EBUSY|EACCES|ENOTEMPTY`).
120
+
121
+ After `worktreeStuckThreshold` consecutive failures the path is
122
+ **escalated**:
123
+
124
+ - A `notes/inbox/engine-worktree-stuck-<basename>-<date>.md` note with
125
+ Windows holder-identification hints is written (deduped per UTC day).
126
+ - Per-tick warns are suppressed for `worktreeStuckSuppressMs`.
127
+ - The retry cadence drops to `worktreeStuckSlowRetryMs`.
128
+ - When the holder finally releases, a
129
+ `worktree-recovered-<basename>` note clears the alert.
130
+
131
+ **Don't fight this with `fs.rmSync({force:true})` outside
132
+ `removeWorktree`** — you'll bypass the retry, escalation, and metrics
133
+ layers.
134
+
135
+ ## Periodic prune (W-mq5o6bvy000x7191)
136
+
137
+ `pruneWorktreesPeriodic` runs `pruneOrphanWorktrees` (in-root) +
138
+ `pruneOrphanWorktreesFromGitRegistry` (out-of-root via
139
+ `git worktree list --porcelain`) per project at
140
+ `ENGINE_DEFAULTS.worktreePruneIntervalTicks` cadence — catches Windows
141
+ EPERM/EBUSY stragglers that the dispatch-end GC couldn't reap and sweeps
142
+ the `git worktree list` registry for OUT-of-root entries the in-root
143
+ scanner is blind to.
144
+
145
+ ## Holder identification + opt-in auto-reap (W-mq6f2fe0000557fa)
146
+
147
+ Orphan-sweep escalations also run `shared.findProcessesWithCwdInside(wt)`
148
+ (cross-platform: PowerShell `Get-CimInstance Win32_Process` on Windows;
149
+ `/proc/*/cwd` walk on Linux; `lsof -d cwd` + `ps` on macOS) and append a
150
+ `## Live holders` section listing pid / cmdline / age to the
151
+ `engine-worktree-stuck-<basename>` escalation note.
152
+
153
+ Setting `engine.autoReapOrphanWorktreeHolders: true`
154
+ (Settings → Worker Pool & Worktrees → "Auto-reap orphan worktree holders")
155
+ additionally kills any holder whose `cmdline` matches `spawn-agent.js`
156
+ AND references the worktree basename AND whose age exceeds
157
+ `engine.agentTimeout * 2`, then retries `removeWorktree` once. Default
158
+ **OFF** — killing a foreign process is destructive.
159
+
160
+ - Scan timeout: `engine.orphanHolderScanTimeoutMs` (default 5000ms,
161
+ clamped 1000–30000).
162
+ - Motivating incident: W-mq1k8z6o003acd89 / stuck `spawn-agent.js`
163
+ PID 16828 that the periodic prune couldn't shake loose without manual
164
+ intervention.
package/engine/ado.js CHANGED
@@ -395,6 +395,24 @@ function applyAdoPrMetadata(pr, prData) {
395
395
  updated = true;
396
396
  }
397
397
 
398
+ // #3079 — Track target ref + clear stale merge-conflict dispatch state
399
+ // when the PR is retargeted. The same-head guard at engine.js:5676 and
400
+ // the MERGE_CONFLICT fingerprint at shared.js:6595 are now target-aware
401
+ // (via shared.prMergeConflictGuardKey), but pre-existing
402
+ // _lastDispatchByCause[MERGE_CONFLICT] / _noOpFixes[MERGE_CONFLICT]
403
+ // records would still block re-dispatch until something else cleared
404
+ // them. Reset them explicitly on retarget. First-poll seeding
405
+ // (no prior targetRefName) is a no-op so we don't strip state immediately
406
+ // after upsert.
407
+ const nextTargetRef = String(prData.targetRefName || '').trim();
408
+ if (nextTargetRef && pr.targetRefName !== nextTargetRef) {
409
+ if (shared.resetMergeConflictStateOnRetarget(pr, nextTargetRef)) {
410
+ log('info', `ADO: PR ${pr.id || '?'} retargeted ${pr.targetRefName} → ${nextTargetRef}, clearing stale MERGE_CONFLICT dispatch records`);
411
+ }
412
+ pr.targetRefName = nextTargetRef;
413
+ updated = true;
414
+ }
415
+
398
416
  return updated;
399
417
  }
400
418
 
@@ -2600,6 +2618,7 @@ module.exports = {
2600
2618
  fetchAdoPrMetadata,
2601
2619
  fetchSinglePrBuildStatus,
2602
2620
  findOpenPrOnBranch,
2621
+ applyAdoPrMetadata, // #3079 — exported for unit tests of targetRefName + retarget reset
2603
2622
  _resetAdoThrottle, // exported for testing
2604
2623
  _setAdoThrottleForTest, // exported for testing
2605
2624
  _setAdoTokenForTest, // exported for testing
package/engine/cli.js CHANGED
@@ -163,8 +163,9 @@ function handleCommand(cmd, args) {
163
163
  //
164
164
  // `minions work --help` used to create ghost work items with title='--help'
165
165
  // because the bare-string `title` was truthy and bypassed the `!title`
166
- // usage check. Same class of bug exists in `spawn`/`plan`/`complete` —
167
- // every command that takes a positional arg and tests it with `if (!arg)`.
166
+ // usage check. The fix lives in per-command guards (`_isHelpArg` /
167
+ // `looksLikeFlagOrHelp`) on `work`/`spawn`/`plan`/`complete`, which print
168
+ // command-specific `Usage:` output. `pr` and `bridge` handle help inline.
168
169
  //
169
170
  // Intercept here so a single guard covers the whole command set. `pr` and
170
171
  // `bridge` already handle `help`/`--help`/`-h` inline (see their own
@@ -400,6 +400,7 @@ function isRetryableFailureReason(reason = '', failureClass = '') {
400
400
  FAILURE_CLASS.WORKTREE_PREFLIGHT, // pre-spawn worktree validation — recompute will produce the same failure
401
401
  FAILURE_CLASS.WORKTREE_DIRTY, // #2996: reused worktree was dirty and could not be auto-healed — non-retryable for this dispatch attempt; the engine quarantined the worktree so the next discovery cycle creates a fresh one
402
402
  FAILURE_CLASS.WORKTREE_DIVERGENT, // #2996: reused worktree's local branch had unpushed commits — engine quarantined + backed up the local ref; non-retryable for this dispatch (next discovery creates fresh worktree on origin/<branch>)
403
+ FAILURE_CLASS.WORKTREE_QUARANTINE_ENV_BLOCKED, // W-mq5n1zx5: quarantine rename couldn't release the worktree dir even after retry + force-remove fallback. Non-retryable at the dispatch level so the per-agent retry counter isn't bumped (environmental, not the agent's fault); the WI auto-recovery loop in engine.js#discoverFromWorkItems re-queues without touching _retriesByAgent.
403
404
  FAILURE_CLASS.INVALID_KEEP_PROCESSES_WORKDIR, // W-mp6k7ywi000fa33c — keep-pids cwd is not a real git worktree; re-running won't fix the structural issue
404
405
  FAILURE_CLASS.INVALID_KEEP_PROCESSES_SCHEMA, // W-mp7i902u000l991f — keep-pids.json failed shape validation; re-running with the same wrong file won't fix it
405
406
  FAILURE_CLASS.INVALID_MANAGED_SPAWN, // W-mpbhxg3b000u8411 — managed-spawn.json failed validation; re-running with the same wrong file won't fix it
@@ -770,6 +771,7 @@ function completeDispatch(id, result = DISPATCH_RESULT.SUCCESS, reason = '', res
770
771
  [FAILURE_CLASS.WORKTREE_PREFLIGHT]: 'worktree preflight rejected (nested in project root or rootDir collapsed to drive root)',
771
772
  [FAILURE_CLASS.WORKTREE_DIRTY]: 'reused worktree had uncommitted edits and could not be auto-healed (#2996) — engine quarantined the dir so the next dispatch creates a fresh worktree',
772
773
  [FAILURE_CLASS.WORKTREE_DIVERGENT]: 'reused worktree had unpushed local commits ahead of origin (#2996) — engine backed up the local ref and quarantined the dir so the next dispatch starts from origin/<branch>',
774
+ [FAILURE_CLASS.WORKTREE_QUARANTINE_ENV_BLOCKED]: 'quarantine rename was blocked by the OS (Windows EBUSY/EPERM) even after retry + force-remove fallback — environmental, not the agent\'s fault; auto-recovery loop will re-queue without bumping per-agent retries',
773
775
  [FAILURE_CLASS.INVALID_KEEP_PROCESSES_WORKDIR]: 'keep_processes cwd is not a real git worktree (rerun in a `git worktree add` directory)',
774
776
  [FAILURE_CLASS.INVALID_KEEP_PROCESSES_SCHEMA]: 'keep-pids.json failed shape validation (wrong keys/types/values — see inbox alert for the canonical shape)',
775
777
  [FAILURE_CLASS.INVALID_MANAGED_SPAWN]: 'managed-spawn.json failed validation (bad schema, workdir, or allowlist — see inbox alert)',
package/engine/github.js CHANGED
@@ -748,6 +748,20 @@ async function pollPrStatus(config) {
748
748
  updated = true;
749
749
  }
750
750
 
751
+ // #3079 — Track GitHub target branch name + clear stale merge-conflict
752
+ // dispatch state when the PR is retargeted (e.g. `gh pr edit --base
753
+ // master` after parent PR merges). Mirrors ado.js applyAdoPrMetadata.
754
+ // First-poll seeding (no prior baseRefName) is a no-op so we don't
755
+ // strip state immediately after upsert.
756
+ const nextBaseRefName = String(prData.base?.ref || '').trim();
757
+ if (nextBaseRefName && pr.baseRefName !== nextBaseRefName) {
758
+ if (shared.resetMergeConflictStateOnRetarget(pr, nextBaseRefName)) {
759
+ log('info', `GitHub: PR ${pr.id} retargeted ${pr.baseRefName} → ${nextBaseRefName}, clearing stale MERGE_CONFLICT dispatch records`);
760
+ }
761
+ pr.baseRefName = nextBaseRefName;
762
+ updated = true;
763
+ }
764
+
751
765
  // P-w1a3f9b2 — Phase 1.1: plumb mergeable / isDraft / mergeStateStatus /
752
766
  // headRefOid onto the PR object so watches captureState (engine/watches.js)
753
767
  // and future predicates (Phase 2.1: head-commit-change, mergeable-flipped,
@@ -2291,6 +2291,15 @@ function recordPrNoOpFixAttempt(target, cause, source, dispatchItem, branchChang
2291
2291
  return out;
2292
2292
  })()
2293
2293
  : {}),
2294
+ // #3079 — MERGE_CONFLICT noops record the composite guard key (source
2295
+ // head + base SHA + target ref name). The same-head guard at
2296
+ // engine.js:5676 reads this back and compares against the current
2297
+ // PR's prMergeConflictGuardKey, so a retarget (target ref change) or
2298
+ // a parent-merge (base SHA change) naturally releases the pause even
2299
+ // when the source head didn't move.
2300
+ ...(cause === shared.PR_FIX_CAUSE.MERGE_CONFLICT
2301
+ ? { mergeConflictKey: shared.prMergeConflictGuardKey(target) }
2302
+ : {}),
2294
2303
  };
2295
2304
  target.lastDispatchedAt = now;
2296
2305
  target.lastDispatchOutcome = 'noop';
@@ -2664,6 +2673,14 @@ function updatePrAfterFixError(pr, project, source, options = {}) {
2664
2673
  || '');
2665
2674
  if (commentKey) next.lastProcessedCommentKey = commentKey;
2666
2675
  }
2676
+ // #3079 — Refresh mergeConflictKey from live target state so a
2677
+ // mid-flight retarget reflects in the agent-error record too. The
2678
+ // engine.js skipConflictFix guard prefers mergeConflictKey over
2679
+ // legacy headSha; without this refresh, the prior record's stale
2680
+ // key could re-fire the suppression even after a retarget.
2681
+ if (cause === shared.PR_FIX_CAUSE.MERGE_CONFLICT) {
2682
+ next.mergeConflictKey = shared.prMergeConflictGuardKey(target);
2683
+ }
2667
2684
  target._lastDispatchByCause[cause] = next;
2668
2685
  result = { cause, indeterminate: true, errorClass };
2669
2686
  log('warn', `Updated ${pr.id} → recorded ${cause} agent-error fix attempt (indeterminate=true) — same-head guard relaxed for next tick${errorMessage ? ` (${errorMessage.slice(0, 80)})` : ''}`);
package/engine/shared.js CHANGED
@@ -2352,6 +2352,30 @@ const ENGINE_DEFAULTS = {
2352
2352
  autoConsolidateMemory: false, // opt-in: periodically spawn engine/kb-sweep-runner.js from the tick loop (4h cadence). Inbox→notes consolidation already runs every tick via consolidateInbox; this flag only controls the KB sweep.
2353
2353
  prNoOpFixPauseAttempts: 2, // pause one PR automation cause after repeated no-op fixes for unchanged evidence
2354
2354
  quarantineAutoRecoveryMax: 2, // #2996 follow-up: cap on auto-flipping WORKTREE_DIRTY/WORKTREE_DIVERGENT failures back to pending (the quarantine is self-healing so the next dispatch starts clean; the cap prevents infinite loops if quarantine itself keeps failing).
2355
+ // W-mq5n1zx5 — Layer 1a/2b: harden the quarantine rename path against
2356
+ // Windows EBUSY/EPERM races where a lingering `git.exe` descendant (leaked
2357
+ // by a status-probe timeout) still holds packfile handles when we try to
2358
+ // rename the worktree dir out of the way. With these on, the engine
2359
+ // retries the rename with jittered backoff (worst case ≈ baseMs * 2^N + N
2360
+ // jitter), then — if everything still fails — asks git to tear down its
2361
+ // own worktree record via `git worktree remove --force` so the next
2362
+ // dispatch isn't blocked by a half-renamed tree. Each knob is overridable
2363
+ // via `config.engine.<name>`.
2364
+ quarantineRenameRetryAttempts: 6,
2365
+ quarantineRenameRetryBaseMs: 250,
2366
+ quarantineForceRemoveFallback: true,
2367
+ // W-mq5n1zx5 — Layer 1b/2a: pre-empt the EBUSY race itself. (1b)
2368
+ // `--no-optional-locks` tells git to skip the .git/index.lock acquire
2369
+ // around the untracked-cache refresh in status; typical probe duration
2370
+ // under AV drops from 6–12s to <500ms, removing the timeout that
2371
+ // creates the leaked child in the first place. (2a) On Windows only,
2372
+ // before attempting the rename we walk live git.exe processes whose
2373
+ // command line points at the worktree path and Stop-Process them, so
2374
+ // a still-alive descendant from a previous status probe doesn't pin
2375
+ // packfile handles. Both gates are belt-and-suspenders — most
2376
+ // incidents are closed by (1b) alone — but both are cheap and idempotent.
2377
+ statusProbeUseNoOptionalLocks: true,
2378
+ statusProbeKillDescendantsWin32: true,
2355
2379
  completionReportRetentionDays: 90, // retain completion report sidecars beyond capped dispatch history
2356
2380
  completionReportMaxFiles: 5000, // hard cap for completion report sidecars during cleanup
2357
2381
  // P-bfa2c-cors-wildcard: extra Origins permitted to receive an
@@ -2538,6 +2562,17 @@ const ENGINE_DEFAULTS = {
2538
2562
  worktreeStuckThreshold: 10, // consecutive removal failures before escalation
2539
2563
  worktreeStuckSuppressMs: 60 * 60 * 1000, // 60min — suppress per-tick warn after escalation
2540
2564
  worktreeStuckSlowRetryMs: 30 * 60 * 1000, // 30min — slow-cadence retry window after escalation
2565
+ // W-mq6f2fe0000557fa — orphan-worktree GC: identify the OS process holding
2566
+ // a stuck worktree's cwd and, behind this opt-in flag, kill it. Default OFF
2567
+ // because killing a foreign process is destructive; flipping on rescues the
2568
+ // engine from the W-mq1k8z6o003acd89 / PID 16828 lock-spin pattern automatically.
2569
+ // Auto-reap only fires for processes whose cmdline matches `spawn-agent.js`
2570
+ // AND whose age exceeds `agentTimeout * 2` AND that have no live dispatch row
2571
+ // — i.e. unambiguous orphan agents the engine forgot about. The holder scan
2572
+ // itself ALWAYS runs on every orphan-sweep escalation (the flag only gates
2573
+ // the kill); holder details are appended to the inbox note either way.
2574
+ autoReapOrphanWorktreeHolders: false,
2575
+ orphanHolderScanTimeoutMs: 5000, // 5s ceiling for the cross-platform holder scan (PowerShell / /proc walk / lsof)
2541
2576
  ccMaxTurns: 50, // max tool-use turns per CC/doc-chat call before CLI stops (per response, not per session)
2542
2577
  ccTurnTimeoutMs: 300000, // W-mpmwxni2000c25c7-b/-d: 5min per-turn no-progress watchdog. The window resets on every liveness signal — token chunk, tool-call notification, tool-update — so an actively-streaming CC/doc-chat turn (long shell command, deep search, sub-agent loop) survives indefinitely up to the outer CC_CALL_TIMEOUT_MS (~1h) ceiling. Only true silence past this window with no progress fires the cancel: the in-flight LLM call is aborted and the handler surfaces `{code:'cc-turn-timeout', retryable:true}` via the typed error envelope so the UI can stop the spinner and offer Retry. Clamped to [10000, 3600000] in the settings POST handler. Independent of CC_CALL_TIMEOUT_MS. Non-streaming doc-chat is the lone wall-clock exception (no progress hooks); see _raceCcDocChatTimeout in dashboard.js for the dual factory/promise shape.
2543
2578
  docSessionMaxEntries: 200, // cap doc-chat session map/disk store by least-recent activity (LRU; sessions are non-expiring otherwise)
@@ -3682,6 +3717,7 @@ const FAILURE_CLASS = {
3682
3717
  WORKTREE_PREFLIGHT: 'worktree-preflight', // Pre-spawn worktree validation rejected (nested-in-project, drive-root collapse) — never retryable
3683
3718
  WORKTREE_DIRTY: 'worktree-dirty', // #2996: reused worktree had uncommitted edits and the engine could not auto-heal (or already quarantined). Non-retryable for this dispatch — next discovery cycle creates a fresh worktree.
3684
3719
  WORKTREE_DIVERGENT: 'worktree-divergent', // #2996: reused worktree's local branch was N commits ahead of origin (unsafe to reset, may contain unpushed agent work). Engine quarantined the worktree + backed up the local ref; non-retryable for this dispatch.
3720
+ WORKTREE_QUARANTINE_ENV_BLOCKED: 'worktree-quarantine-env-blocked', // W-mq5n1zx5: quarantine rename was attempted but the OS refused to release the worktree dir (Windows EBUSY/EPERM/EACCES from a lingering git.exe descendant). Engine retried with jittered backoff and (when enabled) `git worktree remove --force`; if all paths failed we surface this dedicated class so the WI auto-recovery loop can re-queue WITHOUT bumping the per-agent retry counter (the failure is environmental, not the agent's fault).
3685
3721
  DEPENDENCY_MERGE_SETUP: 'dependency-merge-setup', // Dependency pre-merge plumbing (stash/status/reset) failed before a real file conflict was verified. Retryable so a fresh worktree can recover.
3686
3722
  INVALID_KEEP_PROCESSES_WORKDIR: 'invalid-keep-processes-workdir', // W-mp6k7ywi000fa33c: keep-pids.json declared a cwd that is not a real git worktree (likely a selective copy of the repo) — never retryable; agent must rerun in a real worktree
3687
3723
  INVALID_KEEP_PROCESSES_SCHEMA: 'invalid-keep-processes-schema', // W-mp7i902u000l991f: keep-pids.json failed validation for a reason other than workdir (pids-missing, ttl-too-long, expires_at-missing, pids-too-many, port-invalid, etc.) — agent wrote the wrong shape; never retryable until they fix the file
@@ -6239,6 +6275,189 @@ function listAllProcesses() {
6239
6275
  return process.platform === 'win32' ? _winListProcesses() : _unixListProcesses();
6240
6276
  }
6241
6277
 
6278
+ // W-mq6f2fe0000557fa — orphan-worktree GC: identify OS processes whose cwd
6279
+ // is at or inside `dir`. Cross-platform with a fail-open contract: on any
6280
+ // error or timeout we return [] (the caller continues with its existing
6281
+ // escalation path and an unenriched note).
6282
+ //
6283
+ // Return shape: [{ pid, cmdline, startedAt, ageMs }, ...]
6284
+ // - pid: numeric OS pid
6285
+ // - cmdline: full command line as a single string (best-effort; possibly truncated)
6286
+ // - startedAt: ms-since-epoch process start time (0 when unknown)
6287
+ // - ageMs: Date.now() - startedAt (0 when startedAt is 0)
6288
+ //
6289
+ // Platform notes:
6290
+ // - Windows: PowerShell `Get-CimInstance Win32_Process` filtered by the
6291
+ // worktree basename appearing in CommandLine. (Win32_Process does not
6292
+ // expose the working directory; the basename match is the most-reliable
6293
+ // heuristic — every spawn-agent invocation embeds the dispatch id, which
6294
+ // is the worktree basename, in its argv.)
6295
+ // - Linux: walk `/proc/*/cwd` and keep PIDs whose readlinkSync resolves
6296
+ // under `dir`. Cmdline read from `/proc/<pid>/cmdline` (NUL-separated).
6297
+ // Start time derived from stat() mtime of the cwd entry as a proxy.
6298
+ // - macOS: `lsof -F p -d cwd` filtered by directory, then `ps -p <pid>` for
6299
+ // cmdline + start time.
6300
+ function findProcessesWithCwdInside(dir, opts = {}) {
6301
+ if (!dir || typeof dir !== 'string') return [];
6302
+ let resolved;
6303
+ try { resolved = path.resolve(dir); } catch { return []; }
6304
+ if (!resolved) return [];
6305
+ const timeoutMs = Number(opts.timeoutMs) > 0
6306
+ ? Number(opts.timeoutMs)
6307
+ : (ENGINE_DEFAULTS.orphanHolderScanTimeoutMs || 5000);
6308
+ const now = Date.now();
6309
+
6310
+ try {
6311
+ if (process.platform === 'win32') {
6312
+ return _findWindowsProcessesWithCwdInside(resolved, timeoutMs, now);
6313
+ }
6314
+ if (process.platform === 'linux') {
6315
+ return _findLinuxProcessesWithCwdInside(resolved, timeoutMs, now);
6316
+ }
6317
+ return _findMacProcessesWithCwdInside(resolved, timeoutMs, now);
6318
+ } catch { return []; }
6319
+ }
6320
+
6321
+ function _findWindowsProcessesWithCwdInside(resolved, timeoutMs, now) {
6322
+ // Win32_Process exposes CommandLine but not the cwd. Match by basename of
6323
+ // the worktree dir (the dispatch id like W-mq1k8z6o003acd89) appearing in
6324
+ // the cmdline — every spawn-agent prompt-file path embeds it.
6325
+ const basename = path.basename(resolved);
6326
+ if (!basename || basename.length < 3) return [];
6327
+ // PowerShell-quote: outer escape ' as ''
6328
+ const psBasename = basename.replace(/'/g, "''");
6329
+ const psResolved = resolved.replace(/'/g, "''");
6330
+ // Use -like with wildcards on BOTH basename and full path so a process
6331
+ // whose cmdline carries the worktree dir (but a different basename
6332
+ // anywhere in argv) also matches.
6333
+ const script = `Get-CimInstance Win32_Process | Where-Object { ($_.CommandLine -like '*${psBasename}*') -or ($_.CommandLine -like '*${psResolved}*') } | ForEach-Object { [PSCustomObject]@{ pid = $_.ProcessId; cmdline = $_.CommandLine; startedAt = if ($_.CreationDate) { [int64](([datetimeoffset]$_.CreationDate).ToUnixTimeMilliseconds()) } else { 0 } } } | ConvertTo-Json -Compress -Depth 2`;
6334
+ let raw;
6335
+ try {
6336
+ raw = _execSync(
6337
+ `powershell -NoProfile -NonInteractive -Command "${script}"`,
6338
+ { encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: timeoutMs, windowsHide: true, maxBuffer: 4 * 1024 * 1024 }
6339
+ );
6340
+ } catch { return []; }
6341
+ if (!raw || !String(raw).trim()) return [];
6342
+ let parsed;
6343
+ try { parsed = JSON.parse(raw); }
6344
+ catch { return []; }
6345
+ const arr = Array.isArray(parsed) ? parsed : [parsed];
6346
+ const out = [];
6347
+ for (const r of arr) {
6348
+ if (!r) continue;
6349
+ const pid = Number(r.pid);
6350
+ if (!Number.isInteger(pid) || pid <= 0) continue;
6351
+ const startedAt = Number(r.startedAt) || 0;
6352
+ out.push({
6353
+ pid,
6354
+ cmdline: r.cmdline ? String(r.cmdline) : '',
6355
+ startedAt,
6356
+ ageMs: startedAt > 0 ? Math.max(0, now - startedAt) : 0,
6357
+ });
6358
+ }
6359
+ return out;
6360
+ }
6361
+
6362
+ function _findLinuxProcessesWithCwdInside(resolved, _timeoutMs, now) {
6363
+ // /proc walk: scan numeric entries, readlink cwd, keep matches.
6364
+ let entries;
6365
+ try { entries = fs.readdirSync('/proc'); }
6366
+ catch { return []; }
6367
+ const prefix = resolved + path.sep;
6368
+ const out = [];
6369
+ for (const e of entries) {
6370
+ if (!/^\d+$/.test(e)) continue;
6371
+ const pid = Number(e);
6372
+ let cwd;
6373
+ try { cwd = fs.readlinkSync(`/proc/${pid}/cwd`); }
6374
+ catch { continue; }
6375
+ if (!cwd) continue;
6376
+ if (cwd !== resolved && !cwd.startsWith(prefix)) continue;
6377
+ let cmdline = '';
6378
+ try {
6379
+ const buf = fs.readFileSync(`/proc/${pid}/cmdline`);
6380
+ cmdline = buf.toString('utf8').replace(/\0/g, ' ').trim();
6381
+ } catch { /* cmdline read can fail on race */ }
6382
+ let startedAt = 0;
6383
+ try {
6384
+ const st = fs.statSync(`/proc/${pid}`);
6385
+ startedAt = st && st.ctimeMs ? Math.floor(st.ctimeMs) : 0;
6386
+ } catch { /* stat can fail on race */ }
6387
+ out.push({
6388
+ pid,
6389
+ cmdline,
6390
+ startedAt,
6391
+ ageMs: startedAt > 0 ? Math.max(0, now - startedAt) : 0,
6392
+ });
6393
+ }
6394
+ return out;
6395
+ }
6396
+
6397
+ function _findMacProcessesWithCwdInside(resolved, timeoutMs, now) {
6398
+ // `lsof -a -d cwd -F pn` emits PID then cwd path. Filter for dir matches,
6399
+ // then run a single `ps` pass to enrich cmdline + start time.
6400
+ let raw;
6401
+ try {
6402
+ raw = _execSync('lsof -a -d cwd -F pn 2>/dev/null', {
6403
+ encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: timeoutMs, maxBuffer: 4 * 1024 * 1024,
6404
+ });
6405
+ } catch { return []; }
6406
+ if (!raw) return [];
6407
+ const prefix = resolved + path.sep;
6408
+ const pids = new Set();
6409
+ let currentPid = null;
6410
+ for (const line of String(raw).split(/\r?\n/)) {
6411
+ if (!line) continue;
6412
+ const tag = line[0];
6413
+ const val = line.slice(1);
6414
+ if (tag === 'p') {
6415
+ const n = Number(val);
6416
+ currentPid = Number.isInteger(n) && n > 0 ? n : null;
6417
+ } else if (tag === 'n' && currentPid != null) {
6418
+ if (val === resolved || val.startsWith(prefix)) pids.add(currentPid);
6419
+ }
6420
+ }
6421
+ if (pids.size === 0) return [];
6422
+ const out = [];
6423
+ for (const pid of pids) {
6424
+ let line;
6425
+ try {
6426
+ line = String(_execSync(`ps -p ${pid} -o lstart=,command=`, {
6427
+ stdio: ['ignore', 'pipe', 'ignore'], timeout: 2000, encoding: 'utf8',
6428
+ }) || '').trim();
6429
+ } catch { line = ''; }
6430
+ let startedAt = 0;
6431
+ let cmdline = '';
6432
+ if (line) {
6433
+ // ps lstart= emits "Mon Jun 9 07:04:33 2026" (24 chars), then cmdline.
6434
+ const lstart = line.slice(0, 24).trim();
6435
+ cmdline = line.slice(24).trim();
6436
+ const parsed = Date.parse(lstart);
6437
+ if (!Number.isNaN(parsed)) startedAt = parsed;
6438
+ }
6439
+ out.push({
6440
+ pid,
6441
+ cmdline,
6442
+ startedAt,
6443
+ ageMs: startedAt > 0 ? Math.max(0, now - startedAt) : 0,
6444
+ });
6445
+ }
6446
+ return out;
6447
+ }
6448
+
6449
+ // W-mq6f2fe0000557fa — clear the per-path failure cooldown entry so the
6450
+ // post-holder-reap retry can attempt removeWorktree even though the path
6451
+ // has already failed >= 3 times. Without this, the cooldown silently
6452
+ // suppresses the retry and the orphan dir is left stuck.
6453
+ function clearWorktreeFailureCache(wtPath) {
6454
+ if (!wtPath) return false;
6455
+ try {
6456
+ const resolved = path.resolve(wtPath);
6457
+ return _removeWorktreeFailures.delete(resolved);
6458
+ } catch { return false; }
6459
+ }
6460
+
6242
6461
  // Cross-check a single PID's command line for a Minions agent invocation
6243
6462
  // (`claude` or `copilot`, including the `node spawn-agent.js --runtime <name>`
6244
6463
  // wrapper and `gh copilot` fallback). Used by orphan/recycled-PID safety:
@@ -6573,6 +6792,61 @@ function _prHeadSha(pr) {
6573
6792
  return String(pr?.headRefOid || pr?.headSha || pr?._adoSourceCommit || pr?._adoHeadCommit || '').trim();
6574
6793
  }
6575
6794
 
6795
+ // Target-branch base SHA, normalized across hosts. GitHub PRs carry `baseSha`
6796
+ // (engine/github.js:746-749); ADO PRs carry `_adoTargetCommit`
6797
+ // (engine/ado.js:1168-1172, inline in pollPrStatus).
6798
+ function _prBaseSha(pr) {
6799
+ return String(pr?.baseSha || pr?._adoTargetCommit || '').trim();
6800
+ }
6801
+
6802
+ // Target ref name, normalized across hosts. GitHub PRs carry `baseRefName`
6803
+ // (e.g. "master"); ADO PRs carry `targetRefName` (e.g. "refs/heads/main").
6804
+ function _prTargetRefName(pr) {
6805
+ return String(pr?.baseRefName || pr?.targetRefName || '').trim();
6806
+ }
6807
+
6808
+ // #3079 — Composite key the engine uses to gate merge-conflict re-dispatch.
6809
+ // The same-head guard at engine.js:5676 previously compared only source head
6810
+ // SHA, missing the case where a PR is retargeted (e.g. parent branch → main)
6811
+ // without the source moving. Including the base SHA and target ref name makes
6812
+ // the guard target-aware so retargets naturally release the pause.
6813
+ function prMergeConflictGuardKey(pr) {
6814
+ return `${_prHeadSha(pr)}|${_prBaseSha(pr)}|${_prTargetRefName(pr)}`;
6815
+ }
6816
+
6817
+ // #3079 — Helper invoked by the ADO + GitHub poller metadata-apply paths.
6818
+ // When the PR's target ref changes (user retargeted via gh pr edit --base or
6819
+ // ADO PR edit), clear the stale MERGE_CONFLICT records so the next tick can
6820
+ // re-evaluate the conflict against the new target. Other causes
6821
+ // (BUILD_FAILURE, REVIEW_FEEDBACK, HUMAN_FEEDBACK) are untouched — they are
6822
+ // not target-sensitive. First-poll seeding (no prior target ref recorded)
6823
+ // is a no-op so we don't strip state immediately after an upsert.
6824
+ //
6825
+ // Returns true when state was actually mutated (callers may use this to set
6826
+ // the surrounding "updated" flag and trigger a persist).
6827
+ function resetMergeConflictStateOnRetarget(pr, newTargetRef) {
6828
+ if (!pr) return false;
6829
+ const next = String(newTargetRef || '').trim();
6830
+ if (!next) return false;
6831
+ const prior = _prTargetRefName(pr);
6832
+ // First-poll seeding: no prior value → record new value via the caller,
6833
+ // do NOT clear merge-conflict state.
6834
+ if (!prior) return false;
6835
+ if (prior === next) return false;
6836
+ let changed = false;
6837
+ if (pr._lastDispatchByCause && pr._lastDispatchByCause[PR_FIX_CAUSE.MERGE_CONFLICT]) {
6838
+ delete pr._lastDispatchByCause[PR_FIX_CAUSE.MERGE_CONFLICT];
6839
+ if (Object.keys(pr._lastDispatchByCause).length === 0) delete pr._lastDispatchByCause;
6840
+ changed = true;
6841
+ }
6842
+ if (pr._noOpFixes && pr._noOpFixes[PR_FIX_CAUSE.MERGE_CONFLICT]) {
6843
+ delete pr._noOpFixes[PR_FIX_CAUSE.MERGE_CONFLICT];
6844
+ if (Object.keys(pr._noOpFixes).length === 0) delete pr._noOpFixes;
6845
+ changed = true;
6846
+ }
6847
+ return changed;
6848
+ }
6849
+
6576
6850
  function prFixEvidenceFingerprint(pr, cause = PR_FIX_CAUSE.UNKNOWN) {
6577
6851
  const review = pr?.minionsReview || {};
6578
6852
  const feedback = pr?.humanFeedback || {};
@@ -6596,6 +6870,16 @@ function prFixEvidenceFingerprint(pr, cause = PR_FIX_CAUSE.UNKNOWN) {
6596
6870
  evidence.mergeConflict = !!pr?._mergeConflict;
6597
6871
  evidence.mergeStatus = pr?.mergeStatus || '';
6598
6872
  evidence.mergeConflictDetail = pr?._mergeConflictDetail || '';
6873
+ // #3079 — same rationale as BUILD_FAILURE / REVIEW_FEEDBACK in #2979:
6874
+ // without head/base/target fields the fingerprint was sticky across a
6875
+ // rebase + force-push AND across a PR retarget, so existing
6876
+ // _noOpFixes[MERGE_CONFLICT] pauses never released. Adding head + base
6877
+ // SHA + target ref name gives MERGE_CONFLICT the same natural-unsticking
6878
+ // property the other auto-fix causes already have.
6879
+ evidence.headRefOid = _prHeadSha(pr);
6880
+ evidence.baseSha = _prBaseSha(pr);
6881
+ evidence.targetRefName = _prTargetRefName(pr);
6882
+ evidence.lastPushedAt = pr?.lastPushedAt || '';
6599
6883
  } else {
6600
6884
  evidence.reviewStatus = pr?.reviewStatus || '';
6601
6885
  evidence.lastReviewedAt = pr?.lastReviewedAt || '';
@@ -7221,6 +7505,8 @@ module.exports = {
7221
7505
  PR_FIX_CAUSE,
7222
7506
  getPrFixAutomationCause,
7223
7507
  prFixEvidenceFingerprint,
7508
+ prMergeConflictGuardKey,
7509
+ resetMergeConflictStateOnRetarget,
7224
7510
  getPrNoOpFixRecord,
7225
7511
  isPrNoOpFixCausePaused,
7226
7512
  getPrPausedCauses,
@@ -7233,6 +7519,8 @@ module.exports = {
7233
7519
  killByPidsImmediate,
7234
7520
  isProcessCommandLineMatchingAgent,
7235
7521
  listAllProcesses,
7522
+ findProcessesWithCwdInside,
7523
+ clearWorktreeFailureCache,
7236
7524
  listProcessDescendants,
7237
7525
  listProcessReachable,
7238
7526
  removeWorktree,