@yemi33/minions 0.1.2305 → 0.1.2307

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/bin/minions.js +1 -0
  2. package/dashboard/js/refresh.js +17 -13
  3. package/dashboard/js/render-other.js +251 -3
  4. package/dashboard/js/render-plans.js +1 -1
  5. package/dashboard/js/render-prs.js +4 -1
  6. package/dashboard/js/render-utils.js +13 -0
  7. package/dashboard/js/render-work-items.js +11 -13
  8. package/dashboard/js/settings.js +30 -3
  9. package/dashboard/slim/body.html +6 -22
  10. package/dashboard/slim/js/modals-tiles.js +32 -22
  11. package/dashboard/slim/js/plans.js +24 -516
  12. package/dashboard/slim/styles.css +10 -7
  13. package/dashboard/styles.css +58 -0
  14. package/dashboard.js +66 -9
  15. package/docs/completion-reports.md +27 -2
  16. package/docs/copilot-cli-schema.md +31 -1
  17. package/docs/design-state-storage.md +1 -1
  18. package/docs/harness-transparency.md +18 -0
  19. package/docs/live-checkout-mode.md +57 -8
  20. package/engine/ado-comment.js +5 -2
  21. package/engine/ado.js +37 -0
  22. package/engine/cli.js +58 -3
  23. package/engine/comment-format.js +82 -2
  24. package/engine/consolidation.js +59 -2
  25. package/engine/discover-project-skills.js +4 -0
  26. package/engine/dispatch.js +36 -1
  27. package/engine/gh-comment.js +8 -3
  28. package/engine/lifecycle.js +241 -25
  29. package/engine/live-checkout.js +335 -7
  30. package/engine/playbook.js +2 -2
  31. package/engine/pre-dispatch-eval.js +54 -5
  32. package/engine/queries.js +11 -7
  33. package/engine/shared.js +96 -6
  34. package/engine/supervisor.js +144 -19
  35. package/engine/watchdog.js +10 -0
  36. package/engine.js +210 -5
  37. package/package.json +1 -1
  38. package/playbooks/review.md +2 -0
  39. package/playbooks/shared-rules.md +9 -0
package/engine/shared.js CHANGED
@@ -1946,6 +1946,86 @@ function mutateControl(mutator) {
1946
1946
  }, { defaultValue: { state: 'stopped', pid: null }, skipWriteIfUnchanged: true });
1947
1947
  }
1948
1948
 
1949
+ // ── Crash-loop detection (W-mr2c46590003e3ee) ──────────────────────────────
1950
+ // Three independent mechanisms can respawn a dead/frozen engine process:
1951
+ // engine/supervisor.js#checkEngine / #checkEngineHung, engine/watchdog.js
1952
+ // (OS-scheduled external recovery), and dashboard.js's in-process 30s
1953
+ // watchdog. None of them counted or throttled repeated crashes, so a genuine
1954
+ // crash-loop (engine dying every few minutes) silently self-healed forever
1955
+ // with no human-visible signal — see the 2026-07-01 00:20-00:30 incident
1956
+ // (4 respawns in ~5 minutes, notes/inbox/dallas-W-mr2azk6i000y06e0-*.md).
1957
+ //
1958
+ // recordEngineRespawn() gives all three mechanisms a single rolling counter
1959
+ // persisted in control.json so it survives process restarts: every call
1960
+ // appends {ts, source}, prunes entries older than CRASH_LOOP_WINDOW_MS, and
1961
+ // once the count within the window reaches CRASH_LOOP_THRESHOLD it flips a
1962
+ // dashboard-visible `control.crashLoopAlert` flag and writes a deduped
1963
+ // (one-per-day) inbox note.
1964
+ //
1965
+ // Callers MUST invoke this only at the point an actual respawn action is
1966
+ // taken — not on every health-check tick. The existing
1967
+ // RESPAWN_IN_PROGRESS_WINDOW_MS / post-spawn-grace guards already in each
1968
+ // mechanism prevent two watchers from double-spawning for the same crash, so
1969
+ // recording only real spawns keeps a single crash from being counted 2-3x
1970
+ // across the three watchers.
1971
+ const CRASH_LOOP_WINDOW_MS = Number(process.env.MINIONS_CRASH_LOOP_WINDOW_MS) || 10 * 60 * 1000; // 10 min
1972
+ const CRASH_LOOP_THRESHOLD = Number(process.env.MINIONS_CRASH_LOOP_THRESHOLD) || 3;
1973
+
1974
+ function recordEngineRespawn(source) {
1975
+ const now = Date.now();
1976
+ let crashLoop = false;
1977
+ let count = 1;
1978
+ let windowEntries = [];
1979
+ try {
1980
+ mutateControl((control) => {
1981
+ const prior = Array.isArray(control.engineRespawns) ? control.engineRespawns : [];
1982
+ const entries = prior
1983
+ .filter(e => e && Number.isFinite(e.ts) && (now - e.ts) <= CRASH_LOOP_WINDOW_MS)
1984
+ .concat([{ ts: now, source: String(source || 'unknown') }]);
1985
+ // Defensive cap — the window filter above already bounds this in
1986
+ // practice, but guards against unbounded growth under clock skew or an
1987
+ // operator-widened window.
1988
+ control.engineRespawns = entries.slice(-50);
1989
+ windowEntries = control.engineRespawns;
1990
+ count = windowEntries.length;
1991
+ if (count >= CRASH_LOOP_THRESHOLD) {
1992
+ crashLoop = true;
1993
+ control.crashLoopAlert = {
1994
+ count,
1995
+ windowMs: CRASH_LOOP_WINDOW_MS,
1996
+ firstAt: new Date(windowEntries[0].ts).toISOString(),
1997
+ lastAt: new Date(now).toISOString(),
1998
+ sources: windowEntries.map(e => e.source),
1999
+ };
2000
+ } else if (control.crashLoopAlert) {
2001
+ // Window rolled off past incidents — clear a stale alert so the
2002
+ // dashboard doesn't keep showing a resolved crash-loop forever.
2003
+ delete control.crashLoopAlert;
2004
+ }
2005
+ return control;
2006
+ });
2007
+ } catch (e) {
2008
+ log('warn', `recordEngineRespawn: control.json update failed: ${e.message}`);
2009
+ }
2010
+
2011
+ if (crashLoop) {
2012
+ try {
2013
+ const lines = windowEntries.map(e => `- ${e.source} @ ${new Date(e.ts).toISOString()}`).join('\n');
2014
+ const body = `# Engine crash-loop detected\n\n` +
2015
+ `${count} engine respawns in the last ${Math.round(CRASH_LOOP_WINDOW_MS / 60000)} minutes ` +
2016
+ `(threshold: ${CRASH_LOOP_THRESHOLD}).\n\n` +
2017
+ `Respawns:\n${lines}\n\n` +
2018
+ `The engine keeps self-healing (each respawn succeeded) but is crashing repeatedly — ` +
2019
+ `investigate \`engine/engine-stdio.log\` around the timestamps above before this becomes ` +
2020
+ `a silent outage.\n`;
2021
+ writeToInbox('engine', 'crash-loop-detected', body, null, { severity: 'critical' });
2022
+ } catch (e) {
2023
+ log('warn', `recordEngineRespawn: inbox alert failed: ${e.message}`);
2024
+ }
2025
+ }
2026
+ return { count, crashLoop };
2027
+ }
2028
+
1949
2029
  // W-mp60tw0u000j3931: Lock-safe read-modify-write for engine/state.json — the
1950
2030
  // persistent cross-restart engine-level state file (migration markers,
1951
2031
  // one-shot reconciliation versions, etc.). Mirrors mutateControl's shape.
@@ -2904,6 +2984,14 @@ const ENGINE_DEFAULTS = {
2904
2984
  // & Review Loop, or set `engine.autoFixPaused: true` in config.json.
2905
2985
  autoFixPaused: false, // hard-stop kill-switch / master override (see comment above)
2906
2986
  autoConsolidateMemory: false, // opt-in: periodically spawn engine/kb-sweep-runner.js from the tick loop (4h cadence). Inbox→notes consolidation already runs every tick via consolidateInbox; this flag only controls the KB sweep.
2987
+ // W-mqtvnnj1000357fa — fleet-wide fallback for live-checkout auto-stash. When
2988
+ // true, a dirty live-checkout tree is `git stash push --include-untracked`'d
2989
+ // before dispatch instead of failing with FAILURE_CLASS.LIVE_CHECKOUT_DIRTY,
2990
+ // so dispatch proceeds without manual operator intervention. Per-project
2991
+ // `project.liveCheckoutAutoStash` overrides this (resolveLiveCheckoutAutoStash).
2992
+ // The engine never auto-pops the stash — the operator runs `git stash pop`.
2993
+ // Default false so a fresh install behaves identically to before this knob.
2994
+ liveCheckoutAutoStash: false,
2907
2995
  prNoOpFixPauseAttempts: 2, // pause one PR automation cause after repeated no-op fixes for unchanged evidence
2908
2996
  quarantineAutoRecoveryMax: 2, // #2996 follow-up: cap on auto-flipping WORKTREE_DIRTY/WORKTREE_DIVERGENT failures back to pending (the quarantine is self-healing so the next dispatch starts clean; the cap prevents infinite loops if quarantine itself keeps failing).
2909
2997
  // W-mq5n1zx5 — Layer 1a/2b: harden the quarantine rename path against
@@ -3254,11 +3342,12 @@ const ENGINE_DEFAULTS = {
3254
3342
  // When ON, a dirty/broken live-checkout tree is `git fetch origin` +
3255
3343
  // `git reset --hard origin/<branch>`'d before dispatch instead of failing
3256
3344
  // LIVE_CHECKOUT_DIRTY. Per-project `project.liveCheckoutAutoReset` overrides
3257
- // this. Default OFF — auto-reset is destructive (it discards the operator's
3258
- // uncommitted changes), so it is strictly opt-in. Resolution precedence lives
3259
- // in `resolveLiveCheckoutAutoReset`. Fires ONLY on confirmed dirty-tree
3260
- // detection, never on mid-operation / blob-fetch / tooling failures.
3261
- liveCheckoutAutoReset: false,
3345
+ // this. Default ON — discarding uncommitted dirt via reset-to-remote keeps
3346
+ // live-checkout branches from diverging (the old `false` default accumulated
3347
+ // `minions: auto-save agent WIP` commits on every dirty exit). Resolution
3348
+ // precedence lives in `resolveLiveCheckoutAutoReset`. Fires ONLY on confirmed
3349
+ // dirty-tree detection, never on mid-operation / blob-fetch / tooling failures.
3350
+ liveCheckoutAutoReset: true,
3262
3351
  orphanHolderScanTimeoutMs: 5000, // 5s ceiling for the cross-platform holder scan (PowerShell / /proc walk / lsof)
3263
3352
  ccMaxTurns: 50, // max tool-use turns per CC/doc-chat call before CLI stops (per response, not per session)
3264
3353
  ccWorkerIdleTimeoutMs: 30 * 60 * 1000, // W-mr0qs0vw: idle-reaper window for the persistent `copilot --acp` worker pool (engine/cc-worker-pool.js). After this much inactivity with no in-flight turn the warm ACP process is killed; the next message cold-spawns a fresh session with NO memory of prior turns (CC shows a "context cleared after inactivity" notice). Tradeoff: shorter = less idle memory/process footprint, longer = more context durability across gaps between messages. Wired into the pool via ccWorkerPool.setIdleTimeoutMs() on every reloadConfig(); clamped to [60000, 28800000] (1min–8h) in the settings POST handler.
@@ -4555,6 +4644,7 @@ const FAILURE_CLASS = {
4555
4644
  LIVE_CHECKOUT_FAILED: 'live-checkout-failed', // #305 (live-checkout dispatch mode): prepareLiveCheckout THREW before agent spawn (helper guard, ref validation, or a transient `git status`/`rev-parse`/`checkout` failure) — distinct from the confirmed-dirty result (LIVE_CHECKOUT_DIRTY) which the helper returns, not throws. A thrown error is NOT proof the tree is dirty, so it must not be over-classified as dirty. Retryable with bounded backoff (NOT in dispatch.js neverRetry): racy branch-lock handoff, just-finished sibling dispatch, or transient git errors frequently clear on the next attempt; the engine auto-retries up to ENGINE_DEFAULTS.maxRetries before giving up. Genuinely terminal underlying reasons (auth, validation) still short-circuit via the reason-string check in isRetryableFailureReason.
4556
4645
  LIVE_CHECKOUT_MID_OPERATION: 'live-checkout-mid-operation', // P-a7f3c1d9 (live-checkout dispatch mode): spawnAgent could not switch/create the target branch in project.localPath because the operator tree is mid-operation — an in-progress merge/rebase/cherry-pick/bisect or a detached HEAD. Distinct from LIVE_CHECKOUT_DIRTY (uncommitted changes): here the tree may be clean but the branch op cannot proceed. Engine refuses to spawn (it never runs `git reset`/`git clean`/`git rebase --abort` against the operator tree). Non-retryable — operator must finish or abort the in-progress operation, or checkout a branch, before re-dispatch.
4557
4646
  LIVE_CHECKOUT_BLOB_FETCH: 'live-checkout-blob-fetch', // PL-live-checkout-reliability-hardening (live-checkout dispatch mode): `git checkout <existing-branch>` in project.localPath failed because the tree could not be materialized — on a Scalar/GVFS-managed ADO partial (blobless) clone, switching onto a branch whose tree differs from HEAD hydrates the changed paths' blobs through the GVFS cache server (`*.gvfscache.dev.azure.com`), a different endpoint than the main git remote that receives NO auth in the headless engine shell, so the fetch fails DETERMINISTICALLY. Distinct from LIVE_CHECKOUT_FAILED (transient) because retrying reproduces it identically — surfaced once as an operator-actionable refusal instead of retry-storming to the cap. Non-retryable — the operator hydrates the branch once with their own credentials (`git checkout <branch>` interactively, or `scalar prefetch` / `git -C <repo> fetch`), then re-dispatches. The engine NEVER forces, resets, cleans, or stashes the operator tree, and best-effort switches HEAD back to the original ref so the tree is not stranded half-populated.
4647
+ LIVE_CHECKOUT_WORKTREE_CONFLICT: 'live-checkout-worktree-conflict', // W-mr28h2j2000y0de1 (live-checkout dispatch mode): `git checkout <existing-branch>` in project.localPath failed because that exact branch is ALSO checked out in a SECOND worktree elsewhere (a leftover from a prior isolated-worktree dispatch, a manually-created worktree, or a stale worktree left by a checkoutMode change). git refuses deterministically with `fatal: '<branch>' is already used by worktree at '<path>'`. Distinct from LIVE_CHECKOUT_FAILED (transient) because this is a STRUCTURAL conflict — it does NOT clear on retry, ever, until a human or the engine removes/reassigns the other worktree; retrying just reproduces it identically and burns maxRetries. Non-retryable — the operator either `git worktree remove <path>` (freeing the branch) if that worktree is stale, or finishes/commits/pushes from it directly if it holds real WIP. The engine NEVER forces, resets, cleans, or stashes the operator tree, and best-effort switches HEAD back to the original ref so the tree is not stranded.
4558
4648
  INVALID_WORKDIR: 'invalid-workdir', // P-714ef144: dispatch carried a meta.workdir override that failed validation — non-string, absolute path, drive-letter prefix, null byte, ".." segment, or post-resolve containment escape against project.localPath / worktree root. Engine refuses to spawn (the subpath would either be unreachable on disk or point outside the operator's allowed surface). Non-retryable — operator must fix the WI's meta.workdir before re-dispatch. Inbox alert lists the offending value + the resolved-vs-base mismatch.
4559
4649
  MODEL_UNAVAILABLE: 'model-unavailable', // W-mpg6isvy000xca4d: requested model returned overloaded_error / 503 / service_unavailable. Retriable — engine swaps in the runtime-appropriate fallback model on next spawn (Claude leans on --fallback-model already plumbed; Copilot overrides --model with engine.copilotFallbackModel).
4560
4650
  WORKSPACE_MANIFEST_REPO: 'workspace-manifest-repo-forbidden', // W-mq07avbk000m5543: dispatch routed an agent to a project/repo not present in its workspace_manifest.allowed_repos. Structural — never retryable until the manifest is widened or a different agent is chosen.
@@ -9060,7 +9150,7 @@ module.exports = {
9060
9150
  assertStateFileSize,
9061
9151
  withFileLock,
9062
9152
  mutateJsonFileLocked,
9063
- mutateControl,
9153
+ mutateControl, recordEngineRespawn,
9064
9154
  mutateEngineState, // W-mp60tw0u000j3931
9065
9155
  readEngineState, // W-mp60tw0u000j3931
9066
9156
  mutateCooldowns,
@@ -5,7 +5,14 @@
5
5
  * Spawned `detached: true` by `minions start`/`restart` after the engine and
6
6
  * dashboard are up. Polls every SUPERVISOR_INTERVAL_MS:
7
7
  * - engine PID from `engine/control.json`
8
- * - dashboard PID via port-listener probe (port 7331 by default)
8
+ * - dashboard via a port-listener probe (port 7331 by default) AND, when the
9
+ * port is bound, an HTTP `/api/health` probe. The HTTP layer catches the
10
+ * "frozen but still listening" dashboard (wedged event loop) that the
11
+ * port-listen probe alone reports as healthy — it respawns only after
12
+ * DASH_HEALTH_MAX_FAILS consecutive failed probes so a dashboard merely
13
+ * busy with a heavy synchronous getStatus() rebuild isn't killed. What
14
+ * counts as "responsive" (and why engine 'degraded'/'stopped' does NOT mean
15
+ * the dashboard is frozen) is documented at _probeDashboardHealth.
9
16
  *
10
17
  * When either is dead AND the stop-intent flag is NOT set, respawns the dead
11
18
  * one in the same way the CLI does (detached, stdio routed to the engine-
@@ -38,6 +45,17 @@ const MINIONS_DIR = path.resolve(__dirname, '..');
38
45
  function _sharedOrNull() {
39
46
  try { return require('./shared'); } catch { return null; }
40
47
  }
48
+ // Crash-loop counter (W-mr2c46590003e3ee) — best-effort. shared.js owns the
49
+ // counter/threshold/inbox-alert logic (engine/shared.js#recordEngineRespawn)
50
+ // so supervisor.js, engine/watchdog.js, and dashboard.js's in-process
51
+ // watchdog all feed the same rolling window in control.json instead of each
52
+ // keeping (and silently ignoring) their own. Falls through to a no-op when
53
+ // shared.js can't load, matching this file's existing fail-open posture.
54
+ function _recordCrashLoopRespawn(source) {
55
+ const shared = _sharedOrNull();
56
+ if (!shared || typeof shared.recordEngineRespawn !== 'function') return;
57
+ try { shared.recordEngineRespawn(source); } catch { /* best-effort */ }
58
+ }
41
59
  function _engineDir() { return _sharedOrNull()?.ENGINE_DIR || __dirname; }
42
60
  const STATIC_ENGINE_DIR = __dirname; // for require-time paths that can't be lazy
43
61
  function CONTROL_PATH_FN() { return path.join(_engineDir(), 'control.json'); }
@@ -76,6 +94,16 @@ function _resolveDashPort() {
76
94
  const POST_SPAWN_GRACE_MS = Number(process.env.MINIONS_SUPERVISOR_GRACE_MS) || 15000;
77
95
  // #421 — heartbeat age threshold before supervisor considers the engine event loop frozen.
78
96
  const SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS = Number(process.env.MINIONS_SUPERVISOR_STALE_HEARTBEAT_MS) || 180000; // 3 min
97
+ // Dashboard HTTP liveness (frozen-but-listening detection). A dashboard whose
98
+ // event loop is wedged keeps the socket LISTENING, so the port-listen probe
99
+ // alone reports it healthy while the in-browser SPA shows the "unreachable"
100
+ // banner forever. We layer an HTTP /api/health probe on top: only respawn after
101
+ // DASH_HEALTH_MAX_FAILS *consecutive* failed probes so a dashboard that is
102
+ // merely busy with a heavy synchronous getStatus() rebuild (documented to block
103
+ // the loop 15-25s) isn't mistaken for a hang. With the default 30s interval,
104
+ // 3 strikes ≈ 90s of continuous unresponsiveness before we act.
105
+ const DASH_HEALTH_TIMEOUT_MS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_TIMEOUT_MS) || 5000;
106
+ const DASH_HEALTH_MAX_FAILS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_FAILS) || 3;
79
107
  const isWin = process.platform === 'win32';
80
108
 
81
109
  function safeReadJson(p) {
@@ -354,6 +382,39 @@ function spawnDashboard() {
354
382
  // the engine.
355
383
  let _lastEngineRespawnAt = 0;
356
384
  let _lastDashboardRespawnAt = 0;
385
+ // Consecutive failed dashboard HTTP health probes. Reset on any healthy probe,
386
+ // after a respawn, and inside the post-spawn grace window.
387
+ let _dashHealthFailStreak = 0;
388
+
389
+ // Lazy, defensive HTTP health probe. Reuses restart-health.js's httpGetJson
390
+ // (rule: reuse before re-implementing) but never lets a load failure (e.g. a
391
+ // mid-upgrade broken shared.js that restart-health requires at module top)
392
+ // break the supervisor. Returns null when the probe helper is unavailable so
393
+ // the caller can FALL BACK to port-only liveness instead of false-respawning a
394
+ // healthy dashboard.
395
+ function _probeDashboardHealth(port, timeoutMs = DASH_HEALTH_TIMEOUT_MS) {
396
+ let getJson = null;
397
+ try {
398
+ const rh = require('./restart-health');
399
+ if (rh && rh._private && typeof rh._private.httpGetJson === 'function') {
400
+ getJson = rh._private.httpGetJson;
401
+ }
402
+ } catch { /* restart-health/shared unavailable — fall through */ }
403
+ if (!getJson) return Promise.resolve(null);
404
+ return getJson(`http://127.0.0.1:${port}/api/health`, timeoutMs)
405
+ // "Responsive" == the dashboard event loop executed handleHealth and returned
406
+ // a well-formed 2xx envelope, REGARDLESS of the engine-derived status value.
407
+ // /api/health reports 'degraded' when the engine is paused/stopping and
408
+ // 'stopped' when the engine is down — those are legitimate ENGINE states
409
+ // (owned by checkEngine/checkEngineHung), not evidence the DASHBOARD process
410
+ // is frozen. Gating on status==='healthy' would respawn a perfectly
411
+ // responsive dashboard every time an operator pauses the engine, tearing
412
+ // down the socket and tripping the in-browser "unreachable" banner for no
413
+ // reason. Only a failed/timed-out request (res.ok false or a rejection)
414
+ // means the dashboard's own event loop is actually wedged.
415
+ .then(res => !!(res && res.ok && res.json && typeof res.json.status === 'string'))
416
+ .catch(() => false);
417
+ }
357
418
 
358
419
  // Window during which a `null` pid + recent `restarted_at` is interpreted as
359
420
  // "another watchdog is currently respawning the engine — don't double-spawn."
@@ -364,12 +425,29 @@ let _lastDashboardRespawnAt = 0;
364
425
  // to enter cli.js:443 and write its own PID on cold Windows boots.
365
426
  const RESPAWN_IN_PROGRESS_WINDOW_MS = Number(process.env.MINIONS_SUPERVISOR_RESPAWN_WINDOW_MS) || 60000;
366
427
 
428
+ // #3758 — 'degraded' is a FAILURE state (set by dashboard.js's frozen-tick
429
+ // watchdog, see dashboard.js#_markEngineAsDegradedIfFrozen), not a deliberate
430
+ // user-intent state like 'paused'/'stopping'/'stopped'. When a frozen tick loop
431
+ // finally unblocks, engine.js#tickInner reads control.state !== 'running' and
432
+ // calls process.exit(0) — self-exiting exactly BECAUSE it saw 'degraded'. The
433
+ // dashboard.js comment documents the expected recovery: "the PID-dead watchdog
434
+ // then auto-restarts it." Excluding 'degraded' here left that promise unmet —
435
+ // the engine PID goes away and the control plane silently stays down until a
436
+ // manual `minions restart`. Respawn eligibility must include both 'running'
437
+ // (still up) and 'degraded' (crashed/crashing out of a real failure); only
438
+ // 'paused', 'stopping', and 'stopped' reflect actual user intent to stay down.
439
+ function _isRespawnEligibleEngineState(state) {
440
+ return state === 'running' || state === 'degraded';
441
+ }
442
+
367
443
  function checkEngine(now) {
368
444
  if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
369
445
  const control = safeReadJson(CONTROL_PATH_FN());
370
- // Only respawn when control.json says "running" — paused/stopped/stopping
371
- // are legitimate states the supervisor must not override.
372
- if (!control || control.state !== 'running') return;
446
+ // Respawn when control.json says "running" (PID died unexpectedly) or
447
+ // "degraded" (watchdog-flagged failure state — see _isRespawnEligibleEngineState
448
+ // above). "paused"/"stopped"/"stopping" are legitimate user-intent states the
449
+ // supervisor must not override.
450
+ if (!control || !_isRespawnEligibleEngineState(control.state)) return;
373
451
  if (control.pid && isPidAlive(control.pid)) return;
374
452
 
375
453
  // Cross-watchdog race guard: dashboard.js's in-process engine watchdog
@@ -391,6 +469,7 @@ function checkEngine(now) {
391
469
  reapStrayProcesses(path.join(MINIONS_DIR, 'engine.js'), 'engine');
392
470
  const newPid = spawnEngine();
393
471
  _lastEngineRespawnAt = now;
472
+ _recordCrashLoopRespawn('supervisor:dead-pid');
394
473
  console.log(`[supervisor] Engine respawned (new PID: ${newPid})`);
395
474
  }
396
475
 
@@ -403,7 +482,11 @@ function checkEngine(now) {
403
482
  function checkEngineHung(now) {
404
483
  if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
405
484
  const control = safeReadJson(CONTROL_PATH_FN());
406
- if (!control || control.state !== 'running') return;
485
+ // Include 'degraded' (see _isRespawnEligibleEngineState) so a permanently
486
+ // frozen tick loop that never gets a chance to self-exit (tickInner never
487
+ // re-enters) still gets force-restarted once its own heartbeat goes stale,
488
+ // instead of being ignored forever the moment the state flips to 'degraded'.
489
+ if (!control || !_isRespawnEligibleEngineState(control.state)) return;
407
490
  // Only acts when the PID is alive — a dead PID is handled by checkEngine().
408
491
  if (!control.pid || !isPidAlive(control.pid)) return;
409
492
 
@@ -421,34 +504,73 @@ function checkEngineHung(now) {
421
504
  reapStrayProcesses(path.join(MINIONS_DIR, 'engine.js'), 'engine');
422
505
  const newPid = spawnEngine();
423
506
  _lastEngineRespawnAt = now;
507
+ _recordCrashLoopRespawn('supervisor:stale-heartbeat');
424
508
  console.log(`[supervisor] Engine restarted due to stale heartbeat (new PID: ${newPid})`);
425
509
  }
426
510
 
427
- function checkDashboard(now) {
428
- if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) return;
511
+ async function checkDashboard(now) {
512
+ if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) { _dashHealthFailStreak = 0; return; }
429
513
  const dashPort = _resolveDashPort();
430
514
  const pids = listeningPidsForPort(dashPort);
431
- if (pids.length > 0) return;
432
-
433
- console.log(`[supervisor] Dashboard not listening on port ${dashPort} — respawning...`);
434
- // Reap stray dashboards first. Besides bounding the count, this collapses the
435
- // port-desync case: orphan dashboards holding other ports are killed, freeing
436
- // the canonical port for the fresh one whose beacon _resolveDashPort() reads.
437
- reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
438
- const newPid = spawnDashboard();
439
- _lastDashboardRespawnAt = now;
440
- console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
515
+
516
+ const respawn = (why) => {
517
+ console.log(`[supervisor] Dashboard ${why} on port ${dashPort} — respawning...`);
518
+ // Reap stray dashboards first. Besides bounding the count, this collapses the
519
+ // port-desync case: orphan dashboards holding other ports are killed, freeing
520
+ // the canonical port for the fresh one whose beacon _resolveDashPort() reads.
521
+ // Critically, for the frozen-but-listening case the reap also KILLS the wedged
522
+ // process still holding the port so spawnDashboard() can bind it.
523
+ reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
524
+ const newPid = spawnDashboard();
525
+ _lastDashboardRespawnAt = now;
526
+ _dashHealthFailStreak = 0;
527
+ console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
528
+ };
529
+
530
+ // Hard-down: nothing bound to the port → respawn immediately.
531
+ if (pids.length === 0) {
532
+ _dashHealthFailStreak = 0;
533
+ respawn('not listening');
534
+ return;
535
+ }
536
+
537
+ // Port is bound. Verify the event loop is actually responsive — a frozen-but-
538
+ // listening dashboard keeps the socket LISTENING yet serves the in-browser
539
+ // "unreachable" banner. Probe /api/health; only respawn after a streak of
540
+ // failures so a legitimately busy getStatus() rebuild isn't mistaken for a hang.
541
+ const healthy = await _probeDashboardHealth(dashPort);
542
+ if (healthy === null) {
543
+ // Probe helper unavailable — fall back to port-only liveness (the pre-existing
544
+ // behavior). Don't count this as a failure; we won't respawn a bound dashboard
545
+ // we can't actually prove is frozen.
546
+ _dashHealthFailStreak = 0;
547
+ return;
548
+ }
549
+ if (healthy) { _dashHealthFailStreak = 0; return; }
550
+
551
+ _dashHealthFailStreak++;
552
+ console.log(`[supervisor] Dashboard bound but unresponsive on port ${dashPort} (${_dashHealthFailStreak}/${DASH_HEALTH_MAX_FAILS} strikes)`);
553
+ if (_dashHealthFailStreak < DASH_HEALTH_MAX_FAILS) return;
554
+ respawn('frozen (health probe failed)');
441
555
  }
442
556
 
443
- function tick() {
557
+ let _tickInFlight = false;
558
+ async function tick() {
559
+ // The dashboard health probe is async; guard against a slow probe overlapping
560
+ // the next interval fire. setInterval invoking an async fn is fire-and-forget,
561
+ // so this flag is the only thing preventing concurrent ticks.
562
+ if (_tickInFlight) return;
563
+ _tickInFlight = true;
444
564
  try {
445
565
  if (isStopIntentSet()) return;
446
566
  const now = Date.now();
447
567
  checkEngine(now);
448
568
  checkEngineHung(now);
449
- checkDashboard(now);
569
+ await checkDashboard(now);
450
570
  } catch (e) {
451
571
  console.error(`[supervisor] tick error: ${e && e.message}`);
572
+ } finally {
573
+ _tickInFlight = false;
452
574
  }
453
575
  }
454
576
 
@@ -517,8 +639,11 @@ module.exports = {
517
639
  openAppendFd,
518
640
  checkEngine,
519
641
  checkEngineHung,
642
+ _isRespawnEligibleEngineState,
520
643
  checkDashboard,
644
+ _probeDashboardHealth,
521
645
  SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS,
646
+ DASH_HEALTH_MAX_FAILS,
522
647
  tick,
523
648
  // Path getters honor MINIONS_TEST_DIR via shared.ENGINE_DIR, so test
524
649
  // isolation correctly redirects writes/reads under createTestMinionsDir.
@@ -249,6 +249,16 @@ async function tick(opts) {
249
249
  });
250
250
  if (child && typeof child.unref === 'function') child.unref();
251
251
  logLine(minionsHome, `spawned minions ${action} pid=${child && child.pid} (detached)`);
252
+ // Crash-loop counter (W-mr2c46590003e3ee) — best-effort, injected so this
253
+ // module stays dependency-free of shared.js (matches isStopIntentSet /
254
+ // confirmPortUp above). Caller (bin/minions.js) wires opts.recordRespawn
255
+ // to shared.recordEngineRespawn so all three engine-respawn mechanisms
256
+ // (supervisor.js, this watchdog, dashboard.js's in-process watchdog)
257
+ // share one rolling window in control.json instead of each silently
258
+ // ignoring repeated crashes.
259
+ if (typeof opts.recordRespawn === 'function') {
260
+ try { opts.recordRespawn(`watchdog:${action}`); } catch { /* best-effort */ }
261
+ }
252
262
  return { healthy: false, action, spawnedPid: child && child.pid };
253
263
  } catch (err) {
254
264
  logLine(minionsHome, `FAILED to spawn minions ${action}: ${err && err.message || err}`);