@yemi33/minions 0.1.2305 → 0.1.2307
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/minions.js +1 -0
- package/dashboard/js/refresh.js +17 -13
- package/dashboard/js/render-other.js +251 -3
- package/dashboard/js/render-plans.js +1 -1
- package/dashboard/js/render-prs.js +4 -1
- package/dashboard/js/render-utils.js +13 -0
- package/dashboard/js/render-work-items.js +11 -13
- package/dashboard/js/settings.js +30 -3
- package/dashboard/slim/body.html +6 -22
- package/dashboard/slim/js/modals-tiles.js +32 -22
- package/dashboard/slim/js/plans.js +24 -516
- package/dashboard/slim/styles.css +10 -7
- package/dashboard/styles.css +58 -0
- package/dashboard.js +66 -9
- package/docs/completion-reports.md +27 -2
- package/docs/copilot-cli-schema.md +31 -1
- package/docs/design-state-storage.md +1 -1
- package/docs/harness-transparency.md +18 -0
- package/docs/live-checkout-mode.md +57 -8
- package/engine/ado-comment.js +5 -2
- package/engine/ado.js +37 -0
- package/engine/cli.js +58 -3
- package/engine/comment-format.js +82 -2
- package/engine/consolidation.js +59 -2
- package/engine/discover-project-skills.js +4 -0
- package/engine/dispatch.js +36 -1
- package/engine/gh-comment.js +8 -3
- package/engine/lifecycle.js +241 -25
- package/engine/live-checkout.js +335 -7
- package/engine/playbook.js +2 -2
- package/engine/pre-dispatch-eval.js +54 -5
- package/engine/queries.js +11 -7
- package/engine/shared.js +96 -6
- package/engine/supervisor.js +144 -19
- package/engine/watchdog.js +10 -0
- package/engine.js +210 -5
- package/package.json +1 -1
- package/playbooks/review.md +2 -0
- package/playbooks/shared-rules.md +9 -0
package/engine/shared.js
CHANGED
|
@@ -1946,6 +1946,86 @@ function mutateControl(mutator) {
|
|
|
1946
1946
|
}, { defaultValue: { state: 'stopped', pid: null }, skipWriteIfUnchanged: true });
|
|
1947
1947
|
}
|
|
1948
1948
|
|
|
1949
|
+
// ── Crash-loop detection (W-mr2c46590003e3ee) ──────────────────────────────
|
|
1950
|
+
// Three independent mechanisms can respawn a dead/frozen engine process:
|
|
1951
|
+
// engine/supervisor.js#checkEngine / #checkEngineHung, engine/watchdog.js
|
|
1952
|
+
// (OS-scheduled external recovery), and dashboard.js's in-process 30s
|
|
1953
|
+
// watchdog. None of them counted or throttled repeated crashes, so a genuine
|
|
1954
|
+
// crash-loop (engine dying every few minutes) silently self-healed forever
|
|
1955
|
+
// with no human-visible signal — see the 2026-07-01 00:20-00:30 incident
|
|
1956
|
+
// (4 respawns in ~5 minutes, notes/inbox/dallas-W-mr2azk6i000y06e0-*.md).
|
|
1957
|
+
//
|
|
1958
|
+
// recordEngineRespawn() gives all three mechanisms a single rolling counter
|
|
1959
|
+
// persisted in control.json so it survives process restarts: every call
|
|
1960
|
+
// appends {ts, source}, prunes entries older than CRASH_LOOP_WINDOW_MS, and
|
|
1961
|
+
// once the count within the window reaches CRASH_LOOP_THRESHOLD it flips a
|
|
1962
|
+
// dashboard-visible `control.crashLoopAlert` flag and writes a deduped
|
|
1963
|
+
// (one-per-day) inbox note.
|
|
1964
|
+
//
|
|
1965
|
+
// Callers MUST invoke this only at the point an actual respawn action is
|
|
1966
|
+
// taken — not on every health-check tick. The existing
|
|
1967
|
+
// RESPAWN_IN_PROGRESS_WINDOW_MS / post-spawn-grace guards already in each
|
|
1968
|
+
// mechanism prevent two watchers from double-spawning for the same crash, so
|
|
1969
|
+
// recording only real spawns keeps a single crash from being counted 2-3x
|
|
1970
|
+
// across the three watchers.
|
|
1971
|
+
const CRASH_LOOP_WINDOW_MS = Number(process.env.MINIONS_CRASH_LOOP_WINDOW_MS) || 10 * 60 * 1000; // 10 min
|
|
1972
|
+
const CRASH_LOOP_THRESHOLD = Number(process.env.MINIONS_CRASH_LOOP_THRESHOLD) || 3;
|
|
1973
|
+
|
|
1974
|
+
function recordEngineRespawn(source) {
|
|
1975
|
+
const now = Date.now();
|
|
1976
|
+
let crashLoop = false;
|
|
1977
|
+
let count = 1;
|
|
1978
|
+
let windowEntries = [];
|
|
1979
|
+
try {
|
|
1980
|
+
mutateControl((control) => {
|
|
1981
|
+
const prior = Array.isArray(control.engineRespawns) ? control.engineRespawns : [];
|
|
1982
|
+
const entries = prior
|
|
1983
|
+
.filter(e => e && Number.isFinite(e.ts) && (now - e.ts) <= CRASH_LOOP_WINDOW_MS)
|
|
1984
|
+
.concat([{ ts: now, source: String(source || 'unknown') }]);
|
|
1985
|
+
// Defensive cap — the window filter above already bounds this in
|
|
1986
|
+
// practice, but guards against unbounded growth under clock skew or an
|
|
1987
|
+
// operator-widened window.
|
|
1988
|
+
control.engineRespawns = entries.slice(-50);
|
|
1989
|
+
windowEntries = control.engineRespawns;
|
|
1990
|
+
count = windowEntries.length;
|
|
1991
|
+
if (count >= CRASH_LOOP_THRESHOLD) {
|
|
1992
|
+
crashLoop = true;
|
|
1993
|
+
control.crashLoopAlert = {
|
|
1994
|
+
count,
|
|
1995
|
+
windowMs: CRASH_LOOP_WINDOW_MS,
|
|
1996
|
+
firstAt: new Date(windowEntries[0].ts).toISOString(),
|
|
1997
|
+
lastAt: new Date(now).toISOString(),
|
|
1998
|
+
sources: windowEntries.map(e => e.source),
|
|
1999
|
+
};
|
|
2000
|
+
} else if (control.crashLoopAlert) {
|
|
2001
|
+
// Window rolled off past incidents — clear a stale alert so the
|
|
2002
|
+
// dashboard doesn't keep showing a resolved crash-loop forever.
|
|
2003
|
+
delete control.crashLoopAlert;
|
|
2004
|
+
}
|
|
2005
|
+
return control;
|
|
2006
|
+
});
|
|
2007
|
+
} catch (e) {
|
|
2008
|
+
log('warn', `recordEngineRespawn: control.json update failed: ${e.message}`);
|
|
2009
|
+
}
|
|
2010
|
+
|
|
2011
|
+
if (crashLoop) {
|
|
2012
|
+
try {
|
|
2013
|
+
const lines = windowEntries.map(e => `- ${e.source} @ ${new Date(e.ts).toISOString()}`).join('\n');
|
|
2014
|
+
const body = `# Engine crash-loop detected\n\n` +
|
|
2015
|
+
`${count} engine respawns in the last ${Math.round(CRASH_LOOP_WINDOW_MS / 60000)} minutes ` +
|
|
2016
|
+
`(threshold: ${CRASH_LOOP_THRESHOLD}).\n\n` +
|
|
2017
|
+
`Respawns:\n${lines}\n\n` +
|
|
2018
|
+
`The engine keeps self-healing (each respawn succeeded) but is crashing repeatedly — ` +
|
|
2019
|
+
`investigate \`engine/engine-stdio.log\` around the timestamps above before this becomes ` +
|
|
2020
|
+
`a silent outage.\n`;
|
|
2021
|
+
writeToInbox('engine', 'crash-loop-detected', body, null, { severity: 'critical' });
|
|
2022
|
+
} catch (e) {
|
|
2023
|
+
log('warn', `recordEngineRespawn: inbox alert failed: ${e.message}`);
|
|
2024
|
+
}
|
|
2025
|
+
}
|
|
2026
|
+
return { count, crashLoop };
|
|
2027
|
+
}
|
|
2028
|
+
|
|
1949
2029
|
// W-mp60tw0u000j3931: Lock-safe read-modify-write for engine/state.json — the
|
|
1950
2030
|
// persistent cross-restart engine-level state file (migration markers,
|
|
1951
2031
|
// one-shot reconciliation versions, etc.). Mirrors mutateControl's shape.
|
|
@@ -2904,6 +2984,14 @@ const ENGINE_DEFAULTS = {
|
|
|
2904
2984
|
// & Review Loop, or set `engine.autoFixPaused: true` in config.json.
|
|
2905
2985
|
autoFixPaused: false, // hard-stop kill-switch / master override (see comment above)
|
|
2906
2986
|
autoConsolidateMemory: false, // opt-in: periodically spawn engine/kb-sweep-runner.js from the tick loop (4h cadence). Inbox→notes consolidation already runs every tick via consolidateInbox; this flag only controls the KB sweep.
|
|
2987
|
+
// W-mqtvnnj1000357fa — fleet-wide fallback for live-checkout auto-stash. When
|
|
2988
|
+
// true, a dirty live-checkout tree is `git stash push --include-untracked`'d
|
|
2989
|
+
// before dispatch instead of failing with FAILURE_CLASS.LIVE_CHECKOUT_DIRTY,
|
|
2990
|
+
// so dispatch proceeds without manual operator intervention. Per-project
|
|
2991
|
+
// `project.liveCheckoutAutoStash` overrides this (resolveLiveCheckoutAutoStash).
|
|
2992
|
+
// The engine never auto-pops the stash — the operator runs `git stash pop`.
|
|
2993
|
+
// Default false so a fresh install behaves identically to before this knob.
|
|
2994
|
+
liveCheckoutAutoStash: false,
|
|
2907
2995
|
prNoOpFixPauseAttempts: 2, // pause one PR automation cause after repeated no-op fixes for unchanged evidence
|
|
2908
2996
|
quarantineAutoRecoveryMax: 2, // #2996 follow-up: cap on auto-flipping WORKTREE_DIRTY/WORKTREE_DIVERGENT failures back to pending (the quarantine is self-healing so the next dispatch starts clean; the cap prevents infinite loops if quarantine itself keeps failing).
|
|
2909
2997
|
// W-mq5n1zx5 — Layer 1a/2b: harden the quarantine rename path against
|
|
@@ -3254,11 +3342,12 @@ const ENGINE_DEFAULTS = {
|
|
|
3254
3342
|
// When ON, a dirty/broken live-checkout tree is `git fetch origin` +
|
|
3255
3343
|
// `git reset --hard origin/<branch>`'d before dispatch instead of failing
|
|
3256
3344
|
// LIVE_CHECKOUT_DIRTY. Per-project `project.liveCheckoutAutoReset` overrides
|
|
3257
|
-
// this. Default
|
|
3258
|
-
//
|
|
3259
|
-
//
|
|
3260
|
-
//
|
|
3261
|
-
|
|
3345
|
+
// this. Default ON — discarding uncommitted dirt via reset-to-remote keeps
|
|
3346
|
+
// live-checkout branches from diverging (the old `false` default accumulated
|
|
3347
|
+
// `minions: auto-save agent WIP` commits on every dirty exit). Resolution
|
|
3348
|
+
// precedence lives in `resolveLiveCheckoutAutoReset`. Fires ONLY on confirmed
|
|
3349
|
+
// dirty-tree detection, never on mid-operation / blob-fetch / tooling failures.
|
|
3350
|
+
liveCheckoutAutoReset: true,
|
|
3262
3351
|
orphanHolderScanTimeoutMs: 5000, // 5s ceiling for the cross-platform holder scan (PowerShell / /proc walk / lsof)
|
|
3263
3352
|
ccMaxTurns: 50, // max tool-use turns per CC/doc-chat call before CLI stops (per response, not per session)
|
|
3264
3353
|
ccWorkerIdleTimeoutMs: 30 * 60 * 1000, // W-mr0qs0vw: idle-reaper window for the persistent `copilot --acp` worker pool (engine/cc-worker-pool.js). After this much inactivity with no in-flight turn the warm ACP process is killed; the next message cold-spawns a fresh session with NO memory of prior turns (CC shows a "context cleared after inactivity" notice). Tradeoff: shorter = less idle memory/process footprint, longer = more context durability across gaps between messages. Wired into the pool via ccWorkerPool.setIdleTimeoutMs() on every reloadConfig(); clamped to [60000, 28800000] (1min–8h) in the settings POST handler.
|
|
@@ -4555,6 +4644,7 @@ const FAILURE_CLASS = {
|
|
|
4555
4644
|
LIVE_CHECKOUT_FAILED: 'live-checkout-failed', // #305 (live-checkout dispatch mode): prepareLiveCheckout THREW before agent spawn (helper guard, ref validation, or a transient `git status`/`rev-parse`/`checkout` failure) — distinct from the confirmed-dirty result (LIVE_CHECKOUT_DIRTY) which the helper returns, not throws. A thrown error is NOT proof the tree is dirty, so it must not be over-classified as dirty. Retryable with bounded backoff (NOT in dispatch.js neverRetry): racy branch-lock handoff, just-finished sibling dispatch, or transient git errors frequently clear on the next attempt; the engine auto-retries up to ENGINE_DEFAULTS.maxRetries before giving up. Genuinely terminal underlying reasons (auth, validation) still short-circuit via the reason-string check in isRetryableFailureReason.
|
|
4556
4645
|
LIVE_CHECKOUT_MID_OPERATION: 'live-checkout-mid-operation', // P-a7f3c1d9 (live-checkout dispatch mode): spawnAgent could not switch/create the target branch in project.localPath because the operator tree is mid-operation — an in-progress merge/rebase/cherry-pick/bisect or a detached HEAD. Distinct from LIVE_CHECKOUT_DIRTY (uncommitted changes): here the tree may be clean but the branch op cannot proceed. Engine refuses to spawn (it never runs `git reset`/`git clean`/`git rebase --abort` against the operator tree). Non-retryable — operator must finish or abort the in-progress operation, or checkout a branch, before re-dispatch.
|
|
4557
4646
|
LIVE_CHECKOUT_BLOB_FETCH: 'live-checkout-blob-fetch', // PL-live-checkout-reliability-hardening (live-checkout dispatch mode): `git checkout <existing-branch>` in project.localPath failed because the tree could not be materialized — on a Scalar/GVFS-managed ADO partial (blobless) clone, switching onto a branch whose tree differs from HEAD hydrates the changed paths' blobs through the GVFS cache server (`*.gvfscache.dev.azure.com`), a different endpoint than the main git remote that receives NO auth in the headless engine shell, so the fetch fails DETERMINISTICALLY. Distinct from LIVE_CHECKOUT_FAILED (transient) because retrying reproduces it identically — surfaced once as an operator-actionable refusal instead of retry-storming to the cap. Non-retryable — the operator hydrates the branch once with their own credentials (`git checkout <branch>` interactively, or `scalar prefetch` / `git -C <repo> fetch`), then re-dispatches. The engine NEVER forces, resets, cleans, or stashes the operator tree, and best-effort switches HEAD back to the original ref so the tree is not stranded half-populated.
|
|
4647
|
+
LIVE_CHECKOUT_WORKTREE_CONFLICT: 'live-checkout-worktree-conflict', // W-mr28h2j2000y0de1 (live-checkout dispatch mode): `git checkout <existing-branch>` in project.localPath failed because that exact branch is ALSO checked out in a SECOND worktree elsewhere (a leftover from a prior isolated-worktree dispatch, a manually-created worktree, or a stale worktree left by a checkoutMode change). git refuses deterministically with `fatal: '<branch>' is already used by worktree at '<path>'`. Distinct from LIVE_CHECKOUT_FAILED (transient) because this is a STRUCTURAL conflict — it does NOT clear on retry, ever, until a human or the engine removes/reassigns the other worktree; retrying just reproduces it identically and burns maxRetries. Non-retryable — the operator either `git worktree remove <path>` (freeing the branch) if that worktree is stale, or finishes/commits/pushes from it directly if it holds real WIP. The engine NEVER forces, resets, cleans, or stashes the operator tree, and best-effort switches HEAD back to the original ref so the tree is not stranded.
|
|
4558
4648
|
INVALID_WORKDIR: 'invalid-workdir', // P-714ef144: dispatch carried a meta.workdir override that failed validation — non-string, absolute path, drive-letter prefix, null byte, ".." segment, or post-resolve containment escape against project.localPath / worktree root. Engine refuses to spawn (the subpath would either be unreachable on disk or point outside the operator's allowed surface). Non-retryable — operator must fix the WI's meta.workdir before re-dispatch. Inbox alert lists the offending value + the resolved-vs-base mismatch.
|
|
4559
4649
|
MODEL_UNAVAILABLE: 'model-unavailable', // W-mpg6isvy000xca4d: requested model returned overloaded_error / 503 / service_unavailable. Retriable — engine swaps in the runtime-appropriate fallback model on next spawn (Claude leans on --fallback-model already plumbed; Copilot overrides --model with engine.copilotFallbackModel).
|
|
4560
4650
|
WORKSPACE_MANIFEST_REPO: 'workspace-manifest-repo-forbidden', // W-mq07avbk000m5543: dispatch routed an agent to a project/repo not present in its workspace_manifest.allowed_repos. Structural — never retryable until the manifest is widened or a different agent is chosen.
|
|
@@ -9060,7 +9150,7 @@ module.exports = {
|
|
|
9060
9150
|
assertStateFileSize,
|
|
9061
9151
|
withFileLock,
|
|
9062
9152
|
mutateJsonFileLocked,
|
|
9063
|
-
mutateControl,
|
|
9153
|
+
mutateControl, recordEngineRespawn,
|
|
9064
9154
|
mutateEngineState, // W-mp60tw0u000j3931
|
|
9065
9155
|
readEngineState, // W-mp60tw0u000j3931
|
|
9066
9156
|
mutateCooldowns,
|
package/engine/supervisor.js
CHANGED
|
@@ -5,7 +5,14 @@
|
|
|
5
5
|
* Spawned `detached: true` by `minions start`/`restart` after the engine and
|
|
6
6
|
* dashboard are up. Polls every SUPERVISOR_INTERVAL_MS:
|
|
7
7
|
* - engine PID from `engine/control.json`
|
|
8
|
-
* - dashboard
|
|
8
|
+
* - dashboard via a port-listener probe (port 7331 by default) AND, when the
|
|
9
|
+
* port is bound, an HTTP `/api/health` probe. The HTTP layer catches the
|
|
10
|
+
* "frozen but still listening" dashboard (wedged event loop) that the
|
|
11
|
+
* port-listen probe alone reports as healthy — it respawns only after
|
|
12
|
+
* DASH_HEALTH_MAX_FAILS consecutive failed probes so a dashboard merely
|
|
13
|
+
* busy with a heavy synchronous getStatus() rebuild isn't killed. What
|
|
14
|
+
* counts as "responsive" (and why engine 'degraded'/'stopped' does NOT mean
|
|
15
|
+
* the dashboard is frozen) is documented at _probeDashboardHealth.
|
|
9
16
|
*
|
|
10
17
|
* When either is dead AND the stop-intent flag is NOT set, respawns the dead
|
|
11
18
|
* one in the same way the CLI does (detached, stdio routed to the engine-
|
|
@@ -38,6 +45,17 @@ const MINIONS_DIR = path.resolve(__dirname, '..');
|
|
|
38
45
|
function _sharedOrNull() {
|
|
39
46
|
try { return require('./shared'); } catch { return null; }
|
|
40
47
|
}
|
|
48
|
+
// Crash-loop counter (W-mr2c46590003e3ee) — best-effort. shared.js owns the
|
|
49
|
+
// counter/threshold/inbox-alert logic (engine/shared.js#recordEngineRespawn)
|
|
50
|
+
// so supervisor.js, engine/watchdog.js, and dashboard.js's in-process
|
|
51
|
+
// watchdog all feed the same rolling window in control.json instead of each
|
|
52
|
+
// keeping (and silently ignoring) their own. Falls through to a no-op when
|
|
53
|
+
// shared.js can't load, matching this file's existing fail-open posture.
|
|
54
|
+
function _recordCrashLoopRespawn(source) {
|
|
55
|
+
const shared = _sharedOrNull();
|
|
56
|
+
if (!shared || typeof shared.recordEngineRespawn !== 'function') return;
|
|
57
|
+
try { shared.recordEngineRespawn(source); } catch { /* best-effort */ }
|
|
58
|
+
}
|
|
41
59
|
function _engineDir() { return _sharedOrNull()?.ENGINE_DIR || __dirname; }
|
|
42
60
|
const STATIC_ENGINE_DIR = __dirname; // for require-time paths that can't be lazy
|
|
43
61
|
function CONTROL_PATH_FN() { return path.join(_engineDir(), 'control.json'); }
|
|
@@ -76,6 +94,16 @@ function _resolveDashPort() {
|
|
|
76
94
|
const POST_SPAWN_GRACE_MS = Number(process.env.MINIONS_SUPERVISOR_GRACE_MS) || 15000;
|
|
77
95
|
// #421 — heartbeat age threshold before supervisor considers the engine event loop frozen.
|
|
78
96
|
const SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS = Number(process.env.MINIONS_SUPERVISOR_STALE_HEARTBEAT_MS) || 180000; // 3 min
|
|
97
|
+
// Dashboard HTTP liveness (frozen-but-listening detection). A dashboard whose
|
|
98
|
+
// event loop is wedged keeps the socket LISTENING, so the port-listen probe
|
|
99
|
+
// alone reports it healthy while the in-browser SPA shows the "unreachable"
|
|
100
|
+
// banner forever. We layer an HTTP /api/health probe on top: only respawn after
|
|
101
|
+
// DASH_HEALTH_MAX_FAILS *consecutive* failed probes so a dashboard that is
|
|
102
|
+
// merely busy with a heavy synchronous getStatus() rebuild (documented to block
|
|
103
|
+
// the loop 15-25s) isn't mistaken for a hang. With the default 30s interval,
|
|
104
|
+
// 3 strikes ≈ 90s of continuous unresponsiveness before we act.
|
|
105
|
+
const DASH_HEALTH_TIMEOUT_MS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_TIMEOUT_MS) || 5000;
|
|
106
|
+
const DASH_HEALTH_MAX_FAILS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_FAILS) || 3;
|
|
79
107
|
const isWin = process.platform === 'win32';
|
|
80
108
|
|
|
81
109
|
function safeReadJson(p) {
|
|
@@ -354,6 +382,39 @@ function spawnDashboard() {
|
|
|
354
382
|
// the engine.
|
|
355
383
|
let _lastEngineRespawnAt = 0;
|
|
356
384
|
let _lastDashboardRespawnAt = 0;
|
|
385
|
+
// Consecutive failed dashboard HTTP health probes. Reset on any healthy probe,
|
|
386
|
+
// after a respawn, and inside the post-spawn grace window.
|
|
387
|
+
let _dashHealthFailStreak = 0;
|
|
388
|
+
|
|
389
|
+
// Lazy, defensive HTTP health probe. Reuses restart-health.js's httpGetJson
|
|
390
|
+
// (rule: reuse before re-implementing) but never lets a load failure (e.g. a
|
|
391
|
+
// mid-upgrade broken shared.js that restart-health requires at module top)
|
|
392
|
+
// break the supervisor. Returns null when the probe helper is unavailable so
|
|
393
|
+
// the caller can FALL BACK to port-only liveness instead of false-respawning a
|
|
394
|
+
// healthy dashboard.
|
|
395
|
+
function _probeDashboardHealth(port, timeoutMs = DASH_HEALTH_TIMEOUT_MS) {
|
|
396
|
+
let getJson = null;
|
|
397
|
+
try {
|
|
398
|
+
const rh = require('./restart-health');
|
|
399
|
+
if (rh && rh._private && typeof rh._private.httpGetJson === 'function') {
|
|
400
|
+
getJson = rh._private.httpGetJson;
|
|
401
|
+
}
|
|
402
|
+
} catch { /* restart-health/shared unavailable — fall through */ }
|
|
403
|
+
if (!getJson) return Promise.resolve(null);
|
|
404
|
+
return getJson(`http://127.0.0.1:${port}/api/health`, timeoutMs)
|
|
405
|
+
// "Responsive" == the dashboard event loop executed handleHealth and returned
|
|
406
|
+
// a well-formed 2xx envelope, REGARDLESS of the engine-derived status value.
|
|
407
|
+
// /api/health reports 'degraded' when the engine is paused/stopping and
|
|
408
|
+
// 'stopped' when the engine is down — those are legitimate ENGINE states
|
|
409
|
+
// (owned by checkEngine/checkEngineHung), not evidence the DASHBOARD process
|
|
410
|
+
// is frozen. Gating on status==='healthy' would respawn a perfectly
|
|
411
|
+
// responsive dashboard every time an operator pauses the engine, tearing
|
|
412
|
+
// down the socket and tripping the in-browser "unreachable" banner for no
|
|
413
|
+
// reason. Only a failed/timed-out request (res.ok false or a rejection)
|
|
414
|
+
// means the dashboard's own event loop is actually wedged.
|
|
415
|
+
.then(res => !!(res && res.ok && res.json && typeof res.json.status === 'string'))
|
|
416
|
+
.catch(() => false);
|
|
417
|
+
}
|
|
357
418
|
|
|
358
419
|
// Window during which a `null` pid + recent `restarted_at` is interpreted as
|
|
359
420
|
// "another watchdog is currently respawning the engine — don't double-spawn."
|
|
@@ -364,12 +425,29 @@ let _lastDashboardRespawnAt = 0;
|
|
|
364
425
|
// to enter cli.js:443 and write its own PID on cold Windows boots.
|
|
365
426
|
const RESPAWN_IN_PROGRESS_WINDOW_MS = Number(process.env.MINIONS_SUPERVISOR_RESPAWN_WINDOW_MS) || 60000;
|
|
366
427
|
|
|
428
|
+
// #3758 — 'degraded' is a FAILURE state (set by dashboard.js's frozen-tick
|
|
429
|
+
// watchdog, see dashboard.js#_markEngineAsDegradedIfFrozen), not a deliberate
|
|
430
|
+
// user-intent state like 'paused'/'stopping'/'stopped'. When a frozen tick loop
|
|
431
|
+
// finally unblocks, engine.js#tickInner reads control.state !== 'running' and
|
|
432
|
+
// calls process.exit(0) — self-exiting exactly BECAUSE it saw 'degraded'. The
|
|
433
|
+
// dashboard.js comment documents the expected recovery: "the PID-dead watchdog
|
|
434
|
+
// then auto-restarts it." Excluding 'degraded' here left that promise unmet —
|
|
435
|
+
// the engine PID goes away and the control plane silently stays down until a
|
|
436
|
+
// manual `minions restart`. Respawn eligibility must include both 'running'
|
|
437
|
+
// (still up) and 'degraded' (crashed/crashing out of a real failure); only
|
|
438
|
+
// 'paused', 'stopping', and 'stopped' reflect actual user intent to stay down.
|
|
439
|
+
function _isRespawnEligibleEngineState(state) {
|
|
440
|
+
return state === 'running' || state === 'degraded';
|
|
441
|
+
}
|
|
442
|
+
|
|
367
443
|
function checkEngine(now) {
|
|
368
444
|
if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
369
445
|
const control = safeReadJson(CONTROL_PATH_FN());
|
|
370
|
-
//
|
|
371
|
-
//
|
|
372
|
-
|
|
446
|
+
// Respawn when control.json says "running" (PID died unexpectedly) or
|
|
447
|
+
// "degraded" (watchdog-flagged failure state — see _isRespawnEligibleEngineState
|
|
448
|
+
// above). "paused"/"stopped"/"stopping" are legitimate user-intent states the
|
|
449
|
+
// supervisor must not override.
|
|
450
|
+
if (!control || !_isRespawnEligibleEngineState(control.state)) return;
|
|
373
451
|
if (control.pid && isPidAlive(control.pid)) return;
|
|
374
452
|
|
|
375
453
|
// Cross-watchdog race guard: dashboard.js's in-process engine watchdog
|
|
@@ -391,6 +469,7 @@ function checkEngine(now) {
|
|
|
391
469
|
reapStrayProcesses(path.join(MINIONS_DIR, 'engine.js'), 'engine');
|
|
392
470
|
const newPid = spawnEngine();
|
|
393
471
|
_lastEngineRespawnAt = now;
|
|
472
|
+
_recordCrashLoopRespawn('supervisor:dead-pid');
|
|
394
473
|
console.log(`[supervisor] Engine respawned (new PID: ${newPid})`);
|
|
395
474
|
}
|
|
396
475
|
|
|
@@ -403,7 +482,11 @@ function checkEngine(now) {
|
|
|
403
482
|
function checkEngineHung(now) {
|
|
404
483
|
if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
405
484
|
const control = safeReadJson(CONTROL_PATH_FN());
|
|
406
|
-
|
|
485
|
+
// Include 'degraded' (see _isRespawnEligibleEngineState) so a permanently
|
|
486
|
+
// frozen tick loop that never gets a chance to self-exit (tickInner never
|
|
487
|
+
// re-enters) still gets force-restarted once its own heartbeat goes stale,
|
|
488
|
+
// instead of being ignored forever the moment the state flips to 'degraded'.
|
|
489
|
+
if (!control || !_isRespawnEligibleEngineState(control.state)) return;
|
|
407
490
|
// Only acts when the PID is alive — a dead PID is handled by checkEngine().
|
|
408
491
|
if (!control.pid || !isPidAlive(control.pid)) return;
|
|
409
492
|
|
|
@@ -421,34 +504,73 @@ function checkEngineHung(now) {
|
|
|
421
504
|
reapStrayProcesses(path.join(MINIONS_DIR, 'engine.js'), 'engine');
|
|
422
505
|
const newPid = spawnEngine();
|
|
423
506
|
_lastEngineRespawnAt = now;
|
|
507
|
+
_recordCrashLoopRespawn('supervisor:stale-heartbeat');
|
|
424
508
|
console.log(`[supervisor] Engine restarted due to stale heartbeat (new PID: ${newPid})`);
|
|
425
509
|
}
|
|
426
510
|
|
|
427
|
-
function checkDashboard(now) {
|
|
428
|
-
if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
511
|
+
async function checkDashboard(now) {
|
|
512
|
+
if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) { _dashHealthFailStreak = 0; return; }
|
|
429
513
|
const dashPort = _resolveDashPort();
|
|
430
514
|
const pids = listeningPidsForPort(dashPort);
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
515
|
+
|
|
516
|
+
const respawn = (why) => {
|
|
517
|
+
console.log(`[supervisor] Dashboard ${why} on port ${dashPort} — respawning...`);
|
|
518
|
+
// Reap stray dashboards first. Besides bounding the count, this collapses the
|
|
519
|
+
// port-desync case: orphan dashboards holding other ports are killed, freeing
|
|
520
|
+
// the canonical port for the fresh one whose beacon _resolveDashPort() reads.
|
|
521
|
+
// Critically, for the frozen-but-listening case the reap also KILLS the wedged
|
|
522
|
+
// process still holding the port so spawnDashboard() can bind it.
|
|
523
|
+
reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
|
|
524
|
+
const newPid = spawnDashboard();
|
|
525
|
+
_lastDashboardRespawnAt = now;
|
|
526
|
+
_dashHealthFailStreak = 0;
|
|
527
|
+
console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
|
|
528
|
+
};
|
|
529
|
+
|
|
530
|
+
// Hard-down: nothing bound to the port → respawn immediately.
|
|
531
|
+
if (pids.length === 0) {
|
|
532
|
+
_dashHealthFailStreak = 0;
|
|
533
|
+
respawn('not listening');
|
|
534
|
+
return;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
// Port is bound. Verify the event loop is actually responsive — a frozen-but-
|
|
538
|
+
// listening dashboard keeps the socket LISTENING yet serves the in-browser
|
|
539
|
+
// "unreachable" banner. Probe /api/health; only respawn after a streak of
|
|
540
|
+
// failures so a legitimately busy getStatus() rebuild isn't mistaken for a hang.
|
|
541
|
+
const healthy = await _probeDashboardHealth(dashPort);
|
|
542
|
+
if (healthy === null) {
|
|
543
|
+
// Probe helper unavailable — fall back to port-only liveness (the pre-existing
|
|
544
|
+
// behavior). Don't count this as a failure; we won't respawn a bound dashboard
|
|
545
|
+
// we can't actually prove is frozen.
|
|
546
|
+
_dashHealthFailStreak = 0;
|
|
547
|
+
return;
|
|
548
|
+
}
|
|
549
|
+
if (healthy) { _dashHealthFailStreak = 0; return; }
|
|
550
|
+
|
|
551
|
+
_dashHealthFailStreak++;
|
|
552
|
+
console.log(`[supervisor] Dashboard bound but unresponsive on port ${dashPort} (${_dashHealthFailStreak}/${DASH_HEALTH_MAX_FAILS} strikes)`);
|
|
553
|
+
if (_dashHealthFailStreak < DASH_HEALTH_MAX_FAILS) return;
|
|
554
|
+
respawn('frozen (health probe failed)');
|
|
441
555
|
}
|
|
442
556
|
|
|
443
|
-
|
|
557
|
+
let _tickInFlight = false;
|
|
558
|
+
async function tick() {
|
|
559
|
+
// The dashboard health probe is async; guard against a slow probe overlapping
|
|
560
|
+
// the next interval fire. setInterval invoking an async fn is fire-and-forget,
|
|
561
|
+
// so this flag is the only thing preventing concurrent ticks.
|
|
562
|
+
if (_tickInFlight) return;
|
|
563
|
+
_tickInFlight = true;
|
|
444
564
|
try {
|
|
445
565
|
if (isStopIntentSet()) return;
|
|
446
566
|
const now = Date.now();
|
|
447
567
|
checkEngine(now);
|
|
448
568
|
checkEngineHung(now);
|
|
449
|
-
checkDashboard(now);
|
|
569
|
+
await checkDashboard(now);
|
|
450
570
|
} catch (e) {
|
|
451
571
|
console.error(`[supervisor] tick error: ${e && e.message}`);
|
|
572
|
+
} finally {
|
|
573
|
+
_tickInFlight = false;
|
|
452
574
|
}
|
|
453
575
|
}
|
|
454
576
|
|
|
@@ -517,8 +639,11 @@ module.exports = {
|
|
|
517
639
|
openAppendFd,
|
|
518
640
|
checkEngine,
|
|
519
641
|
checkEngineHung,
|
|
642
|
+
_isRespawnEligibleEngineState,
|
|
520
643
|
checkDashboard,
|
|
644
|
+
_probeDashboardHealth,
|
|
521
645
|
SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS,
|
|
646
|
+
DASH_HEALTH_MAX_FAILS,
|
|
522
647
|
tick,
|
|
523
648
|
// Path getters honor MINIONS_TEST_DIR via shared.ENGINE_DIR, so test
|
|
524
649
|
// isolation correctly redirects writes/reads under createTestMinionsDir.
|
package/engine/watchdog.js
CHANGED
|
@@ -249,6 +249,16 @@ async function tick(opts) {
|
|
|
249
249
|
});
|
|
250
250
|
if (child && typeof child.unref === 'function') child.unref();
|
|
251
251
|
logLine(minionsHome, `spawned minions ${action} pid=${child && child.pid} (detached)`);
|
|
252
|
+
// Crash-loop counter (W-mr2c46590003e3ee) — best-effort, injected so this
|
|
253
|
+
// module stays dependency-free of shared.js (matches isStopIntentSet /
|
|
254
|
+
// confirmPortUp above). Caller (bin/minions.js) wires opts.recordRespawn
|
|
255
|
+
// to shared.recordEngineRespawn so all three engine-respawn mechanisms
|
|
256
|
+
// (supervisor.js, this watchdog, dashboard.js's in-process watchdog)
|
|
257
|
+
// share one rolling window in control.json instead of each silently
|
|
258
|
+
// ignoring repeated crashes.
|
|
259
|
+
if (typeof opts.recordRespawn === 'function') {
|
|
260
|
+
try { opts.recordRespawn(`watchdog:${action}`); } catch { /* best-effort */ }
|
|
261
|
+
}
|
|
252
262
|
return { healthy: false, action, spawnedPid: child && child.pid };
|
|
253
263
|
} catch (err) {
|
|
254
264
|
logLine(minionsHome, `FAILED to spawn minions ${action}: ${err && err.message || err}`);
|