@yemi33/minions 0.1.2304 → 0.1.2306

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,7 +5,14 @@
5
5
  * Spawned `detached: true` by `minions start`/`restart` after the engine and
6
6
  * dashboard are up. Polls every SUPERVISOR_INTERVAL_MS:
7
7
  * - engine PID from `engine/control.json`
8
- * - dashboard PID via port-listener probe (port 7331 by default)
8
+ * - dashboard via a port-listener probe (port 7331 by default) AND, when the
9
+ * port is bound, an HTTP `/api/health` probe. The HTTP layer catches the
10
+ * "frozen but still listening" dashboard (wedged event loop) that the
11
+ * port-listen probe alone reports as healthy — it respawns only after
12
+ * DASH_HEALTH_MAX_FAILS consecutive failed probes so a dashboard merely
13
+ * busy with a heavy synchronous getStatus() rebuild isn't killed. What
14
+ * counts as "responsive" (and why engine 'degraded'/'stopped' does NOT mean
15
+ * the dashboard is frozen) is documented at _probeDashboardHealth.
9
16
  *
10
17
  * When either is dead AND the stop-intent flag is NOT set, respawns the dead
11
18
  * one in the same way the CLI does (detached, stdio routed to the engine-
@@ -76,6 +83,16 @@ function _resolveDashPort() {
76
83
  const POST_SPAWN_GRACE_MS = Number(process.env.MINIONS_SUPERVISOR_GRACE_MS) || 15000;
77
84
  // #421 — heartbeat age threshold before supervisor considers the engine event loop frozen.
78
85
  const SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS = Number(process.env.MINIONS_SUPERVISOR_STALE_HEARTBEAT_MS) || 180000; // 3 min
86
+ // Dashboard HTTP liveness (frozen-but-listening detection). A dashboard whose
87
+ // event loop is wedged keeps the socket LISTENING, so the port-listen probe
88
+ // alone reports it healthy while the in-browser SPA shows the "unreachable"
89
+ // banner forever. We layer an HTTP /api/health probe on top: only respawn after
90
+ // DASH_HEALTH_MAX_FAILS *consecutive* failed probes so a dashboard that is
91
+ // merely busy with a heavy synchronous getStatus() rebuild (documented to block
92
+ // the loop 15-25s) isn't mistaken for a hang. With the default 30s interval,
93
+ // 3 strikes ≈ 90s of continuous unresponsiveness before we act.
94
+ const DASH_HEALTH_TIMEOUT_MS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_TIMEOUT_MS) || 5000;
95
+ const DASH_HEALTH_MAX_FAILS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_FAILS) || 3;
79
96
  const isWin = process.platform === 'win32';
80
97
 
81
98
  function safeReadJson(p) {
@@ -354,6 +371,39 @@ function spawnDashboard() {
354
371
  // the engine.
355
372
  let _lastEngineRespawnAt = 0;
356
373
  let _lastDashboardRespawnAt = 0;
374
+ // Consecutive failed dashboard HTTP health probes. Reset on any healthy probe,
375
+ // after a respawn, and inside the post-spawn grace window.
376
+ let _dashHealthFailStreak = 0;
377
+
378
+ // Lazy, defensive HTTP health probe. Reuses restart-health.js's httpGetJson
379
+ // (rule: reuse before re-implementing) but never lets a load failure (e.g. a
380
+ // mid-upgrade broken shared.js that restart-health requires at module top)
381
+ // break the supervisor. Returns null when the probe helper is unavailable so
382
+ // the caller can FALL BACK to port-only liveness instead of false-respawning a
383
+ // healthy dashboard.
384
+ function _probeDashboardHealth(port, timeoutMs = DASH_HEALTH_TIMEOUT_MS) {
385
+ let getJson = null;
386
+ try {
387
+ const rh = require('./restart-health');
388
+ if (rh && rh._private && typeof rh._private.httpGetJson === 'function') {
389
+ getJson = rh._private.httpGetJson;
390
+ }
391
+ } catch { /* restart-health/shared unavailable — fall through */ }
392
+ if (!getJson) return Promise.resolve(null);
393
+ return getJson(`http://127.0.0.1:${port}/api/health`, timeoutMs)
394
+ // "Responsive" == the dashboard event loop executed handleHealth and returned
395
+ // a well-formed 2xx envelope, REGARDLESS of the engine-derived status value.
396
+ // /api/health reports 'degraded' when the engine is paused/stopping and
397
+ // 'stopped' when the engine is down — those are legitimate ENGINE states
398
+ // (owned by checkEngine/checkEngineHung), not evidence the DASHBOARD process
399
+ // is frozen. Gating on status==='healthy' would respawn a perfectly
400
+ // responsive dashboard every time an operator pauses the engine, tearing
401
+ // down the socket and tripping the in-browser "unreachable" banner for no
402
+ // reason. Only a failed/timed-out request (res.ok false or a rejection)
403
+ // means the dashboard's own event loop is actually wedged.
404
+ .then(res => !!(res && res.ok && res.json && typeof res.json.status === 'string'))
405
+ .catch(() => false);
406
+ }
357
407
 
358
408
  // Window during which a `null` pid + recent `restarted_at` is interpreted as
359
409
  // "another watchdog is currently respawning the engine — don't double-spawn."
@@ -364,12 +414,29 @@ let _lastDashboardRespawnAt = 0;
364
414
  // to enter cli.js:443 and write its own PID on cold Windows boots.
365
415
  const RESPAWN_IN_PROGRESS_WINDOW_MS = Number(process.env.MINIONS_SUPERVISOR_RESPAWN_WINDOW_MS) || 60000;
366
416
 
417
+ // #3758 — 'degraded' is a FAILURE state (set by dashboard.js's frozen-tick
418
+ // watchdog, see dashboard.js#_markEngineAsDegradedIfFrozen), not a deliberate
419
+ // user-intent state like 'paused'/'stopping'/'stopped'. When a frozen tick loop
420
+ // finally unblocks, engine.js#tickInner reads control.state !== 'running' and
421
+ // calls process.exit(0) — self-exiting exactly BECAUSE it saw 'degraded'. The
422
+ // dashboard.js comment documents the expected recovery: "the PID-dead watchdog
423
+ // then auto-restarts it." Excluding 'degraded' here left that promise unmet —
424
+ // the engine PID goes away and the control plane silently stays down until a
425
+ // manual `minions restart`. Respawn eligibility must include both 'running'
426
+ // (still up) and 'degraded' (crashed/crashing out of a real failure); only
427
+ // 'paused', 'stopping', and 'stopped' reflect actual user intent to stay down.
428
+ function _isRespawnEligibleEngineState(state) {
429
+ return state === 'running' || state === 'degraded';
430
+ }
431
+
367
432
  function checkEngine(now) {
368
433
  if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
369
434
  const control = safeReadJson(CONTROL_PATH_FN());
370
- // Only respawn when control.json says "running" — paused/stopped/stopping
371
- // are legitimate states the supervisor must not override.
372
- if (!control || control.state !== 'running') return;
435
+ // Respawn when control.json says "running" (PID died unexpectedly) or
436
+ // "degraded" (watchdog-flagged failure state — see _isRespawnEligibleEngineState
437
+ // above). "paused"/"stopped"/"stopping" are legitimate user-intent states the
438
+ // supervisor must not override.
439
+ if (!control || !_isRespawnEligibleEngineState(control.state)) return;
373
440
  if (control.pid && isPidAlive(control.pid)) return;
374
441
 
375
442
  // Cross-watchdog race guard: dashboard.js's in-process engine watchdog
@@ -403,7 +470,11 @@ function checkEngine(now) {
403
470
  function checkEngineHung(now) {
404
471
  if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
405
472
  const control = safeReadJson(CONTROL_PATH_FN());
406
- if (!control || control.state !== 'running') return;
473
+ // Include 'degraded' (see _isRespawnEligibleEngineState) so a permanently
474
+ // frozen tick loop that never gets a chance to self-exit (tickInner never
475
+ // re-enters) still gets force-restarted once its own heartbeat goes stale,
476
+ // instead of being ignored forever the moment the state flips to 'degraded'.
477
+ if (!control || !_isRespawnEligibleEngineState(control.state)) return;
407
478
  // Only acts when the PID is alive — a dead PID is handled by checkEngine().
408
479
  if (!control.pid || !isPidAlive(control.pid)) return;
409
480
 
@@ -424,31 +495,69 @@ function checkEngineHung(now) {
424
495
  console.log(`[supervisor] Engine restarted due to stale heartbeat (new PID: ${newPid})`);
425
496
  }
426
497
 
427
- function checkDashboard(now) {
428
- if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) return;
498
+ async function checkDashboard(now) {
499
+ if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) { _dashHealthFailStreak = 0; return; }
429
500
  const dashPort = _resolveDashPort();
430
501
  const pids = listeningPidsForPort(dashPort);
431
- if (pids.length > 0) return;
432
-
433
- console.log(`[supervisor] Dashboard not listening on port ${dashPort} — respawning...`);
434
- // Reap stray dashboards first. Besides bounding the count, this collapses the
435
- // port-desync case: orphan dashboards holding other ports are killed, freeing
436
- // the canonical port for the fresh one whose beacon _resolveDashPort() reads.
437
- reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
438
- const newPid = spawnDashboard();
439
- _lastDashboardRespawnAt = now;
440
- console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
502
+
503
+ const respawn = (why) => {
504
+ console.log(`[supervisor] Dashboard ${why} on port ${dashPort} — respawning...`);
505
+ // Reap stray dashboards first. Besides bounding the count, this collapses the
506
+ // port-desync case: orphan dashboards holding other ports are killed, freeing
507
+ // the canonical port for the fresh one whose beacon _resolveDashPort() reads.
508
+ // Critically, for the frozen-but-listening case the reap also KILLS the wedged
509
+ // process still holding the port so spawnDashboard() can bind it.
510
+ reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
511
+ const newPid = spawnDashboard();
512
+ _lastDashboardRespawnAt = now;
513
+ _dashHealthFailStreak = 0;
514
+ console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
515
+ };
516
+
517
+ // Hard-down: nothing bound to the port → respawn immediately.
518
+ if (pids.length === 0) {
519
+ _dashHealthFailStreak = 0;
520
+ respawn('not listening');
521
+ return;
522
+ }
523
+
524
+ // Port is bound. Verify the event loop is actually responsive — a frozen-but-
525
+ // listening dashboard keeps the socket LISTENING yet serves the in-browser
526
+ // "unreachable" banner. Probe /api/health; only respawn after a streak of
527
+ // failures so a legitimately busy getStatus() rebuild isn't mistaken for a hang.
528
+ const healthy = await _probeDashboardHealth(dashPort);
529
+ if (healthy === null) {
530
+ // Probe helper unavailable — fall back to port-only liveness (the pre-existing
531
+ // behavior). Don't count this as a failure; we won't respawn a bound dashboard
532
+ // we can't actually prove is frozen.
533
+ _dashHealthFailStreak = 0;
534
+ return;
535
+ }
536
+ if (healthy) { _dashHealthFailStreak = 0; return; }
537
+
538
+ _dashHealthFailStreak++;
539
+ console.log(`[supervisor] Dashboard bound but unresponsive on port ${dashPort} (${_dashHealthFailStreak}/${DASH_HEALTH_MAX_FAILS} strikes)`);
540
+ if (_dashHealthFailStreak < DASH_HEALTH_MAX_FAILS) return;
541
+ respawn('frozen (health probe failed)');
441
542
  }
442
543
 
443
- function tick() {
544
+ let _tickInFlight = false;
545
+ async function tick() {
546
+ // The dashboard health probe is async; guard against a slow probe overlapping
547
+ // the next interval fire. setInterval invoking an async fn is fire-and-forget,
548
+ // so this flag is the only thing preventing concurrent ticks.
549
+ if (_tickInFlight) return;
550
+ _tickInFlight = true;
444
551
  try {
445
552
  if (isStopIntentSet()) return;
446
553
  const now = Date.now();
447
554
  checkEngine(now);
448
555
  checkEngineHung(now);
449
- checkDashboard(now);
556
+ await checkDashboard(now);
450
557
  } catch (e) {
451
558
  console.error(`[supervisor] tick error: ${e && e.message}`);
559
+ } finally {
560
+ _tickInFlight = false;
452
561
  }
453
562
  }
454
563
 
@@ -517,8 +626,11 @@ module.exports = {
517
626
  openAppendFd,
518
627
  checkEngine,
519
628
  checkEngineHung,
629
+ _isRespawnEligibleEngineState,
520
630
  checkDashboard,
631
+ _probeDashboardHealth,
521
632
  SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS,
633
+ DASH_HEALTH_MAX_FAILS,
522
634
  tick,
523
635
  // Path getters honor MINIONS_TEST_DIR via shared.ENGINE_DIR, so test
524
636
  // isolation correctly redirects writes/reads under createTestMinionsDir.
package/engine.js CHANGED
@@ -138,7 +138,7 @@ const worktreePool = require('./engine/worktree-pool');
138
138
  // ─── State Readers (delegated to engine/queries.js) ─────────────────────────
139
139
 
140
140
  const { getConfig, getControl, getDispatch, getNotes,
141
- getAgentStatus, getAgentCharter, getInboxFiles,
141
+ getAgentStatus, getInboxFiles,
142
142
  collectSkillFiles, getSkillIndex, getKnowledgeBaseIndex,
143
143
  getPrs, SKILLS_DIR } = queries;
144
144
 
@@ -10564,7 +10564,7 @@ module.exports = {
10564
10564
 
10565
10565
  // State readers/writers
10566
10566
  getConfig, getControl, getDispatch, getRouting, getNotes,
10567
- getAgentStatus, getAgentCharter, getInboxFiles, getPrs,
10567
+ getAgentStatus, getInboxFiles, getPrs,
10568
10568
  validateConfig,
10569
10569
 
10570
10570
  // Dispatch management (re-exported from engine/dispatch.js)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yemi33/minions",
3
- "version": "0.1.2304",
3
+ "version": "0.1.2306",
4
4
  "description": "Multi-agent AI dev team that runs from ~/.minions/ — five autonomous agents share a single engine, dashboard, and knowledge base",
5
5
  "bin": {
6
6
  "minions": "bin/minions.js"