@yemi33/minions 0.1.2304 → 0.1.2306
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dashboard/js/refresh.js +17 -13
- package/dashboard/js/render-other.js +251 -3
- package/dashboard/js/render-plans.js +1 -1
- package/dashboard/js/render-prs.js +4 -1
- package/dashboard/js/render-utils.js +13 -0
- package/dashboard/js/render-work-items.js +13 -14
- package/dashboard/js/settings.js +3 -3
- package/dashboard/layout.html +13 -0
- package/dashboard/slim/js/modals-tiles.js +21 -130
- package/dashboard/slim/styles.css +9 -14
- package/dashboard/styles.css +79 -0
- package/dashboard.js +41 -9
- package/docs/completion-reports.md +25 -0
- package/docs/copilot-cli-schema.md +31 -1
- package/docs/design-state-storage.md +1 -1
- package/docs/live-checkout-mode.md +5 -3
- package/engine/ado.js +37 -0
- package/engine/consolidation.js +59 -2
- package/engine/lifecycle.js +149 -0
- package/engine/live-checkout.js +290 -79
- package/engine/playbook.js +2 -2
- package/engine/queries.js +11 -7
- package/engine/shared.js +6 -5
- package/engine/supervisor.js +131 -19
- package/engine.js +2 -2
- package/package.json +1 -1
package/engine/supervisor.js
CHANGED
|
@@ -5,7 +5,14 @@
|
|
|
5
5
|
* Spawned `detached: true` by `minions start`/`restart` after the engine and
|
|
6
6
|
* dashboard are up. Polls every SUPERVISOR_INTERVAL_MS:
|
|
7
7
|
* - engine PID from `engine/control.json`
|
|
8
|
-
* - dashboard
|
|
8
|
+
* - dashboard via a port-listener probe (port 7331 by default) AND, when the
|
|
9
|
+
* port is bound, an HTTP `/api/health` probe. The HTTP layer catches the
|
|
10
|
+
* "frozen but still listening" dashboard (wedged event loop) that the
|
|
11
|
+
* port-listen probe alone reports as healthy — it respawns only after
|
|
12
|
+
* DASH_HEALTH_MAX_FAILS consecutive failed probes so a dashboard merely
|
|
13
|
+
* busy with a heavy synchronous getStatus() rebuild isn't killed. What
|
|
14
|
+
* counts as "responsive" (and why engine 'degraded'/'stopped' does NOT mean
|
|
15
|
+
* the dashboard is frozen) is documented at _probeDashboardHealth.
|
|
9
16
|
*
|
|
10
17
|
* When either is dead AND the stop-intent flag is NOT set, respawns the dead
|
|
11
18
|
* one in the same way the CLI does (detached, stdio routed to the engine-
|
|
@@ -76,6 +83,16 @@ function _resolveDashPort() {
|
|
|
76
83
|
const POST_SPAWN_GRACE_MS = Number(process.env.MINIONS_SUPERVISOR_GRACE_MS) || 15000;
|
|
77
84
|
// #421 — heartbeat age threshold before supervisor considers the engine event loop frozen.
|
|
78
85
|
const SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS = Number(process.env.MINIONS_SUPERVISOR_STALE_HEARTBEAT_MS) || 180000; // 3 min
|
|
86
|
+
// Dashboard HTTP liveness (frozen-but-listening detection). A dashboard whose
|
|
87
|
+
// event loop is wedged keeps the socket LISTENING, so the port-listen probe
|
|
88
|
+
// alone reports it healthy while the in-browser SPA shows the "unreachable"
|
|
89
|
+
// banner forever. We layer an HTTP /api/health probe on top: only respawn after
|
|
90
|
+
// DASH_HEALTH_MAX_FAILS *consecutive* failed probes so a dashboard that is
|
|
91
|
+
// merely busy with a heavy synchronous getStatus() rebuild (documented to block
|
|
92
|
+
// the loop 15-25s) isn't mistaken for a hang. With the default 30s interval,
|
|
93
|
+
// 3 strikes ≈ 90s of continuous unresponsiveness before we act.
|
|
94
|
+
const DASH_HEALTH_TIMEOUT_MS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_TIMEOUT_MS) || 5000;
|
|
95
|
+
const DASH_HEALTH_MAX_FAILS = Number(process.env.MINIONS_SUPERVISOR_DASH_HEALTH_FAILS) || 3;
|
|
79
96
|
const isWin = process.platform === 'win32';
|
|
80
97
|
|
|
81
98
|
function safeReadJson(p) {
|
|
@@ -354,6 +371,39 @@ function spawnDashboard() {
|
|
|
354
371
|
// the engine.
|
|
355
372
|
let _lastEngineRespawnAt = 0;
|
|
356
373
|
let _lastDashboardRespawnAt = 0;
|
|
374
|
+
// Consecutive failed dashboard HTTP health probes. Reset on any healthy probe,
|
|
375
|
+
// after a respawn, and inside the post-spawn grace window.
|
|
376
|
+
let _dashHealthFailStreak = 0;
|
|
377
|
+
|
|
378
|
+
// Lazy, defensive HTTP health probe. Reuses restart-health.js's httpGetJson
|
|
379
|
+
// (rule: reuse before re-implementing) but never lets a load failure (e.g. a
|
|
380
|
+
// mid-upgrade broken shared.js that restart-health requires at module top)
|
|
381
|
+
// break the supervisor. Returns null when the probe helper is unavailable so
|
|
382
|
+
// the caller can FALL BACK to port-only liveness instead of false-respawning a
|
|
383
|
+
// healthy dashboard.
|
|
384
|
+
function _probeDashboardHealth(port, timeoutMs = DASH_HEALTH_TIMEOUT_MS) {
|
|
385
|
+
let getJson = null;
|
|
386
|
+
try {
|
|
387
|
+
const rh = require('./restart-health');
|
|
388
|
+
if (rh && rh._private && typeof rh._private.httpGetJson === 'function') {
|
|
389
|
+
getJson = rh._private.httpGetJson;
|
|
390
|
+
}
|
|
391
|
+
} catch { /* restart-health/shared unavailable — fall through */ }
|
|
392
|
+
if (!getJson) return Promise.resolve(null);
|
|
393
|
+
return getJson(`http://127.0.0.1:${port}/api/health`, timeoutMs)
|
|
394
|
+
// "Responsive" == the dashboard event loop executed handleHealth and returned
|
|
395
|
+
// a well-formed 2xx envelope, REGARDLESS of the engine-derived status value.
|
|
396
|
+
// /api/health reports 'degraded' when the engine is paused/stopping and
|
|
397
|
+
// 'stopped' when the engine is down — those are legitimate ENGINE states
|
|
398
|
+
// (owned by checkEngine/checkEngineHung), not evidence the DASHBOARD process
|
|
399
|
+
// is frozen. Gating on status==='healthy' would respawn a perfectly
|
|
400
|
+
// responsive dashboard every time an operator pauses the engine, tearing
|
|
401
|
+
// down the socket and tripping the in-browser "unreachable" banner for no
|
|
402
|
+
// reason. Only a failed/timed-out request (res.ok false or a rejection)
|
|
403
|
+
// means the dashboard's own event loop is actually wedged.
|
|
404
|
+
.then(res => !!(res && res.ok && res.json && typeof res.json.status === 'string'))
|
|
405
|
+
.catch(() => false);
|
|
406
|
+
}
|
|
357
407
|
|
|
358
408
|
// Window during which a `null` pid + recent `restarted_at` is interpreted as
|
|
359
409
|
// "another watchdog is currently respawning the engine — don't double-spawn."
|
|
@@ -364,12 +414,29 @@ let _lastDashboardRespawnAt = 0;
|
|
|
364
414
|
// to enter cli.js:443 and write its own PID on cold Windows boots.
|
|
365
415
|
const RESPAWN_IN_PROGRESS_WINDOW_MS = Number(process.env.MINIONS_SUPERVISOR_RESPAWN_WINDOW_MS) || 60000;
|
|
366
416
|
|
|
417
|
+
// #3758 — 'degraded' is a FAILURE state (set by dashboard.js's frozen-tick
|
|
418
|
+
// watchdog, see dashboard.js#_markEngineAsDegradedIfFrozen), not a deliberate
|
|
419
|
+
// user-intent state like 'paused'/'stopping'/'stopped'. When a frozen tick loop
|
|
420
|
+
// finally unblocks, engine.js#tickInner reads control.state !== 'running' and
|
|
421
|
+
// calls process.exit(0) — self-exiting exactly BECAUSE it saw 'degraded'. The
|
|
422
|
+
// dashboard.js comment documents the expected recovery: "the PID-dead watchdog
|
|
423
|
+
// then auto-restarts it." Excluding 'degraded' here left that promise unmet —
|
|
424
|
+
// the engine PID goes away and the control plane silently stays down until a
|
|
425
|
+
// manual `minions restart`. Respawn eligibility must include both 'running'
|
|
426
|
+
// (still up) and 'degraded' (crashed/crashing out of a real failure); only
|
|
427
|
+
// 'paused', 'stopping', and 'stopped' reflect actual user intent to stay down.
|
|
428
|
+
function _isRespawnEligibleEngineState(state) {
|
|
429
|
+
return state === 'running' || state === 'degraded';
|
|
430
|
+
}
|
|
431
|
+
|
|
367
432
|
function checkEngine(now) {
|
|
368
433
|
if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
369
434
|
const control = safeReadJson(CONTROL_PATH_FN());
|
|
370
|
-
//
|
|
371
|
-
//
|
|
372
|
-
|
|
435
|
+
// Respawn when control.json says "running" (PID died unexpectedly) or
|
|
436
|
+
// "degraded" (watchdog-flagged failure state — see _isRespawnEligibleEngineState
|
|
437
|
+
// above). "paused"/"stopped"/"stopping" are legitimate user-intent states the
|
|
438
|
+
// supervisor must not override.
|
|
439
|
+
if (!control || !_isRespawnEligibleEngineState(control.state)) return;
|
|
373
440
|
if (control.pid && isPidAlive(control.pid)) return;
|
|
374
441
|
|
|
375
442
|
// Cross-watchdog race guard: dashboard.js's in-process engine watchdog
|
|
@@ -403,7 +470,11 @@ function checkEngine(now) {
|
|
|
403
470
|
function checkEngineHung(now) {
|
|
404
471
|
if (now - _lastEngineRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
405
472
|
const control = safeReadJson(CONTROL_PATH_FN());
|
|
406
|
-
|
|
473
|
+
// Include 'degraded' (see _isRespawnEligibleEngineState) so a permanently
|
|
474
|
+
// frozen tick loop that never gets a chance to self-exit (tickInner never
|
|
475
|
+
// re-enters) still gets force-restarted once its own heartbeat goes stale,
|
|
476
|
+
// instead of being ignored forever the moment the state flips to 'degraded'.
|
|
477
|
+
if (!control || !_isRespawnEligibleEngineState(control.state)) return;
|
|
407
478
|
// Only acts when the PID is alive — a dead PID is handled by checkEngine().
|
|
408
479
|
if (!control.pid || !isPidAlive(control.pid)) return;
|
|
409
480
|
|
|
@@ -424,31 +495,69 @@ function checkEngineHung(now) {
|
|
|
424
495
|
console.log(`[supervisor] Engine restarted due to stale heartbeat (new PID: ${newPid})`);
|
|
425
496
|
}
|
|
426
497
|
|
|
427
|
-
function checkDashboard(now) {
|
|
428
|
-
if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) return;
|
|
498
|
+
async function checkDashboard(now) {
|
|
499
|
+
if (now - _lastDashboardRespawnAt < POST_SPAWN_GRACE_MS) { _dashHealthFailStreak = 0; return; }
|
|
429
500
|
const dashPort = _resolveDashPort();
|
|
430
501
|
const pids = listeningPidsForPort(dashPort);
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
502
|
+
|
|
503
|
+
const respawn = (why) => {
|
|
504
|
+
console.log(`[supervisor] Dashboard ${why} on port ${dashPort} — respawning...`);
|
|
505
|
+
// Reap stray dashboards first. Besides bounding the count, this collapses the
|
|
506
|
+
// port-desync case: orphan dashboards holding other ports are killed, freeing
|
|
507
|
+
// the canonical port for the fresh one whose beacon _resolveDashPort() reads.
|
|
508
|
+
// Critically, for the frozen-but-listening case the reap also KILLS the wedged
|
|
509
|
+
// process still holding the port so spawnDashboard() can bind it.
|
|
510
|
+
reapStrayProcesses(path.join(MINIONS_DIR, 'dashboard.js'), 'dashboard');
|
|
511
|
+
const newPid = spawnDashboard();
|
|
512
|
+
_lastDashboardRespawnAt = now;
|
|
513
|
+
_dashHealthFailStreak = 0;
|
|
514
|
+
console.log(`[supervisor] Dashboard respawned (new PID: ${newPid})`);
|
|
515
|
+
};
|
|
516
|
+
|
|
517
|
+
// Hard-down: nothing bound to the port → respawn immediately.
|
|
518
|
+
if (pids.length === 0) {
|
|
519
|
+
_dashHealthFailStreak = 0;
|
|
520
|
+
respawn('not listening');
|
|
521
|
+
return;
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
// Port is bound. Verify the event loop is actually responsive — a frozen-but-
|
|
525
|
+
// listening dashboard keeps the socket LISTENING yet serves the in-browser
|
|
526
|
+
// "unreachable" banner. Probe /api/health; only respawn after a streak of
|
|
527
|
+
// failures so a legitimately busy getStatus() rebuild isn't mistaken for a hang.
|
|
528
|
+
const healthy = await _probeDashboardHealth(dashPort);
|
|
529
|
+
if (healthy === null) {
|
|
530
|
+
// Probe helper unavailable — fall back to port-only liveness (the pre-existing
|
|
531
|
+
// behavior). Don't count this as a failure; we won't respawn a bound dashboard
|
|
532
|
+
// we can't actually prove is frozen.
|
|
533
|
+
_dashHealthFailStreak = 0;
|
|
534
|
+
return;
|
|
535
|
+
}
|
|
536
|
+
if (healthy) { _dashHealthFailStreak = 0; return; }
|
|
537
|
+
|
|
538
|
+
_dashHealthFailStreak++;
|
|
539
|
+
console.log(`[supervisor] Dashboard bound but unresponsive on port ${dashPort} (${_dashHealthFailStreak}/${DASH_HEALTH_MAX_FAILS} strikes)`);
|
|
540
|
+
if (_dashHealthFailStreak < DASH_HEALTH_MAX_FAILS) return;
|
|
541
|
+
respawn('frozen (health probe failed)');
|
|
441
542
|
}
|
|
442
543
|
|
|
443
|
-
|
|
544
|
+
let _tickInFlight = false;
|
|
545
|
+
async function tick() {
|
|
546
|
+
// The dashboard health probe is async; guard against a slow probe overlapping
|
|
547
|
+
// the next interval fire. setInterval invoking an async fn is fire-and-forget,
|
|
548
|
+
// so this flag is the only thing preventing concurrent ticks.
|
|
549
|
+
if (_tickInFlight) return;
|
|
550
|
+
_tickInFlight = true;
|
|
444
551
|
try {
|
|
445
552
|
if (isStopIntentSet()) return;
|
|
446
553
|
const now = Date.now();
|
|
447
554
|
checkEngine(now);
|
|
448
555
|
checkEngineHung(now);
|
|
449
|
-
checkDashboard(now);
|
|
556
|
+
await checkDashboard(now);
|
|
450
557
|
} catch (e) {
|
|
451
558
|
console.error(`[supervisor] tick error: ${e && e.message}`);
|
|
559
|
+
} finally {
|
|
560
|
+
_tickInFlight = false;
|
|
452
561
|
}
|
|
453
562
|
}
|
|
454
563
|
|
|
@@ -517,8 +626,11 @@ module.exports = {
|
|
|
517
626
|
openAppendFd,
|
|
518
627
|
checkEngine,
|
|
519
628
|
checkEngineHung,
|
|
629
|
+
_isRespawnEligibleEngineState,
|
|
520
630
|
checkDashboard,
|
|
631
|
+
_probeDashboardHealth,
|
|
521
632
|
SUPERVISOR_STALE_ENGINE_HEARTBEAT_MS,
|
|
633
|
+
DASH_HEALTH_MAX_FAILS,
|
|
522
634
|
tick,
|
|
523
635
|
// Path getters honor MINIONS_TEST_DIR via shared.ENGINE_DIR, so test
|
|
524
636
|
// isolation correctly redirects writes/reads under createTestMinionsDir.
|
package/engine.js
CHANGED
|
@@ -138,7 +138,7 @@ const worktreePool = require('./engine/worktree-pool');
|
|
|
138
138
|
// ─── State Readers (delegated to engine/queries.js) ─────────────────────────
|
|
139
139
|
|
|
140
140
|
const { getConfig, getControl, getDispatch, getNotes,
|
|
141
|
-
getAgentStatus,
|
|
141
|
+
getAgentStatus, getInboxFiles,
|
|
142
142
|
collectSkillFiles, getSkillIndex, getKnowledgeBaseIndex,
|
|
143
143
|
getPrs, SKILLS_DIR } = queries;
|
|
144
144
|
|
|
@@ -10564,7 +10564,7 @@ module.exports = {
|
|
|
10564
10564
|
|
|
10565
10565
|
// State readers/writers
|
|
10566
10566
|
getConfig, getControl, getDispatch, getRouting, getNotes,
|
|
10567
|
-
getAgentStatus,
|
|
10567
|
+
getAgentStatus, getInboxFiles, getPrs,
|
|
10568
10568
|
validateConfig,
|
|
10569
10569
|
|
|
10570
10570
|
// Dispatch management (re-exported from engine/dispatch.js)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@yemi33/minions",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.2306",
|
|
4
4
|
"description": "Multi-agent AI dev team that runs from ~/.minions/ — five autonomous agents share a single engine, dashboard, and knowledge base",
|
|
5
5
|
"bin": {
|
|
6
6
|
"minions": "bin/minions.js"
|