c8ctl-plugin-nano 1.19.0 → 1.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +72 -0
- package/c8ctl-plugin.js +1328 -30
- package/package.json +8 -8
package/c8ctl-plugin.js
CHANGED
|
@@ -35,6 +35,7 @@ import {
|
|
|
35
35
|
openSync,
|
|
36
36
|
readFileSync,
|
|
37
37
|
writeFileSync,
|
|
38
|
+
appendFileSync,
|
|
38
39
|
rmSync,
|
|
39
40
|
readdirSync,
|
|
40
41
|
chmodSync,
|
|
@@ -43,13 +44,18 @@ import {
|
|
|
43
44
|
statfsSync,
|
|
44
45
|
lstatSync,
|
|
45
46
|
mkdtempSync,
|
|
47
|
+
closeSync,
|
|
48
|
+
watchFile,
|
|
49
|
+
unwatchFile,
|
|
46
50
|
} from 'node:fs';
|
|
47
|
-
import {
|
|
48
|
-
import {
|
|
51
|
+
import { createConnection, createServer } from 'node:net';
|
|
52
|
+
import { randomUUID, createHash } from 'node:crypto';
|
|
53
|
+
import { homedir, platform as osPlatform, devNull, tmpdir } from 'node:os';
|
|
49
54
|
import { join, isAbsolute, resolve as resolvePath, dirname, basename, sep } from 'node:path';
|
|
50
55
|
import { createRequire } from 'node:module';
|
|
51
56
|
import { fileURLToPath } from 'node:url';
|
|
52
57
|
import { createInterface } from 'node:readline/promises';
|
|
58
|
+
import { createInterface as createReadline } from 'node:readline';
|
|
53
59
|
import { platformForHost } from './platforms.mjs';
|
|
54
60
|
|
|
55
61
|
const requireFromHere = createRequire(import.meta.url);
|
|
@@ -117,6 +123,7 @@ const READINESS_POLL_MS = 500;
|
|
|
117
123
|
const HEALTH_TIMEOUT_MS = 1_500;
|
|
118
124
|
const STOP_GRACE_MS = 8_000;
|
|
119
125
|
const PROCESSOS_STATE_FILE = 'processos.json';
|
|
126
|
+
const SUPERVISOR_STATE_FILE = 'supervisor.json';
|
|
120
127
|
const PROCESSOS_DEFAULT_PORT = 8090;
|
|
121
128
|
const DEFAULT_NANO_URL = 'http://localhost:8080';
|
|
122
129
|
|
|
@@ -191,6 +198,34 @@ function getLogDir() {
|
|
|
191
198
|
return join(getStateHome(), 'logs');
|
|
192
199
|
}
|
|
193
200
|
|
|
201
|
+
// ---------------------------------------------------------------------------
|
|
202
|
+
// Worker supervisor paths (see the `supervisor` command). The supervisor is a
|
|
203
|
+
// detached daemon that manages a fleet of `nano work` child processes; it keeps
|
|
204
|
+
// its own state file, a control socket, and per-worker + daemon log files.
|
|
205
|
+
// ---------------------------------------------------------------------------
|
|
206
|
+
|
|
207
|
+
function getSupervisorStateFile() {
|
|
208
|
+
return join(getStateHome(), SUPERVISOR_STATE_FILE);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
function getSupervisorLogDir() {
|
|
212
|
+
return join(getLogDir(), 'supervisor');
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/**
|
|
216
|
+
* Deterministic control-socket path shared by the daemon and every client.
|
|
217
|
+
* Derived from a hash of the (possibly overridden) state home so distinct
|
|
218
|
+
* C8CTL_NANO_HOME instances get distinct sockets, and kept SHORT to stay under
|
|
219
|
+
* the ~104-byte AF_UNIX `sun_path` limit on macOS regardless of username. On
|
|
220
|
+
* Windows a named pipe is used instead. The chosen path is also recorded in the
|
|
221
|
+
* state file so clients can prefer the daemon's own reported path.
|
|
222
|
+
*/
|
|
223
|
+
function getSupervisorSocketPath() {
|
|
224
|
+
const hash = createHash('sha1').update(getStateHome()).digest('hex').slice(0, 8);
|
|
225
|
+
if (osPlatform() === 'win32') return `\\\\.\\pipe\\c8ctl-nano-sup-${hash}`;
|
|
226
|
+
return join(tmpdir(), `c8ctl-nano-sup-${hash}.sock`);
|
|
227
|
+
}
|
|
228
|
+
|
|
194
229
|
// ---------------------------------------------------------------------------
|
|
195
230
|
// Persistent plugin config (config.json) — user settings that survive across
|
|
196
231
|
// clusters: the binary path and the workspace (models/workers) location.
|
|
@@ -200,20 +235,39 @@ function getConfigFile() {
|
|
|
200
235
|
return join(getStateHome(), CONFIG_FILE);
|
|
201
236
|
}
|
|
202
237
|
|
|
203
|
-
function
|
|
238
|
+
function readConfigStrict() {
|
|
204
239
|
const file = getConfigFile();
|
|
205
240
|
if (!existsSync(file)) return {};
|
|
241
|
+
const cfg = JSON.parse(readFileSync(file, 'utf-8'));
|
|
242
|
+
return cfg && typeof cfg === 'object' ? cfg : {};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
function readConfig() {
|
|
206
246
|
try {
|
|
207
|
-
|
|
208
|
-
return cfg && typeof cfg === 'object' ? cfg : {};
|
|
247
|
+
return readConfigStrict();
|
|
209
248
|
} catch {
|
|
249
|
+
// A malformed/torn config.json is swallowed here so ordinary callers get an
|
|
250
|
+
// empty map; callers that must tell "absent" from "unreadable" apart use
|
|
251
|
+
// readConfigStrict() directly and handle the throw.
|
|
210
252
|
return {};
|
|
211
253
|
}
|
|
212
254
|
}
|
|
213
255
|
|
|
214
256
|
function writeConfig(cfg) {
|
|
215
257
|
mkdirSync(getStateHome(), { recursive: true });
|
|
216
|
-
|
|
258
|
+
// Atomic write: serialize to a temp file in the same dir, then rename over the
|
|
259
|
+
// target. A rename is atomic on a POSIX filesystem, so a concurrent reader
|
|
260
|
+
// (e.g. `work`'s profile watcher, or another `assign`) never observes a
|
|
261
|
+
// half-written config.json and JSON.parse never sees a torn file.
|
|
262
|
+
const target = getConfigFile();
|
|
263
|
+
const tmp = `${target}.${process.pid}.${Date.now()}.tmp`;
|
|
264
|
+
writeFileSync(tmp, JSON.stringify(cfg, null, 2));
|
|
265
|
+
try {
|
|
266
|
+
renameSync(tmp, target);
|
|
267
|
+
} catch (err) {
|
|
268
|
+
try { rmSync(tmp, { force: true }); } catch { /* best effort */ }
|
|
269
|
+
throw err;
|
|
270
|
+
}
|
|
217
271
|
}
|
|
218
272
|
|
|
219
273
|
/**
|
|
@@ -368,7 +422,7 @@ function launcherEnvMarkers(resolved) {
|
|
|
368
422
|
// Argument parsing
|
|
369
423
|
// ---------------------------------------------------------------------------
|
|
370
424
|
|
|
371
|
-
const VALID_SUBCOMMANDS = ['start', 'stop', 'status', 'logs', 'log', 'restart', 'pause', 'resume', 'clean', 'set', 'config', 'update', 'hire', 'assign', 'work'];
|
|
425
|
+
const VALID_SUBCOMMANDS = ['start', 'stop', 'status', 'logs', 'log', 'restart', 'pause', 'resume', 'clean', 'set', 'config', 'update', 'hire', 'assign', 'work', 'supervisor'];
|
|
372
426
|
|
|
373
427
|
/**
|
|
374
428
|
* Parse positional args + flags into a normalized request.
|
|
@@ -1565,6 +1619,21 @@ function jobTypeMatrix(rank, capabilities) {
|
|
|
1565
1619
|
return [...new Set(tokens)];
|
|
1566
1620
|
}
|
|
1567
1621
|
|
|
1622
|
+
/**
|
|
1623
|
+
* Diff a running set of job-type pollers against a desired set. Pure so the
|
|
1624
|
+
* profile-watch reconcile in `work` (which starts pollers for `added` types and
|
|
1625
|
+
* gracefully drains pollers for `removed` types) is unit-testable. Order in the
|
|
1626
|
+
* returned arrays is stable (desired order for `added`, current order for
|
|
1627
|
+
* `removed`) for deterministic logging.
|
|
1628
|
+
*/
|
|
1629
|
+
function diffJobTypes(current, desired) {
|
|
1630
|
+
const cur = new Set(current);
|
|
1631
|
+
const want = new Set(desired);
|
|
1632
|
+
const added = [...want].filter((t) => !cur.has(t));
|
|
1633
|
+
const removed = [...cur].filter((t) => !want.has(t));
|
|
1634
|
+
return { added, removed };
|
|
1635
|
+
}
|
|
1636
|
+
|
|
1568
1637
|
/** All persisted hire profiles, keyed by name. */
|
|
1569
1638
|
function readHires() {
|
|
1570
1639
|
const cfg = readConfig();
|
|
@@ -1573,6 +1642,16 @@ function readHires() {
|
|
|
1573
1642
|
return cfg.hires && typeof cfg.hires === 'object' && !Array.isArray(cfg.hires) ? cfg.hires : {};
|
|
1574
1643
|
}
|
|
1575
1644
|
|
|
1645
|
+
/**
|
|
1646
|
+
* Like readHires(), but propagates a malformed-config parse error instead of
|
|
1647
|
+
* swallowing it. Lets a caller distinguish "profile genuinely removed" from
|
|
1648
|
+
* "config temporarily unreadable/torn" so it can report an accurate reason.
|
|
1649
|
+
*/
|
|
1650
|
+
function readHiresStrict() {
|
|
1651
|
+
const cfg = readConfigStrict();
|
|
1652
|
+
return cfg.hires && typeof cfg.hires === 'object' && !Array.isArray(cfg.hires) ? cfg.hires : {};
|
|
1653
|
+
}
|
|
1654
|
+
|
|
1576
1655
|
/** Persist a single hire profile into config.json under `hires`. */
|
|
1577
1656
|
function writeHire(profile) {
|
|
1578
1657
|
const cfg = readConfig();
|
|
@@ -2975,7 +3054,10 @@ async function workAgent(req, flags) {
|
|
|
2975
3054
|
logger.info(` max parallel: ${maxParallelJobs}; job timeout: ${jobKillMs}ms; activation lock: ${jobLockMs}ms; poll timeout: ${pollTimeoutMs}ms`);
|
|
2976
3055
|
logger.info('Polling for work — press Ctrl-C to stop.');
|
|
2977
3056
|
|
|
2978
|
-
|
|
3057
|
+
// A per-job-type worker factory. Captures all the CLI-local + profile context
|
|
3058
|
+
// in closure scope so the profile watcher below can (re)spawn a poller for any
|
|
3059
|
+
// job type on demand without re-reading the flags.
|
|
3060
|
+
const makeWorker = (jobType) =>
|
|
2979
3061
|
camunda.createJobWorker({
|
|
2980
3062
|
jobType,
|
|
2981
3063
|
workerName: `${name}:${jobType}`,
|
|
@@ -3156,35 +3238,172 @@ async function workAgent(req, flags) {
|
|
|
3156
3238
|
variables: { [AGENT_RESULT_KEY]: resultEnvelope },
|
|
3157
3239
|
});
|
|
3158
3240
|
},
|
|
3159
|
-
})
|
|
3160
|
-
|
|
3241
|
+
});
|
|
3242
|
+
|
|
3243
|
+
// Live worker registry keyed by job type, so the profile watcher can add or
|
|
3244
|
+
// drain individual pollers without disturbing the others. `draining` is the
|
|
3245
|
+
// shutdown latch (shared with the watcher so a reconcile can't race a stop).
|
|
3246
|
+
const workers = new Map();
|
|
3247
|
+
let draining = false;
|
|
3248
|
+
|
|
3249
|
+
const drainWorker = async (w) => {
|
|
3250
|
+
try {
|
|
3251
|
+
if (typeof w.stopGracefully === 'function') {
|
|
3252
|
+
await w.stopGracefully({ waitUpToMs: STOP_GRACE_MS });
|
|
3253
|
+
} else if (typeof w.stop === 'function') {
|
|
3254
|
+
await w.stop();
|
|
3255
|
+
}
|
|
3256
|
+
return true;
|
|
3257
|
+
} catch {
|
|
3258
|
+
return false; // best-effort: never let one worker's stop failure hang us
|
|
3259
|
+
}
|
|
3260
|
+
};
|
|
3261
|
+
|
|
3262
|
+
const spawnJobType = (jobType) => {
|
|
3263
|
+
if (workers.has(jobType)) return false;
|
|
3264
|
+
workers.set(jobType, makeWorker(jobType));
|
|
3265
|
+
return true;
|
|
3266
|
+
};
|
|
3267
|
+
|
|
3268
|
+
for (const jobType of jobTypes) spawnJobType(jobType);
|
|
3269
|
+
|
|
3270
|
+
// ---- Live profile watch: reconcile the poller set when the watched profile's
|
|
3271
|
+
// job types change (e.g. `c8ctl nano assign <name> …`) — start pollers for
|
|
3272
|
+
// added types, gracefully drain pollers for removed types — without a restart
|
|
3273
|
+
// and without disturbing unchanged types' in-flight work. ----
|
|
3274
|
+
const configFile = getConfigFile();
|
|
3275
|
+
const WATCH_INTERVAL_MS = 1500;
|
|
3276
|
+
let reconciling = false;
|
|
3277
|
+
// Set when a profile change arrives while a reconcile is already in flight, so
|
|
3278
|
+
// we run one more pass after the current drain completes instead of dropping
|
|
3279
|
+
// the update until the next change fires.
|
|
3280
|
+
let reconcileRequested = false;
|
|
3281
|
+
// Handle to the in-flight reconcile so shutdown can wait for it to finish
|
|
3282
|
+
// before snapshotting `workers` (avoids double-stops / missed drains).
|
|
3283
|
+
let inFlightReconcile = null;
|
|
3284
|
+
|
|
3285
|
+
// Desired job types from the CURRENT on-disk profile (matrix ∪ --job-type
|
|
3286
|
+
// extras). Returns { skip } for a transient/torn read, a vanished profile, or
|
|
3287
|
+
// an invalid edit — callers must then KEEP the running set, never tear down.
|
|
3288
|
+
const desiredJobTypes = () => {
|
|
3289
|
+
let stored;
|
|
3290
|
+
try {
|
|
3291
|
+
stored = readHiresStrict()[name];
|
|
3292
|
+
} catch {
|
|
3293
|
+
// config.json exists but doesn't parse (e.g. a torn write): the profile is
|
|
3294
|
+
// NOT necessarily gone, so don't claim it was deleted — skip this pass.
|
|
3295
|
+
return { skip: 'config unreadable' };
|
|
3296
|
+
}
|
|
3297
|
+
if (!stored) return { skip: 'deleted' };
|
|
3298
|
+
const norm = normalizeStoredProfile(name, stored);
|
|
3299
|
+
if (norm.error) return { skip: norm.error };
|
|
3300
|
+
const m = jobTypeMatrix(norm.profile.rank, norm.profile.capabilities);
|
|
3301
|
+
return { jobTypes: [...new Set([...m, ...extraJobTypes])] };
|
|
3302
|
+
};
|
|
3303
|
+
|
|
3304
|
+
const reconcile = () => {
|
|
3305
|
+
if (draining) return inFlightReconcile || Promise.resolve();
|
|
3306
|
+
if (reconciling) {
|
|
3307
|
+
// A change landed mid-reconcile — remember it so the current pass loops
|
|
3308
|
+
// once more rather than leaving the worker set stale until the next edit.
|
|
3309
|
+
// Return the ACTUAL in-flight promise (not a fresh short-lived one) so a
|
|
3310
|
+
// caller — including shutdown — waits for the real reconcile to finish.
|
|
3311
|
+
reconcileRequested = true;
|
|
3312
|
+
return inFlightReconcile || Promise.resolve();
|
|
3313
|
+
}
|
|
3314
|
+
reconciling = true;
|
|
3315
|
+
reconcileRequested = false;
|
|
3316
|
+
inFlightReconcile = (async () => {
|
|
3317
|
+
try {
|
|
3318
|
+
do {
|
|
3319
|
+
reconcileRequested = false;
|
|
3320
|
+
await runReconcilePass();
|
|
3321
|
+
} while (reconcileRequested && !draining);
|
|
3322
|
+
} finally {
|
|
3323
|
+
reconciling = false;
|
|
3324
|
+
inFlightReconcile = null;
|
|
3325
|
+
}
|
|
3326
|
+
})();
|
|
3327
|
+
return inFlightReconcile;
|
|
3328
|
+
};
|
|
3329
|
+
|
|
3330
|
+
const runReconcilePass = async () => {
|
|
3331
|
+
const desired = desiredJobTypes();
|
|
3332
|
+
if (desired.skip) {
|
|
3333
|
+
if (desired.skip === 'deleted') {
|
|
3334
|
+
logger.warn(`Profile "${name}" is gone from config — keeping the current ${workers.size} worker(s) running.`);
|
|
3335
|
+
} else {
|
|
3336
|
+
logger.warn(`Profile "${name}" reload skipped — ${desired.skip}; keeping current workers.`);
|
|
3337
|
+
}
|
|
3338
|
+
return;
|
|
3339
|
+
}
|
|
3340
|
+
const { added, removed } = diffJobTypes([...workers.keys()], desired.jobTypes);
|
|
3341
|
+
if (added.length === 0 && removed.length === 0) return;
|
|
3342
|
+
logger.info(`Profile "${name}" changed — reconciling job types (+${added.length} / -${removed.length}).`);
|
|
3343
|
+
for (const jt of added) {
|
|
3344
|
+
spawnJobType(jt);
|
|
3345
|
+
logger.info(` + now listening on ${jt}`);
|
|
3346
|
+
}
|
|
3347
|
+
await Promise.all(
|
|
3348
|
+
removed.map(async (jt) => {
|
|
3349
|
+
const w = workers.get(jt);
|
|
3350
|
+
logger.info(` - draining ${jt} …`);
|
|
3351
|
+
const ok = await drainWorker(w);
|
|
3352
|
+
if (ok) {
|
|
3353
|
+
// Only drop it from the registry once it has actually stopped, so a
|
|
3354
|
+
// failed drain stays tracked and gets retried on the next reconcile
|
|
3355
|
+
// pass (or on shutdown) instead of leaking an untracked poller.
|
|
3356
|
+
workers.delete(jt);
|
|
3357
|
+
logger.info(` - stopped ${jt}`);
|
|
3358
|
+
} else {
|
|
3359
|
+
logger.warn(` - ${jt} did not stop cleanly; keeping it tracked so it is retried on the next reconcile or shutdown.`);
|
|
3360
|
+
}
|
|
3361
|
+
}),
|
|
3362
|
+
);
|
|
3363
|
+
logger.info(` now listening on ${workers.size} job type(s): ${[...workers.keys()].join(' ')}`);
|
|
3364
|
+
};
|
|
3365
|
+
|
|
3366
|
+
// `watchFile` (polling stat) is deliberate over `fs.watch`: it survives the
|
|
3367
|
+
// atomic temp+rename that `writeConfig` does (fs.watch would rebind to the old
|
|
3368
|
+
// inode and go silent), and it's uniform across platforms. Profile edits are
|
|
3369
|
+
// rare + manual, so a ~1.5s poll latency is fine.
|
|
3370
|
+
watchFile(configFile, { interval: WATCH_INTERVAL_MS }, (curr, prev) => {
|
|
3371
|
+
// Fires each interval; act only on real changes. Compare mtime, ctime and
|
|
3372
|
+
// size, not mtime alone: on filesystems with coarse mtime resolution (or two
|
|
3373
|
+
// edits within one mtime tick) mtimeMs can be unchanged while size/ctimeMs
|
|
3374
|
+
// differ, and an mtime-only guard would skip a genuine profile update.
|
|
3375
|
+
if (
|
|
3376
|
+
curr.mtimeMs === prev.mtimeMs &&
|
|
3377
|
+
curr.ctimeMs === prev.ctimeMs &&
|
|
3378
|
+
curr.size === prev.size
|
|
3379
|
+
) return;
|
|
3380
|
+
// `reconcile()` owns the `inFlightReconcile` handle: a change arriving while
|
|
3381
|
+
// a reconcile is already running coalesces into the current pass and returns
|
|
3382
|
+
// that same in-flight promise, so shutdown always waits for the real one.
|
|
3383
|
+
reconcile().catch((err) => logger.warn(`profile reload failed: ${err?.message || err}`));
|
|
3384
|
+
});
|
|
3161
3385
|
|
|
3162
3386
|
// Keep the process alive until a stop signal, then drain gracefully.
|
|
3163
3387
|
await new Promise((resolve) => {
|
|
3164
|
-
let stopping = false;
|
|
3165
3388
|
const stop = async (signal) => {
|
|
3166
|
-
if (
|
|
3167
|
-
|
|
3168
|
-
|
|
3389
|
+
if (draining) return;
|
|
3390
|
+
draining = true;
|
|
3391
|
+
// Stop watching first so no new reconcile can be triggered, then wait for
|
|
3392
|
+
// any in-flight reconcile to finish before snapshotting `workers` — this
|
|
3393
|
+
// prevents double-stops, missed drains, or a wrong worker count on exit.
|
|
3394
|
+
unwatchFile(configFile);
|
|
3395
|
+
if (inFlightReconcile) {
|
|
3396
|
+
logger.info('Waiting for in-flight profile reconcile to finish before shutdown…');
|
|
3397
|
+
await inFlightReconcile;
|
|
3398
|
+
}
|
|
3399
|
+
const list = [...workers.values()];
|
|
3400
|
+
logger.info(`Received ${signal} — stopping ${list.length} worker(s)...`);
|
|
3169
3401
|
if (reaperTimer) clearInterval(reaperTimer);
|
|
3170
3402
|
if (runDirTimer) clearInterval(runDirTimer);
|
|
3171
|
-
|
|
3172
|
-
|
|
3173
|
-
workers.map(async (w) => {
|
|
3174
|
-
try {
|
|
3175
|
-
if (typeof w.stopGracefully === 'function') {
|
|
3176
|
-
await w.stopGracefully({ waitUpToMs: STOP_GRACE_MS });
|
|
3177
|
-
} else if (typeof w.stop === 'function') {
|
|
3178
|
-
await w.stop();
|
|
3179
|
-
}
|
|
3180
|
-
} catch {
|
|
3181
|
-
// best-effort: never let one worker's stop failure hang shutdown
|
|
3182
|
-
stopFailures += 1;
|
|
3183
|
-
}
|
|
3184
|
-
}),
|
|
3185
|
-
);
|
|
3403
|
+
const results = await Promise.all(list.map(drainWorker));
|
|
3404
|
+
const stopFailures = results.filter((ok) => !ok).length;
|
|
3186
3405
|
if (stopFailures > 0) {
|
|
3187
|
-
logger.warn(`${stopFailures} of ${
|
|
3406
|
+
logger.warn(`${stopFailures} of ${list.length} worker(s) did not stop cleanly; some connections may still be open.`);
|
|
3188
3407
|
} else {
|
|
3189
3408
|
logger.info('All workers stopped.');
|
|
3190
3409
|
}
|
|
@@ -3195,6 +3414,1051 @@ async function workAgent(req, flags) {
|
|
|
3195
3414
|
});
|
|
3196
3415
|
}
|
|
3197
3416
|
|
|
3417
|
+
// ---------------------------------------------------------------------------
|
|
3418
|
+
// supervisor — run & manage a fleet of `nano work` children from one terminal.
|
|
3419
|
+
//
|
|
3420
|
+
// `nano work` needs the c8ctl host runtime (createClient), so worker loops
|
|
3421
|
+
// cannot run inside a bare detached process. The supervisor is therefore a
|
|
3422
|
+
// process *manager*: a detached daemon spawns one `c8ctl nano work <profile>`
|
|
3423
|
+
// child per worker, restarts crashed children with capped backoff, and serves a
|
|
3424
|
+
// control socket (newline-delimited JSON) used by both the management
|
|
3425
|
+
// subcommands (status/add/remove/restart/stop/logs — no interactive surface
|
|
3426
|
+
// needed) and the interactive `attach` console, which can be detached from
|
|
3427
|
+
// (leaving the daemon running) or used to `stop` the whole fleet.
|
|
3428
|
+
// ---------------------------------------------------------------------------
|
|
3429
|
+
|
|
3430
|
+
const SUPERVISOR_BACKOFF_BASE_MS = 1_000;
|
|
3431
|
+
const SUPERVISOR_BACKOFF_MAX_MS = 30_000;
|
|
3432
|
+
// A child that stayed up at least this long before exiting is not crash-looping,
|
|
3433
|
+
// so its restart backoff is reset to zero.
|
|
3434
|
+
const SUPERVISOR_HEALTHY_UPTIME_MS = 60_000;
|
|
3435
|
+
const SUPERVISOR_CONNECT_TIMEOUT_MS = 6_000;
|
|
3436
|
+
// End-to-end deadline for a single request: once connected, a wedged/incompatible
|
|
3437
|
+
// daemon that accepts but never sends a `final` frame must not hang the client.
|
|
3438
|
+
const SUPERVISOR_RESPONSE_TIMEOUT_MS = 15_000;
|
|
3439
|
+
// Tighter end-to-end deadline for quick liveness probes (status checks used by
|
|
3440
|
+
// liveSupervisor/ensureSupervisor). Without this, a daemon that accepts the
|
|
3441
|
+
// connection but never returns a `final` frame would still block the "fast"
|
|
3442
|
+
// probe for the full SUPERVISOR_RESPONSE_TIMEOUT_MS, hanging stop/remove/restart.
|
|
3443
|
+
const SUPERVISOR_PROBE_RESPONSE_TIMEOUT_MS = 2_000;
|
|
3444
|
+
// Hard cap on a single connection's inbound buffer, so a misbehaving client
|
|
3445
|
+
// can't grow the daemon's memory without bound with a newline-free frame.
|
|
3446
|
+
const SUPERVISOR_MAX_FRAME_BYTES = 1 << 20; // 1 MiB
|
|
3447
|
+
|
|
3448
|
+
// The `nano work` flags forwarded verbatim to each spawned child.
|
|
3449
|
+
// kind: 'value' → `--flag v`; 'boolean' → `--flag`; 'list' → repeated `--flag v`.
|
|
3450
|
+
const WORK_FORWARD_FLAGS = {
|
|
3451
|
+
'max-parallel': 'value',
|
|
3452
|
+
'job-timeout': 'value',
|
|
3453
|
+
'lock-grace': 'value',
|
|
3454
|
+
'poll-timeout': 'value',
|
|
3455
|
+
sandbox: 'value',
|
|
3456
|
+
image: 'value',
|
|
3457
|
+
'secret-resolver': 'value',
|
|
3458
|
+
'reap-age': 'value',
|
|
3459
|
+
'reap-interval': 'value',
|
|
3460
|
+
'min-free-mb': 'value',
|
|
3461
|
+
'clone-timeout': 'value',
|
|
3462
|
+
'keep-runs': 'boolean',
|
|
3463
|
+
stream: 'boolean',
|
|
3464
|
+
arg: 'list',
|
|
3465
|
+
env: 'list',
|
|
3466
|
+
'job-type': 'list',
|
|
3467
|
+
};
|
|
3468
|
+
|
|
3469
|
+
/**
|
|
3470
|
+
* Reconstruct the `work` argv tail from a parsed flags object, so `supervisor
|
|
3471
|
+
* add <profile> [work flags]` forwards those flags to the spawned child. Pure.
|
|
3472
|
+
*/
|
|
3473
|
+
function reconstructWorkArgs(flags) {
|
|
3474
|
+
const out = [];
|
|
3475
|
+
if (!flags || typeof flags !== 'object') return out;
|
|
3476
|
+
for (const [name, kind] of Object.entries(WORK_FORWARD_FLAGS)) {
|
|
3477
|
+
const v = flags[name];
|
|
3478
|
+
if (v === undefined || v === null) continue;
|
|
3479
|
+
if (kind === 'boolean') {
|
|
3480
|
+
if (v === true || v === 'true') out.push(`--${name}`);
|
|
3481
|
+
} else if (kind === 'list') {
|
|
3482
|
+
const items = Array.isArray(v) ? v : [v];
|
|
3483
|
+
for (const item of items) {
|
|
3484
|
+
if (item === undefined || item === null) continue;
|
|
3485
|
+
out.push(`--${name}`, String(item));
|
|
3486
|
+
}
|
|
3487
|
+
} else if (v !== '') {
|
|
3488
|
+
out.push(`--${name}`, String(v));
|
|
3489
|
+
}
|
|
3490
|
+
}
|
|
3491
|
+
return out;
|
|
3492
|
+
}
|
|
3493
|
+
|
|
3494
|
+
/** Assign a unique, stable worker id from a profile name (pure). */
|
|
3495
|
+
function supervisorWorkerId(profile, taken) {
|
|
3496
|
+
const base = String(profile || '').trim() || 'worker';
|
|
3497
|
+
const set = taken instanceof Set ? taken : new Set(taken || []);
|
|
3498
|
+
if (!set.has(base)) return base;
|
|
3499
|
+
for (let i = 2; ; i++) {
|
|
3500
|
+
const candidate = `${base}#${i}`;
|
|
3501
|
+
if (!set.has(candidate)) return candidate;
|
|
3502
|
+
}
|
|
3503
|
+
}
|
|
3504
|
+
|
|
3505
|
+
/**
|
|
3506
|
+
* Redact sensitive values from a reconstructed `work` argv before logging, so
|
|
3507
|
+
* supervisor logs never capture secrets. Both `--env NAME=VALUE` and the
|
|
3508
|
+
* inline `--env=NAME=VALUE` form become `NAME=***` (the value passed to
|
|
3509
|
+
* `nano work` is untouched). Pure.
|
|
3510
|
+
*/
|
|
3511
|
+
function redactWorkArgs(args) {
|
|
3512
|
+
const out = [];
|
|
3513
|
+
const list = Array.isArray(args) ? args : [];
|
|
3514
|
+
const redactPair = (pair) => {
|
|
3515
|
+
const eq = pair.indexOf('=');
|
|
3516
|
+
return eq === -1 ? '***' : `${pair.slice(0, eq)}=***`;
|
|
3517
|
+
};
|
|
3518
|
+
for (let i = 0; i < list.length; i++) {
|
|
3519
|
+
const tok = String(list[i]);
|
|
3520
|
+
if (tok === '--env' && i + 1 < list.length) {
|
|
3521
|
+
out.push(tok, redactPair(String(list[i + 1])));
|
|
3522
|
+
i++;
|
|
3523
|
+
} else if (tok.startsWith('--env=')) {
|
|
3524
|
+
out.push(`--env=${redactPair(tok.slice('--env='.length))}`);
|
|
3525
|
+
} else {
|
|
3526
|
+
out.push(tok);
|
|
3527
|
+
}
|
|
3528
|
+
}
|
|
3529
|
+
return out;
|
|
3530
|
+
}
|
|
3531
|
+
|
|
3532
|
+
/** Capped exponential restart backoff for a crash-looping child (pure). */
|
|
3533
|
+
function supervisorBackoffMs(restarts, base = SUPERVISOR_BACKOFF_BASE_MS, max = SUPERVISOR_BACKOFF_MAX_MS) {
|
|
3534
|
+
const n = Math.max(0, Number(restarts) || 0);
|
|
3535
|
+
return Math.min(max, base * 2 ** Math.min(n, 20));
|
|
3536
|
+
}
|
|
3537
|
+
|
|
3538
|
+
/** Newline-delimited JSON framing for the control socket (pure). */
|
|
3539
|
+
function encodeFrame(obj) {
|
|
3540
|
+
return JSON.stringify(obj) + '\n';
|
|
3541
|
+
}
|
|
3542
|
+
|
|
3543
|
+
/** Split a buffered string into complete JSON frames + a remainder (pure). */
|
|
3544
|
+
function decodeFrames(buffer) {
|
|
3545
|
+
const frames = [];
|
|
3546
|
+
let rest = String(buffer ?? '');
|
|
3547
|
+
let idx;
|
|
3548
|
+
while ((idx = rest.indexOf('\n')) >= 0) {
|
|
3549
|
+
const line = rest.slice(0, idx).trim();
|
|
3550
|
+
rest = rest.slice(idx + 1);
|
|
3551
|
+
if (!line) continue;
|
|
3552
|
+
try { frames.push(JSON.parse(line)); } catch { /* skip malformed frame */ }
|
|
3553
|
+
}
|
|
3554
|
+
return { frames, rest };
|
|
3555
|
+
}
|
|
3556
|
+
|
|
3557
|
+
/** Humanise a millisecond duration compactly (pure). */
|
|
3558
|
+
function formatDuration(ms) {
|
|
3559
|
+
const s = Math.floor((Number(ms) || 0) / 1000);
|
|
3560
|
+
if (s < 60) return `${s}s`;
|
|
3561
|
+
const m = Math.floor(s / 60);
|
|
3562
|
+
if (m < 60) return `${m}m${s % 60}s`;
|
|
3563
|
+
const h = Math.floor(m / 60);
|
|
3564
|
+
if (h < 24) return `${h}h${m % 60}m`;
|
|
3565
|
+
const d = Math.floor(h / 24);
|
|
3566
|
+
return `${d}d${h % 24}h`;
|
|
3567
|
+
}
|
|
3568
|
+
|
|
3569
|
+
/** Project a live/stored worker record to a status row (pure w.r.t. `now`). */
|
|
3570
|
+
function summarizeSupervisorWorker(w, now = Date.now()) {
|
|
3571
|
+
const alive = isPidAlive(w.pid);
|
|
3572
|
+
const uptimeMs = alive && w.startedAt ? Math.max(0, now - new Date(w.startedAt).getTime()) : 0;
|
|
3573
|
+
return {
|
|
3574
|
+
id: w.id,
|
|
3575
|
+
profile: w.profile,
|
|
3576
|
+
pid: alive ? w.pid : null,
|
|
3577
|
+
state: w.stopping ? 'stopping' : alive ? 'running' : 'down',
|
|
3578
|
+
restarts: Number(w.restarts) || 0,
|
|
3579
|
+
uptimeMs,
|
|
3580
|
+
lastExit: w.lastExit ?? null,
|
|
3581
|
+
args: Array.isArray(w.args) ? w.args : [],
|
|
3582
|
+
};
|
|
3583
|
+
}
|
|
3584
|
+
|
|
3585
|
+
/** Render a supervisor status object as an aligned text table. */
|
|
3586
|
+
function formatSupervisorStatus(status) {
|
|
3587
|
+
const lines = [];
|
|
3588
|
+
const d = status.daemon || {};
|
|
3589
|
+
const alive = d.pid ? isPidAlive(d.pid) : false;
|
|
3590
|
+
lines.push('Supervisor:');
|
|
3591
|
+
lines.push(` daemon pid: ${d.pid ?? '-'} ${alive ? '(alive)' : '(dead — stale state)'}`);
|
|
3592
|
+
if (d.startedAt) lines.push(` started: ${d.startedAt}`);
|
|
3593
|
+
if (d.socket) lines.push(` control: ${d.socket}`);
|
|
3594
|
+
const workers = Array.isArray(status.workers) ? status.workers : [];
|
|
3595
|
+
lines.push('');
|
|
3596
|
+
if (workers.length === 0) {
|
|
3597
|
+
lines.push(' No workers. Add one with: c8ctl nano supervisor add <profile>');
|
|
3598
|
+
return lines.join('\n');
|
|
3599
|
+
}
|
|
3600
|
+
const rows = workers.map((w) => ({
|
|
3601
|
+
id: String(w.id),
|
|
3602
|
+
profile: String(w.profile),
|
|
3603
|
+
state: String(w.state),
|
|
3604
|
+
pid: w.pid ? String(w.pid) : '-',
|
|
3605
|
+
restarts: String(w.restarts),
|
|
3606
|
+
uptime: w.state === 'running' ? formatDuration(w.uptimeMs) : '-',
|
|
3607
|
+
last: w.lastExit ? String(w.lastExit) : '-',
|
|
3608
|
+
}));
|
|
3609
|
+
const head = { id: 'ID', profile: 'PROFILE', state: 'STATE', pid: 'PID', restarts: 'RESTARTS', uptime: 'UPTIME', last: 'LAST EXIT' };
|
|
3610
|
+
const cols = ['id', 'profile', 'state', 'pid', 'restarts', 'uptime', 'last'];
|
|
3611
|
+
const width = {};
|
|
3612
|
+
for (const c of cols) width[c] = Math.max(head[c].length, ...rows.map((r) => r[c].length));
|
|
3613
|
+
const fmt = (r) => ' ' + cols.map((c) => r[c].padEnd(width[c])).join(' ');
|
|
3614
|
+
lines.push(fmt(head));
|
|
3615
|
+
for (const r of rows) lines.push(fmt(r));
|
|
3616
|
+
return lines.join('\n');
|
|
3617
|
+
}
|
|
3618
|
+
|
|
3619
|
+
function readSupervisorState() {
|
|
3620
|
+
const file = getSupervisorStateFile();
|
|
3621
|
+
if (!existsSync(file)) return null;
|
|
3622
|
+
try {
|
|
3623
|
+
return JSON.parse(readFileSync(file, 'utf-8'));
|
|
3624
|
+
} catch {
|
|
3625
|
+
return null;
|
|
3626
|
+
}
|
|
3627
|
+
}
|
|
3628
|
+
|
|
3629
|
+
function writeSupervisorState(state) {
|
|
3630
|
+
mkdirSync(getStateHome(), { recursive: true });
|
|
3631
|
+
// Atomic + owner-only: write to a same-dir temp file (mode 0600) then rename
|
|
3632
|
+
// over the target, so a concurrent reader never sees a torn file and the
|
|
3633
|
+
// state (which records worker argv) isn't world-readable.
|
|
3634
|
+
const target = getSupervisorStateFile();
|
|
3635
|
+
const tmp = `${target}.${process.pid}.${Date.now()}.tmp`;
|
|
3636
|
+
writeFileSync(tmp, JSON.stringify(state, null, 2), { mode: 0o600 });
|
|
3637
|
+
try { renameSync(tmp, target); }
|
|
3638
|
+
catch (err) { try { rmSync(tmp, { force: true }); } catch { /* best effort */ } throw err; }
|
|
3639
|
+
}
|
|
3640
|
+
|
|
3641
|
+
function clearSupervisorState() {
|
|
3642
|
+
const file = getSupervisorStateFile();
|
|
3643
|
+
try { if (existsSync(file)) rmSync(file); } catch { /* best effort */ }
|
|
3644
|
+
}
|
|
3645
|
+
|
|
3646
|
+
/** Running daemon state (pid alive) or null. */
|
|
3647
|
+
function runningSupervisor() {
|
|
3648
|
+
const state = readSupervisorState();
|
|
3649
|
+
return state && isPidAlive(state.pid) ? state : null;
|
|
3650
|
+
}
|
|
3651
|
+
|
|
3652
|
+
/** Synthesize a state-file-shaped object from a live `status` response. */
|
|
3653
|
+
function stateFromStatus(res, socketPath) {
|
|
3654
|
+
return {
|
|
3655
|
+
pid: res.daemon?.pid,
|
|
3656
|
+
startedAt: res.daemon?.startedAt,
|
|
3657
|
+
socket: res.daemon?.socket || socketPath,
|
|
3658
|
+
logFile: res.daemon?.logFile,
|
|
3659
|
+
workers: res.workers || [],
|
|
3660
|
+
};
|
|
3661
|
+
}
|
|
3662
|
+
|
|
3663
|
+
/**
|
|
3664
|
+
* Resolve a live supervisor, healing a missing/stale state file. Returns the
|
|
3665
|
+
* running state (pid alive) if present; otherwise probes the deterministic
|
|
3666
|
+
* control socket and, if a daemon answers, re-persists and returns its state so
|
|
3667
|
+
* management commands still work when supervisor.json was deleted/cleaned.
|
|
3668
|
+
* Returns null when nothing is listening.
|
|
3669
|
+
*/
|
|
3670
|
+
async function liveSupervisor() {
|
|
3671
|
+
const running = runningSupervisor();
|
|
3672
|
+
if (running) return running;
|
|
3673
|
+
try {
|
|
3674
|
+
const socketPath = getSupervisorSocketPath();
|
|
3675
|
+
const res = await supervisorRequest({ op: 'status' }, { socketPath, timeoutMs: 500, responseTimeoutMs: SUPERVISOR_PROBE_RESPONSE_TIMEOUT_MS });
|
|
3676
|
+
if (res && res.ok) {
|
|
3677
|
+
const state = stateFromStatus(res, socketPath);
|
|
3678
|
+
try { writeSupervisorState(state); } catch { /* best effort */ }
|
|
3679
|
+
return state;
|
|
3680
|
+
}
|
|
3681
|
+
} catch { /* no live daemon on the socket */ }
|
|
3682
|
+
return null;
|
|
3683
|
+
}
|
|
3684
|
+
|
|
3685
|
+
/** How to re-invoke the c8ctl CLI to spawn the daemon + `work` children. */
|
|
3686
|
+
function c8ctlInvocation() {
|
|
3687
|
+
const entry = process.env.C8CTL_NANO_ENTRY || process.argv[1];
|
|
3688
|
+
return { exec: process.execPath, entry };
|
|
3689
|
+
}
|
|
3690
|
+
|
|
3691
|
+
function supervisorDaemonLogFile() {
|
|
3692
|
+
return join(getSupervisorLogDir(), 'daemon.log');
|
|
3693
|
+
}
|
|
3694
|
+
|
|
3695
|
+
function supervisorWorkerLogFile(id) {
|
|
3696
|
+
return join(getSupervisorLogDir(), `worker-${String(id).replace(/[^\w.#-]/g, '_')}.log`);
|
|
3697
|
+
}
|
|
3698
|
+
|
|
3699
|
+
function waitForChildExit(child, timeoutMs) {
|
|
3700
|
+
return new Promise((resolve) => {
|
|
3701
|
+
if (!child || child.exitCode !== null || child.signalCode !== null) return resolve();
|
|
3702
|
+
let done = false;
|
|
3703
|
+
const finish = () => { if (done) return; done = true; clearTimeout(t); resolve(); };
|
|
3704
|
+
const t = setTimeout(finish, timeoutMs);
|
|
3705
|
+
child.once('exit', finish);
|
|
3706
|
+
});
|
|
3707
|
+
}
|
|
3708
|
+
|
|
3709
|
+
// --- Daemon ----------------------------------------------------------------
|
|
3710
|
+
|
|
3711
|
+
/**
|
|
3712
|
+
* The supervisor daemon body. Runs under `c8ctl nano supervisor __daemon`,
|
|
3713
|
+
* spawned detached by `startSupervisorDaemon`. Never returns — it runs until a
|
|
3714
|
+
* stop request or SIGTERM, then drains children and exits.
|
|
3715
|
+
*/
|
|
3716
|
+
async function runSupervisorDaemon() {
|
|
3717
|
+
const startedAt = new Date().toISOString();
|
|
3718
|
+
const { exec, entry } = c8ctlInvocation();
|
|
3719
|
+
const socketPath = getSupervisorSocketPath();
|
|
3720
|
+
const daemonLogFile = supervisorDaemonLogFile();
|
|
3721
|
+
mkdirSync(getSupervisorLogDir(), { recursive: true });
|
|
3722
|
+
|
|
3723
|
+
const workers = new Map();
|
|
3724
|
+
const attachClients = new Set();
|
|
3725
|
+
let shuttingDown = false;
|
|
3726
|
+
|
|
3727
|
+
// Daemon-wide mutation serialization: `add`/`remove`/`restart` must not
|
|
3728
|
+
// interleave, or two clients racing the same worker could each spawn an
|
|
3729
|
+
// untracked child. Every mutation runs to completion before the next starts.
|
|
3730
|
+
let opQueue = Promise.resolve();
|
|
3731
|
+
const serializeOp = (fn) => {
|
|
3732
|
+
const run = opQueue.then(fn, fn);
|
|
3733
|
+
opQueue = run.then(() => {}, () => {});
|
|
3734
|
+
return run;
|
|
3735
|
+
};
|
|
3736
|
+
|
|
3737
|
+
const dlog = (msg) => {
|
|
3738
|
+
try { appendFileSync(daemonLogFile, `[${new Date().toISOString()}] ${msg}\n`); } catch { /* best effort */ }
|
|
3739
|
+
};
|
|
3740
|
+
|
|
3741
|
+
const workerPublic = (w) => summarizeSupervisorWorker(w);
|
|
3742
|
+
|
|
3743
|
+
const persist = () => {
|
|
3744
|
+
try {
|
|
3745
|
+
writeSupervisorState({
|
|
3746
|
+
pid: process.pid,
|
|
3747
|
+
startedAt,
|
|
3748
|
+
socket: socketPath,
|
|
3749
|
+
logFile: daemonLogFile,
|
|
3750
|
+
workers: [...workers.values()].map((w) => ({
|
|
3751
|
+
id: w.id, profile: w.profile, args: w.args, pid: isPidAlive(w.pid) ? w.pid : null,
|
|
3752
|
+
startedAt: w.startedAt || null, restarts: w.restarts, lastExit: w.lastExit ?? null,
|
|
3753
|
+
stopping: !!w.stopping, logFile: w.logFile,
|
|
3754
|
+
})),
|
|
3755
|
+
});
|
|
3756
|
+
} catch { /* best effort */ }
|
|
3757
|
+
};
|
|
3758
|
+
|
|
3759
|
+
const broadcast = (frame) => {
|
|
3760
|
+
const data = encodeFrame(frame);
|
|
3761
|
+
for (const sock of attachClients) {
|
|
3762
|
+
try { sock.write(data); } catch { /* client gone */ }
|
|
3763
|
+
}
|
|
3764
|
+
};
|
|
3765
|
+
|
|
3766
|
+
const startWorker = (w) => {
|
|
3767
|
+
let fd;
|
|
3768
|
+
try { fd = openSync(w.logFile, 'a'); } catch { fd = 'ignore'; }
|
|
3769
|
+
const child = spawn(exec, [entry, 'nano', 'work', w.profile, ...w.args], {
|
|
3770
|
+
env: process.env,
|
|
3771
|
+
stdio: ['ignore', fd, fd],
|
|
3772
|
+
});
|
|
3773
|
+
if (typeof fd === 'number') { try { closeSync(fd); } catch { /* dup'd into child */ } }
|
|
3774
|
+
w.child = child;
|
|
3775
|
+
w.pid = child.pid || null;
|
|
3776
|
+
w.startedAt = new Date().toISOString();
|
|
3777
|
+
w.spawnedAt = Date.now();
|
|
3778
|
+
dlog(`worker '${w.id}' (profile ${w.profile}) started pid ${w.pid}: work ${[w.profile, ...redactWorkArgs(w.args)].join(' ')}`);
|
|
3779
|
+
broadcast({ type: 'event', event: 'worker-start', worker: workerPublic(w) });
|
|
3780
|
+
|
|
3781
|
+
// A spawn failure (ENOENT/EMFILE/…) emits only 'error' with no 'exit', so
|
|
3782
|
+
// both paths funnel through one death handler that schedules a restart.
|
|
3783
|
+
// `settled` guards the error+exit double-fire; the `w.child !== child` check
|
|
3784
|
+
// ignores a stale child's late exit after `restart` swapped in a new one
|
|
3785
|
+
// (which would otherwise clobber the live pid and leak a duplicate worker).
|
|
3786
|
+
let settled = false;
|
|
3787
|
+
const handleDeath = (reason) => {
|
|
3788
|
+
if (w.child !== child || settled) return;
|
|
3789
|
+
settled = true;
|
|
3790
|
+
w.pid = null;
|
|
3791
|
+
w.lastExit = reason;
|
|
3792
|
+
const ranMs = Date.now() - (w.spawnedAt || Date.now());
|
|
3793
|
+
if (ranMs >= SUPERVISOR_HEALTHY_UPTIME_MS) w.restarts = 0;
|
|
3794
|
+
if (w.stopping || shuttingDown || !workers.has(w.id)) { persist(); return; }
|
|
3795
|
+
const delay = supervisorBackoffMs(w.restarts);
|
|
3796
|
+
w.restarts += 1;
|
|
3797
|
+
dlog(`worker '${w.id}' down (${reason}); restarting in ${delay}ms (restart #${w.restarts})`);
|
|
3798
|
+
broadcast({ type: 'event', event: 'worker-exit', worker: workerPublic(w), restartInMs: delay });
|
|
3799
|
+
w.restartTimer = setTimeout(() => {
|
|
3800
|
+
w.restartTimer = null;
|
|
3801
|
+
if (!w.stopping && !shuttingDown && workers.has(w.id)) startWorker(w);
|
|
3802
|
+
}, delay);
|
|
3803
|
+
if (typeof w.restartTimer.unref === 'function') w.restartTimer.unref();
|
|
3804
|
+
persist();
|
|
3805
|
+
};
|
|
3806
|
+
child.on('error', (err) => handleDeath(`spawn error: ${err.message}`));
|
|
3807
|
+
child.on('exit', (code, signal) => handleDeath(signal ? `signal ${signal}` : `code ${code}`));
|
|
3808
|
+
persist();
|
|
3809
|
+
};
|
|
3810
|
+
|
|
3811
|
+
const addWorker = (profile, args) => {
|
|
3812
|
+
const id = supervisorWorkerId(profile, new Set(workers.keys()));
|
|
3813
|
+
const w = {
|
|
3814
|
+
id, profile: String(profile), args: Array.isArray(args) ? args.map(String) : [],
|
|
3815
|
+
restarts: 0, stopping: false, lastExit: null, logFile: supervisorWorkerLogFile(id),
|
|
3816
|
+
};
|
|
3817
|
+
workers.set(id, w);
|
|
3818
|
+
startWorker(w);
|
|
3819
|
+
return w;
|
|
3820
|
+
};
|
|
3821
|
+
|
|
3822
|
+
const stopWorker = async (id) => {
|
|
3823
|
+
const w = workers.get(id);
|
|
3824
|
+
if (!w) return false;
|
|
3825
|
+
w.stopping = true;
|
|
3826
|
+
if (w.restartTimer) { clearTimeout(w.restartTimer); w.restartTimer = null; }
|
|
3827
|
+
const pid = w.pid;
|
|
3828
|
+
if (w.child && pid) {
|
|
3829
|
+
try { process.kill(pid, 'SIGTERM'); } catch { /* already gone */ }
|
|
3830
|
+
await waitForChildExit(w.child, STOP_GRACE_MS);
|
|
3831
|
+
if (isPidAlive(pid)) { try { process.kill(pid, 'SIGKILL'); } catch { /* ignore */ } }
|
|
3832
|
+
}
|
|
3833
|
+
return true;
|
|
3834
|
+
};
|
|
3835
|
+
|
|
3836
|
+
const removeWorker = async (id) => {
|
|
3837
|
+
if (!workers.has(id)) return false;
|
|
3838
|
+
await stopWorker(id);
|
|
3839
|
+
workers.delete(id);
|
|
3840
|
+
dlog(`worker '${id}' removed`);
|
|
3841
|
+
broadcast({ type: 'event', event: 'worker-remove', id });
|
|
3842
|
+
persist();
|
|
3843
|
+
return true;
|
|
3844
|
+
};
|
|
3845
|
+
|
|
3846
|
+
const restartWorker = async (id) => {
|
|
3847
|
+
const w = workers.get(id);
|
|
3848
|
+
if (!w) return false;
|
|
3849
|
+
await stopWorker(id);
|
|
3850
|
+
w.stopping = false;
|
|
3851
|
+
w.restarts = 0;
|
|
3852
|
+
startWorker(w);
|
|
3853
|
+
dlog(`worker '${id}' restarted`);
|
|
3854
|
+
return true;
|
|
3855
|
+
};
|
|
3856
|
+
|
|
3857
|
+
// Resolve a target token to worker ids: exact id, else all with that profile.
|
|
3858
|
+
const resolveTargets = (target) => {
|
|
3859
|
+
const t = String(target || '').trim();
|
|
3860
|
+
if (!t) return [];
|
|
3861
|
+
if (t === 'all' || t === '*') return [...workers.keys()];
|
|
3862
|
+
if (workers.has(t)) return [t];
|
|
3863
|
+
return [...workers.values()].filter((w) => w.profile === t).map((w) => w.id);
|
|
3864
|
+
};
|
|
3865
|
+
|
|
3866
|
+
const statusFrame = (final) => ({
|
|
3867
|
+
ok: true,
|
|
3868
|
+
type: 'status',
|
|
3869
|
+
daemon: { pid: process.pid, startedAt, socket: socketPath, logFile: daemonLogFile },
|
|
3870
|
+
workers: [...workers.values()].map(workerPublic),
|
|
3871
|
+
...(final ? { final: true } : {}),
|
|
3872
|
+
});
|
|
3873
|
+
|
|
3874
|
+
const shutdown = async (signal) => {
|
|
3875
|
+
if (shuttingDown) return;
|
|
3876
|
+
shuttingDown = true;
|
|
3877
|
+
// Let any in-flight mutation finish before we snapshot the worker set, so
|
|
3878
|
+
// an add/restart racing the shutdown can't leave an orphaned child behind.
|
|
3879
|
+
try { await opQueue; } catch { /* mutation already logged */ }
|
|
3880
|
+
dlog(`received ${signal || 'stop'} — stopping ${workers.size} worker(s)`);
|
|
3881
|
+
await Promise.all([...workers.keys()].map((id) => stopWorker(id)));
|
|
3882
|
+
broadcast({ type: 'event', event: 'daemon-stop' });
|
|
3883
|
+
try { server.close(); } catch { /* ignore */ }
|
|
3884
|
+
if (osPlatform() !== 'win32') { try { rmSync(socketPath, { force: true }); } catch { /* ignore */ } }
|
|
3885
|
+
clearSupervisorState();
|
|
3886
|
+
process.exit(0);
|
|
3887
|
+
};
|
|
3888
|
+
|
|
3889
|
+
const handleRequest = async (req, sock) => {
|
|
3890
|
+
const op = req && req.op;
|
|
3891
|
+
try {
|
|
3892
|
+
switch (op) {
|
|
3893
|
+
case 'status':
|
|
3894
|
+
sock.write(encodeFrame(statusFrame(true)));
|
|
3895
|
+
break;
|
|
3896
|
+
case 'add': {
|
|
3897
|
+
if (shuttingDown) { sock.write(encodeFrame({ ok: false, error: 'supervisor is shutting down', final: true })); break; }
|
|
3898
|
+
if (!req.profile) { sock.write(encodeFrame({ ok: false, error: 'add requires a profile', final: true })); break; }
|
|
3899
|
+
// `--name` selects a DIFFERENT hire inside `nano work`, so a worker
|
|
3900
|
+
// added as profile X but carrying `--name Y` would run Y while status
|
|
3901
|
+
// and logs report X. Reject it — the supervisor id derives from the
|
|
3902
|
+
// positional profile and that must be what actually runs.
|
|
3903
|
+
// `req.args` comes from untrusted JSON and may be non-array (e.g. a
|
|
3904
|
+
// string or object). Coerce to an array of string tokens before
|
|
3905
|
+
// scanning/forwarding so a malformed payload yields a clean rejection
|
|
3906
|
+
// instead of throwing a generic request error.
|
|
3907
|
+
const args = Array.isArray(req.args) ? req.args.filter((a) => typeof a === 'string') : [];
|
|
3908
|
+
const badName = args.find((a) => a === '--name' || a === '-n' || /^--name=/.test(a) || /^-n=/.test(a));
|
|
3909
|
+
if (badName) { sock.write(encodeFrame({ ok: false, error: `--name is not allowed for a supervised worker (it would run a different hire than the reported profile "${req.profile}")`, final: true })); break; }
|
|
3910
|
+
const stored = readHires()[String(req.profile)];
|
|
3911
|
+
if (!stored) { sock.write(encodeFrame({ ok: false, error: `no hire named "${req.profile}"`, final: true })); break; }
|
|
3912
|
+
const w = await serializeOp(() => addWorker(req.profile, args));
|
|
3913
|
+
sock.write(encodeFrame({ ok: true, type: 'added', worker: workerPublic(w), final: true }));
|
|
3914
|
+
break;
|
|
3915
|
+
}
|
|
3916
|
+
case 'remove': {
|
|
3917
|
+
if (shuttingDown) { sock.write(encodeFrame({ ok: false, error: 'supervisor is shutting down', final: true })); break; }
|
|
3918
|
+
const removed = await serializeOp(async () => {
|
|
3919
|
+
const ids = resolveTargets(req.target);
|
|
3920
|
+
for (const id of ids) await removeWorker(id);
|
|
3921
|
+
return ids;
|
|
3922
|
+
});
|
|
3923
|
+
sock.write(encodeFrame({ ok: true, type: 'removed', removed, final: true }));
|
|
3924
|
+
break;
|
|
3925
|
+
}
|
|
3926
|
+
case 'restart': {
|
|
3927
|
+
if (shuttingDown) { sock.write(encodeFrame({ ok: false, error: 'supervisor is shutting down', final: true })); break; }
|
|
3928
|
+
const restarted = await serializeOp(async () => {
|
|
3929
|
+
const ids = resolveTargets(req.target);
|
|
3930
|
+
for (const id of ids) await restartWorker(id);
|
|
3931
|
+
return ids;
|
|
3932
|
+
});
|
|
3933
|
+
sock.write(encodeFrame({ ok: true, type: 'restarted', restarted, final: true }));
|
|
3934
|
+
break;
|
|
3935
|
+
}
|
|
3936
|
+
case 'attach':
|
|
3937
|
+
attachClients.add(sock);
|
|
3938
|
+
sock.write(encodeFrame(statusFrame(false)));
|
|
3939
|
+
break;
|
|
3940
|
+
case 'stop':
|
|
3941
|
+
sock.write(encodeFrame({ ok: true, type: 'stopping', final: true }));
|
|
3942
|
+
setTimeout(() => shutdown('stop'), 50);
|
|
3943
|
+
break;
|
|
3944
|
+
default:
|
|
3945
|
+
sock.write(encodeFrame({ ok: false, error: `unknown op "${op}"`, final: true }));
|
|
3946
|
+
}
|
|
3947
|
+
} catch (err) {
|
|
3948
|
+
try { sock.write(encodeFrame({ ok: false, error: String(err && err.message || err), final: true })); } catch { /* ignore */ }
|
|
3949
|
+
}
|
|
3950
|
+
};
|
|
3951
|
+
|
|
3952
|
+
// Bind the control socket. A stale unix socket file from a crashed daemon
|
|
3953
|
+
// would make listen() fail with EADDRINUSE even though nobody is listening;
|
|
3954
|
+
// remove it first (we already know no live daemon owns our state).
|
|
3955
|
+
if (osPlatform() !== 'win32') { try { rmSync(socketPath, { force: true }); } catch { /* ignore */ } }
|
|
3956
|
+
|
|
3957
|
+
const server = createServer((sock) => {
|
|
3958
|
+
sock.setEncoding('utf8');
|
|
3959
|
+
let buf = '';
|
|
3960
|
+
// Serialize requests per connection: handleRequest is async and mutates the
|
|
3961
|
+
// shared workers map, so a second 'data' event arriving mid-await must not
|
|
3962
|
+
// interleave add/remove/restart. Chain each frame onto a per-socket queue.
|
|
3963
|
+
let queue = Promise.resolve();
|
|
3964
|
+
sock.on('data', (chunk) => {
|
|
3965
|
+
buf += chunk;
|
|
3966
|
+
// Cap by UTF-8 byte length, not string length: buf is a decoded string
|
|
3967
|
+
// whose .length counts UTF-16 code units, so multibyte input could hold
|
|
3968
|
+
// far more than SUPERVISOR_MAX_FRAME_BYTES in memory before being dropped.
|
|
3969
|
+
if (Buffer.byteLength(buf, 'utf8') > SUPERVISOR_MAX_FRAME_BYTES) {
|
|
3970
|
+
dlog(`control connection exceeded ${SUPERVISOR_MAX_FRAME_BYTES} bytes without a complete frame — dropping`);
|
|
3971
|
+
try { sock.destroy(); } catch { /* ignore */ }
|
|
3972
|
+
buf = '';
|
|
3973
|
+
return;
|
|
3974
|
+
}
|
|
3975
|
+
const { frames, rest } = decodeFrames(buf);
|
|
3976
|
+
buf = rest;
|
|
3977
|
+
for (const req of frames) {
|
|
3978
|
+
queue = queue.then(() => handleRequest(req, sock)).catch((err) => dlog(`request error: ${err?.message || err}`));
|
|
3979
|
+
}
|
|
3980
|
+
});
|
|
3981
|
+
sock.on('close', () => attachClients.delete(sock));
|
|
3982
|
+
sock.on('error', () => attachClients.delete(sock));
|
|
3983
|
+
});
|
|
3984
|
+
|
|
3985
|
+
// Create the control socket owner-only from the start. The socket file lives
|
|
3986
|
+
// in shared tmpdir(); libuv binds it synchronously inside listen(), so a
|
|
3987
|
+
// restrictive umask around that call closes the TOCTOU window where another
|
|
3988
|
+
// local user could connect before the chmod below lands. Restore the previous
|
|
3989
|
+
// umask immediately after — the listen() bind is synchronous, so no unrelated
|
|
3990
|
+
// file creation can interleave. Unix only; on Windows umask/mode are no-ops.
|
|
3991
|
+
const isWin = osPlatform() === 'win32';
|
|
3992
|
+
const prevUmask = isWin ? null : process.umask(0o177);
|
|
3993
|
+
try {
|
|
3994
|
+
await new Promise((resolve, reject) => {
|
|
3995
|
+
server.once('error', reject);
|
|
3996
|
+
server.listen(socketPath, resolve);
|
|
3997
|
+
});
|
|
3998
|
+
} catch (err) {
|
|
3999
|
+
dlog(`failed to bind control socket ${socketPath}: ${err.message}`);
|
|
4000
|
+
process.exit(1);
|
|
4001
|
+
} finally {
|
|
4002
|
+
if (!isWin) { try { process.umask(prevUmask); } catch { /* ignore */ } }
|
|
4003
|
+
}
|
|
4004
|
+
|
|
4005
|
+
// Lock the control socket to the owner so another local user can't drive the
|
|
4006
|
+
// supervisor (stop/add/remove). Unix only — Windows named pipes are secured
|
|
4007
|
+
// by their own ACLs, not filesystem mode bits. This chmod is now a backstop
|
|
4008
|
+
// for the owner-only umask applied around listen() above.
|
|
4009
|
+
if (!isWin) {
|
|
4010
|
+
try { chmodSync(socketPath, 0o600); } catch (err) { dlog(`could not chmod control socket: ${err.message}`); }
|
|
4011
|
+
}
|
|
4012
|
+
|
|
4013
|
+
process.once('SIGTERM', () => shutdown('SIGTERM'));
|
|
4014
|
+
process.once('SIGINT', () => shutdown('SIGINT'));
|
|
4015
|
+
dlog(`supervisor daemon up (pid ${process.pid}) — control ${socketPath}`);
|
|
4016
|
+
persist();
|
|
4017
|
+
|
|
4018
|
+
// Keep the event loop alive indefinitely; the server holds it, but add an
|
|
4019
|
+
// explicit never-resolving guard so a transient server close can't exit us.
|
|
4020
|
+
await new Promise(() => {});
|
|
4021
|
+
}
|
|
4022
|
+
|
|
4023
|
+
// --- Client (management subcommands + attach) ------------------------------
|
|
4024
|
+
|
|
4025
|
+
/** Connect to the control socket, resolving with the socket once connected. */
|
|
4026
|
+
function supervisorConnect(socketPath, { timeoutMs = SUPERVISOR_CONNECT_TIMEOUT_MS } = {}) {
|
|
4027
|
+
return new Promise((resolve, reject) => {
|
|
4028
|
+
const sock = createConnection(socketPath);
|
|
4029
|
+
let settled = false;
|
|
4030
|
+
const timer = setTimeout(() => {
|
|
4031
|
+
if (settled) return;
|
|
4032
|
+
settled = true;
|
|
4033
|
+
sock.destroy();
|
|
4034
|
+
reject(new Error(`timed out connecting to supervisor at ${socketPath}`));
|
|
4035
|
+
}, timeoutMs);
|
|
4036
|
+
sock.once('connect', () => {
|
|
4037
|
+
if (settled) return;
|
|
4038
|
+
settled = true;
|
|
4039
|
+
clearTimeout(timer);
|
|
4040
|
+
sock.setEncoding('utf8');
|
|
4041
|
+
resolve(sock);
|
|
4042
|
+
});
|
|
4043
|
+
sock.once('error', (err) => {
|
|
4044
|
+
if (settled) return;
|
|
4045
|
+
settled = true;
|
|
4046
|
+
clearTimeout(timer);
|
|
4047
|
+
reject(err);
|
|
4048
|
+
});
|
|
4049
|
+
});
|
|
4050
|
+
}
|
|
4051
|
+
|
|
4052
|
+
/** Send one request and collect frames until a `final:true` frame arrives. */
|
|
4053
|
+
function supervisorRequest(req, { socketPath, timeoutMs, responseTimeoutMs = SUPERVISOR_RESPONSE_TIMEOUT_MS } = {}) {
|
|
4054
|
+
const path = socketPath || (readSupervisorState()?.socket) || getSupervisorSocketPath();
|
|
4055
|
+
return new Promise((resolve, reject) => {
|
|
4056
|
+
supervisorConnect(path, { timeoutMs }).then((sock) => {
|
|
4057
|
+
let buf = '';
|
|
4058
|
+
let settled = false;
|
|
4059
|
+
const finish = (fn, arg) => { if (settled) return; settled = true; clearTimeout(timer); try { sock.end(); } catch { /* ignore */ } fn(arg); };
|
|
4060
|
+
const done = (result) => finish(resolve, result);
|
|
4061
|
+
const fail = (err) => finish(reject, err);
|
|
4062
|
+
// End-to-end response deadline: a daemon that accepts the connection but
|
|
4063
|
+
// never sends a `final` frame must not hang the caller forever.
|
|
4064
|
+
const timer = setTimeout(() => {
|
|
4065
|
+
if (settled) return;
|
|
4066
|
+
settled = true;
|
|
4067
|
+
try { sock.destroy(); } catch { /* ignore */ }
|
|
4068
|
+
reject(new Error(`timed out waiting for supervisor response from ${path}`));
|
|
4069
|
+
}, responseTimeoutMs);
|
|
4070
|
+
sock.on('data', (chunk) => {
|
|
4071
|
+
buf += chunk;
|
|
4072
|
+
const { frames, rest } = decodeFrames(buf);
|
|
4073
|
+
buf = rest;
|
|
4074
|
+
for (const frame of frames) {
|
|
4075
|
+
if (frame.final) return done(frame);
|
|
4076
|
+
}
|
|
4077
|
+
});
|
|
4078
|
+
sock.on('error', fail);
|
|
4079
|
+
sock.on('close', () => done({ ok: false, error: 'connection closed before response' }));
|
|
4080
|
+
sock.write(encodeFrame(req));
|
|
4081
|
+
}).catch(reject);
|
|
4082
|
+
});
|
|
4083
|
+
}
|
|
4084
|
+
|
|
4085
|
+
/**
|
|
4086
|
+
* Ensure a daemon is running, spawning it detached if not, and return its
|
|
4087
|
+
* running state. Polls the control socket until it answers a status request.
|
|
4088
|
+
*/
|
|
4089
|
+
async function startSupervisorDaemon() {
|
|
4090
|
+
const existing = runningSupervisor();
|
|
4091
|
+
if (existing) return existing;
|
|
4092
|
+
|
|
4093
|
+
const socketPath = getSupervisorSocketPath();
|
|
4094
|
+
// The state file may be missing (deleted, cleaned up, or not yet written)
|
|
4095
|
+
// while a daemon is still listening on the deterministic socket. Adopt that
|
|
4096
|
+
// live daemon instead of spawning a second one that would orphan the
|
|
4097
|
+
// original and its workers.
|
|
4098
|
+
try {
|
|
4099
|
+
const res = await supervisorRequest({ op: 'status' }, { socketPath, timeoutMs: 500, responseTimeoutMs: SUPERVISOR_PROBE_RESPONSE_TIMEOUT_MS });
|
|
4100
|
+
if (res && res.ok) {
|
|
4101
|
+
// Re-persist the adopted daemon's state so subsequent pid-based checks
|
|
4102
|
+
// (runningSupervisor()) work immediately, instead of staying broken until
|
|
4103
|
+
// some later command happens to heal supervisor.json.
|
|
4104
|
+
const adopted = runningSupervisor() || stateFromStatus(res, socketPath);
|
|
4105
|
+
try { writeSupervisorState(adopted); } catch { /* best effort */ }
|
|
4106
|
+
return adopted;
|
|
4107
|
+
}
|
|
4108
|
+
} catch { /* no live daemon on the socket — safe to (re)spawn */ }
|
|
4109
|
+
|
|
4110
|
+
clearSupervisorState(); // clear any stale marker from a dead daemon
|
|
4111
|
+
|
|
4112
|
+
const { exec, entry } = c8ctlInvocation();
|
|
4113
|
+
mkdirSync(getSupervisorLogDir(), { recursive: true });
|
|
4114
|
+
const logFile = supervisorDaemonLogFile();
|
|
4115
|
+
let fd;
|
|
4116
|
+
try { fd = openSync(logFile, 'a'); } catch { fd = 'ignore'; }
|
|
4117
|
+
const child = spawn(exec, [entry, 'nano', 'supervisor', '__daemon'], {
|
|
4118
|
+
env: process.env,
|
|
4119
|
+
detached: true,
|
|
4120
|
+
stdio: ['ignore', fd, fd],
|
|
4121
|
+
});
|
|
4122
|
+
child.unref();
|
|
4123
|
+
if (typeof fd === 'number') { try { closeSync(fd); } catch { /* ignore */ } }
|
|
4124
|
+
if (typeof child.pid !== 'number') throw new Error('failed to spawn supervisor daemon');
|
|
4125
|
+
|
|
4126
|
+
|
|
4127
|
+
const deadline = Date.now() + SUPERVISOR_CONNECT_TIMEOUT_MS;
|
|
4128
|
+
while (Date.now() < deadline) {
|
|
4129
|
+
try {
|
|
4130
|
+
const res = await supervisorRequest({ op: 'status' }, { socketPath, timeoutMs: 750, responseTimeoutMs: SUPERVISOR_PROBE_RESPONSE_TIMEOUT_MS });
|
|
4131
|
+
if (res && res.ok) {
|
|
4132
|
+
// The daemon can answer `status` on the socket a beat before it has
|
|
4133
|
+
// written supervisor.json. Fall back to the live status response so
|
|
4134
|
+
// callers always get a state object with a usable pid.
|
|
4135
|
+
return runningSupervisor() || readSupervisorState() || stateFromStatus(res, socketPath);
|
|
4136
|
+
}
|
|
4137
|
+
} catch { /* not up yet */ }
|
|
4138
|
+
await new Promise((r) => setTimeout(r, 150));
|
|
4139
|
+
}
|
|
4140
|
+
throw new Error(`supervisor daemon did not become ready (see ${logFile})`);
|
|
4141
|
+
}
|
|
4142
|
+
|
|
4143
|
+
async function supervisorStartCmd(req, flags) {
|
|
4144
|
+
const logger = getLogger();
|
|
4145
|
+
const state = await startSupervisorDaemon();
|
|
4146
|
+
logger.info(`Supervisor daemon running (pid ${state.pid}).`);
|
|
4147
|
+
|
|
4148
|
+
const specs = normalizeArgList(flags?.worker);
|
|
4149
|
+
const workArgs = reconstructWorkArgs(flags);
|
|
4150
|
+
for (const profile of specs) {
|
|
4151
|
+
const res = await supervisorRequest({ op: 'add', profile, args: workArgs });
|
|
4152
|
+
if (res.ok) logger.info(` + worker "${res.worker.id}" (profile ${profile})`);
|
|
4153
|
+
else logger.error(` ! could not add "${profile}": ${res.error}`);
|
|
4154
|
+
}
|
|
4155
|
+
|
|
4156
|
+
if (coerceBool(flags?.attach, false)) {
|
|
4157
|
+
await attachSupervisorConsole(runningSupervisor() || state);
|
|
4158
|
+
return;
|
|
4159
|
+
}
|
|
4160
|
+
await supervisorStatusCmd();
|
|
4161
|
+
logger.info('');
|
|
4162
|
+
logger.info('Attach an interactive console with: c8ctl nano supervisor');
|
|
4163
|
+
logger.info('Manage without it: c8ctl nano supervisor add|remove|restart|status|stop');
|
|
4164
|
+
}
|
|
4165
|
+
|
|
4166
|
+
async function supervisorStatusCmd() {
|
|
4167
|
+
const logger = getLogger();
|
|
4168
|
+
const running = runningSupervisor();
|
|
4169
|
+
if (!running) {
|
|
4170
|
+
// The state file may be missing (deleted/cleaned) while a daemon is still
|
|
4171
|
+
// listening on the deterministic socket — same case startSupervisorDaemon
|
|
4172
|
+
// adopts. Probe it before declaring the supervisor down, and re-persist so
|
|
4173
|
+
// the state file is healed for later pid-based checks.
|
|
4174
|
+
try {
|
|
4175
|
+
const res = await supervisorRequest({ op: 'status' }, { socketPath: getSupervisorSocketPath(), timeoutMs: 500, responseTimeoutMs: SUPERVISOR_PROBE_RESPONSE_TIMEOUT_MS });
|
|
4176
|
+
if (res && res.ok) {
|
|
4177
|
+
try { writeSupervisorState(stateFromStatus(res, getSupervisorSocketPath())); } catch { /* best effort */ }
|
|
4178
|
+
logger.info(formatSupervisorStatus(res));
|
|
4179
|
+
return;
|
|
4180
|
+
}
|
|
4181
|
+
} catch { /* no live daemon on the socket — genuinely down */ }
|
|
4182
|
+
const stale = readSupervisorState();
|
|
4183
|
+
if (stale) {
|
|
4184
|
+
logger.info('Supervisor: not running (stale state — daemon pid is dead).');
|
|
4185
|
+
logger.info(' Start it with: c8ctl nano supervisor start');
|
|
4186
|
+
} else {
|
|
4187
|
+
logger.info('Supervisor: not running.');
|
|
4188
|
+
logger.info(' Start it with: c8ctl nano supervisor start (or attach: c8ctl nano supervisor)');
|
|
4189
|
+
}
|
|
4190
|
+
return;
|
|
4191
|
+
}
|
|
4192
|
+
try {
|
|
4193
|
+
const res = await supervisorRequest({ op: 'status' });
|
|
4194
|
+
if (res.ok) { logger.info(formatSupervisorStatus(res)); return; }
|
|
4195
|
+
} catch { /* fall back to state file below */ }
|
|
4196
|
+
// Socket unreachable but pid alive — render from the last persisted state.
|
|
4197
|
+
logger.info(formatSupervisorStatus({
|
|
4198
|
+
daemon: { pid: running.pid, startedAt: running.startedAt, socket: running.socket },
|
|
4199
|
+
workers: (running.workers || []).map((w) => summarizeSupervisorWorker(w)),
|
|
4200
|
+
}));
|
|
4201
|
+
}
|
|
4202
|
+
|
|
4203
|
+
async function supervisorAddCmd(req, flags) {
|
|
4204
|
+
const logger = getLogger();
|
|
4205
|
+
// Use only the positional profile. `--name` is a documented hire/work/assign
|
|
4206
|
+
// flag, so honouring it here would make `supervisor add reviewer --name foo`
|
|
4207
|
+
// surprisingly add `foo` instead of `reviewer`.
|
|
4208
|
+
const profile = req.positional[1];
|
|
4209
|
+
if (!profile) { logger.error('Usage: c8ctl nano supervisor add <profile> [work flags]'); process.exit(1); }
|
|
4210
|
+
await startSupervisorDaemon();
|
|
4211
|
+
const res = await supervisorRequest({ op: 'add', profile, args: reconstructWorkArgs(flags) });
|
|
4212
|
+
if (res.ok) logger.info(`Added worker "${res.worker.id}" (profile ${profile}); pid ${res.worker.pid ?? 'starting'}.`);
|
|
4213
|
+
else { logger.error(`Could not add "${profile}": ${res.error}`); process.exit(1); }
|
|
4214
|
+
}
|
|
4215
|
+
|
|
4216
|
+
async function supervisorRemoveCmd(req) {
|
|
4217
|
+
const logger = getLogger();
|
|
4218
|
+
const target = req.positional[1];
|
|
4219
|
+
if (!target) { logger.error('Usage: c8ctl nano supervisor remove <id|profile|all>'); process.exit(1); }
|
|
4220
|
+
if (!await liveSupervisor()) { logger.error('Supervisor is not running.'); process.exit(1); }
|
|
4221
|
+
const res = await supervisorRequest({ op: 'remove', target });
|
|
4222
|
+
if (res.ok && res.removed.length > 0) logger.info(`Removed worker(s): ${res.removed.join(', ')}.`);
|
|
4223
|
+
else if (res.ok) { logger.warn(`No worker matched "${target}".`); }
|
|
4224
|
+
else { logger.error(res.error); process.exit(1); }
|
|
4225
|
+
}
|
|
4226
|
+
|
|
4227
|
+
async function supervisorRestartCmd(req) {
|
|
4228
|
+
const logger = getLogger();
|
|
4229
|
+
const target = req.positional[1];
|
|
4230
|
+
if (!target) { logger.error('Usage: c8ctl nano supervisor restart <id|profile|all>'); process.exit(1); }
|
|
4231
|
+
if (!await liveSupervisor()) { logger.error('Supervisor is not running.'); process.exit(1); }
|
|
4232
|
+
const res = await supervisorRequest({ op: 'restart', target });
|
|
4233
|
+
if (res.ok && res.restarted.length > 0) logger.info(`Restarted worker(s): ${res.restarted.join(', ')}.`);
|
|
4234
|
+
else if (res.ok) { logger.warn(`No worker matched "${target}".`); }
|
|
4235
|
+
else { logger.error(res.error); process.exit(1); }
|
|
4236
|
+
}
|
|
4237
|
+
|
|
4238
|
+
async function supervisorStopCmd() {
|
|
4239
|
+
const logger = getLogger();
|
|
4240
|
+
const running = await liveSupervisor();
|
|
4241
|
+
if (!running) {
|
|
4242
|
+
if (readSupervisorState()) { clearSupervisorState(); logger.info('Cleared stale supervisor state.'); }
|
|
4243
|
+
else logger.warn('Supervisor is not running — nothing to stop.');
|
|
4244
|
+
return;
|
|
4245
|
+
}
|
|
4246
|
+
try {
|
|
4247
|
+
await supervisorRequest({ op: 'stop' });
|
|
4248
|
+
} catch {
|
|
4249
|
+
// Socket unreachable — fall back to signalling the daemon pid directly.
|
|
4250
|
+
try { process.kill(running.pid, 'SIGTERM'); } catch { /* already gone */ }
|
|
4251
|
+
}
|
|
4252
|
+
// Gate the wait loop and the SIGKILL fallback on the daemon pid we captured,
|
|
4253
|
+
// not on runningSupervisor()/the state file: the daemon clears its state file
|
|
4254
|
+
// as part of shutting down (and liveSupervisor()/external cleanup can remove
|
|
4255
|
+
// it too), so a state-file check can report "gone" while the process is still
|
|
4256
|
+
// alive — which would break the loop early and skip the SIGKILL fallback,
|
|
4257
|
+
// leaving a wedged daemon and its worker process group running.
|
|
4258
|
+
const deadline = Date.now() + STOP_GRACE_MS + 2_000;
|
|
4259
|
+
while (Date.now() < deadline) {
|
|
4260
|
+
if (!isPidAlive(running.pid)) break;
|
|
4261
|
+
await new Promise((r) => setTimeout(r, 150));
|
|
4262
|
+
}
|
|
4263
|
+
if (isPidAlive(running.pid)) {
|
|
4264
|
+
logger.warn(`Supervisor (pid ${running.pid}) did not stop gracefully — sending SIGKILL.`);
|
|
4265
|
+
// The daemon is spawned detached (a process-group leader) and its workers
|
|
4266
|
+
// are children in that group, so SIGKILL the whole group to avoid orphaning
|
|
4267
|
+
// `nano work` processes. Fall back to the bare pid (e.g. on Windows, or if
|
|
4268
|
+
// the daemon isn't a group leader).
|
|
4269
|
+
let killedGroup = false;
|
|
4270
|
+
if (osPlatform() !== 'win32') {
|
|
4271
|
+
try { process.kill(-running.pid, 'SIGKILL'); killedGroup = true; } catch { /* fall back below */ }
|
|
4272
|
+
}
|
|
4273
|
+
if (!killedGroup) { try { process.kill(running.pid, 'SIGKILL'); } catch { /* ignore */ } }
|
|
4274
|
+
clearSupervisorState();
|
|
4275
|
+
}
|
|
4276
|
+
logger.info('Supervisor stopped.');
|
|
4277
|
+
}
|
|
4278
|
+
|
|
4279
|
+
function supervisorLogsCmd(req) {
|
|
4280
|
+
const logger = getLogger();
|
|
4281
|
+
const id = req.positional[1];
|
|
4282
|
+
const file = id ? supervisorWorkerLogFile(id) : supervisorDaemonLogFile();
|
|
4283
|
+
if (!existsSync(file)) {
|
|
4284
|
+
logger.error(`No log file at ${file}.${id ? ` (unknown worker "${id}"?)` : ''}`);
|
|
4285
|
+
process.exit(1);
|
|
4286
|
+
}
|
|
4287
|
+
const follow = Boolean(req.follow);
|
|
4288
|
+
const tailArgs = follow ? ['-n', '200', '-F', file] : ['-n', '200', file];
|
|
4289
|
+
const proc = spawn('tail', tailArgs, { stdio: ['ignore', 'inherit', 'inherit'] });
|
|
4290
|
+
proc.on('error', () => {
|
|
4291
|
+
// tail unavailable (e.g. Windows): print the tail ourselves, no follow.
|
|
4292
|
+
if (follow) logger.warn('`--follow` is not supported without `tail` on this platform; printing the current tail only.');
|
|
4293
|
+
try {
|
|
4294
|
+
const lines = readFileSync(file, 'utf-8').split('\n');
|
|
4295
|
+
logger.info(lines.slice(-200).join('\n'));
|
|
4296
|
+
} catch (err) { logger.error(`Could not read ${file}: ${err.message}`); }
|
|
4297
|
+
});
|
|
4298
|
+
}
|
|
4299
|
+
|
|
4300
|
+
/**
|
|
4301
|
+
* Interactive attach console. Streams live events from the daemon and accepts
|
|
4302
|
+
* line commands. `detach` (or Ctrl-D) disconnects but leaves the daemon
|
|
4303
|
+
* running; `stop` tears the fleet down.
|
|
4304
|
+
*/
|
|
4305
|
+
async function attachSupervisorConsole(state) {
|
|
4306
|
+
const logger = getLogger();
|
|
4307
|
+
const socketPath = state?.socket || getSupervisorSocketPath();
|
|
4308
|
+
let sock;
|
|
4309
|
+
try {
|
|
4310
|
+
sock = await supervisorConnect(socketPath);
|
|
4311
|
+
} catch (err) {
|
|
4312
|
+
logger.error(`Could not attach to supervisor: ${err.message}`);
|
|
4313
|
+
process.exit(1);
|
|
4314
|
+
}
|
|
4315
|
+
|
|
4316
|
+
const out = (s) => process.stdout.write(s + '\n');
|
|
4317
|
+
out('Attached to nano worker supervisor. Type "help" for commands.');
|
|
4318
|
+
out('Detach (leave it running) with "detach" or Ctrl-D; tear it down with "stop".');
|
|
4319
|
+
sock.write(encodeFrame({ op: 'attach' }));
|
|
4320
|
+
|
|
4321
|
+
let buf = '';
|
|
4322
|
+
sock.on('data', (chunk) => {
|
|
4323
|
+
buf += chunk;
|
|
4324
|
+
const { frames, rest } = decodeFrames(buf);
|
|
4325
|
+
buf = rest;
|
|
4326
|
+
for (const frame of frames) {
|
|
4327
|
+
if (frame.type === 'status') {
|
|
4328
|
+
out('');
|
|
4329
|
+
out(formatSupervisorStatus(frame));
|
|
4330
|
+
} else if (frame.type === 'event') {
|
|
4331
|
+
const w = frame.worker;
|
|
4332
|
+
if (frame.event === 'worker-start') out(`• worker ${w.id} started (pid ${w.pid}).`);
|
|
4333
|
+
else if (frame.event === 'worker-exit') out(`• worker ${w.id} exited (${w.lastExit}); restarting in ${formatDuration(frame.restartInMs)}.`);
|
|
4334
|
+
else if (frame.event === 'worker-remove') out(`• worker ${frame.id} removed.`);
|
|
4335
|
+
else if (frame.event === 'daemon-stop') out('• supervisor stopping.');
|
|
4336
|
+
} else if (frame.type === 'added') {
|
|
4337
|
+
out(`• added worker ${frame.worker.id}.`);
|
|
4338
|
+
} else if (frame.type === 'removed') {
|
|
4339
|
+
out(`• removed: ${frame.removed.join(', ') || '(none matched)'}.`);
|
|
4340
|
+
} else if (frame.type === 'restarted') {
|
|
4341
|
+
out(`• restarted: ${frame.restarted.join(', ') || '(none matched)'}.`);
|
|
4342
|
+
} else if (frame.ok === false) {
|
|
4343
|
+
out(`! ${frame.error}`);
|
|
4344
|
+
}
|
|
4345
|
+
}
|
|
4346
|
+
});
|
|
4347
|
+
|
|
4348
|
+
const rl = createReadline({ input: process.stdin, output: process.stdout, prompt: 'supervisor> ' });
|
|
4349
|
+
rl.prompt();
|
|
4350
|
+
|
|
4351
|
+
await new Promise((resolve) => {
|
|
4352
|
+
let stopping = false;
|
|
4353
|
+
const finish = () => { try { rl.close(); } catch { /* ignore */ } try { sock.end(); } catch { /* ignore */ } resolve(); };
|
|
4354
|
+
|
|
4355
|
+
sock.on('close', () => { if (!stopping) out('\nSupervisor connection closed.'); finish(); });
|
|
4356
|
+
|
|
4357
|
+
rl.on('line', (line) => {
|
|
4358
|
+
const parts = String(line).trim().split(/\s+/).filter(Boolean);
|
|
4359
|
+
const cmd = (parts.shift() || '').toLowerCase();
|
|
4360
|
+
switch (cmd) {
|
|
4361
|
+
case '': break;
|
|
4362
|
+
case 'help':
|
|
4363
|
+
out('Commands: status | add <profile> [work flags] | remove <id|profile|all> |');
|
|
4364
|
+
out(' restart <id|profile|all> | logs [id] | detach | stop | help');
|
|
4365
|
+
break;
|
|
4366
|
+
case 'status': sock.write(encodeFrame({ op: 'status' })); break;
|
|
4367
|
+
case 'add': {
|
|
4368
|
+
const profile = parts.shift();
|
|
4369
|
+
if (!profile) { out('usage: add <profile> [work flags]'); break; }
|
|
4370
|
+
sock.write(encodeFrame({ op: 'add', profile, args: parts }));
|
|
4371
|
+
break;
|
|
4372
|
+
}
|
|
4373
|
+
case 'remove': case 'rm': {
|
|
4374
|
+
const target = parts.shift();
|
|
4375
|
+
if (!target) { out('usage: remove <id|profile|all>'); break; }
|
|
4376
|
+
sock.write(encodeFrame({ op: 'remove', target }));
|
|
4377
|
+
break;
|
|
4378
|
+
}
|
|
4379
|
+
case 'restart': {
|
|
4380
|
+
const target = parts.shift();
|
|
4381
|
+
if (!target) { out('usage: restart <id|profile|all>'); break; }
|
|
4382
|
+
sock.write(encodeFrame({ op: 'restart', target }));
|
|
4383
|
+
break;
|
|
4384
|
+
}
|
|
4385
|
+
case 'logs': case 'log': {
|
|
4386
|
+
const file = parts[0] ? supervisorWorkerLogFile(parts[0]) : supervisorDaemonLogFile();
|
|
4387
|
+
try {
|
|
4388
|
+
const lines = readFileSync(file, 'utf-8').split('\n');
|
|
4389
|
+
out(lines.slice(-30).join('\n'));
|
|
4390
|
+
} catch { out(`no log at ${file}`); }
|
|
4391
|
+
break;
|
|
4392
|
+
}
|
|
4393
|
+
case 'detach': case 'quit': case 'exit':
|
|
4394
|
+
out('Detaching — supervisor keeps running. Reattach with: c8ctl nano supervisor');
|
|
4395
|
+
finish();
|
|
4396
|
+
return;
|
|
4397
|
+
case 'stop':
|
|
4398
|
+
stopping = true;
|
|
4399
|
+
out('Stopping supervisor…');
|
|
4400
|
+
sock.write(encodeFrame({ op: 'stop' }));
|
|
4401
|
+
setTimeout(finish, 500);
|
|
4402
|
+
return;
|
|
4403
|
+
default:
|
|
4404
|
+
out(`unknown command "${cmd}" — type "help"`);
|
|
4405
|
+
}
|
|
4406
|
+
rl.prompt();
|
|
4407
|
+
});
|
|
4408
|
+
|
|
4409
|
+
// Ctrl-D (EOF) detaches, leaving the daemon running.
|
|
4410
|
+
rl.on('close', () => {
|
|
4411
|
+
if (stopping) return;
|
|
4412
|
+
out('\nDetaching — supervisor keeps running. Reattach with: c8ctl nano supervisor');
|
|
4413
|
+
finish();
|
|
4414
|
+
});
|
|
4415
|
+
});
|
|
4416
|
+
}
|
|
4417
|
+
|
|
4418
|
+
/** Dispatch the `supervisor` subcommand's action. */
|
|
4419
|
+
async function supervisorCommand(req, flags) {
|
|
4420
|
+
const action = (req.positional[0] || '').toLowerCase();
|
|
4421
|
+
switch (action) {
|
|
4422
|
+
case '__daemon':
|
|
4423
|
+
await runSupervisorDaemon();
|
|
4424
|
+
return;
|
|
4425
|
+
case '':
|
|
4426
|
+
case 'attach': {
|
|
4427
|
+
const state = await startSupervisorDaemon();
|
|
4428
|
+
await attachSupervisorConsole(runningSupervisor() || state);
|
|
4429
|
+
return;
|
|
4430
|
+
}
|
|
4431
|
+
case 'start':
|
|
4432
|
+
await supervisorStartCmd(req, flags);
|
|
4433
|
+
return;
|
|
4434
|
+
case 'status':
|
|
4435
|
+
case 'list':
|
|
4436
|
+
case 'ls':
|
|
4437
|
+
await supervisorStatusCmd();
|
|
4438
|
+
return;
|
|
4439
|
+
case 'add':
|
|
4440
|
+
await supervisorAddCmd(req, flags);
|
|
4441
|
+
return;
|
|
4442
|
+
case 'remove':
|
|
4443
|
+
case 'rm':
|
|
4444
|
+
await supervisorRemoveCmd(req);
|
|
4445
|
+
return;
|
|
4446
|
+
case 'restart':
|
|
4447
|
+
await supervisorRestartCmd(req);
|
|
4448
|
+
return;
|
|
4449
|
+
case 'stop':
|
|
4450
|
+
await supervisorStopCmd();
|
|
4451
|
+
return;
|
|
4452
|
+
case 'logs':
|
|
4453
|
+
case 'log':
|
|
4454
|
+
supervisorLogsCmd(req);
|
|
4455
|
+
return;
|
|
4456
|
+
default:
|
|
4457
|
+
getLogger().error(`Unknown supervisor action "${action}". Use: start|status|add|remove|restart|stop|logs|attach`);
|
|
4458
|
+
process.exit(1);
|
|
4459
|
+
}
|
|
4460
|
+
}
|
|
4461
|
+
|
|
3198
4462
|
// ---------------------------------------------------------------------------
|
|
3199
4463
|
// update — pull a new nanobpmn release onto a machine with an existing install.
|
|
3200
4464
|
// The plugin (and the bundled server binary, shipped via the matching platform
|
|
@@ -4547,6 +5811,7 @@ export {
|
|
|
4547
5811
|
applyAssign,
|
|
4548
5812
|
resolveAssignInputs,
|
|
4549
5813
|
jobTypeMatrix,
|
|
5814
|
+
diffJobTypes,
|
|
4550
5815
|
parseJobTypeFlags,
|
|
4551
5816
|
deriveJobLockMs,
|
|
4552
5817
|
derivePollTimeoutMs,
|
|
@@ -4556,6 +5821,26 @@ export {
|
|
|
4556
5821
|
RESERVED_RESULT_KEYS,
|
|
4557
5822
|
SANDBOXES,
|
|
4558
5823
|
};
|
|
5824
|
+
export {
|
|
5825
|
+
reconstructWorkArgs,
|
|
5826
|
+
supervisorWorkerId,
|
|
5827
|
+
redactWorkArgs,
|
|
5828
|
+
supervisorBackoffMs,
|
|
5829
|
+
encodeFrame,
|
|
5830
|
+
decodeFrames,
|
|
5831
|
+
formatDuration,
|
|
5832
|
+
summarizeSupervisorWorker,
|
|
5833
|
+
formatSupervisorStatus,
|
|
5834
|
+
WORK_FORWARD_FLAGS,
|
|
5835
|
+
runSupervisorDaemon,
|
|
5836
|
+
startSupervisorDaemon,
|
|
5837
|
+
supervisorRequest,
|
|
5838
|
+
runningSupervisor,
|
|
5839
|
+
readSupervisorState,
|
|
5840
|
+
clearSupervisorState,
|
|
5841
|
+
getSupervisorSocketPath,
|
|
5842
|
+
getSupervisorStateFile,
|
|
5843
|
+
};
|
|
4559
5844
|
|
|
4560
5845
|
export const metadata = {
|
|
4561
5846
|
name: 'c8ctl-plugin-nano',
|
|
@@ -4595,6 +5880,12 @@ export const metadata = {
|
|
|
4595
5880
|
{ command: 'c8ctl nano hire --name coder --rank senior --command "agent-harness" --sandbox docker --image ghcr.io/acme/agent:1', description: 'Create a profile that runs each job in a throwaway Docker container' },
|
|
4596
5881
|
{ command: 'c8ctl nano work reviewer', description: 'Spawn Nano job workers for the "reviewer" profile and poll for work' },
|
|
4597
5882
|
{ command: 'c8ctl nano work coder --sandbox docker --image ghcr.io/acme/agent:1', description: 'Run jobs in isolated containers with disk-hygiene reaping' },
|
|
5883
|
+
{ command: 'c8ctl nano supervisor start --worker reviewer --worker coder', description: 'Start a detached supervisor managing several workers from one terminal' },
|
|
5884
|
+
{ command: 'c8ctl nano supervisor', description: 'Attach an interactive console to the supervisor (detach with Ctrl-D, leaving it running)' },
|
|
5885
|
+
{ command: 'c8ctl nano supervisor status', description: 'List supervised workers (pid, state, restarts, uptime) without the console' },
|
|
5886
|
+
{ command: 'c8ctl nano supervisor add decider --max-parallel 2', description: 'Add a supervised worker (forwarding work flags) to the running supervisor' },
|
|
5887
|
+
{ command: 'c8ctl nano supervisor restart reviewer', description: 'Restart a supervised worker by id or profile' },
|
|
5888
|
+
{ command: 'c8ctl nano supervisor stop', description: 'Stop the supervisor daemon and all its workers' },
|
|
4598
5889
|
],
|
|
4599
5890
|
},
|
|
4600
5891
|
processos: {
|
|
@@ -4657,6 +5948,8 @@ export const commands = {
|
|
|
4657
5948
|
'lock-grace': { type: 'string', description: 'work: extra ms added to --job-timeout to derive the broker activation lock, so the worker reports before the lock lapses (default 120000)' },
|
|
4658
5949
|
'poll-timeout': { type: 'string', description: 'work: broker long-poll window in ms each activateJobs request is held open (fewer reconnects → fewer transient connect errors); default 30000, 0 = broker default, negative = return immediately' },
|
|
4659
5950
|
'job-type': { type: 'string', multiple: true, description: 'work: extra job type to service alongside the rank×capability matrix (repeatable)' },
|
|
5951
|
+
worker: { type: 'string', multiple: true, description: 'supervisor start: profile to launch as a supervised worker (repeatable)' },
|
|
5952
|
+
attach: { type: 'boolean', description: 'supervisor start: attach the interactive console after starting the daemon' },
|
|
4660
5953
|
},
|
|
4661
5954
|
handler: async (args, flags) => {
|
|
4662
5955
|
const logger = getLogger();
|
|
@@ -4714,6 +6007,9 @@ export const commands = {
|
|
|
4714
6007
|
case 'work':
|
|
4715
6008
|
await workAgent(req, flags);
|
|
4716
6009
|
break;
|
|
6010
|
+
case 'supervisor':
|
|
6011
|
+
await supervisorCommand(req, flags);
|
|
6012
|
+
break;
|
|
4717
6013
|
}
|
|
4718
6014
|
} catch (error) {
|
|
4719
6015
|
logger.error(`nano ${req.subcommand} failed: ${error instanceof Error ? error.message : error}`);
|
|
@@ -4805,6 +6101,7 @@ function printUsage() {
|
|
|
4805
6101
|
console.log(' c8ctl nano hire [--name <n>] [--rank <r>] [--command <c>] [--arg <switch> ...] [--model <m>] [--capabilities <a,b>] [--sandbox none|docker|podman] [--image <ref>] [--env NAME=VALUE ...] [--list]');
|
|
4806
6102
|
console.log(' c8ctl nano assign <profileName> [<capability> ...] [--name <n>] [--capabilities <a,b>]');
|
|
4807
6103
|
console.log(' c8ctl nano work <profileName> [--arg <switch> ...] [--max-parallel <n>] [--job-timeout <ms>] [--lock-grace <ms>] [--poll-timeout <ms>] [--job-type <token> ...] [--sandbox none|docker|podman] [--image <ref>] [--env NAME=VALUE ...] [--secret-resolver host] [--min-free-mb <n>] [--clone-timeout <ms>] [--keep-runs] [--stream]');
|
|
6104
|
+
console.log(' c8ctl nano supervisor [start|status|add|remove|restart|stop|logs|attach] ... (manage many workers from one terminal)');
|
|
4808
6105
|
console.log('');
|
|
4809
6106
|
console.log('Subcommands:');
|
|
4810
6107
|
console.log(' start Spawn an N-node local cluster wired to talk to each other on localhost');
|
|
@@ -4821,6 +6118,7 @@ function printUsage() {
|
|
|
4821
6118
|
console.log(' hire Create a CLI agent worker profile (rank + capabilities → job-type matrix)');
|
|
4822
6119
|
console.log(' assign Grant new capabilities (roles) to an existing hire (additive)');
|
|
4823
6120
|
console.log(' work Run a hired profile as Nano job workers, polling for work until Ctrl-C');
|
|
6121
|
+
console.log(' supervisor Run/manage a fleet of workers from one terminal (detachable console + non-interactive control)');
|
|
4824
6122
|
console.log('');
|
|
4825
6123
|
console.log('Options:');
|
|
4826
6124
|
console.log(' <nodes> Number of nodes to start (default 1)');
|