@triflux/remote 10.0.0 → 10.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -290,4 +290,4 @@ export function createNotifier(opts = {}) {
290
290
  });
291
291
 
292
292
  return createNotifierInstance(normalizeChannels(opts.channels, env), deps);
293
- }
293
+ }
@@ -5,6 +5,7 @@
5
5
  import { execFileSync } from 'node:child_process';
6
6
  import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
7
7
  import { basename, join, posix as posixPath, win32 as win32Path } from 'node:path';
8
+ import { execSshWithRetry } from '@triflux/core/hub/lib/ssh-retry.mjs';
8
9
 
9
10
  const REMOTE_ENV_TTL_MS = 86_400_000; // 24h
10
11
  const REMOTE_STAGE_ROOT = 'tfx-remote';
@@ -84,8 +85,9 @@ function probeRemoteEnvViaPwsh(host) {
84
85
  ].join('; ');
85
86
 
86
87
  try {
87
- const output = execFileSync('ssh', [host, 'pwsh', '-NoProfile', '-Command', command], {
88
+ const output = execSshWithRetry([host, 'pwsh', '-NoProfile', '-Command', command], {
88
89
  encoding: 'utf8', timeout: 15000, stdio: ['pipe', 'pipe', 'pipe'],
90
+ maxRetries: 2, baseDelayMs: 1000,
89
91
  });
90
92
  return normalizePwshProbeEnv(parseProbeLines(output));
91
93
  } catch {
@@ -102,8 +104,9 @@ function probeRemoteEnvViaPosix(host) {
102
104
  ].join('\n');
103
105
 
104
106
  try {
105
- const output = execFileSync('ssh', [host, 'sh'], {
107
+ const output = execSshWithRetry([host, 'sh'], {
106
108
  encoding: 'utf8', timeout: 15000, input: script,
109
+ maxRetries: 2, baseDelayMs: 1000,
107
110
  });
108
111
  return normalizePosixProbeEnv(parseProbeLines(output));
109
112
  } catch {
@@ -0,0 +1,613 @@
1
+ // hub/team/swarm-hypervisor.mjs — Multi-model swarm orchestration hypervisor
2
+ // Consumes a SwarmPlan (from swarm-planner.mjs) and orchestrates parallel
3
+ // conductor sessions with file-lease enforcement, result validation,
4
+ // and ordered integration.
5
+ //
6
+ // Failure modes handled:
7
+ // F1: Worker crash → conductor auto-restart (maxRestarts)
8
+ // F2: Rate limit → account-broker cooldown + fallback agent
9
+ // F3: Stall → health probe L1 detection + kill + restart
10
+ // F4: File lease violation → revert worker changes, flag shard as failed
11
+ // F5: Merge conflict → retry integration with conflict resolution
12
+
13
+ import { EventEmitter } from 'node:events';
14
+ import { join } from 'node:path';
15
+ import { mkdirSync, readFileSync, existsSync } from 'node:fs';
16
+ import { execSync } from 'node:child_process';
17
+
18
+ import { createConductor, STATES } from './conductor.mjs';
19
+ import { createSwarmLocks } from './swarm-locks.mjs';
20
+ import { createEventLog } from './event-log.mjs';
21
+ import { probeRemoteEnv, resolveRemoteDir } from './remote-session.mjs';
22
+ import { fetchRemoteShard } from './worktree-lifecycle.mjs';
23
+ import { getHostConfig } from '@triflux/core/hub/lib/ssh-command.mjs';
24
+
25
+ // ── Swarm states ──────────────────────────────────────────────
26
+
27
+ export const SWARM_STATES = Object.freeze({
28
+ PLANNING: 'planning',
29
+ LAUNCHING: 'launching',
30
+ RUNNING: 'running',
31
+ INTEGRATING: 'integrating',
32
+ VALIDATING: 'validating',
33
+ COMPLETED: 'completed',
34
+ FAILED: 'failed',
35
+ });
36
+
37
+ // ── Failure mode classification ───────────────────────────────
38
+
39
+ const FAILURE_MODES = Object.freeze({
40
+ F1_CRASH: 'F1_crash',
41
+ F2_RATE_LIMIT: 'F2_rate_limit',
42
+ F3_STALL: 'F3_stall',
43
+ F4_LEASE_VIOLATION: 'F4_lease_violation',
44
+ F5_MERGE_CONFLICT: 'F5_merge_conflict',
45
+ });
46
+
47
+ const FALLBACK_AGENTS = Object.freeze({
48
+ codex: 'gemini',
49
+ gemini: 'codex',
50
+ claude: 'codex',
51
+ });
52
+
53
+ /**
54
+ * Create a swarm hypervisor.
55
+ * @param {object} opts
56
+ * @param {string} opts.workdir — repository root / working directory
57
+ * @param {string} opts.logsDir — base directory for all logs
58
+ * @param {number} [opts.maxRestarts=2] — per-shard max restarts
59
+ * @param {number} [opts.graceMs=10000] — conductor shutdown grace period
60
+ * @param {number} [opts.integrationTimeoutMs=60000] — max time for integration phase
61
+ * @param {object} [opts.probeOpts] — health probe overrides
62
+ * @param {object} [opts.deps] — dependency injection for testing
63
+ * @returns {SwarmHypervisor}
64
+ */
65
+ export function createSwarmHypervisor(opts) {
66
+ const {
67
+ workdir,
68
+ logsDir,
69
+ maxRestarts = 2,
70
+ graceMs = 10_000,
71
+ integrationTimeoutMs = 60_000,
72
+ probeOpts = {},
73
+ deps = {},
74
+ } = opts;
75
+
76
+ if (!workdir) throw new Error('workdir is required');
77
+ if (!logsDir) throw new Error('logsDir is required');
78
+
79
+ mkdirSync(logsDir, { recursive: true });
80
+
81
+ const emitter = new EventEmitter();
82
+ const eventLog = createEventLog(join(logsDir, 'swarm-events.jsonl'));
83
+
84
+ let state = SWARM_STATES.PLANNING;
85
+ let plan = null;
86
+ let lockManager = null;
87
+
88
+ /** @type {Map<string, { conductor, shardConfig, result, status }>} */
89
+ const workers = new Map();
90
+
91
+ /** @type {Map<string, { conductor, shardConfig }>} redundant workers for critical shards */
92
+ const redundantWorkers = new Map();
93
+
94
+ const results = new Map(); // shardName → validated result
95
+ const failures = new Map(); // shardName → failure info
96
+
97
+ // ── State machine ───────────────────────────────────────────
98
+
99
+ function setState(next, reason = '') {
100
+ const prev = state;
101
+ state = next;
102
+ eventLog.append('swarm_state', { from: prev, to: next, reason });
103
+ emitter.emit('stateChange', { from: prev, to: next, reason });
104
+ }
105
+
106
+ // ── Worker lifecycle ────────────────────────────────────────
107
+
108
+ function buildSessionConfig(shard) {
109
+ const config = {
110
+ id: `swarm-${shard.name}-${Date.now()}`,
111
+ agent: shard.agent,
112
+ prompt: shard.prompt,
113
+ workdir,
114
+ mcpServers: shard.mcp,
115
+ };
116
+
117
+ // Remote shard: add conductor remote fields
118
+ if (shard.host && shard._remoteEnv) {
119
+ const remoteDir = resolveRemoteDir(workdir, shard._remoteEnv);
120
+ return {
121
+ ...config,
122
+ remote: true,
123
+ host: shard.host,
124
+ sessionName: `swarm-${shard.name}-${Date.now()}`,
125
+ paneTarget: `swarm-${shard.name}-${Date.now()}:0.0`,
126
+ workdir: remoteDir,
127
+ };
128
+ }
129
+
130
+ return config;
131
+ }
132
+
133
+ function launchShard(shard, isRedundant = false) {
134
+ const shardLogsDir = join(logsDir, isRedundant ? `${shard.name}-redundant` : shard.name);
135
+ mkdirSync(shardLogsDir, { recursive: true });
136
+
137
+ // Remote shard: probe environment before conductor creation
138
+ if (shard.host && !shard._remoteEnv) {
139
+ try {
140
+ shard._remoteEnv = probeRemoteEnv(shard.host);
141
+ if (!shard._remoteEnv.claudePath) {
142
+ eventLog.append('remote_probe_no_claude', { shard: shard.name, host: shard.host });
143
+ failures.set(shard.name, { mode: FAILURE_MODES.F1_CRASH, reason: `claude not found on ${shard.host}` });
144
+ return null;
145
+ }
146
+ eventLog.append('remote_probe_ok', { shard: shard.name, host: shard.host, env: shard._remoteEnv });
147
+ } catch (err) {
148
+ eventLog.append('remote_probe_failed', { shard: shard.name, host: shard.host, error: err.message });
149
+ failures.set(shard.name, { mode: FAILURE_MODES.F1_CRASH, reason: `remote probe failed: ${err.message}` });
150
+ return null;
151
+ }
152
+ }
153
+
154
+ const conductor = createConductor({
155
+ logsDir: shardLogsDir,
156
+ maxRestarts,
157
+ graceMs,
158
+ probeOpts,
159
+ onCompleted: (sessionId) => handleShardCompleted(shard.name, sessionId, isRedundant),
160
+ });
161
+
162
+ const sessionConfig = buildSessionConfig(shard);
163
+
164
+ // Acquire file leases
165
+ if (!isRedundant) {
166
+ const leaseResult = lockManager.acquire(shard.name, shard.files);
167
+ if (!leaseResult.ok) {
168
+ eventLog.append('lease_denied', {
169
+ shard: shard.name,
170
+ conflicts: leaseResult.conflicts,
171
+ });
172
+ failures.set(shard.name, {
173
+ mode: FAILURE_MODES.F4_LEASE_VIOLATION,
174
+ conflicts: leaseResult.conflicts,
175
+ });
176
+ return null;
177
+ }
178
+ }
179
+
180
+ conductor.spawnSession(sessionConfig);
181
+
182
+ eventLog.append('shard_launched', {
183
+ shard: shard.name,
184
+ agent: shard.agent,
185
+ sessionId: sessionConfig.id,
186
+ isRedundant,
187
+ files: shard.files,
188
+ remote: Boolean(shard.host),
189
+ host: shard.host || null,
190
+ });
191
+
192
+ const entry = { conductor, shardConfig: shard, sessionConfig, startedAt: Date.now() };
193
+
194
+ if (isRedundant) {
195
+ redundantWorkers.set(shard.name, entry);
196
+ } else {
197
+ workers.set(shard.name, entry);
198
+ }
199
+
200
+ // Listen for dead events (F1/F2/F3)
201
+ conductor.on('dead', ({ sessionId, reason }) => {
202
+ handleShardFailed(shard.name, sessionId, reason, isRedundant);
203
+ });
204
+
205
+ return entry;
206
+ }
207
+
208
+ // ── Completion handling ─────────────────────────────────────
209
+
210
+ function handleShardCompleted(shardName, sessionId, isRedundant) {
211
+ eventLog.append('shard_completed', { shard: shardName, sessionId, isRedundant });
212
+
213
+ if (isRedundant) {
214
+ // Redundant worker completed first — kill primary if still running
215
+ const primary = workers.get(shardName);
216
+ if (primary && !isTerminal(primary)) {
217
+ eventLog.append('redundant_wins', { shard: shardName });
218
+ void primary.conductor.shutdown('redundant_completed_first');
219
+ }
220
+ } else {
221
+ // Primary completed — kill redundant if exists
222
+ const redundant = redundantWorkers.get(shardName);
223
+ if (redundant) {
224
+ void redundant.conductor.shutdown('primary_completed_first');
225
+ }
226
+ }
227
+
228
+ emitter.emit('shardCompleted', { shardName, sessionId, isRedundant });
229
+ checkAllShardsCompleted();
230
+ }
231
+
232
+ function handleShardFailed(shardName, sessionId, reason, isRedundant) {
233
+ const failureMode = classifyFailure(reason);
234
+
235
+ eventLog.append('shard_failed', {
236
+ shard: shardName,
237
+ sessionId,
238
+ reason,
239
+ failureMode,
240
+ isRedundant,
241
+ });
242
+
243
+ if (isRedundant) return; // redundant failure is non-critical
244
+
245
+ // F2: Rate limit — try fallback agent
246
+ if (failureMode === FAILURE_MODES.F2_RATE_LIMIT) {
247
+ const shard = plan.shards.find((s) => s.name === shardName);
248
+ if (shard) {
249
+ const fallbackAgent = FALLBACK_AGENTS[shard.agent];
250
+ if (fallbackAgent) {
251
+ eventLog.append('fallback_agent', {
252
+ shard: shardName,
253
+ from: shard.agent,
254
+ to: fallbackAgent,
255
+ });
256
+ const fallbackShard = { ...shard, agent: fallbackAgent };
257
+ lockManager.release(shardName);
258
+ launchShard(fallbackShard);
259
+ return;
260
+ }
261
+ }
262
+ }
263
+
264
+ failures.set(shardName, { mode: failureMode, reason, sessionId });
265
+ lockManager.release(shardName);
266
+
267
+ emitter.emit('shardFailed', { shardName, failureMode, reason });
268
+ checkAllShardsCompleted();
269
+ }
270
+
271
+ function classifyFailure(reason) {
272
+ if (!reason) return FAILURE_MODES.F1_CRASH;
273
+ const r = String(reason).toLowerCase();
274
+ if (/rate.?limit|cooldown/u.test(r)) return FAILURE_MODES.F2_RATE_LIMIT;
275
+ if (/stall|l1_stall|timeout/u.test(r)) return FAILURE_MODES.F3_STALL;
276
+ if (/lease|violation/u.test(r)) return FAILURE_MODES.F4_LEASE_VIOLATION;
277
+ if (/merge|conflict/u.test(r)) return FAILURE_MODES.F5_MERGE_CONFLICT;
278
+ return FAILURE_MODES.F1_CRASH;
279
+ }
280
+
281
+ function isTerminal(entry) {
282
+ const snap = entry.conductor.getSnapshot();
283
+ return snap.every((s) => s.state === STATES.COMPLETED || s.state === STATES.DEAD);
284
+ }
285
+
286
+ // ── Integration ─────────────────────────────────────────────
287
+
288
+ function checkAllShardsCompleted() {
289
+ if (state !== SWARM_STATES.RUNNING) return;
290
+
291
+ const allDone = plan.mergeOrder.every((name) => {
292
+ const w = workers.get(name);
293
+ return (w && isTerminal(w)) || failures.has(name);
294
+ });
295
+
296
+ if (allDone) {
297
+ void integrateResults();
298
+ }
299
+ }
300
+
301
+ /**
302
+ * Validate a shard's output — check for file lease violations.
303
+ * @param {string} shardName
304
+ * @param {string[]} changedFiles — files the shard actually modified
305
+ * @returns {{ ok: boolean, violations: Array }}
306
+ */
307
+ function validateResult(shardName, changedFiles) {
308
+ const violations = lockManager.validateChanges(shardName, changedFiles);
309
+
310
+ eventLog.append('validate_result', {
311
+ shard: shardName,
312
+ changedFiles,
313
+ violations,
314
+ ok: violations.length === 0,
315
+ });
316
+
317
+ return {
318
+ ok: violations.length === 0,
319
+ violations,
320
+ };
321
+ }
322
+
323
+ /**
324
+ * Integrate results from all completed shards in merge order.
325
+ * Uses git operations for conflict detection.
326
+ */
327
+ async function integrateResults() {
328
+ setState(SWARM_STATES.INTEGRATING, 'all_shards_done');
329
+
330
+ const integrated = [];
331
+ const integrationFailures = [];
332
+
333
+ for (const shardName of plan.mergeOrder) {
334
+ if (failures.has(shardName)) {
335
+ eventLog.append('skip_failed_shard', { shard: shardName });
336
+ continue;
337
+ }
338
+
339
+ const worker = workers.get(shardName);
340
+ if (!worker) continue;
341
+
342
+ // Fetch remote shard branch to local (push-blocked hosts like Ultra4)
343
+ const shard = plan.shards.find((s) => s.name === shardName);
344
+ if (shard?.host && shard._remoteEnv) {
345
+ const hostConfig = getHostConfig(shard.host, config.rootDir);
346
+ const sshUser = hostConfig?.ssh_user || shard.host;
347
+ const remoteRepoPath = resolveRemoteDir(config.rootDir || process.cwd(), shard._remoteEnv);
348
+ const fetchResult = await fetchRemoteShard({
349
+ host: shard.host,
350
+ sshUser,
351
+ remoteRepoPath,
352
+ branchName: worker.branchName || `swarm/${config.runId}/${shardName}`,
353
+ rootDir: config.rootDir || process.cwd(),
354
+ });
355
+
356
+ if (!fetchResult.ok) {
357
+ eventLog.append('remote_fetch_failed', { shard: shardName, error: fetchResult.error });
358
+ integrationFailures.push(shardName);
359
+ continue;
360
+ }
361
+ eventLog.append('remote_fetch_ok', { shard: shardName, headCommit: fetchResult.headCommit });
362
+ }
363
+
364
+ // Read shard output log for changed files
365
+ const changedFiles = detectChangedFiles(shardName, worker);
366
+
367
+ // Validate against lease map
368
+ const validation = validateResult(shardName, changedFiles);
369
+ if (!validation.ok) {
370
+ failures.set(shardName, {
371
+ mode: FAILURE_MODES.F4_LEASE_VIOLATION,
372
+ violations: validation.violations,
373
+ });
374
+ eventLog.append('lease_violation_revert', {
375
+ shard: shardName,
376
+ violations: validation.violations,
377
+ });
378
+ integrationFailures.push(shardName);
379
+ continue;
380
+ }
381
+
382
+ results.set(shardName, {
383
+ shard: shardName,
384
+ changedFiles,
385
+ completedAt: Date.now(),
386
+ });
387
+ integrated.push(shardName);
388
+ }
389
+
390
+ eventLog.append('integration_complete', {
391
+ integrated,
392
+ failed: integrationFailures,
393
+ skipped: [...failures.keys()].filter((n) => !integrationFailures.includes(n)),
394
+ });
395
+
396
+ if (integrationFailures.length > 0 && integrated.length === 0) {
397
+ setState(SWARM_STATES.FAILED, 'all_shards_failed_integration');
398
+ } else {
399
+ setState(SWARM_STATES.COMPLETED, `${integrated.length}/${plan.shards.length} integrated`);
400
+ }
401
+
402
+ emitter.emit('integrationComplete', {
403
+ integrated,
404
+ failed: integrationFailures,
405
+ results: [...results.values()],
406
+ });
407
+ }
408
+
409
+ /**
410
+ * Detect which files a shard modified by reading its output logs.
411
+ * Falls back to an empty list if detection fails.
412
+ * @param {string} shardName
413
+ * @param {object} worker
414
+ * @returns {string[]}
415
+ */
416
+ function detectChangedFiles(shardName, worker) {
417
+ // Best-effort: parse output log for file paths
418
+ const outPath = join(logsDir, shardName);
419
+ try {
420
+ const snap = worker.conductor.getSnapshot();
421
+ for (const session of snap) {
422
+ if (session.outPath && existsSync(session.outPath)) {
423
+ const output = readFileSync(session.outPath, 'utf8');
424
+ return extractFilePathsFromOutput(output, plan.leaseMap.get(shardName) || []);
425
+ }
426
+ }
427
+ } catch { /* best-effort */ }
428
+
429
+ // Fallback: trust the lease map (shard was allowed these files)
430
+ return plan.leaseMap.get(shardName) || [];
431
+ }
432
+
433
+ /**
434
+ * Extract modified file paths from worker output text.
435
+ * Looks for common patterns: "wrote file.mjs", "modified file.mjs", diff headers.
436
+ * @param {string} output
437
+ * @param {string[]} allowedFiles — lease map files to match against
438
+ * @returns {string[]}
439
+ */
440
+ function extractFilePathsFromOutput(output, allowedFiles) {
441
+ if (!output) return allowedFiles;
442
+
443
+ const found = new Set();
444
+ const lines = output.split(/\r?\n/);
445
+
446
+ for (const line of lines) {
447
+ // Match common patterns
448
+ const patterns = [
449
+ /(?:wrote|created|modified|updated|edited)\s+['"]?([^\s'"]+\.\w+)/i,
450
+ /^[+-]{3}\s+[ab]\/(.+)/, // diff headers
451
+ /^diff --git a\/(.+)\s+b\//, // git diff headers
452
+ ];
453
+
454
+ for (const re of patterns) {
455
+ const match = line.match(re);
456
+ if (match) found.add(match[1]);
457
+ }
458
+ }
459
+
460
+ // Intersect with allowed files if we found anything
461
+ if (found.size > 0) {
462
+ return [...found].filter((f) => allowedFiles.some(
463
+ (a) => f.endsWith(a) || a.endsWith(f) || f === a,
464
+ ));
465
+ }
466
+
467
+ return allowedFiles;
468
+ }
469
+
470
+ // ── Status monitor ──────────────────────────────────────────
471
+
472
+ /**
473
+ * Get current swarm status snapshot.
474
+ * @returns {SwarmStatus}
475
+ */
476
+ function getStatus() {
477
+ const workerStatuses = [];
478
+
479
+ for (const [name, w] of workers) {
480
+ const snap = w.conductor.getSnapshot();
481
+ workerStatuses.push({
482
+ shard: name,
483
+ agent: w.shardConfig.agent,
484
+ sessions: snap,
485
+ failed: failures.has(name),
486
+ failureInfo: failures.get(name) || null,
487
+ integrated: results.has(name),
488
+ });
489
+ }
490
+
491
+ return Object.freeze({
492
+ state,
493
+ totalShards: plan?.shards.length || 0,
494
+ completedShards: results.size,
495
+ failedShards: failures.size,
496
+ workers: workerStatuses,
497
+ mergeOrder: plan?.mergeOrder || [],
498
+ criticalShards: plan?.criticalShards || [],
499
+ locks: lockManager?.snapshot() || [],
500
+ });
501
+ }
502
+
503
+ // ── Public API ──────────────────────────────────────────────
504
+
505
+ /**
506
+ * Launch the swarm from a pre-built plan.
507
+ * @param {SwarmPlan} swarmPlan — from planSwarm()
508
+ * @returns {SwarmStatus}
509
+ */
510
+ function launch(swarmPlan) {
511
+ if (state !== SWARM_STATES.PLANNING) {
512
+ throw new Error(`Cannot launch in state "${state}"`);
513
+ }
514
+
515
+ plan = swarmPlan;
516
+
517
+ // Warn about file conflicts but don't block
518
+ if (plan.conflicts.length > 0) {
519
+ eventLog.append('file_conflicts_warning', { conflicts: plan.conflicts });
520
+ emitter.emit('warning', {
521
+ type: 'file_conflicts',
522
+ conflicts: plan.conflicts,
523
+ });
524
+ }
525
+
526
+ // Initialize lock manager
527
+ lockManager = createSwarmLocks({
528
+ repoRoot: workdir,
529
+ persistPath: join(workdir, '.triflux', 'swarm-locks.json'),
530
+ });
531
+
532
+ setState(SWARM_STATES.LAUNCHING, `${plan.shards.length} shards`);
533
+
534
+ // Launch shards respecting dependency order
535
+ const launched = new Set();
536
+ const pending = new Set(plan.mergeOrder);
537
+
538
+ function launchReady() {
539
+ for (const name of pending) {
540
+ const shard = plan.shards.find((s) => s.name === name);
541
+ if (!shard) continue;
542
+
543
+ // Check all dependencies are launched (not necessarily completed)
544
+ const depsReady = shard.depends.every((d) => launched.has(d));
545
+ if (!depsReady) continue;
546
+
547
+ pending.delete(name);
548
+ launched.add(name);
549
+ launchShard(shard);
550
+
551
+ // Launch redundant worker for critical shards
552
+ if (shard.critical) {
553
+ const redundantShard = {
554
+ ...shard,
555
+ agent: FALLBACK_AGENTS[shard.agent] || shard.agent,
556
+ };
557
+ launchShard(redundantShard, true);
558
+ }
559
+ }
560
+ }
561
+
562
+ launchReady();
563
+
564
+ // Re-check pending on each shard completion (dependency chains)
565
+ emitter.on('shardCompleted', () => {
566
+ if (pending.size > 0) launchReady();
567
+ });
568
+
569
+ setState(SWARM_STATES.RUNNING, `${launched.size} launched, ${pending.size} pending deps`);
570
+
571
+ return getStatus();
572
+ }
573
+
574
+ /**
575
+ * Graceful shutdown — kill all workers and release locks.
576
+ * @param {string} [reason]
577
+ */
578
+ async function shutdown(reason = 'shutdown') {
579
+ eventLog.append('swarm_shutdown', { reason, state });
580
+
581
+ const shutdowns = [];
582
+ for (const [, w] of workers) {
583
+ shutdowns.push(w.conductor.shutdown(reason));
584
+ }
585
+ for (const [, w] of redundantWorkers) {
586
+ shutdowns.push(w.conductor.shutdown(reason));
587
+ }
588
+
589
+ await Promise.allSettled(shutdowns);
590
+
591
+ lockManager?.releaseAll();
592
+ await eventLog.flush();
593
+ await eventLog.close();
594
+
595
+ if (state !== SWARM_STATES.COMPLETED && state !== SWARM_STATES.FAILED) {
596
+ setState(SWARM_STATES.FAILED, reason);
597
+ }
598
+
599
+ emitter.emit('shutdown', { reason });
600
+ }
601
+
602
+ return Object.freeze({
603
+ launch,
604
+ shutdown,
605
+ getStatus,
606
+ validateResult,
607
+ on: emitter.on.bind(emitter),
608
+ off: emitter.off.bind(emitter),
609
+ get state() { return state; },
610
+ get plan() { return plan; },
611
+ get eventLogPath() { return eventLog.filePath; },
612
+ });
613
+ }