@songsid/agend 2.1.4-beta.48 → 2.1.4-beta.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,6 +22,7 @@ import { ClassicChannelManager } from "./classic-channel-manager.js";
22
22
  import type { InstanceState } from "./backend/types.js";
23
23
  import { StormWindow } from "./storm-window.js";
24
24
  import { SpawnGate } from "./spawn-gate.js";
25
+ import { BackendOutageTracker } from "./backend-outage.js";
25
26
  export declare function resolveReplyThreadId(argsThreadId: unknown, instanceConfig?: InstanceConfig): string | undefined;
26
27
  /**
27
28
  * Pure warm-cap victim selection (extracted for testability). Given the current
@@ -73,6 +74,8 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
73
74
  readonly lifecycle: InstanceLifecycle;
74
75
  readonly stormWindow: StormWindow;
75
76
  readonly spawnGate: SpawnGate;
77
+ /** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
78
+ readonly backendOutage: BackendOutageTracker;
76
79
  /** Live view of lifecycle.daemons — used throughout; not deprecated. */
77
80
  get daemons(): Map<string, import("./daemon.js").Daemon>;
78
81
  fleetConfig: FleetConfig | null;
@@ -143,6 +146,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
143
146
  private ipcWaitTails;
144
147
  /** instanceName → restart currently executing; concurrent callers join it. */
145
148
  private restartsInFlight;
149
+ /**
150
+ * Delayed automatic retries for instances whose startup failed. Before this a
151
+ * failed start was logged once and the instance stayed `stopped` until an
152
+ * operator noticed (2026-09-03: 4 kiro instances, all victims of the same
153
+ * backend outage during a post-update herd). instanceName → pending retry.
154
+ */
155
+ private startupRetries;
156
+ /** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
157
+ private explicitStopGeneration;
158
+ /** instanceName → outage hand-off currently executing; a repeat joins it. */
159
+ private handOffsInFlight;
160
+ /** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
161
+ private stopsInFlight;
162
+ /** Aggregation window for the "N instances failed to start" notice. */
163
+ private startupRetryNotices;
164
+ /** Backoff between automatic startup retries; the last step repeats while the backend is down. */
165
+ static readonly STARTUP_RETRY_BACKOFF_MS: number[];
166
+ /** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
167
+ static readonly STARTUP_RETRY_MAX_ATTEMPTS = 6;
168
+ /** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
169
+ static readonly STARTUP_RETRY_STORM_DEFER_MS = 60000;
170
+ static readonly STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1000;
146
171
  private lastInboundMsg;
147
172
  private topicArchiver;
148
173
  controlClient: TmuxControlClient | null;
@@ -368,6 +393,65 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
368
393
  * TODO: per-instance startup timeout (existing issue, not introduced here)
369
394
  */
370
395
  private startInstancesWithConcurrency;
396
+ /**
397
+ * Start from an UNATTENDED path (fleet startup, full restart, config
398
+ * reconcile): nobody is watching the result, so a failure is logged and
399
+ * handed to the delayed automatic retry instead of leaving the instance
400
+ * `stopped` forever. Explicit operator/API starts call startInstance directly
401
+ * and keep their synchronous error. Returns whether the instance is up.
402
+ */
403
+ private startInstanceUnattended;
404
+ /**
405
+ * Schedule attempt `attempt` (0-based) of the automatic startup retry for an
406
+ * instance whose start just failed. Backoff 1 → 5 → 15 min; while the
407
+ * instance's backend is known to be unreachable the 15-min step repeats up to
408
+ * STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
409
+ * retry re-checks the world (still configured, not running, not paused, no
410
+ * tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
411
+ * failed instances comes back at the gate's concurrency, never all at once.
412
+ */
413
+ scheduleStartupRetry(name: string, attempt: number): void;
414
+ /** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
415
+ cancelStartupRetry(name: string): void;
416
+ /** Pending automatic retry, if any (status display / tests). */
417
+ pendingStartupRetry(name: string): {
418
+ attempt: number;
419
+ } | null;
420
+ /**
421
+ * A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
422
+ * stop THAT daemon and schedule the delayed retry. Serialized against the
423
+ * operator paths: an in-flight restart is awaited first (it replaces the
424
+ * daemon itself), the stop is identity-checked so a fresh daemon registered
425
+ * meanwhile is never deleted, and an explicit stop/restart that began during
426
+ * the hand-off (generation bump) owns the outcome — no retry is scheduled
427
+ * behind an operator's back.
428
+ */
429
+ handOffToStartupRetry(name: string, daemon: unknown): Promise<void>;
430
+ private runStartupRetry;
431
+ /** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
432
+ private isConfiguredInstance;
433
+ /**
434
+ * Kind-aware start for the automatic retry: fleet-topic instances come from
435
+ * fleet.yaml, ClassicBot instances exist only in the classic channel manager
436
+ * and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
437
+ * silently dropped them — the retry timer fired and nothing happened).
438
+ */
439
+ private startConfiguredInstance;
440
+ /**
441
+ * Unattended ClassicBot start (fleet startup batch, full-restart batch, the
442
+ * classicBot.yaml reconcile): same contract as startInstanceUnattended —
443
+ * failures are logged and handed to the delayed automatic retry, whose
444
+ * kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
445
+ * whether the instance is up.
446
+ */
447
+ private startClassicInstanceUnattended;
448
+ private backendNameOf;
449
+ /**
450
+ * One fleet-level notice per burst, not one per instance: a post-update herd
451
+ * fails many instances within the same second. Two notices per incident at
452
+ * most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
453
+ */
454
+ private queueStartupRetryNotice;
371
455
  private runnableStartupCount;
372
456
  private configuredStartupInstanceNames;
373
457
  private restartProgressTarget;
@@ -62,6 +62,7 @@ import { RestartProgress } from "./restart-progress.js";
62
62
  import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
63
63
  import { StormWindow } from "./storm-window.js";
64
64
  import { SpawnGate } from "./spawn-gate.js";
65
+ import { BackendOutageTracker } from "./backend-outage.js";
65
66
  import { canUnlockAdvancedTips, DailyTipScheduler, selectTip, visibleTipLevels, } from "./tips.js";
66
67
  import { getTmuxSession } from "./config.js";
67
68
  export function resolveReplyThreadId(argsThreadId, instanceConfig) {
@@ -233,6 +234,8 @@ export class FleetManager {
233
234
  lifecycle;
234
235
  stormWindow;
235
236
  spawnGate;
237
+ /** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
238
+ backendOutage = new BackendOutageTracker();
236
239
  /** Live view of lifecycle.daemons — used throughout; not deprecated. */
237
240
  get daemons() { return this.lifecycle.daemons; }
238
241
  fleetConfig = null;
@@ -314,6 +317,28 @@ export class FleetManager {
314
317
  ipcWaitTails = new Map();
315
318
  /** instanceName → restart currently executing; concurrent callers join it. */
316
319
  restartsInFlight = new Map();
320
+ /**
321
+ * Delayed automatic retries for instances whose startup failed. Before this a
322
+ * failed start was logged once and the instance stayed `stopped` until an
323
+ * operator noticed (2026-09-03: 4 kiro instances, all victims of the same
324
+ * backend outage during a post-update herd). instanceName → pending retry.
325
+ */
326
+ startupRetries = new Map();
327
+ /** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
328
+ explicitStopGeneration = new Map();
329
+ /** instanceName → outage hand-off currently executing; a repeat joins it. */
330
+ handOffsInFlight = new Map();
331
+ /** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
332
+ stopsInFlight = new Map();
333
+ /** Aggregation window for the "N instances failed to start" notice. */
334
+ startupRetryNotices = new Map();
335
+ /** Backoff between automatic startup retries; the last step repeats while the backend is down. */
336
+ static STARTUP_RETRY_BACKOFF_MS = [60_000, 5 * 60_000, 15 * 60_000];
337
+ /** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
338
+ static STARTUP_RETRY_MAX_ATTEMPTS = 6;
339
+ /** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
340
+ static STARTUP_RETRY_STORM_DEFER_MS = 60_000;
341
+ static STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1_000;
317
342
  // Last user message delivered to each instance — used to react ✅ on completion.
318
343
  lastInboundMsg = new Map();
319
344
  topicArchiver;
@@ -1325,6 +1350,9 @@ export class FleetManager {
1325
1350
  * across a fleet restart.
1326
1351
  */
1327
1352
  resumePaused = false) {
1353
+ // Any start supersedes a pending automatic retry (it would otherwise fire
1354
+ // into a running instance — harmless, but noisy — or race this start).
1355
+ this.cancelStartupRetry(name);
1328
1356
  if (resumePaused && this.lifecycle.isPaused(name)) {
1329
1357
  await this.lifecycle.wake(name, 30_000);
1330
1358
  // A successful wake clears the persisted pause marker and produces a
@@ -1432,15 +1460,231 @@ export class FleetManager {
1432
1460
  workingDirectory: config.working_directory,
1433
1461
  reason: "startup",
1434
1462
  }, async () => {
1435
- try {
1436
- await this.startInstance(name, config, topicMode);
1437
- if (this.daemons.has(name))
1438
- onReady?.(name);
1463
+ if (await this.startInstanceUnattended(name, config, topicMode, "instance"))
1464
+ onReady?.(name);
1465
+ })));
1466
+ }
1467
+ /**
1468
+ * Start from an UNATTENDED path (fleet startup, full restart, config
1469
+ * reconcile): nobody is watching the result, so a failure is logged and
1470
+ * handed to the delayed automatic retry instead of leaving the instance
1471
+ * `stopped` forever. Explicit operator/API starts call startInstance directly
1472
+ * and keep their synchronous error. Returns whether the instance is up.
1473
+ */
1474
+ async startInstanceUnattended(name, config, topicMode, what) {
1475
+ try {
1476
+ await this.startInstance(name, config, topicMode);
1477
+ return this.daemons.has(name);
1478
+ }
1479
+ catch (err) {
1480
+ this.logger.error({ err, name }, `Failed to start ${what}`);
1481
+ this.scheduleStartupRetry(name, 0);
1482
+ return false;
1483
+ }
1484
+ }
1485
+ // ── Delayed automatic startup retries ──────────────────────────────────
1486
+ /**
1487
+ * Schedule attempt `attempt` (0-based) of the automatic startup retry for an
1488
+ * instance whose start just failed. Backoff 1 → 5 → 15 min; while the
1489
+ * instance's backend is known to be unreachable the 15-min step repeats up to
1490
+ * STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
1491
+ * retry re-checks the world (still configured, not running, not paused, no
1492
+ * tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
1493
+ * failed instances comes back at the gate's concurrency, never all at once.
1494
+ */
1495
+ scheduleStartupRetry(name, attempt) {
1496
+ if (this.shuttingDown)
1497
+ return;
1498
+ if (this.startupRetries.has(name))
1499
+ return;
1500
+ const backoff = FleetManager.STARTUP_RETRY_BACKOFF_MS;
1501
+ const outage = this.backendOutage.isActive(this.backendNameOf(name));
1502
+ const exhausted = attempt >= backoff.length && !(outage && attempt < FleetManager.STARTUP_RETRY_MAX_ATTEMPTS);
1503
+ if (attempt >= FleetManager.STARTUP_RETRY_MAX_ATTEMPTS || exhausted) {
1504
+ this.logger.error({ name, attempts: attempt }, "Giving up automatic startup retries");
1505
+ this.queueStartupRetryNotice("gave_up", name, 0);
1506
+ return;
1507
+ }
1508
+ const delayMs = backoff[Math.min(attempt, backoff.length - 1)];
1509
+ const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, delayMs);
1510
+ timer.unref?.();
1511
+ this.startupRetries.set(name, { attempt, timer });
1512
+ this.logger.warn({ name, attempt: attempt + 1, delayMs, backendOutage: outage }, "Startup failed — automatic retry scheduled");
1513
+ if (attempt === 0)
1514
+ this.queueStartupRetryNotice("scheduled", name, delayMs);
1515
+ }
1516
+ /** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
1517
+ cancelStartupRetry(name) {
1518
+ const pending = this.startupRetries.get(name);
1519
+ if (!pending)
1520
+ return;
1521
+ clearTimeout(pending.timer);
1522
+ this.startupRetries.delete(name);
1523
+ this.logger.info({ name }, "Pending automatic startup retry cancelled");
1524
+ }
1525
+ /** Pending automatic retry, if any (status display / tests). */
1526
+ pendingStartupRetry(name) {
1527
+ const pending = this.startupRetries.get(name);
1528
+ return pending ? { attempt: pending.attempt } : null;
1529
+ }
1530
+ /**
1531
+ * A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
1532
+ * stop THAT daemon and schedule the delayed retry. Serialized against the
1533
+ * operator paths: an in-flight restart is awaited first (it replaces the
1534
+ * daemon itself), the stop is identity-checked so a fresh daemon registered
1535
+ * meanwhile is never deleted, and an explicit stop/restart that began during
1536
+ * the hand-off (generation bump) owns the outcome — no retry is scheduled
1537
+ * behind an operator's back.
1538
+ */
1539
+ async handOffToStartupRetry(name, daemon) {
1540
+ const joined = this.handOffsInFlight.get(name);
1541
+ if (joined)
1542
+ return joined;
1543
+ const run = (async () => {
1544
+ // Let any operator restart/stop that is already executing finish first:
1545
+ // a restart replaces the daemon itself, a stop removes it — either way
1546
+ // the identity check below then sees "not ours any more" and we do
1547
+ // nothing. Loop: one operation can be chained behind another.
1548
+ for (;;) {
1549
+ const inFlight = this.restartsInFlight.get(name) ?? this.stopsInFlight.get(name);
1550
+ if (!inFlight)
1551
+ break;
1552
+ await inFlight.catch(() => { });
1439
1553
  }
1440
- catch (err) {
1441
- this.logger.error({ err, name }, "Failed to start instance");
1554
+ const stopGen = this.explicitStopGeneration.get(name) ?? 0;
1555
+ if (this.lifecycle.daemons.get(name) !== daemon)
1556
+ return; // already replaced or stopped
1557
+ await this.lifecycle.stopIfCurrent(name, daemon);
1558
+ if (this.shuttingDown)
1559
+ return;
1560
+ if ((this.explicitStopGeneration.get(name) ?? 0) !== stopGen) {
1561
+ this.logger.info({ name }, "Outage hand-off superseded by an explicit stop/restart — no automatic retry");
1562
+ return;
1442
1563
  }
1443
- })));
1564
+ if (this.lifecycle.daemons.has(name) || this.lifecycle.isPaused(name))
1565
+ return;
1566
+ this.scheduleStartupRetry(name, 0);
1567
+ })().finally(() => this.handOffsInFlight.delete(name));
1568
+ this.handOffsInFlight.set(name, run);
1569
+ return run;
1570
+ }
1571
+ async runStartupRetry(name, attempt) {
1572
+ this.startupRetries.delete(name);
1573
+ if (this.shuttingDown)
1574
+ return;
1575
+ if (!this.isConfiguredInstance(name))
1576
+ return; // removed from fleet.yaml / classic channel meanwhile
1577
+ if (this.lifecycle.isPaused(name) || this.daemons.has(name))
1578
+ return; // operator acted meanwhile
1579
+ if (this.stormWindow.isSpawnBlocked()) {
1580
+ // The tmux server is being restarted; joining that herd is what we are
1581
+ // trying to avoid. Same attempt again after the storm's own recovery.
1582
+ const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, FleetManager.STARTUP_RETRY_STORM_DEFER_MS);
1583
+ timer.unref?.();
1584
+ this.startupRetries.set(name, { attempt, timer });
1585
+ this.logger.info({ name, attempt: attempt + 1 }, "Startup retry deferred — tmux storm window is blocking spawns");
1586
+ return;
1587
+ }
1588
+ const topicMode = this.fleetConfig?.channel?.mode === "topic"
1589
+ || !!this.fleetConfig?.channels?.some(channel => channel.mode === "topic");
1590
+ try {
1591
+ await this.spawnGate.run({
1592
+ instanceName: name,
1593
+ workingDirectory: this.fleetConfig?.instances[name]?.working_directory || this.getInstanceDir(name),
1594
+ reason: "recovery",
1595
+ }, () => this.startConfiguredInstance(name, topicMode));
1596
+ if (this.daemons.has(name)) {
1597
+ this.logger.info({ name, attempt: attempt + 1 }, "Automatic startup retry succeeded");
1598
+ }
1599
+ }
1600
+ catch (err) {
1601
+ this.logger.error({ err, name, attempt: attempt + 1 }, "Automatic startup retry failed");
1602
+ this.scheduleStartupRetry(name, attempt + 1);
1603
+ }
1604
+ }
1605
+ /** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
1606
+ isConfiguredInstance(name) {
1607
+ if (this.fleetConfig?.instances[name])
1608
+ return true;
1609
+ return !!this.classicChannels?.getAll().some(channel => channel.instanceName === name);
1610
+ }
1611
+ /**
1612
+ * Kind-aware start for the automatic retry: fleet-topic instances come from
1613
+ * fleet.yaml, ClassicBot instances exist only in the classic channel manager
1614
+ * and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
1615
+ * silently dropped them — the retry timer fired and nothing happened).
1616
+ */
1617
+ async startConfiguredInstance(name, topicMode) {
1618
+ const config = this.fleetConfig?.instances[name];
1619
+ if (config) {
1620
+ await this.startInstance(name, config, topicMode);
1621
+ return;
1622
+ }
1623
+ const channel = this.classicChannels?.getAll().find(item => item.instanceName === name);
1624
+ if (!channel || !this.classicChannels)
1625
+ throw new Error(`Instance '${name}' is no longer configured`);
1626
+ await this.startClassicInstance(name, this.classicChannels.getBackendByInstance(name, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(channel.channelId, channel.adapterId), this.classicChannels.getModel(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
1627
+ }
1628
+ /**
1629
+ * Unattended ClassicBot start (fleet startup batch, full-restart batch, the
1630
+ * classicBot.yaml reconcile): same contract as startInstanceUnattended —
1631
+ * failures are logged and handed to the delayed automatic retry, whose
1632
+ * kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
1633
+ * whether the instance is up.
1634
+ */
1635
+ async startClassicInstanceUnattended(ch, what) {
1636
+ try {
1637
+ await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
1638
+ return this.daemons.has(ch.instanceName);
1639
+ }
1640
+ catch (err) {
1641
+ this.logger.warn({ err, instanceName: ch.instanceName }, `Failed to start ${what}`);
1642
+ this.scheduleStartupRetry(ch.instanceName, 0);
1643
+ return false;
1644
+ }
1645
+ }
1646
+ backendNameOf(name) {
1647
+ const fleetDefault = this.fleetConfig?.defaults?.backend;
1648
+ const configured = this.fleetConfig?.instances[name]?.backend;
1649
+ if (configured)
1650
+ return configured;
1651
+ // ClassicBot channels pick their own backend; the fleet default is only the fallback.
1652
+ if (this.classicChannels?.getAll().some(channel => channel.instanceName === name)) {
1653
+ return this.classicChannels.getBackendByInstance(name, fleetDefault);
1654
+ }
1655
+ return fleetDefault ?? "claude-code";
1656
+ }
1657
+ /**
1658
+ * One fleet-level notice per burst, not one per instance: a post-update herd
1659
+ * fails many instances within the same second. Two notices per incident at
1660
+ * most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
1661
+ */
1662
+ queueStartupRetryNotice(kind, name, delayMs) {
1663
+ const pending = this.startupRetryNotices.get(kind);
1664
+ if (pending) {
1665
+ if (!pending.names.includes(name))
1666
+ pending.names.push(name);
1667
+ return;
1668
+ }
1669
+ const timer = setTimeout(() => {
1670
+ const entry = this.startupRetryNotices.get(kind);
1671
+ this.startupRetryNotices.delete(kind);
1672
+ if (!entry)
1673
+ return;
1674
+ const list = entry.names.join(", ");
1675
+ if (kind === "scheduled") {
1676
+ let text = t("fleet.startup_retry_scheduled", entry.names.length, list, this.formatStormDelay(entry.delayMs));
1677
+ const downBackends = [...new Set(entry.names.map(n => this.backendNameOf(n)))].filter(b => this.backendOutage.isActive(b));
1678
+ if (downBackends.length)
1679
+ text += `\n${t("fleet.startup_retry_outage", downBackends.join(", "))}`;
1680
+ this.notifyFleetError(text);
1681
+ }
1682
+ else {
1683
+ this.notifyFleetError(t("fleet.startup_retry_gave_up", entry.names.length, list));
1684
+ }
1685
+ }, FleetManager.STARTUP_RETRY_NOTICE_AGGREGATE_MS);
1686
+ timer.unref?.();
1687
+ this.startupRetryNotices.set(kind, { names: [name], delayMs, timer });
1444
1688
  }
1445
1689
  runnableStartupCount(fleet, includeClassic) {
1446
1690
  const names = this.configuredStartupInstanceNames(fleet, includeClassic);
@@ -1475,6 +1719,8 @@ export class FleetManager {
1475
1719
  };
1476
1720
  }
1477
1721
  async stopInstance(name) {
1722
+ this.explicitStopGeneration.set(name, (this.explicitStopGeneration.get(name) ?? 0) + 1);
1723
+ this.cancelStartupRetry(name);
1478
1724
  this.failoverActive.delete(name);
1479
1725
  this.cancelIdleButtonRetirement(name);
1480
1726
  this.instanceStateCache.delete(name);
@@ -1486,12 +1732,20 @@ export class FleetManager {
1486
1732
  // interactive-prompt / clean-exit events whose handlers post fresh prompts
1487
1733
  // while the stop is still awaiting (the TOCTOU sol's review called out).
1488
1734
  this.clearNoncePromptsForInstance(name);
1489
- try {
1490
- return await this.lifecycle.stop(name);
1491
- }
1492
- finally {
1493
- this.clearNoncePromptsForInstance(name);
1494
- }
1735
+ // Published so the outage hand-off can wait for an explicit stop that began
1736
+ // BEFORE it (the generation fence alone only catches stops that begin after
1737
+ // the hand-off snapshotted it).
1738
+ const run = (async () => {
1739
+ try {
1740
+ await this.lifecycle.stop(name);
1741
+ }
1742
+ finally {
1743
+ this.clearNoncePromptsForInstance(name);
1744
+ }
1745
+ })().finally(() => { if (this.stopsInFlight.get(name) === run)
1746
+ this.stopsInFlight.delete(name); });
1747
+ this.stopsInFlight.set(name, run);
1748
+ return run;
1495
1749
  }
1496
1750
  /** Restart a single instance, reloading fleet.yaml first to pick up config changes. */
1497
1751
  async restartSingleInstance(name, opts) {
@@ -1781,7 +2035,10 @@ export class FleetManager {
1781
2035
  await this.stopInstance(ch.instanceName).catch(() => { });
1782
2036
  // Small delay to let tmux window clean up
1783
2037
  await new Promise(r => setTimeout(r, 2000));
1784
- await this.startClassicInstance(ch.instanceName, newBackend, this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), newModel, newAutoPause).catch(err => this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to restart classic instance"));
2038
+ // The manager already holds the new backend/model/auto-pause; the
2039
+ // unattended helper reads them from it and schedules the delayed
2040
+ // retry on failure like every other unattended start.
2041
+ await this.startClassicInstanceUnattended(ch, "classic instance after backend/model change");
1785
2042
  }
1786
2043
  }
1787
2044
  }
@@ -1968,6 +2225,9 @@ export class FleetManager {
1968
2225
  }
1969
2226
  catch (err) {
1970
2227
  this.logger.error({ err, name }, "Failed to start general instance");
2228
+ // General is the most important instance to bring back: it also gets
2229
+ // the delayed automatic retry (the topic notice below still goes out).
2230
+ this.scheduleStartupRetry(name, 0);
1971
2231
  const errorMsg = err instanceof Error ? err.message : String(err);
1972
2232
  const topicId = cfg.topic_id ? String(cfg.topic_id) : undefined;
1973
2233
  if (this.adapter && topicId) {
@@ -2087,14 +2347,8 @@ export class FleetManager {
2087
2347
  while (idx < channels.length) {
2088
2348
  const batch = channels.slice(idx, idx + concurrency);
2089
2349
  await Promise.allSettled(batch.map(async (ch) => {
2090
- try {
2091
- await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, fleetBackend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
2092
- if (this.daemons.has(ch.instanceName))
2093
- startupProgress.markReady();
2094
- }
2095
- catch (err) {
2096
- this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
2097
- }
2350
+ if (await this.startClassicInstanceUnattended(ch, "classic instance"))
2351
+ startupProgress.markReady();
2098
2352
  }));
2099
2353
  idx += concurrency;
2100
2354
  }
@@ -8672,6 +8926,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
8672
8926
  // make `agend stop` wait forever for its own queue.
8673
8927
  this.stormWindow.shutdown();
8674
8928
  this.spawnGate.shutdown();
8929
+ for (const pending of this.startupRetries.values())
8930
+ clearTimeout(pending.timer);
8931
+ this.startupRetries.clear();
8932
+ for (const pending of this.startupRetryNotices.values())
8933
+ clearTimeout(pending.timer);
8934
+ this.startupRetryNotices.clear();
8675
8935
  if (this.stormOpenNotifyTimer) {
8676
8936
  clearTimeout(this.stormOpenNotifyTimer);
8677
8937
  this.stormOpenNotifyTimer = null;
@@ -8997,7 +9257,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
8997
9257
  if (!this.daemons.has(name)) {
8998
9258
  // New instance — startInstance already calls connectIpcToInstance
8999
9259
  this.logger.info({ name }, "New instance in config — starting");
9000
- await this.startInstance(name, config, topicMode).catch(err => this.logger.error({ err, name }, "Failed to start new instance"));
9260
+ await this.startInstanceUnattended(name, config, topicMode, "new instance");
9001
9261
  }
9002
9262
  else if (oldConfig?.instances[name]) {
9003
9263
  const daemon = this.daemons.get(name);
@@ -9008,7 +9268,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
9008
9268
  if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
9009
9269
  this.logger.info({ name }, "Instance config changed — restarting");
9010
9270
  await this.stopInstance(name).catch(() => { });
9011
- await this.startInstance(name, config, topicMode).catch(err => this.logger.error({ err, name }, "Failed to restart modified instance"));
9271
+ await this.startInstanceUnattended(name, config, topicMode, "modified instance");
9012
9272
  }
9013
9273
  else if (!isDeepStrictEqual(oldParts.hot, newParts.hot)) {
9014
9274
  const update = hotConfigUpdate(config);
@@ -9107,14 +9367,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
9107
9367
  const restartGenerals = restartEntries.filter(([_, cfg]) => cfg.general_topic);
9108
9368
  const restartOthers = restartEntries.filter(([_, cfg]) => !cfg.general_topic);
9109
9369
  for (const [name, cfg] of restartGenerals) {
9110
- try {
9111
- await this.startInstance(name, cfg, topicMode);
9112
- if (this.daemons.has(name))
9113
- restartProgress.markReady();
9114
- }
9115
- catch (err) {
9116
- this.logger.error({ err, name }, "Failed to start general instance");
9117
- }
9370
+ if (await this.startInstanceUnattended(name, cfg, topicMode, "general instance"))
9371
+ restartProgress.markReady();
9118
9372
  }
9119
9373
  // General is ready again; now its topic can own the live progress message.
9120
9374
  await restartProgress.start(progressTarget);
@@ -9135,14 +9389,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
9135
9389
  while (idx < channels.length) {
9136
9390
  const batch = channels.slice(idx, idx + concurrency);
9137
9391
  await Promise.allSettled(batch.map(async (ch) => {
9138
- try {
9139
- await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, fleetBackend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
9140
- if (this.daemons.has(ch.instanceName))
9141
- restartProgress.markReady();
9142
- }
9143
- catch (err) {
9144
- this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
9145
- }
9392
+ if (await this.startClassicInstanceUnattended(ch, "classic instance"))
9393
+ restartProgress.markReady();
9146
9394
  }));
9147
9395
  idx += concurrency;
9148
9396
  }