@songsid/agend 2.1.4-beta.47 → 2.1.4-beta.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,6 +62,7 @@ import { RestartProgress } from "./restart-progress.js";
62
62
  import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
63
63
  import { StormWindow } from "./storm-window.js";
64
64
  import { SpawnGate } from "./spawn-gate.js";
65
+ import { BackendOutageTracker } from "./backend-outage.js";
65
66
  import { canUnlockAdvancedTips, DailyTipScheduler, selectTip, visibleTipLevels, } from "./tips.js";
66
67
  import { getTmuxSession } from "./config.js";
67
68
  export function resolveReplyThreadId(argsThreadId, instanceConfig) {
@@ -216,6 +217,7 @@ const CLEAR_CONFIRM_CALLBACK_PREFIX = "clear-confirm:";
216
217
  const TIP_DISMISS_CALLBACK_PREFIX = "tip-dismiss:";
217
218
  const TIP_UNLOCK_CALLBACK_PREFIX = "tip-unlock:";
218
219
  const LOGIN_CALLBACK_PREFIX = "login:";
220
+ const INSTALL_CALLBACK_PREFIX = "install-select:";
219
221
  const LOGIN_MENU_CALLBACK_PREFIX = "login-menu:";
220
222
  const LOGIN_CONFIRM_CALLBACK_PREFIX = "login-confirm:";
221
223
  const INSTALL_LOGIN_CALLBACK_PREFIX = "install-login:";
@@ -232,6 +234,8 @@ export class FleetManager {
232
234
  lifecycle;
233
235
  stormWindow;
234
236
  spawnGate;
237
+ /** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
238
+ backendOutage = new BackendOutageTracker();
235
239
  /** Live view of lifecycle.daemons — used throughout; not deprecated. */
236
240
  get daemons() { return this.lifecycle.daemons; }
237
241
  fleetConfig = null;
@@ -313,6 +317,28 @@ export class FleetManager {
313
317
  ipcWaitTails = new Map();
314
318
  /** instanceName → restart currently executing; concurrent callers join it. */
315
319
  restartsInFlight = new Map();
320
+ /**
321
+ * Delayed automatic retries for instances whose startup failed. Before this a
322
+ * failed start was logged once and the instance stayed `stopped` until an
323
+ * operator noticed (2026-09-03: 4 kiro instances, all victims of the same
324
+ * backend outage during a post-update herd). instanceName → pending retry.
325
+ */
326
+ startupRetries = new Map();
327
+ /** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
328
+ explicitStopGeneration = new Map();
329
+ /** instanceName → outage hand-off currently executing; a repeat joins it. */
330
+ handOffsInFlight = new Map();
331
+ /** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
332
+ stopsInFlight = new Map();
333
+ /** Aggregation window for the "N instances failed to start" notice. */
334
+ startupRetryNotices = new Map();
335
+ /** Backoff between automatic startup retries; the last step repeats while the backend is down. */
336
+ static STARTUP_RETRY_BACKOFF_MS = [60_000, 5 * 60_000, 15 * 60_000];
337
+ /** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
338
+ static STARTUP_RETRY_MAX_ATTEMPTS = 6;
339
+ /** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
340
+ static STARTUP_RETRY_STORM_DEFER_MS = 60_000;
341
+ static STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1_000;
316
342
  // Last user message delivered to each instance — used to react ✅ on completion.
317
343
  lastInboundMsg = new Map();
318
344
  topicArchiver;
@@ -1324,6 +1350,9 @@ export class FleetManager {
1324
1350
  * across a fleet restart.
1325
1351
  */
1326
1352
  resumePaused = false) {
1353
+ // Any start supersedes a pending automatic retry (it would otherwise fire
1354
+ // into a running instance — harmless, but noisy — or race this start).
1355
+ this.cancelStartupRetry(name);
1327
1356
  if (resumePaused && this.lifecycle.isPaused(name)) {
1328
1357
  await this.lifecycle.wake(name, 30_000);
1329
1358
  // A successful wake clears the persisted pause marker and produces a
@@ -1431,15 +1460,231 @@ export class FleetManager {
1431
1460
  workingDirectory: config.working_directory,
1432
1461
  reason: "startup",
1433
1462
  }, async () => {
1434
- try {
1435
- await this.startInstance(name, config, topicMode);
1436
- if (this.daemons.has(name))
1437
- onReady?.(name);
1463
+ if (await this.startInstanceUnattended(name, config, topicMode, "instance"))
1464
+ onReady?.(name);
1465
+ })));
1466
+ }
1467
+ /**
1468
+ * Start from an UNATTENDED path (fleet startup, full restart, config
1469
+ * reconcile): nobody is watching the result, so a failure is logged and
1470
+ * handed to the delayed automatic retry instead of leaving the instance
1471
+ * `stopped` forever. Explicit operator/API starts call startInstance directly
1472
+ * and keep their synchronous error. Returns whether the instance is up.
1473
+ */
1474
+ async startInstanceUnattended(name, config, topicMode, what) {
1475
+ try {
1476
+ await this.startInstance(name, config, topicMode);
1477
+ return this.daemons.has(name);
1478
+ }
1479
+ catch (err) {
1480
+ this.logger.error({ err, name }, `Failed to start ${what}`);
1481
+ this.scheduleStartupRetry(name, 0);
1482
+ return false;
1483
+ }
1484
+ }
1485
+ // ── Delayed automatic startup retries ──────────────────────────────────
1486
+ /**
1487
+ * Schedule attempt `attempt` (0-based) of the automatic startup retry for an
1488
+ * instance whose start just failed. Backoff 1 → 5 → 15 min; while the
1489
+ * instance's backend is known to be unreachable the 15-min step repeats up to
1490
+ * STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
1491
+ * retry re-checks the world (still configured, not running, not paused, no
1492
+ * tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
1493
+ * failed instances comes back at the gate's concurrency, never all at once.
1494
+ */
1495
+ scheduleStartupRetry(name, attempt) {
1496
+ if (this.shuttingDown)
1497
+ return;
1498
+ if (this.startupRetries.has(name))
1499
+ return;
1500
+ const backoff = FleetManager.STARTUP_RETRY_BACKOFF_MS;
1501
+ const outage = this.backendOutage.isActive(this.backendNameOf(name));
1502
+ const exhausted = attempt >= backoff.length && !(outage && attempt < FleetManager.STARTUP_RETRY_MAX_ATTEMPTS);
1503
+ if (attempt >= FleetManager.STARTUP_RETRY_MAX_ATTEMPTS || exhausted) {
1504
+ this.logger.error({ name, attempts: attempt }, "Giving up automatic startup retries");
1505
+ this.queueStartupRetryNotice("gave_up", name, 0);
1506
+ return;
1507
+ }
1508
+ const delayMs = backoff[Math.min(attempt, backoff.length - 1)];
1509
+ const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, delayMs);
1510
+ timer.unref?.();
1511
+ this.startupRetries.set(name, { attempt, timer });
1512
+ this.logger.warn({ name, attempt: attempt + 1, delayMs, backendOutage: outage }, "Startup failed — automatic retry scheduled");
1513
+ if (attempt === 0)
1514
+ this.queueStartupRetryNotice("scheduled", name, delayMs);
1515
+ }
1516
+ /** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
1517
+ cancelStartupRetry(name) {
1518
+ const pending = this.startupRetries.get(name);
1519
+ if (!pending)
1520
+ return;
1521
+ clearTimeout(pending.timer);
1522
+ this.startupRetries.delete(name);
1523
+ this.logger.info({ name }, "Pending automatic startup retry cancelled");
1524
+ }
1525
+ /** Pending automatic retry, if any (status display / tests). */
1526
+ pendingStartupRetry(name) {
1527
+ const pending = this.startupRetries.get(name);
1528
+ return pending ? { attempt: pending.attempt } : null;
1529
+ }
1530
+ /**
1531
+ * A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
1532
+ * stop THAT daemon and schedule the delayed retry. Serialized against the
1533
+ * operator paths: an in-flight restart is awaited first (it replaces the
1534
+ * daemon itself), the stop is identity-checked so a fresh daemon registered
1535
+ * meanwhile is never deleted, and an explicit stop/restart that began during
1536
+ * the hand-off (generation bump) owns the outcome — no retry is scheduled
1537
+ * behind an operator's back.
1538
+ */
1539
+ async handOffToStartupRetry(name, daemon) {
1540
+ const joined = this.handOffsInFlight.get(name);
1541
+ if (joined)
1542
+ return joined;
1543
+ const run = (async () => {
1544
+ // Let any operator restart/stop that is already executing finish first:
1545
+ // a restart replaces the daemon itself, a stop removes it — either way
1546
+ // the identity check below then sees "not ours any more" and we do
1547
+ // nothing. Loop: one operation can be chained behind another.
1548
+ for (;;) {
1549
+ const inFlight = this.restartsInFlight.get(name) ?? this.stopsInFlight.get(name);
1550
+ if (!inFlight)
1551
+ break;
1552
+ await inFlight.catch(() => { });
1438
1553
  }
1439
- catch (err) {
1440
- this.logger.error({ err, name }, "Failed to start instance");
1554
+ const stopGen = this.explicitStopGeneration.get(name) ?? 0;
1555
+ if (this.lifecycle.daemons.get(name) !== daemon)
1556
+ return; // already replaced or stopped
1557
+ await this.lifecycle.stopIfCurrent(name, daemon);
1558
+ if (this.shuttingDown)
1559
+ return;
1560
+ if ((this.explicitStopGeneration.get(name) ?? 0) !== stopGen) {
1561
+ this.logger.info({ name }, "Outage hand-off superseded by an explicit stop/restart — no automatic retry");
1562
+ return;
1441
1563
  }
1442
- })));
1564
+ if (this.lifecycle.daemons.has(name) || this.lifecycle.isPaused(name))
1565
+ return;
1566
+ this.scheduleStartupRetry(name, 0);
1567
+ })().finally(() => this.handOffsInFlight.delete(name));
1568
+ this.handOffsInFlight.set(name, run);
1569
+ return run;
1570
+ }
1571
+ async runStartupRetry(name, attempt) {
1572
+ this.startupRetries.delete(name);
1573
+ if (this.shuttingDown)
1574
+ return;
1575
+ if (!this.isConfiguredInstance(name))
1576
+ return; // removed from fleet.yaml / classic channel meanwhile
1577
+ if (this.lifecycle.isPaused(name) || this.daemons.has(name))
1578
+ return; // operator acted meanwhile
1579
+ if (this.stormWindow.isSpawnBlocked()) {
1580
+ // The tmux server is being restarted; joining that herd is what we are
1581
+ // trying to avoid. Same attempt again after the storm's own recovery.
1582
+ const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, FleetManager.STARTUP_RETRY_STORM_DEFER_MS);
1583
+ timer.unref?.();
1584
+ this.startupRetries.set(name, { attempt, timer });
1585
+ this.logger.info({ name, attempt: attempt + 1 }, "Startup retry deferred — tmux storm window is blocking spawns");
1586
+ return;
1587
+ }
1588
+ const topicMode = this.fleetConfig?.channel?.mode === "topic"
1589
+ || !!this.fleetConfig?.channels?.some(channel => channel.mode === "topic");
1590
+ try {
1591
+ await this.spawnGate.run({
1592
+ instanceName: name,
1593
+ workingDirectory: this.fleetConfig?.instances[name]?.working_directory || this.getInstanceDir(name),
1594
+ reason: "recovery",
1595
+ }, () => this.startConfiguredInstance(name, topicMode));
1596
+ if (this.daemons.has(name)) {
1597
+ this.logger.info({ name, attempt: attempt + 1 }, "Automatic startup retry succeeded");
1598
+ }
1599
+ }
1600
+ catch (err) {
1601
+ this.logger.error({ err, name, attempt: attempt + 1 }, "Automatic startup retry failed");
1602
+ this.scheduleStartupRetry(name, attempt + 1);
1603
+ }
1604
+ }
1605
+ /** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
1606
+ isConfiguredInstance(name) {
1607
+ if (this.fleetConfig?.instances[name])
1608
+ return true;
1609
+ return !!this.classicChannels?.getAll().some(channel => channel.instanceName === name);
1610
+ }
1611
+ /**
1612
+ * Kind-aware start for the automatic retry: fleet-topic instances come from
1613
+ * fleet.yaml, ClassicBot instances exist only in the classic channel manager
1614
+ * and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
1615
+ * silently dropped them — the retry timer fired and nothing happened).
1616
+ */
1617
+ async startConfiguredInstance(name, topicMode) {
1618
+ const config = this.fleetConfig?.instances[name];
1619
+ if (config) {
1620
+ await this.startInstance(name, config, topicMode);
1621
+ return;
1622
+ }
1623
+ const channel = this.classicChannels?.getAll().find(item => item.instanceName === name);
1624
+ if (!channel || !this.classicChannels)
1625
+ throw new Error(`Instance '${name}' is no longer configured`);
1626
+ await this.startClassicInstance(name, this.classicChannels.getBackendByInstance(name, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(channel.channelId, channel.adapterId), this.classicChannels.getModel(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
1627
+ }
1628
+ /**
1629
+ * Unattended ClassicBot start (fleet startup batch, full-restart batch, the
1630
+ * classicBot.yaml reconcile): same contract as startInstanceUnattended —
1631
+ * failures are logged and handed to the delayed automatic retry, whose
1632
+ * kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
1633
+ * whether the instance is up.
1634
+ */
1635
+ async startClassicInstanceUnattended(ch, what) {
1636
+ try {
1637
+ await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
1638
+ return this.daemons.has(ch.instanceName);
1639
+ }
1640
+ catch (err) {
1641
+ this.logger.warn({ err, instanceName: ch.instanceName }, `Failed to start ${what}`);
1642
+ this.scheduleStartupRetry(ch.instanceName, 0);
1643
+ return false;
1644
+ }
1645
+ }
1646
+ backendNameOf(name) {
1647
+ const fleetDefault = this.fleetConfig?.defaults?.backend;
1648
+ const configured = this.fleetConfig?.instances[name]?.backend;
1649
+ if (configured)
1650
+ return configured;
1651
+ // ClassicBot channels pick their own backend; the fleet default is only the fallback.
1652
+ if (this.classicChannels?.getAll().some(channel => channel.instanceName === name)) {
1653
+ return this.classicChannels.getBackendByInstance(name, fleetDefault);
1654
+ }
1655
+ return fleetDefault ?? "claude-code";
1656
+ }
1657
+ /**
1658
+ * One fleet-level notice per burst, not one per instance: a post-update herd
1659
+ * fails many instances within the same second. Two notices per incident at
1660
+ * most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
1661
+ */
1662
+ queueStartupRetryNotice(kind, name, delayMs) {
1663
+ const pending = this.startupRetryNotices.get(kind);
1664
+ if (pending) {
1665
+ if (!pending.names.includes(name))
1666
+ pending.names.push(name);
1667
+ return;
1668
+ }
1669
+ const timer = setTimeout(() => {
1670
+ const entry = this.startupRetryNotices.get(kind);
1671
+ this.startupRetryNotices.delete(kind);
1672
+ if (!entry)
1673
+ return;
1674
+ const list = entry.names.join(", ");
1675
+ if (kind === "scheduled") {
1676
+ let text = t("fleet.startup_retry_scheduled", entry.names.length, list, this.formatStormDelay(entry.delayMs));
1677
+ const downBackends = [...new Set(entry.names.map(n => this.backendNameOf(n)))].filter(b => this.backendOutage.isActive(b));
1678
+ if (downBackends.length)
1679
+ text += `\n${t("fleet.startup_retry_outage", downBackends.join(", "))}`;
1680
+ this.notifyFleetError(text);
1681
+ }
1682
+ else {
1683
+ this.notifyFleetError(t("fleet.startup_retry_gave_up", entry.names.length, list));
1684
+ }
1685
+ }, FleetManager.STARTUP_RETRY_NOTICE_AGGREGATE_MS);
1686
+ timer.unref?.();
1687
+ this.startupRetryNotices.set(kind, { names: [name], delayMs, timer });
1443
1688
  }
1444
1689
  runnableStartupCount(fleet, includeClassic) {
1445
1690
  const names = this.configuredStartupInstanceNames(fleet, includeClassic);
@@ -1474,6 +1719,8 @@ export class FleetManager {
1474
1719
  };
1475
1720
  }
1476
1721
  async stopInstance(name) {
1722
+ this.explicitStopGeneration.set(name, (this.explicitStopGeneration.get(name) ?? 0) + 1);
1723
+ this.cancelStartupRetry(name);
1477
1724
  this.failoverActive.delete(name);
1478
1725
  this.cancelIdleButtonRetirement(name);
1479
1726
  this.instanceStateCache.delete(name);
@@ -1485,12 +1732,20 @@ export class FleetManager {
1485
1732
  // interactive-prompt / clean-exit events whose handlers post fresh prompts
1486
1733
  // while the stop is still awaiting (the TOCTOU sol's review called out).
1487
1734
  this.clearNoncePromptsForInstance(name);
1488
- try {
1489
- return await this.lifecycle.stop(name);
1490
- }
1491
- finally {
1492
- this.clearNoncePromptsForInstance(name);
1493
- }
1735
+ // Published so the outage hand-off can wait for an explicit stop that began
1736
+ // BEFORE it (the generation fence alone only catches stops that begin after
1737
+ // the hand-off snapshotted it).
1738
+ const run = (async () => {
1739
+ try {
1740
+ await this.lifecycle.stop(name);
1741
+ }
1742
+ finally {
1743
+ this.clearNoncePromptsForInstance(name);
1744
+ }
1745
+ })().finally(() => { if (this.stopsInFlight.get(name) === run)
1746
+ this.stopsInFlight.delete(name); });
1747
+ this.stopsInFlight.set(name, run);
1748
+ return run;
1494
1749
  }
1495
1750
  /** Restart a single instance, reloading fleet.yaml first to pick up config changes. */
1496
1751
  async restartSingleInstance(name, opts) {
@@ -1780,7 +2035,10 @@ export class FleetManager {
1780
2035
  await this.stopInstance(ch.instanceName).catch(() => { });
1781
2036
  // Small delay to let tmux window clean up
1782
2037
  await new Promise(r => setTimeout(r, 2000));
1783
- await this.startClassicInstance(ch.instanceName, newBackend, this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), newModel, newAutoPause).catch(err => this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to restart classic instance"));
2038
+ // The manager already holds the new backend/model/auto-pause; the
2039
+ // unattended helper reads them from it and schedules the delayed
2040
+ // retry on failure like every other unattended start.
2041
+ await this.startClassicInstanceUnattended(ch, "classic instance after backend/model change");
1784
2042
  }
1785
2043
  }
1786
2044
  }
@@ -1967,6 +2225,9 @@ export class FleetManager {
1967
2225
  }
1968
2226
  catch (err) {
1969
2227
  this.logger.error({ err, name }, "Failed to start general instance");
2228
+ // General is the most important instance to bring back: it also gets
2229
+ // the delayed automatic retry (the topic notice below still goes out).
2230
+ this.scheduleStartupRetry(name, 0);
1970
2231
  const errorMsg = err instanceof Error ? err.message : String(err);
1971
2232
  const topicId = cfg.topic_id ? String(cfg.topic_id) : undefined;
1972
2233
  if (this.adapter && topicId) {
@@ -2086,14 +2347,8 @@ export class FleetManager {
2086
2347
  while (idx < channels.length) {
2087
2348
  const batch = channels.slice(idx, idx + concurrency);
2088
2349
  await Promise.allSettled(batch.map(async (ch) => {
2089
- try {
2090
- await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, fleetBackend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
2091
- if (this.daemons.has(ch.instanceName))
2092
- startupProgress.markReady();
2093
- }
2094
- catch (err) {
2095
- this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
2096
- }
2350
+ if (await this.startClassicInstanceUnattended(ch, "classic instance"))
2351
+ startupProgress.markReady();
2097
2352
  }));
2098
2353
  idx += concurrency;
2099
2354
  }
@@ -2419,6 +2674,8 @@ export class FleetManager {
2419
2674
  return;
2420
2675
  if (await this.handleLoginBackendSelect(data, adapterId, this.adapter ?? undefined))
2421
2676
  return;
2677
+ if (await this.handleInstallBackendSelect(data, adapterId, this.adapter ?? undefined))
2678
+ return;
2422
2679
  if (await this.handleLoginMenuSelect(data, adapterId, this.adapter ?? undefined))
2423
2680
  return;
2424
2681
  if (await this.handleLoginConfirm(data, adapterId, this.adapter ?? undefined))
@@ -2813,6 +3070,8 @@ export class FleetManager {
2813
3070
  return;
2814
3071
  if (await this.handleLoginBackendSelect(data, adapterId, adapter))
2815
3072
  return;
3073
+ if (await this.handleInstallBackendSelect(data, adapterId, adapter))
3074
+ return;
2816
3075
  if (await this.handleLoginMenuSelect(data, adapterId, adapter))
2817
3076
  return;
2818
3077
  if (await this.handleLoginConfirm(data, adapterId, adapter))
@@ -5170,8 +5429,13 @@ export class FleetManager {
5170
5429
  */
5171
5430
  async postNonceButtonPrompt(opts) {
5172
5431
  // 16 bytes = the 128-bit capability the design claims. Telegram's 64-byte
5173
- // callback_data cap still holds: longest id is "interactive-assist:" (19)
5174
- // + 32 hex + ":confirm" (8) = 59 bytes.
5432
+ // callback_data cap still holds, with two prefixes tied at the longest:
5433
+ // "interactive-assist:" (19) + 32 hex + ":confirm" (8) = 59, and
5434
+ // "install-select:" (15) + 32 hex + ":" + the longest backend name
5435
+ // ("claude-code"/"antigravity", 11) = 59. Only 5 bytes of headroom: a
5436
+ // backend name of 17+ characters, or a longer prefix, would be silently
5437
+ // rejected by Telegram — see the callback_data assertion in
5438
+ // install-backend-menu.test.ts.
5175
5439
  const nonce = randomBytes(16).toString("hex");
5176
5440
  const entry = {
5177
5441
  prefix: opts.prefix,
@@ -6410,6 +6674,56 @@ export class FleetManager {
6410
6674
  expiredText: t("buttons.stale"),
6411
6675
  });
6412
6676
  }
6677
+ /**
6678
+ * Backend chooser for a bare `/install-cli`, mirroring promptLoginBackends so
6679
+ * both commands feel the same. Built on postNonceButtonPrompt rather than the
6680
+ * `/model` selection coordinator: that is the mechanism `/login` already uses,
6681
+ * and the one whose canonical-address binding (#682) makes the buttons answer
6682
+ * in a Telegram General topic.
6683
+ *
6684
+ * Unlike the login chooser this does NOT filter to backends the fleet already
6685
+ * runs. Installing is how you get a backend you do not have yet, so filtering
6686
+ * by configured backends would hide the only entry the admin came for.
6687
+ *
6688
+ * gemini-cli is omitted: it is deprecated (see backend/factory.ts). Typing
6689
+ * `/install-cli gemini-cli` still works — this only stops recommending it.
6690
+ */
6691
+ async promptInstallBackends(chat) {
6692
+ const choices = Object.keys(BACKEND_INSTALLATION_INFO)
6693
+ .filter(backend => backend !== "gemini-cli")
6694
+ .map(backend => ({ action: backend, label: backend }));
6695
+ await this.postNonceButtonPrompt({
6696
+ prefix: INSTALL_CALLBACK_PREFIX,
6697
+ alertType: "install",
6698
+ instanceName: "install",
6699
+ adapter: chat.adapter,
6700
+ adapterId: chat.adapterId,
6701
+ chatId: chat.chatId,
6702
+ threadId: chat.threadId,
6703
+ message: t("install.choose_backend"),
6704
+ choices,
6705
+ expiredText: t("buttons.stale"),
6706
+ });
6707
+ }
6708
+ /** Backend chooser button → start that backend's install session. */
6709
+ async handleInstallBackendSelect(data, callbackAdapterId, receivingAdapter) {
6710
+ const claimed = this.consumeNonceCallback(INSTALL_CALLBACK_PREFIX, /^install-select:([0-9a-f]+):([a-z][a-z-]*)$/, data, callbackAdapterId, receivingAdapter);
6711
+ if (claimed === null)
6712
+ return false;
6713
+ if (claimed === "consumed")
6714
+ return true;
6715
+ const { entry, action: backend } = claimed;
6716
+ await this.retireNonceButtons(entry, entry.messageId ?? data.messageId, t("install.starting_backend", backend));
6717
+ const text = await this.startInstallSession(backend, {
6718
+ adapter: entry.adapter,
6719
+ adapterId: entry.adapterId,
6720
+ chatId: entry.chatId,
6721
+ threadId: entry.threadId,
6722
+ });
6723
+ if (text)
6724
+ await entry.adapter.sendText(entry.chatId, text, { threadId: entry.threadId }).catch(() => { });
6725
+ return true;
6726
+ }
6413
6727
  /**
6414
6728
  * Start a login session for one backend. Caller enforces admin.
6415
6729
  * Returns a status line to post, or null when a confirmation prompt was
@@ -6801,7 +7115,11 @@ export class FleetManager {
6801
7115
  }
6802
7116
  const backend = String(data.options?.backend ?? "").trim();
6803
7117
  if (!backend) {
6804
- await data.respond(t("install.usage"));
7118
+ // Bare call: offer the backends instead of printing a usage line the user
7119
+ // then has to retype. A Discord slash that DID pick the native `backend`
7120
+ // choice never lands here, so the two paths cannot both fire.
7121
+ await this.promptInstallBackends({ adapter, adapterId, chatId: data.channelId });
7122
+ await data.respond(t("install.chooser_posted"));
6805
7123
  return;
6806
7124
  }
6807
7125
  await data.respond(await this.startInstallSession(backend, { adapter, adapterId, chatId: data.channelId }));
@@ -8608,6 +8926,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
8608
8926
  // make `agend stop` wait forever for its own queue.
8609
8927
  this.stormWindow.shutdown();
8610
8928
  this.spawnGate.shutdown();
8929
+ for (const pending of this.startupRetries.values())
8930
+ clearTimeout(pending.timer);
8931
+ this.startupRetries.clear();
8932
+ for (const pending of this.startupRetryNotices.values())
8933
+ clearTimeout(pending.timer);
8934
+ this.startupRetryNotices.clear();
8611
8935
  if (this.stormOpenNotifyTimer) {
8612
8936
  clearTimeout(this.stormOpenNotifyTimer);
8613
8937
  this.stormOpenNotifyTimer = null;
@@ -8933,7 +9257,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
8933
9257
  if (!this.daemons.has(name)) {
8934
9258
  // New instance — startInstance already calls connectIpcToInstance
8935
9259
  this.logger.info({ name }, "New instance in config — starting");
8936
- await this.startInstance(name, config, topicMode).catch(err => this.logger.error({ err, name }, "Failed to start new instance"));
9260
+ await this.startInstanceUnattended(name, config, topicMode, "new instance");
8937
9261
  }
8938
9262
  else if (oldConfig?.instances[name]) {
8939
9263
  const daemon = this.daemons.get(name);
@@ -8944,7 +9268,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
8944
9268
  if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
8945
9269
  this.logger.info({ name }, "Instance config changed — restarting");
8946
9270
  await this.stopInstance(name).catch(() => { });
8947
- await this.startInstance(name, config, topicMode).catch(err => this.logger.error({ err, name }, "Failed to restart modified instance"));
9271
+ await this.startInstanceUnattended(name, config, topicMode, "modified instance");
8948
9272
  }
8949
9273
  else if (!isDeepStrictEqual(oldParts.hot, newParts.hot)) {
8950
9274
  const update = hotConfigUpdate(config);
@@ -9043,14 +9367,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
9043
9367
  const restartGenerals = restartEntries.filter(([_, cfg]) => cfg.general_topic);
9044
9368
  const restartOthers = restartEntries.filter(([_, cfg]) => !cfg.general_topic);
9045
9369
  for (const [name, cfg] of restartGenerals) {
9046
- try {
9047
- await this.startInstance(name, cfg, topicMode);
9048
- if (this.daemons.has(name))
9049
- restartProgress.markReady();
9050
- }
9051
- catch (err) {
9052
- this.logger.error({ err, name }, "Failed to start general instance");
9053
- }
9370
+ if (await this.startInstanceUnattended(name, cfg, topicMode, "general instance"))
9371
+ restartProgress.markReady();
9054
9372
  }
9055
9373
  // General is ready again; now its topic can own the live progress message.
9056
9374
  await restartProgress.start(progressTarget);
@@ -9071,14 +9389,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
9071
9389
  while (idx < channels.length) {
9072
9390
  const batch = channels.slice(idx, idx + concurrency);
9073
9391
  await Promise.allSettled(batch.map(async (ch) => {
9074
- try {
9075
- await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, fleetBackend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
9076
- if (this.daemons.has(ch.instanceName))
9077
- restartProgress.markReady();
9078
- }
9079
- catch (err) {
9080
- this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
9081
- }
9392
+ if (await this.startClassicInstanceUnattended(ch, "classic instance"))
9393
+ restartProgress.markReady();
9082
9394
  }));
9083
9395
  idx += concurrency;
9084
9396
  }