@songsid/agend 2.1.4-beta.47 → 2.1.4-beta.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/kiro.d.ts +14 -0
- package/dist/backend/kiro.js +48 -0
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/types.d.ts +26 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/backend-outage.d.ts +61 -0
- package/dist/backend-outage.js +71 -0
- package/dist/backend-outage.js.map +1 -0
- package/dist/channel/types.d.ts +1 -1
- package/dist/daemon.d.ts +58 -1
- package/dist/daemon.js +177 -8
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-context.d.ts +7 -0
- package/dist/fleet-manager.d.ts +106 -0
- package/dist/fleet-manager.js +355 -43
- package/dist/fleet-manager.js.map +1 -1
- package/dist/instance-lifecycle.d.ts +32 -0
- package/dist/instance-lifecycle.js +67 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +14 -0
- package/dist/locale.js.map +1 -1
- package/dist/topic-commands.js +11 -0
- package/dist/topic-commands.js.map +1 -1
- package/package.json +1 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -62,6 +62,7 @@ import { RestartProgress } from "./restart-progress.js";
|
|
|
62
62
|
import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
|
|
63
63
|
import { StormWindow } from "./storm-window.js";
|
|
64
64
|
import { SpawnGate } from "./spawn-gate.js";
|
|
65
|
+
import { BackendOutageTracker } from "./backend-outage.js";
|
|
65
66
|
import { canUnlockAdvancedTips, DailyTipScheduler, selectTip, visibleTipLevels, } from "./tips.js";
|
|
66
67
|
import { getTmuxSession } from "./config.js";
|
|
67
68
|
export function resolveReplyThreadId(argsThreadId, instanceConfig) {
|
|
@@ -216,6 +217,7 @@ const CLEAR_CONFIRM_CALLBACK_PREFIX = "clear-confirm:";
|
|
|
216
217
|
const TIP_DISMISS_CALLBACK_PREFIX = "tip-dismiss:";
|
|
217
218
|
const TIP_UNLOCK_CALLBACK_PREFIX = "tip-unlock:";
|
|
218
219
|
const LOGIN_CALLBACK_PREFIX = "login:";
|
|
220
|
+
const INSTALL_CALLBACK_PREFIX = "install-select:";
|
|
219
221
|
const LOGIN_MENU_CALLBACK_PREFIX = "login-menu:";
|
|
220
222
|
const LOGIN_CONFIRM_CALLBACK_PREFIX = "login-confirm:";
|
|
221
223
|
const INSTALL_LOGIN_CALLBACK_PREFIX = "install-login:";
|
|
@@ -232,6 +234,8 @@ export class FleetManager {
|
|
|
232
234
|
lifecycle;
|
|
233
235
|
stormWindow;
|
|
234
236
|
spawnGate;
|
|
237
|
+
/** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
|
|
238
|
+
backendOutage = new BackendOutageTracker();
|
|
235
239
|
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
236
240
|
get daemons() { return this.lifecycle.daemons; }
|
|
237
241
|
fleetConfig = null;
|
|
@@ -313,6 +317,28 @@ export class FleetManager {
|
|
|
313
317
|
ipcWaitTails = new Map();
|
|
314
318
|
/** instanceName → restart currently executing; concurrent callers join it. */
|
|
315
319
|
restartsInFlight = new Map();
|
|
320
|
+
/**
|
|
321
|
+
* Delayed automatic retries for instances whose startup failed. Before this a
|
|
322
|
+
* failed start was logged once and the instance stayed `stopped` until an
|
|
323
|
+
* operator noticed (2026-09-03: 4 kiro instances, all victims of the same
|
|
324
|
+
* backend outage during a post-update herd). instanceName → pending retry.
|
|
325
|
+
*/
|
|
326
|
+
startupRetries = new Map();
|
|
327
|
+
/** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
|
|
328
|
+
explicitStopGeneration = new Map();
|
|
329
|
+
/** instanceName → outage hand-off currently executing; a repeat joins it. */
|
|
330
|
+
handOffsInFlight = new Map();
|
|
331
|
+
/** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
|
|
332
|
+
stopsInFlight = new Map();
|
|
333
|
+
/** Aggregation window for the "N instances failed to start" notice. */
|
|
334
|
+
startupRetryNotices = new Map();
|
|
335
|
+
/** Backoff between automatic startup retries; the last step repeats while the backend is down. */
|
|
336
|
+
static STARTUP_RETRY_BACKOFF_MS = [60_000, 5 * 60_000, 15 * 60_000];
|
|
337
|
+
/** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
|
|
338
|
+
static STARTUP_RETRY_MAX_ATTEMPTS = 6;
|
|
339
|
+
/** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
|
|
340
|
+
static STARTUP_RETRY_STORM_DEFER_MS = 60_000;
|
|
341
|
+
static STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1_000;
|
|
316
342
|
// Last user message delivered to each instance — used to react ✅ on completion.
|
|
317
343
|
lastInboundMsg = new Map();
|
|
318
344
|
topicArchiver;
|
|
@@ -1324,6 +1350,9 @@ export class FleetManager {
|
|
|
1324
1350
|
* across a fleet restart.
|
|
1325
1351
|
*/
|
|
1326
1352
|
resumePaused = false) {
|
|
1353
|
+
// Any start supersedes a pending automatic retry (it would otherwise fire
|
|
1354
|
+
// into a running instance — harmless, but noisy — or race this start).
|
|
1355
|
+
this.cancelStartupRetry(name);
|
|
1327
1356
|
if (resumePaused && this.lifecycle.isPaused(name)) {
|
|
1328
1357
|
await this.lifecycle.wake(name, 30_000);
|
|
1329
1358
|
// A successful wake clears the persisted pause marker and produces a
|
|
@@ -1431,15 +1460,231 @@ export class FleetManager {
|
|
|
1431
1460
|
workingDirectory: config.working_directory,
|
|
1432
1461
|
reason: "startup",
|
|
1433
1462
|
}, async () => {
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1463
|
+
if (await this.startInstanceUnattended(name, config, topicMode, "instance"))
|
|
1464
|
+
onReady?.(name);
|
|
1465
|
+
})));
|
|
1466
|
+
}
|
|
1467
|
+
/**
|
|
1468
|
+
* Start from an UNATTENDED path (fleet startup, full restart, config
|
|
1469
|
+
* reconcile): nobody is watching the result, so a failure is logged and
|
|
1470
|
+
* handed to the delayed automatic retry instead of leaving the instance
|
|
1471
|
+
* `stopped` forever. Explicit operator/API starts call startInstance directly
|
|
1472
|
+
* and keep their synchronous error. Returns whether the instance is up.
|
|
1473
|
+
*/
|
|
1474
|
+
async startInstanceUnattended(name, config, topicMode, what) {
|
|
1475
|
+
try {
|
|
1476
|
+
await this.startInstance(name, config, topicMode);
|
|
1477
|
+
return this.daemons.has(name);
|
|
1478
|
+
}
|
|
1479
|
+
catch (err) {
|
|
1480
|
+
this.logger.error({ err, name }, `Failed to start ${what}`);
|
|
1481
|
+
this.scheduleStartupRetry(name, 0);
|
|
1482
|
+
return false;
|
|
1483
|
+
}
|
|
1484
|
+
}
|
|
1485
|
+
// ── Delayed automatic startup retries ──────────────────────────────────
|
|
1486
|
+
/**
|
|
1487
|
+
* Schedule attempt `attempt` (0-based) of the automatic startup retry for an
|
|
1488
|
+
* instance whose start just failed. Backoff 1 → 5 → 15 min; while the
|
|
1489
|
+
* instance's backend is known to be unreachable the 15-min step repeats up to
|
|
1490
|
+
* STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
|
|
1491
|
+
* retry re-checks the world (still configured, not running, not paused, no
|
|
1492
|
+
* tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
|
|
1493
|
+
* failed instances comes back at the gate's concurrency, never all at once.
|
|
1494
|
+
*/
|
|
1495
|
+
scheduleStartupRetry(name, attempt) {
|
|
1496
|
+
if (this.shuttingDown)
|
|
1497
|
+
return;
|
|
1498
|
+
if (this.startupRetries.has(name))
|
|
1499
|
+
return;
|
|
1500
|
+
const backoff = FleetManager.STARTUP_RETRY_BACKOFF_MS;
|
|
1501
|
+
const outage = this.backendOutage.isActive(this.backendNameOf(name));
|
|
1502
|
+
const exhausted = attempt >= backoff.length && !(outage && attempt < FleetManager.STARTUP_RETRY_MAX_ATTEMPTS);
|
|
1503
|
+
if (attempt >= FleetManager.STARTUP_RETRY_MAX_ATTEMPTS || exhausted) {
|
|
1504
|
+
this.logger.error({ name, attempts: attempt }, "Giving up automatic startup retries");
|
|
1505
|
+
this.queueStartupRetryNotice("gave_up", name, 0);
|
|
1506
|
+
return;
|
|
1507
|
+
}
|
|
1508
|
+
const delayMs = backoff[Math.min(attempt, backoff.length - 1)];
|
|
1509
|
+
const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, delayMs);
|
|
1510
|
+
timer.unref?.();
|
|
1511
|
+
this.startupRetries.set(name, { attempt, timer });
|
|
1512
|
+
this.logger.warn({ name, attempt: attempt + 1, delayMs, backendOutage: outage }, "Startup failed — automatic retry scheduled");
|
|
1513
|
+
if (attempt === 0)
|
|
1514
|
+
this.queueStartupRetryNotice("scheduled", name, delayMs);
|
|
1515
|
+
}
|
|
1516
|
+
/** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
|
|
1517
|
+
cancelStartupRetry(name) {
|
|
1518
|
+
const pending = this.startupRetries.get(name);
|
|
1519
|
+
if (!pending)
|
|
1520
|
+
return;
|
|
1521
|
+
clearTimeout(pending.timer);
|
|
1522
|
+
this.startupRetries.delete(name);
|
|
1523
|
+
this.logger.info({ name }, "Pending automatic startup retry cancelled");
|
|
1524
|
+
}
|
|
1525
|
+
/** Pending automatic retry, if any (status display / tests). */
|
|
1526
|
+
pendingStartupRetry(name) {
|
|
1527
|
+
const pending = this.startupRetries.get(name);
|
|
1528
|
+
return pending ? { attempt: pending.attempt } : null;
|
|
1529
|
+
}
|
|
1530
|
+
/**
|
|
1531
|
+
* A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
|
|
1532
|
+
* stop THAT daemon and schedule the delayed retry. Serialized against the
|
|
1533
|
+
* operator paths: an in-flight restart is awaited first (it replaces the
|
|
1534
|
+
* daemon itself), the stop is identity-checked so a fresh daemon registered
|
|
1535
|
+
* meanwhile is never deleted, and an explicit stop/restart that began during
|
|
1536
|
+
* the hand-off (generation bump) owns the outcome — no retry is scheduled
|
|
1537
|
+
* behind an operator's back.
|
|
1538
|
+
*/
|
|
1539
|
+
async handOffToStartupRetry(name, daemon) {
|
|
1540
|
+
const joined = this.handOffsInFlight.get(name);
|
|
1541
|
+
if (joined)
|
|
1542
|
+
return joined;
|
|
1543
|
+
const run = (async () => {
|
|
1544
|
+
// Let any operator restart/stop that is already executing finish first:
|
|
1545
|
+
// a restart replaces the daemon itself, a stop removes it — either way
|
|
1546
|
+
// the identity check below then sees "not ours any more" and we do
|
|
1547
|
+
// nothing. Loop: one operation can be chained behind another.
|
|
1548
|
+
for (;;) {
|
|
1549
|
+
const inFlight = this.restartsInFlight.get(name) ?? this.stopsInFlight.get(name);
|
|
1550
|
+
if (!inFlight)
|
|
1551
|
+
break;
|
|
1552
|
+
await inFlight.catch(() => { });
|
|
1438
1553
|
}
|
|
1439
|
-
|
|
1440
|
-
|
|
1554
|
+
const stopGen = this.explicitStopGeneration.get(name) ?? 0;
|
|
1555
|
+
if (this.lifecycle.daemons.get(name) !== daemon)
|
|
1556
|
+
return; // already replaced or stopped
|
|
1557
|
+
await this.lifecycle.stopIfCurrent(name, daemon);
|
|
1558
|
+
if (this.shuttingDown)
|
|
1559
|
+
return;
|
|
1560
|
+
if ((this.explicitStopGeneration.get(name) ?? 0) !== stopGen) {
|
|
1561
|
+
this.logger.info({ name }, "Outage hand-off superseded by an explicit stop/restart — no automatic retry");
|
|
1562
|
+
return;
|
|
1441
1563
|
}
|
|
1442
|
-
|
|
1564
|
+
if (this.lifecycle.daemons.has(name) || this.lifecycle.isPaused(name))
|
|
1565
|
+
return;
|
|
1566
|
+
this.scheduleStartupRetry(name, 0);
|
|
1567
|
+
})().finally(() => this.handOffsInFlight.delete(name));
|
|
1568
|
+
this.handOffsInFlight.set(name, run);
|
|
1569
|
+
return run;
|
|
1570
|
+
}
|
|
1571
|
+
async runStartupRetry(name, attempt) {
|
|
1572
|
+
this.startupRetries.delete(name);
|
|
1573
|
+
if (this.shuttingDown)
|
|
1574
|
+
return;
|
|
1575
|
+
if (!this.isConfiguredInstance(name))
|
|
1576
|
+
return; // removed from fleet.yaml / classic channel meanwhile
|
|
1577
|
+
if (this.lifecycle.isPaused(name) || this.daemons.has(name))
|
|
1578
|
+
return; // operator acted meanwhile
|
|
1579
|
+
if (this.stormWindow.isSpawnBlocked()) {
|
|
1580
|
+
// The tmux server is being restarted; joining that herd is what we are
|
|
1581
|
+
// trying to avoid. Same attempt again after the storm's own recovery.
|
|
1582
|
+
const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, FleetManager.STARTUP_RETRY_STORM_DEFER_MS);
|
|
1583
|
+
timer.unref?.();
|
|
1584
|
+
this.startupRetries.set(name, { attempt, timer });
|
|
1585
|
+
this.logger.info({ name, attempt: attempt + 1 }, "Startup retry deferred — tmux storm window is blocking spawns");
|
|
1586
|
+
return;
|
|
1587
|
+
}
|
|
1588
|
+
const topicMode = this.fleetConfig?.channel?.mode === "topic"
|
|
1589
|
+
|| !!this.fleetConfig?.channels?.some(channel => channel.mode === "topic");
|
|
1590
|
+
try {
|
|
1591
|
+
await this.spawnGate.run({
|
|
1592
|
+
instanceName: name,
|
|
1593
|
+
workingDirectory: this.fleetConfig?.instances[name]?.working_directory || this.getInstanceDir(name),
|
|
1594
|
+
reason: "recovery",
|
|
1595
|
+
}, () => this.startConfiguredInstance(name, topicMode));
|
|
1596
|
+
if (this.daemons.has(name)) {
|
|
1597
|
+
this.logger.info({ name, attempt: attempt + 1 }, "Automatic startup retry succeeded");
|
|
1598
|
+
}
|
|
1599
|
+
}
|
|
1600
|
+
catch (err) {
|
|
1601
|
+
this.logger.error({ err, name, attempt: attempt + 1 }, "Automatic startup retry failed");
|
|
1602
|
+
this.scheduleStartupRetry(name, attempt + 1);
|
|
1603
|
+
}
|
|
1604
|
+
}
|
|
1605
|
+
/** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
|
|
1606
|
+
isConfiguredInstance(name) {
|
|
1607
|
+
if (this.fleetConfig?.instances[name])
|
|
1608
|
+
return true;
|
|
1609
|
+
return !!this.classicChannels?.getAll().some(channel => channel.instanceName === name);
|
|
1610
|
+
}
|
|
1611
|
+
/**
|
|
1612
|
+
* Kind-aware start for the automatic retry: fleet-topic instances come from
|
|
1613
|
+
* fleet.yaml, ClassicBot instances exist only in the classic channel manager
|
|
1614
|
+
* and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
|
|
1615
|
+
* silently dropped them — the retry timer fired and nothing happened).
|
|
1616
|
+
*/
|
|
1617
|
+
async startConfiguredInstance(name, topicMode) {
|
|
1618
|
+
const config = this.fleetConfig?.instances[name];
|
|
1619
|
+
if (config) {
|
|
1620
|
+
await this.startInstance(name, config, topicMode);
|
|
1621
|
+
return;
|
|
1622
|
+
}
|
|
1623
|
+
const channel = this.classicChannels?.getAll().find(item => item.instanceName === name);
|
|
1624
|
+
if (!channel || !this.classicChannels)
|
|
1625
|
+
throw new Error(`Instance '${name}' is no longer configured`);
|
|
1626
|
+
await this.startClassicInstance(name, this.classicChannels.getBackendByInstance(name, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(channel.channelId, channel.adapterId), this.classicChannels.getModel(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
|
|
1627
|
+
}
|
|
1628
|
+
/**
|
|
1629
|
+
* Unattended ClassicBot start (fleet startup batch, full-restart batch, the
|
|
1630
|
+
* classicBot.yaml reconcile): same contract as startInstanceUnattended —
|
|
1631
|
+
* failures are logged and handed to the delayed automatic retry, whose
|
|
1632
|
+
* kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
|
|
1633
|
+
* whether the instance is up.
|
|
1634
|
+
*/
|
|
1635
|
+
async startClassicInstanceUnattended(ch, what) {
|
|
1636
|
+
try {
|
|
1637
|
+
await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
|
|
1638
|
+
return this.daemons.has(ch.instanceName);
|
|
1639
|
+
}
|
|
1640
|
+
catch (err) {
|
|
1641
|
+
this.logger.warn({ err, instanceName: ch.instanceName }, `Failed to start ${what}`);
|
|
1642
|
+
this.scheduleStartupRetry(ch.instanceName, 0);
|
|
1643
|
+
return false;
|
|
1644
|
+
}
|
|
1645
|
+
}
|
|
1646
|
+
backendNameOf(name) {
|
|
1647
|
+
const fleetDefault = this.fleetConfig?.defaults?.backend;
|
|
1648
|
+
const configured = this.fleetConfig?.instances[name]?.backend;
|
|
1649
|
+
if (configured)
|
|
1650
|
+
return configured;
|
|
1651
|
+
// ClassicBot channels pick their own backend; the fleet default is only the fallback.
|
|
1652
|
+
if (this.classicChannels?.getAll().some(channel => channel.instanceName === name)) {
|
|
1653
|
+
return this.classicChannels.getBackendByInstance(name, fleetDefault);
|
|
1654
|
+
}
|
|
1655
|
+
return fleetDefault ?? "claude-code";
|
|
1656
|
+
}
|
|
1657
|
+
/**
|
|
1658
|
+
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
1659
|
+
* fails many instances within the same second. Two notices per incident at
|
|
1660
|
+
* most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
|
|
1661
|
+
*/
|
|
1662
|
+
queueStartupRetryNotice(kind, name, delayMs) {
|
|
1663
|
+
const pending = this.startupRetryNotices.get(kind);
|
|
1664
|
+
if (pending) {
|
|
1665
|
+
if (!pending.names.includes(name))
|
|
1666
|
+
pending.names.push(name);
|
|
1667
|
+
return;
|
|
1668
|
+
}
|
|
1669
|
+
const timer = setTimeout(() => {
|
|
1670
|
+
const entry = this.startupRetryNotices.get(kind);
|
|
1671
|
+
this.startupRetryNotices.delete(kind);
|
|
1672
|
+
if (!entry)
|
|
1673
|
+
return;
|
|
1674
|
+
const list = entry.names.join(", ");
|
|
1675
|
+
if (kind === "scheduled") {
|
|
1676
|
+
let text = t("fleet.startup_retry_scheduled", entry.names.length, list, this.formatStormDelay(entry.delayMs));
|
|
1677
|
+
const downBackends = [...new Set(entry.names.map(n => this.backendNameOf(n)))].filter(b => this.backendOutage.isActive(b));
|
|
1678
|
+
if (downBackends.length)
|
|
1679
|
+
text += `\n${t("fleet.startup_retry_outage", downBackends.join(", "))}`;
|
|
1680
|
+
this.notifyFleetError(text);
|
|
1681
|
+
}
|
|
1682
|
+
else {
|
|
1683
|
+
this.notifyFleetError(t("fleet.startup_retry_gave_up", entry.names.length, list));
|
|
1684
|
+
}
|
|
1685
|
+
}, FleetManager.STARTUP_RETRY_NOTICE_AGGREGATE_MS);
|
|
1686
|
+
timer.unref?.();
|
|
1687
|
+
this.startupRetryNotices.set(kind, { names: [name], delayMs, timer });
|
|
1443
1688
|
}
|
|
1444
1689
|
runnableStartupCount(fleet, includeClassic) {
|
|
1445
1690
|
const names = this.configuredStartupInstanceNames(fleet, includeClassic);
|
|
@@ -1474,6 +1719,8 @@ export class FleetManager {
|
|
|
1474
1719
|
};
|
|
1475
1720
|
}
|
|
1476
1721
|
async stopInstance(name) {
|
|
1722
|
+
this.explicitStopGeneration.set(name, (this.explicitStopGeneration.get(name) ?? 0) + 1);
|
|
1723
|
+
this.cancelStartupRetry(name);
|
|
1477
1724
|
this.failoverActive.delete(name);
|
|
1478
1725
|
this.cancelIdleButtonRetirement(name);
|
|
1479
1726
|
this.instanceStateCache.delete(name);
|
|
@@ -1485,12 +1732,20 @@ export class FleetManager {
|
|
|
1485
1732
|
// interactive-prompt / clean-exit events whose handlers post fresh prompts
|
|
1486
1733
|
// while the stop is still awaiting (the TOCTOU sol's review called out).
|
|
1487
1734
|
this.clearNoncePromptsForInstance(name);
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1735
|
+
// Published so the outage hand-off can wait for an explicit stop that began
|
|
1736
|
+
// BEFORE it (the generation fence alone only catches stops that begin after
|
|
1737
|
+
// the hand-off snapshotted it).
|
|
1738
|
+
const run = (async () => {
|
|
1739
|
+
try {
|
|
1740
|
+
await this.lifecycle.stop(name);
|
|
1741
|
+
}
|
|
1742
|
+
finally {
|
|
1743
|
+
this.clearNoncePromptsForInstance(name);
|
|
1744
|
+
}
|
|
1745
|
+
})().finally(() => { if (this.stopsInFlight.get(name) === run)
|
|
1746
|
+
this.stopsInFlight.delete(name); });
|
|
1747
|
+
this.stopsInFlight.set(name, run);
|
|
1748
|
+
return run;
|
|
1494
1749
|
}
|
|
1495
1750
|
/** Restart a single instance, reloading fleet.yaml first to pick up config changes. */
|
|
1496
1751
|
async restartSingleInstance(name, opts) {
|
|
@@ -1780,7 +2035,10 @@ export class FleetManager {
|
|
|
1780
2035
|
await this.stopInstance(ch.instanceName).catch(() => { });
|
|
1781
2036
|
// Small delay to let tmux window clean up
|
|
1782
2037
|
await new Promise(r => setTimeout(r, 2000));
|
|
1783
|
-
|
|
2038
|
+
// The manager already holds the new backend/model/auto-pause; the
|
|
2039
|
+
// unattended helper reads them from it and schedules the delayed
|
|
2040
|
+
// retry on failure like every other unattended start.
|
|
2041
|
+
await this.startClassicInstanceUnattended(ch, "classic instance after backend/model change");
|
|
1784
2042
|
}
|
|
1785
2043
|
}
|
|
1786
2044
|
}
|
|
@@ -1967,6 +2225,9 @@ export class FleetManager {
|
|
|
1967
2225
|
}
|
|
1968
2226
|
catch (err) {
|
|
1969
2227
|
this.logger.error({ err, name }, "Failed to start general instance");
|
|
2228
|
+
// General is the most important instance to bring back: it also gets
|
|
2229
|
+
// the delayed automatic retry (the topic notice below still goes out).
|
|
2230
|
+
this.scheduleStartupRetry(name, 0);
|
|
1970
2231
|
const errorMsg = err instanceof Error ? err.message : String(err);
|
|
1971
2232
|
const topicId = cfg.topic_id ? String(cfg.topic_id) : undefined;
|
|
1972
2233
|
if (this.adapter && topicId) {
|
|
@@ -2086,14 +2347,8 @@ export class FleetManager {
|
|
|
2086
2347
|
while (idx < channels.length) {
|
|
2087
2348
|
const batch = channels.slice(idx, idx + concurrency);
|
|
2088
2349
|
await Promise.allSettled(batch.map(async (ch) => {
|
|
2089
|
-
|
|
2090
|
-
|
|
2091
|
-
if (this.daemons.has(ch.instanceName))
|
|
2092
|
-
startupProgress.markReady();
|
|
2093
|
-
}
|
|
2094
|
-
catch (err) {
|
|
2095
|
-
this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
|
|
2096
|
-
}
|
|
2350
|
+
if (await this.startClassicInstanceUnattended(ch, "classic instance"))
|
|
2351
|
+
startupProgress.markReady();
|
|
2097
2352
|
}));
|
|
2098
2353
|
idx += concurrency;
|
|
2099
2354
|
}
|
|
@@ -2419,6 +2674,8 @@ export class FleetManager {
|
|
|
2419
2674
|
return;
|
|
2420
2675
|
if (await this.handleLoginBackendSelect(data, adapterId, this.adapter ?? undefined))
|
|
2421
2676
|
return;
|
|
2677
|
+
if (await this.handleInstallBackendSelect(data, adapterId, this.adapter ?? undefined))
|
|
2678
|
+
return;
|
|
2422
2679
|
if (await this.handleLoginMenuSelect(data, adapterId, this.adapter ?? undefined))
|
|
2423
2680
|
return;
|
|
2424
2681
|
if (await this.handleLoginConfirm(data, adapterId, this.adapter ?? undefined))
|
|
@@ -2813,6 +3070,8 @@ export class FleetManager {
|
|
|
2813
3070
|
return;
|
|
2814
3071
|
if (await this.handleLoginBackendSelect(data, adapterId, adapter))
|
|
2815
3072
|
return;
|
|
3073
|
+
if (await this.handleInstallBackendSelect(data, adapterId, adapter))
|
|
3074
|
+
return;
|
|
2816
3075
|
if (await this.handleLoginMenuSelect(data, adapterId, adapter))
|
|
2817
3076
|
return;
|
|
2818
3077
|
if (await this.handleLoginConfirm(data, adapterId, adapter))
|
|
@@ -5170,8 +5429,13 @@ export class FleetManager {
|
|
|
5170
5429
|
*/
|
|
5171
5430
|
async postNonceButtonPrompt(opts) {
|
|
5172
5431
|
// 16 bytes = the 128-bit capability the design claims. Telegram's 64-byte
|
|
5173
|
-
// callback_data cap still holds
|
|
5174
|
-
// + 32 hex + ":confirm" (8) = 59
|
|
5432
|
+
// callback_data cap still holds, with two prefixes tied at the longest:
|
|
5433
|
+
// "interactive-assist:" (19) + 32 hex + ":confirm" (8) = 59, and
|
|
5434
|
+
// "install-select:" (15) + 32 hex + ":" + the longest backend name
|
|
5435
|
+
// ("claude-code"/"antigravity", 11) = 59. Only 5 bytes of headroom: a
|
|
5436
|
+
// backend name of 17+ characters, or a longer prefix, would be silently
|
|
5437
|
+
// rejected by Telegram — see the callback_data assertion in
|
|
5438
|
+
// install-backend-menu.test.ts.
|
|
5175
5439
|
const nonce = randomBytes(16).toString("hex");
|
|
5176
5440
|
const entry = {
|
|
5177
5441
|
prefix: opts.prefix,
|
|
@@ -6410,6 +6674,56 @@ export class FleetManager {
|
|
|
6410
6674
|
expiredText: t("buttons.stale"),
|
|
6411
6675
|
});
|
|
6412
6676
|
}
|
|
6677
|
+
/**
|
|
6678
|
+
* Backend chooser for a bare `/install-cli`, mirroring promptLoginBackends so
|
|
6679
|
+
* both commands feel the same. Built on postNonceButtonPrompt rather than the
|
|
6680
|
+
* `/model` selection coordinator: that is the mechanism `/login` already uses,
|
|
6681
|
+
* and the one whose canonical-address binding (#682) makes the buttons answer
|
|
6682
|
+
* in a Telegram General topic.
|
|
6683
|
+
*
|
|
6684
|
+
* Unlike the login chooser this does NOT filter to backends the fleet already
|
|
6685
|
+
* runs. Installing is how you get a backend you do not have yet, so filtering
|
|
6686
|
+
* by configured backends would hide the only entry the admin came for.
|
|
6687
|
+
*
|
|
6688
|
+
* gemini-cli is omitted: it is deprecated (see backend/factory.ts). Typing
|
|
6689
|
+
* `/install-cli gemini-cli` still works — this only stops recommending it.
|
|
6690
|
+
*/
|
|
6691
|
+
async promptInstallBackends(chat) {
|
|
6692
|
+
const choices = Object.keys(BACKEND_INSTALLATION_INFO)
|
|
6693
|
+
.filter(backend => backend !== "gemini-cli")
|
|
6694
|
+
.map(backend => ({ action: backend, label: backend }));
|
|
6695
|
+
await this.postNonceButtonPrompt({
|
|
6696
|
+
prefix: INSTALL_CALLBACK_PREFIX,
|
|
6697
|
+
alertType: "install",
|
|
6698
|
+
instanceName: "install",
|
|
6699
|
+
adapter: chat.adapter,
|
|
6700
|
+
adapterId: chat.adapterId,
|
|
6701
|
+
chatId: chat.chatId,
|
|
6702
|
+
threadId: chat.threadId,
|
|
6703
|
+
message: t("install.choose_backend"),
|
|
6704
|
+
choices,
|
|
6705
|
+
expiredText: t("buttons.stale"),
|
|
6706
|
+
});
|
|
6707
|
+
}
|
|
6708
|
+
/** Backend chooser button → start that backend's install session. */
|
|
6709
|
+
async handleInstallBackendSelect(data, callbackAdapterId, receivingAdapter) {
|
|
6710
|
+
const claimed = this.consumeNonceCallback(INSTALL_CALLBACK_PREFIX, /^install-select:([0-9a-f]+):([a-z][a-z-]*)$/, data, callbackAdapterId, receivingAdapter);
|
|
6711
|
+
if (claimed === null)
|
|
6712
|
+
return false;
|
|
6713
|
+
if (claimed === "consumed")
|
|
6714
|
+
return true;
|
|
6715
|
+
const { entry, action: backend } = claimed;
|
|
6716
|
+
await this.retireNonceButtons(entry, entry.messageId ?? data.messageId, t("install.starting_backend", backend));
|
|
6717
|
+
const text = await this.startInstallSession(backend, {
|
|
6718
|
+
adapter: entry.adapter,
|
|
6719
|
+
adapterId: entry.adapterId,
|
|
6720
|
+
chatId: entry.chatId,
|
|
6721
|
+
threadId: entry.threadId,
|
|
6722
|
+
});
|
|
6723
|
+
if (text)
|
|
6724
|
+
await entry.adapter.sendText(entry.chatId, text, { threadId: entry.threadId }).catch(() => { });
|
|
6725
|
+
return true;
|
|
6726
|
+
}
|
|
6413
6727
|
/**
|
|
6414
6728
|
* Start a login session for one backend. Caller enforces admin.
|
|
6415
6729
|
* Returns a status line to post, or null when a confirmation prompt was
|
|
@@ -6801,7 +7115,11 @@ export class FleetManager {
|
|
|
6801
7115
|
}
|
|
6802
7116
|
const backend = String(data.options?.backend ?? "").trim();
|
|
6803
7117
|
if (!backend) {
|
|
6804
|
-
|
|
7118
|
+
// Bare call: offer the backends instead of printing a usage line the user
|
|
7119
|
+
// then has to retype. A Discord slash that DID pick the native `backend`
|
|
7120
|
+
// choice never lands here, so the two paths cannot both fire.
|
|
7121
|
+
await this.promptInstallBackends({ adapter, adapterId, chatId: data.channelId });
|
|
7122
|
+
await data.respond(t("install.chooser_posted"));
|
|
6805
7123
|
return;
|
|
6806
7124
|
}
|
|
6807
7125
|
await data.respond(await this.startInstallSession(backend, { adapter, adapterId, chatId: data.channelId }));
|
|
@@ -8608,6 +8926,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8608
8926
|
// make `agend stop` wait forever for its own queue.
|
|
8609
8927
|
this.stormWindow.shutdown();
|
|
8610
8928
|
this.spawnGate.shutdown();
|
|
8929
|
+
for (const pending of this.startupRetries.values())
|
|
8930
|
+
clearTimeout(pending.timer);
|
|
8931
|
+
this.startupRetries.clear();
|
|
8932
|
+
for (const pending of this.startupRetryNotices.values())
|
|
8933
|
+
clearTimeout(pending.timer);
|
|
8934
|
+
this.startupRetryNotices.clear();
|
|
8611
8935
|
if (this.stormOpenNotifyTimer) {
|
|
8612
8936
|
clearTimeout(this.stormOpenNotifyTimer);
|
|
8613
8937
|
this.stormOpenNotifyTimer = null;
|
|
@@ -8933,7 +9257,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8933
9257
|
if (!this.daemons.has(name)) {
|
|
8934
9258
|
// New instance — startInstance already calls connectIpcToInstance
|
|
8935
9259
|
this.logger.info({ name }, "New instance in config — starting");
|
|
8936
|
-
await this.
|
|
9260
|
+
await this.startInstanceUnattended(name, config, topicMode, "new instance");
|
|
8937
9261
|
}
|
|
8938
9262
|
else if (oldConfig?.instances[name]) {
|
|
8939
9263
|
const daemon = this.daemons.get(name);
|
|
@@ -8944,7 +9268,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8944
9268
|
if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
|
|
8945
9269
|
this.logger.info({ name }, "Instance config changed — restarting");
|
|
8946
9270
|
await this.stopInstance(name).catch(() => { });
|
|
8947
|
-
await this.
|
|
9271
|
+
await this.startInstanceUnattended(name, config, topicMode, "modified instance");
|
|
8948
9272
|
}
|
|
8949
9273
|
else if (!isDeepStrictEqual(oldParts.hot, newParts.hot)) {
|
|
8950
9274
|
const update = hotConfigUpdate(config);
|
|
@@ -9043,14 +9367,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9043
9367
|
const restartGenerals = restartEntries.filter(([_, cfg]) => cfg.general_topic);
|
|
9044
9368
|
const restartOthers = restartEntries.filter(([_, cfg]) => !cfg.general_topic);
|
|
9045
9369
|
for (const [name, cfg] of restartGenerals) {
|
|
9046
|
-
|
|
9047
|
-
|
|
9048
|
-
if (this.daemons.has(name))
|
|
9049
|
-
restartProgress.markReady();
|
|
9050
|
-
}
|
|
9051
|
-
catch (err) {
|
|
9052
|
-
this.logger.error({ err, name }, "Failed to start general instance");
|
|
9053
|
-
}
|
|
9370
|
+
if (await this.startInstanceUnattended(name, cfg, topicMode, "general instance"))
|
|
9371
|
+
restartProgress.markReady();
|
|
9054
9372
|
}
|
|
9055
9373
|
// General is ready again; now its topic can own the live progress message.
|
|
9056
9374
|
await restartProgress.start(progressTarget);
|
|
@@ -9071,14 +9389,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9071
9389
|
while (idx < channels.length) {
|
|
9072
9390
|
const batch = channels.slice(idx, idx + concurrency);
|
|
9073
9391
|
await Promise.allSettled(batch.map(async (ch) => {
|
|
9074
|
-
|
|
9075
|
-
|
|
9076
|
-
if (this.daemons.has(ch.instanceName))
|
|
9077
|
-
restartProgress.markReady();
|
|
9078
|
-
}
|
|
9079
|
-
catch (err) {
|
|
9080
|
-
this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
|
|
9081
|
-
}
|
|
9392
|
+
if (await this.startClassicInstanceUnattended(ch, "classic instance"))
|
|
9393
|
+
restartProgress.markReady();
|
|
9082
9394
|
}));
|
|
9083
9395
|
idx += concurrency;
|
|
9084
9396
|
}
|