@songsid/agend 2.1.4-beta.48 → 2.1.4-beta.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/kiro.d.ts +14 -0
- package/dist/backend/kiro.js +48 -0
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/types.d.ts +26 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/backend-outage.d.ts +61 -0
- package/dist/backend-outage.js +71 -0
- package/dist/backend-outage.js.map +1 -0
- package/dist/daemon.d.ts +58 -1
- package/dist/daemon.js +177 -8
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-manager.d.ts +84 -0
- package/dist/fleet-manager.js +288 -40
- package/dist/fleet-manager.js.map +1 -1
- package/dist/instance-lifecycle.d.ts +32 -0
- package/dist/instance-lifecycle.js +67 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +8 -0
- package/dist/locale.js.map +1 -1
- package/package.json +1 -1
package/dist/fleet-manager.d.ts
CHANGED
|
@@ -22,6 +22,7 @@ import { ClassicChannelManager } from "./classic-channel-manager.js";
|
|
|
22
22
|
import type { InstanceState } from "./backend/types.js";
|
|
23
23
|
import { StormWindow } from "./storm-window.js";
|
|
24
24
|
import { SpawnGate } from "./spawn-gate.js";
|
|
25
|
+
import { BackendOutageTracker } from "./backend-outage.js";
|
|
25
26
|
export declare function resolveReplyThreadId(argsThreadId: unknown, instanceConfig?: InstanceConfig): string | undefined;
|
|
26
27
|
/**
|
|
27
28
|
* Pure warm-cap victim selection (extracted for testability). Given the current
|
|
@@ -73,6 +74,8 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
73
74
|
readonly lifecycle: InstanceLifecycle;
|
|
74
75
|
readonly stormWindow: StormWindow;
|
|
75
76
|
readonly spawnGate: SpawnGate;
|
|
77
|
+
/** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
|
|
78
|
+
readonly backendOutage: BackendOutageTracker;
|
|
76
79
|
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
77
80
|
get daemons(): Map<string, import("./daemon.js").Daemon>;
|
|
78
81
|
fleetConfig: FleetConfig | null;
|
|
@@ -143,6 +146,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
143
146
|
private ipcWaitTails;
|
|
144
147
|
/** instanceName → restart currently executing; concurrent callers join it. */
|
|
145
148
|
private restartsInFlight;
|
|
149
|
+
/**
|
|
150
|
+
* Delayed automatic retries for instances whose startup failed. Before this a
|
|
151
|
+
* failed start was logged once and the instance stayed `stopped` until an
|
|
152
|
+
* operator noticed (2026-09-03: 4 kiro instances, all victims of the same
|
|
153
|
+
* backend outage during a post-update herd). instanceName → pending retry.
|
|
154
|
+
*/
|
|
155
|
+
private startupRetries;
|
|
156
|
+
/** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
|
|
157
|
+
private explicitStopGeneration;
|
|
158
|
+
/** instanceName → outage hand-off currently executing; a repeat joins it. */
|
|
159
|
+
private handOffsInFlight;
|
|
160
|
+
/** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
|
|
161
|
+
private stopsInFlight;
|
|
162
|
+
/** Aggregation window for the "N instances failed to start" notice. */
|
|
163
|
+
private startupRetryNotices;
|
|
164
|
+
/** Backoff between automatic startup retries; the last step repeats while the backend is down. */
|
|
165
|
+
static readonly STARTUP_RETRY_BACKOFF_MS: number[];
|
|
166
|
+
/** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
|
|
167
|
+
static readonly STARTUP_RETRY_MAX_ATTEMPTS = 6;
|
|
168
|
+
/** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
|
|
169
|
+
static readonly STARTUP_RETRY_STORM_DEFER_MS = 60000;
|
|
170
|
+
static readonly STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1000;
|
|
146
171
|
private lastInboundMsg;
|
|
147
172
|
private topicArchiver;
|
|
148
173
|
controlClient: TmuxControlClient | null;
|
|
@@ -368,6 +393,65 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
368
393
|
* TODO: per-instance startup timeout (existing issue, not introduced here)
|
|
369
394
|
*/
|
|
370
395
|
private startInstancesWithConcurrency;
|
|
396
|
+
/**
|
|
397
|
+
* Start from an UNATTENDED path (fleet startup, full restart, config
|
|
398
|
+
* reconcile): nobody is watching the result, so a failure is logged and
|
|
399
|
+
* handed to the delayed automatic retry instead of leaving the instance
|
|
400
|
+
* `stopped` forever. Explicit operator/API starts call startInstance directly
|
|
401
|
+
* and keep their synchronous error. Returns whether the instance is up.
|
|
402
|
+
*/
|
|
403
|
+
private startInstanceUnattended;
|
|
404
|
+
/**
|
|
405
|
+
* Schedule attempt `attempt` (0-based) of the automatic startup retry for an
|
|
406
|
+
* instance whose start just failed. Backoff 1 → 5 → 15 min; while the
|
|
407
|
+
* instance's backend is known to be unreachable the 15-min step repeats up to
|
|
408
|
+
* STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
|
|
409
|
+
* retry re-checks the world (still configured, not running, not paused, no
|
|
410
|
+
* tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
|
|
411
|
+
* failed instances comes back at the gate's concurrency, never all at once.
|
|
412
|
+
*/
|
|
413
|
+
scheduleStartupRetry(name: string, attempt: number): void;
|
|
414
|
+
/** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
|
|
415
|
+
cancelStartupRetry(name: string): void;
|
|
416
|
+
/** Pending automatic retry, if any (status display / tests). */
|
|
417
|
+
pendingStartupRetry(name: string): {
|
|
418
|
+
attempt: number;
|
|
419
|
+
} | null;
|
|
420
|
+
/**
|
|
421
|
+
* A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
|
|
422
|
+
* stop THAT daemon and schedule the delayed retry. Serialized against the
|
|
423
|
+
* operator paths: an in-flight restart is awaited first (it replaces the
|
|
424
|
+
* daemon itself), the stop is identity-checked so a fresh daemon registered
|
|
425
|
+
* meanwhile is never deleted, and an explicit stop/restart that began during
|
|
426
|
+
* the hand-off (generation bump) owns the outcome — no retry is scheduled
|
|
427
|
+
* behind an operator's back.
|
|
428
|
+
*/
|
|
429
|
+
handOffToStartupRetry(name: string, daemon: unknown): Promise<void>;
|
|
430
|
+
private runStartupRetry;
|
|
431
|
+
/** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
|
|
432
|
+
private isConfiguredInstance;
|
|
433
|
+
/**
|
|
434
|
+
* Kind-aware start for the automatic retry: fleet-topic instances come from
|
|
435
|
+
* fleet.yaml, ClassicBot instances exist only in the classic channel manager
|
|
436
|
+
* and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
|
|
437
|
+
* silently dropped them — the retry timer fired and nothing happened).
|
|
438
|
+
*/
|
|
439
|
+
private startConfiguredInstance;
|
|
440
|
+
/**
|
|
441
|
+
* Unattended ClassicBot start (fleet startup batch, full-restart batch, the
|
|
442
|
+
* classicBot.yaml reconcile): same contract as startInstanceUnattended —
|
|
443
|
+
* failures are logged and handed to the delayed automatic retry, whose
|
|
444
|
+
* kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
|
|
445
|
+
* whether the instance is up.
|
|
446
|
+
*/
|
|
447
|
+
private startClassicInstanceUnattended;
|
|
448
|
+
private backendNameOf;
|
|
449
|
+
/**
|
|
450
|
+
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
451
|
+
* fails many instances within the same second. Two notices per incident at
|
|
452
|
+
* most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
|
|
453
|
+
*/
|
|
454
|
+
private queueStartupRetryNotice;
|
|
371
455
|
private runnableStartupCount;
|
|
372
456
|
private configuredStartupInstanceNames;
|
|
373
457
|
private restartProgressTarget;
|
package/dist/fleet-manager.js
CHANGED
|
@@ -62,6 +62,7 @@ import { RestartProgress } from "./restart-progress.js";
|
|
|
62
62
|
import { collectRedundantInstanceDefaultPaths } from "./fleet-yaml-slim.js";
|
|
63
63
|
import { StormWindow } from "./storm-window.js";
|
|
64
64
|
import { SpawnGate } from "./spawn-gate.js";
|
|
65
|
+
import { BackendOutageTracker } from "./backend-outage.js";
|
|
65
66
|
import { canUnlockAdvancedTips, DailyTipScheduler, selectTip, visibleTipLevels, } from "./tips.js";
|
|
66
67
|
import { getTmuxSession } from "./config.js";
|
|
67
68
|
export function resolveReplyThreadId(argsThreadId, instanceConfig) {
|
|
@@ -233,6 +234,8 @@ export class FleetManager {
|
|
|
233
234
|
lifecycle;
|
|
234
235
|
stormWindow;
|
|
235
236
|
spawnGate;
|
|
237
|
+
/** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
|
|
238
|
+
backendOutage = new BackendOutageTracker();
|
|
236
239
|
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
237
240
|
get daemons() { return this.lifecycle.daemons; }
|
|
238
241
|
fleetConfig = null;
|
|
@@ -314,6 +317,28 @@ export class FleetManager {
|
|
|
314
317
|
ipcWaitTails = new Map();
|
|
315
318
|
/** instanceName → restart currently executing; concurrent callers join it. */
|
|
316
319
|
restartsInFlight = new Map();
|
|
320
|
+
/**
|
|
321
|
+
* Delayed automatic retries for instances whose startup failed. Before this a
|
|
322
|
+
* failed start was logged once and the instance stayed `stopped` until an
|
|
323
|
+
* operator noticed (2026-09-03: 4 kiro instances, all victims of the same
|
|
324
|
+
* backend outage during a post-update herd). instanceName → pending retry.
|
|
325
|
+
*/
|
|
326
|
+
startupRetries = new Map();
|
|
327
|
+
/** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
|
|
328
|
+
explicitStopGeneration = new Map();
|
|
329
|
+
/** instanceName → outage hand-off currently executing; a repeat joins it. */
|
|
330
|
+
handOffsInFlight = new Map();
|
|
331
|
+
/** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
|
|
332
|
+
stopsInFlight = new Map();
|
|
333
|
+
/** Aggregation window for the "N instances failed to start" notice. */
|
|
334
|
+
startupRetryNotices = new Map();
|
|
335
|
+
/** Backoff between automatic startup retries; the last step repeats while the backend is down. */
|
|
336
|
+
static STARTUP_RETRY_BACKOFF_MS = [60_000, 5 * 60_000, 15 * 60_000];
|
|
337
|
+
/** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
|
|
338
|
+
static STARTUP_RETRY_MAX_ATTEMPTS = 6;
|
|
339
|
+
/** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
|
|
340
|
+
static STARTUP_RETRY_STORM_DEFER_MS = 60_000;
|
|
341
|
+
static STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1_000;
|
|
317
342
|
// Last user message delivered to each instance — used to react ✅ on completion.
|
|
318
343
|
lastInboundMsg = new Map();
|
|
319
344
|
topicArchiver;
|
|
@@ -1325,6 +1350,9 @@ export class FleetManager {
|
|
|
1325
1350
|
* across a fleet restart.
|
|
1326
1351
|
*/
|
|
1327
1352
|
resumePaused = false) {
|
|
1353
|
+
// Any start supersedes a pending automatic retry (it would otherwise fire
|
|
1354
|
+
// into a running instance — harmless, but noisy — or race this start).
|
|
1355
|
+
this.cancelStartupRetry(name);
|
|
1328
1356
|
if (resumePaused && this.lifecycle.isPaused(name)) {
|
|
1329
1357
|
await this.lifecycle.wake(name, 30_000);
|
|
1330
1358
|
// A successful wake clears the persisted pause marker and produces a
|
|
@@ -1432,15 +1460,231 @@ export class FleetManager {
|
|
|
1432
1460
|
workingDirectory: config.working_directory,
|
|
1433
1461
|
reason: "startup",
|
|
1434
1462
|
}, async () => {
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
|
|
1438
|
-
|
|
1463
|
+
if (await this.startInstanceUnattended(name, config, topicMode, "instance"))
|
|
1464
|
+
onReady?.(name);
|
|
1465
|
+
})));
|
|
1466
|
+
}
|
|
1467
|
+
/**
|
|
1468
|
+
* Start from an UNATTENDED path (fleet startup, full restart, config
|
|
1469
|
+
* reconcile): nobody is watching the result, so a failure is logged and
|
|
1470
|
+
* handed to the delayed automatic retry instead of leaving the instance
|
|
1471
|
+
* `stopped` forever. Explicit operator/API starts call startInstance directly
|
|
1472
|
+
* and keep their synchronous error. Returns whether the instance is up.
|
|
1473
|
+
*/
|
|
1474
|
+
async startInstanceUnattended(name, config, topicMode, what) {
|
|
1475
|
+
try {
|
|
1476
|
+
await this.startInstance(name, config, topicMode);
|
|
1477
|
+
return this.daemons.has(name);
|
|
1478
|
+
}
|
|
1479
|
+
catch (err) {
|
|
1480
|
+
this.logger.error({ err, name }, `Failed to start ${what}`);
|
|
1481
|
+
this.scheduleStartupRetry(name, 0);
|
|
1482
|
+
return false;
|
|
1483
|
+
}
|
|
1484
|
+
}
|
|
1485
|
+
// ── Delayed automatic startup retries ──────────────────────────────────
|
|
1486
|
+
/**
|
|
1487
|
+
* Schedule attempt `attempt` (0-based) of the automatic startup retry for an
|
|
1488
|
+
* instance whose start just failed. Backoff 1 → 5 → 15 min; while the
|
|
1489
|
+
* instance's backend is known to be unreachable the 15-min step repeats up to
|
|
1490
|
+
* STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
|
|
1491
|
+
* retry re-checks the world (still configured, not running, not paused, no
|
|
1492
|
+
* tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
|
|
1493
|
+
* failed instances comes back at the gate's concurrency, never all at once.
|
|
1494
|
+
*/
|
|
1495
|
+
scheduleStartupRetry(name, attempt) {
|
|
1496
|
+
if (this.shuttingDown)
|
|
1497
|
+
return;
|
|
1498
|
+
if (this.startupRetries.has(name))
|
|
1499
|
+
return;
|
|
1500
|
+
const backoff = FleetManager.STARTUP_RETRY_BACKOFF_MS;
|
|
1501
|
+
const outage = this.backendOutage.isActive(this.backendNameOf(name));
|
|
1502
|
+
const exhausted = attempt >= backoff.length && !(outage && attempt < FleetManager.STARTUP_RETRY_MAX_ATTEMPTS);
|
|
1503
|
+
if (attempt >= FleetManager.STARTUP_RETRY_MAX_ATTEMPTS || exhausted) {
|
|
1504
|
+
this.logger.error({ name, attempts: attempt }, "Giving up automatic startup retries");
|
|
1505
|
+
this.queueStartupRetryNotice("gave_up", name, 0);
|
|
1506
|
+
return;
|
|
1507
|
+
}
|
|
1508
|
+
const delayMs = backoff[Math.min(attempt, backoff.length - 1)];
|
|
1509
|
+
const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, delayMs);
|
|
1510
|
+
timer.unref?.();
|
|
1511
|
+
this.startupRetries.set(name, { attempt, timer });
|
|
1512
|
+
this.logger.warn({ name, attempt: attempt + 1, delayMs, backendOutage: outage }, "Startup failed — automatic retry scheduled");
|
|
1513
|
+
if (attempt === 0)
|
|
1514
|
+
this.queueStartupRetryNotice("scheduled", name, delayMs);
|
|
1515
|
+
}
|
|
1516
|
+
/** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
|
|
1517
|
+
cancelStartupRetry(name) {
|
|
1518
|
+
const pending = this.startupRetries.get(name);
|
|
1519
|
+
if (!pending)
|
|
1520
|
+
return;
|
|
1521
|
+
clearTimeout(pending.timer);
|
|
1522
|
+
this.startupRetries.delete(name);
|
|
1523
|
+
this.logger.info({ name }, "Pending automatic startup retry cancelled");
|
|
1524
|
+
}
|
|
1525
|
+
/** Pending automatic retry, if any (status display / tests). */
|
|
1526
|
+
pendingStartupRetry(name) {
|
|
1527
|
+
const pending = this.startupRetries.get(name);
|
|
1528
|
+
return pending ? { attempt: pending.attempt } : null;
|
|
1529
|
+
}
|
|
1530
|
+
/**
|
|
1531
|
+
* A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
|
|
1532
|
+
* stop THAT daemon and schedule the delayed retry. Serialized against the
|
|
1533
|
+
* operator paths: an in-flight restart is awaited first (it replaces the
|
|
1534
|
+
* daemon itself), the stop is identity-checked so a fresh daemon registered
|
|
1535
|
+
* meanwhile is never deleted, and an explicit stop/restart that began during
|
|
1536
|
+
* the hand-off (generation bump) owns the outcome — no retry is scheduled
|
|
1537
|
+
* behind an operator's back.
|
|
1538
|
+
*/
|
|
1539
|
+
async handOffToStartupRetry(name, daemon) {
|
|
1540
|
+
const joined = this.handOffsInFlight.get(name);
|
|
1541
|
+
if (joined)
|
|
1542
|
+
return joined;
|
|
1543
|
+
const run = (async () => {
|
|
1544
|
+
// Let any operator restart/stop that is already executing finish first:
|
|
1545
|
+
// a restart replaces the daemon itself, a stop removes it — either way
|
|
1546
|
+
// the identity check below then sees "not ours any more" and we do
|
|
1547
|
+
// nothing. Loop: one operation can be chained behind another.
|
|
1548
|
+
for (;;) {
|
|
1549
|
+
const inFlight = this.restartsInFlight.get(name) ?? this.stopsInFlight.get(name);
|
|
1550
|
+
if (!inFlight)
|
|
1551
|
+
break;
|
|
1552
|
+
await inFlight.catch(() => { });
|
|
1439
1553
|
}
|
|
1440
|
-
|
|
1441
|
-
|
|
1554
|
+
const stopGen = this.explicitStopGeneration.get(name) ?? 0;
|
|
1555
|
+
if (this.lifecycle.daemons.get(name) !== daemon)
|
|
1556
|
+
return; // already replaced or stopped
|
|
1557
|
+
await this.lifecycle.stopIfCurrent(name, daemon);
|
|
1558
|
+
if (this.shuttingDown)
|
|
1559
|
+
return;
|
|
1560
|
+
if ((this.explicitStopGeneration.get(name) ?? 0) !== stopGen) {
|
|
1561
|
+
this.logger.info({ name }, "Outage hand-off superseded by an explicit stop/restart — no automatic retry");
|
|
1562
|
+
return;
|
|
1442
1563
|
}
|
|
1443
|
-
|
|
1564
|
+
if (this.lifecycle.daemons.has(name) || this.lifecycle.isPaused(name))
|
|
1565
|
+
return;
|
|
1566
|
+
this.scheduleStartupRetry(name, 0);
|
|
1567
|
+
})().finally(() => this.handOffsInFlight.delete(name));
|
|
1568
|
+
this.handOffsInFlight.set(name, run);
|
|
1569
|
+
return run;
|
|
1570
|
+
}
|
|
1571
|
+
async runStartupRetry(name, attempt) {
|
|
1572
|
+
this.startupRetries.delete(name);
|
|
1573
|
+
if (this.shuttingDown)
|
|
1574
|
+
return;
|
|
1575
|
+
if (!this.isConfiguredInstance(name))
|
|
1576
|
+
return; // removed from fleet.yaml / classic channel meanwhile
|
|
1577
|
+
if (this.lifecycle.isPaused(name) || this.daemons.has(name))
|
|
1578
|
+
return; // operator acted meanwhile
|
|
1579
|
+
if (this.stormWindow.isSpawnBlocked()) {
|
|
1580
|
+
// The tmux server is being restarted; joining that herd is what we are
|
|
1581
|
+
// trying to avoid. Same attempt again after the storm's own recovery.
|
|
1582
|
+
const timer = setTimeout(() => { void this.runStartupRetry(name, attempt); }, FleetManager.STARTUP_RETRY_STORM_DEFER_MS);
|
|
1583
|
+
timer.unref?.();
|
|
1584
|
+
this.startupRetries.set(name, { attempt, timer });
|
|
1585
|
+
this.logger.info({ name, attempt: attempt + 1 }, "Startup retry deferred — tmux storm window is blocking spawns");
|
|
1586
|
+
return;
|
|
1587
|
+
}
|
|
1588
|
+
const topicMode = this.fleetConfig?.channel?.mode === "topic"
|
|
1589
|
+
|| !!this.fleetConfig?.channels?.some(channel => channel.mode === "topic");
|
|
1590
|
+
try {
|
|
1591
|
+
await this.spawnGate.run({
|
|
1592
|
+
instanceName: name,
|
|
1593
|
+
workingDirectory: this.fleetConfig?.instances[name]?.working_directory || this.getInstanceDir(name),
|
|
1594
|
+
reason: "recovery",
|
|
1595
|
+
}, () => this.startConfiguredInstance(name, topicMode));
|
|
1596
|
+
if (this.daemons.has(name)) {
|
|
1597
|
+
this.logger.info({ name, attempt: attempt + 1 }, "Automatic startup retry succeeded");
|
|
1598
|
+
}
|
|
1599
|
+
}
|
|
1600
|
+
catch (err) {
|
|
1601
|
+
this.logger.error({ err, name, attempt: attempt + 1 }, "Automatic startup retry failed");
|
|
1602
|
+
this.scheduleStartupRetry(name, attempt + 1);
|
|
1603
|
+
}
|
|
1604
|
+
}
|
|
1605
|
+
/** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
|
|
1606
|
+
isConfiguredInstance(name) {
|
|
1607
|
+
if (this.fleetConfig?.instances[name])
|
|
1608
|
+
return true;
|
|
1609
|
+
return !!this.classicChannels?.getAll().some(channel => channel.instanceName === name);
|
|
1610
|
+
}
|
|
1611
|
+
/**
|
|
1612
|
+
* Kind-aware start for the automatic retry: fleet-topic instances come from
|
|
1613
|
+
* fleet.yaml, ClassicBot instances exist only in the classic channel manager
|
|
1614
|
+
* and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
|
|
1615
|
+
* silently dropped them — the retry timer fired and nothing happened).
|
|
1616
|
+
*/
|
|
1617
|
+
async startConfiguredInstance(name, topicMode) {
|
|
1618
|
+
const config = this.fleetConfig?.instances[name];
|
|
1619
|
+
if (config) {
|
|
1620
|
+
await this.startInstance(name, config, topicMode);
|
|
1621
|
+
return;
|
|
1622
|
+
}
|
|
1623
|
+
const channel = this.classicChannels?.getAll().find(item => item.instanceName === name);
|
|
1624
|
+
if (!channel || !this.classicChannels)
|
|
1625
|
+
throw new Error(`Instance '${name}' is no longer configured`);
|
|
1626
|
+
await this.startClassicInstance(name, this.classicChannels.getBackendByInstance(name, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(channel.channelId, channel.adapterId), this.classicChannels.getModel(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
|
|
1627
|
+
}
|
|
1628
|
+
/**
|
|
1629
|
+
* Unattended ClassicBot start (fleet startup batch, full-restart batch, the
|
|
1630
|
+
* classicBot.yaml reconcile): same contract as startInstanceUnattended —
|
|
1631
|
+
* failures are logged and handed to the delayed automatic retry, whose
|
|
1632
|
+
* kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
|
|
1633
|
+
* whether the instance is up.
|
|
1634
|
+
*/
|
|
1635
|
+
async startClassicInstanceUnattended(ch, what) {
|
|
1636
|
+
try {
|
|
1637
|
+
await this.startClassicInstance(ch.instanceName, this.classicChannels.getBackendByInstance(ch.instanceName, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(ch.channelId, ch.adapterId), this.classicChannels.getModel(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(ch.channelId, ch.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
|
|
1638
|
+
return this.daemons.has(ch.instanceName);
|
|
1639
|
+
}
|
|
1640
|
+
catch (err) {
|
|
1641
|
+
this.logger.warn({ err, instanceName: ch.instanceName }, `Failed to start ${what}`);
|
|
1642
|
+
this.scheduleStartupRetry(ch.instanceName, 0);
|
|
1643
|
+
return false;
|
|
1644
|
+
}
|
|
1645
|
+
}
|
|
1646
|
+
backendNameOf(name) {
|
|
1647
|
+
const fleetDefault = this.fleetConfig?.defaults?.backend;
|
|
1648
|
+
const configured = this.fleetConfig?.instances[name]?.backend;
|
|
1649
|
+
if (configured)
|
|
1650
|
+
return configured;
|
|
1651
|
+
// ClassicBot channels pick their own backend; the fleet default is only the fallback.
|
|
1652
|
+
if (this.classicChannels?.getAll().some(channel => channel.instanceName === name)) {
|
|
1653
|
+
return this.classicChannels.getBackendByInstance(name, fleetDefault);
|
|
1654
|
+
}
|
|
1655
|
+
return fleetDefault ?? "claude-code";
|
|
1656
|
+
}
|
|
1657
|
+
/**
|
|
1658
|
+
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
1659
|
+
* fails many instances within the same second. Two notices per incident at
|
|
1660
|
+
* most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
|
|
1661
|
+
*/
|
|
1662
|
+
queueStartupRetryNotice(kind, name, delayMs) {
|
|
1663
|
+
const pending = this.startupRetryNotices.get(kind);
|
|
1664
|
+
if (pending) {
|
|
1665
|
+
if (!pending.names.includes(name))
|
|
1666
|
+
pending.names.push(name);
|
|
1667
|
+
return;
|
|
1668
|
+
}
|
|
1669
|
+
const timer = setTimeout(() => {
|
|
1670
|
+
const entry = this.startupRetryNotices.get(kind);
|
|
1671
|
+
this.startupRetryNotices.delete(kind);
|
|
1672
|
+
if (!entry)
|
|
1673
|
+
return;
|
|
1674
|
+
const list = entry.names.join(", ");
|
|
1675
|
+
if (kind === "scheduled") {
|
|
1676
|
+
let text = t("fleet.startup_retry_scheduled", entry.names.length, list, this.formatStormDelay(entry.delayMs));
|
|
1677
|
+
const downBackends = [...new Set(entry.names.map(n => this.backendNameOf(n)))].filter(b => this.backendOutage.isActive(b));
|
|
1678
|
+
if (downBackends.length)
|
|
1679
|
+
text += `\n${t("fleet.startup_retry_outage", downBackends.join(", "))}`;
|
|
1680
|
+
this.notifyFleetError(text);
|
|
1681
|
+
}
|
|
1682
|
+
else {
|
|
1683
|
+
this.notifyFleetError(t("fleet.startup_retry_gave_up", entry.names.length, list));
|
|
1684
|
+
}
|
|
1685
|
+
}, FleetManager.STARTUP_RETRY_NOTICE_AGGREGATE_MS);
|
|
1686
|
+
timer.unref?.();
|
|
1687
|
+
this.startupRetryNotices.set(kind, { names: [name], delayMs, timer });
|
|
1444
1688
|
}
|
|
1445
1689
|
runnableStartupCount(fleet, includeClassic) {
|
|
1446
1690
|
const names = this.configuredStartupInstanceNames(fleet, includeClassic);
|
|
@@ -1475,6 +1719,8 @@ export class FleetManager {
|
|
|
1475
1719
|
};
|
|
1476
1720
|
}
|
|
1477
1721
|
async stopInstance(name) {
|
|
1722
|
+
this.explicitStopGeneration.set(name, (this.explicitStopGeneration.get(name) ?? 0) + 1);
|
|
1723
|
+
this.cancelStartupRetry(name);
|
|
1478
1724
|
this.failoverActive.delete(name);
|
|
1479
1725
|
this.cancelIdleButtonRetirement(name);
|
|
1480
1726
|
this.instanceStateCache.delete(name);
|
|
@@ -1486,12 +1732,20 @@ export class FleetManager {
|
|
|
1486
1732
|
// interactive-prompt / clean-exit events whose handlers post fresh prompts
|
|
1487
1733
|
// while the stop is still awaiting (the TOCTOU sol's review called out).
|
|
1488
1734
|
this.clearNoncePromptsForInstance(name);
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1492
|
-
|
|
1493
|
-
|
|
1494
|
-
|
|
1735
|
+
// Published so the outage hand-off can wait for an explicit stop that began
|
|
1736
|
+
// BEFORE it (the generation fence alone only catches stops that begin after
|
|
1737
|
+
// the hand-off snapshotted it).
|
|
1738
|
+
const run = (async () => {
|
|
1739
|
+
try {
|
|
1740
|
+
await this.lifecycle.stop(name);
|
|
1741
|
+
}
|
|
1742
|
+
finally {
|
|
1743
|
+
this.clearNoncePromptsForInstance(name);
|
|
1744
|
+
}
|
|
1745
|
+
})().finally(() => { if (this.stopsInFlight.get(name) === run)
|
|
1746
|
+
this.stopsInFlight.delete(name); });
|
|
1747
|
+
this.stopsInFlight.set(name, run);
|
|
1748
|
+
return run;
|
|
1495
1749
|
}
|
|
1496
1750
|
/** Restart a single instance, reloading fleet.yaml first to pick up config changes. */
|
|
1497
1751
|
async restartSingleInstance(name, opts) {
|
|
@@ -1781,7 +2035,10 @@ export class FleetManager {
|
|
|
1781
2035
|
await this.stopInstance(ch.instanceName).catch(() => { });
|
|
1782
2036
|
// Small delay to let tmux window clean up
|
|
1783
2037
|
await new Promise(r => setTimeout(r, 2000));
|
|
1784
|
-
|
|
2038
|
+
// The manager already holds the new backend/model/auto-pause; the
|
|
2039
|
+
// unattended helper reads them from it and schedules the delayed
|
|
2040
|
+
// retry on failure like every other unattended start.
|
|
2041
|
+
await this.startClassicInstanceUnattended(ch, "classic instance after backend/model change");
|
|
1785
2042
|
}
|
|
1786
2043
|
}
|
|
1787
2044
|
}
|
|
@@ -1968,6 +2225,9 @@ export class FleetManager {
|
|
|
1968
2225
|
}
|
|
1969
2226
|
catch (err) {
|
|
1970
2227
|
this.logger.error({ err, name }, "Failed to start general instance");
|
|
2228
|
+
// General is the most important instance to bring back: it also gets
|
|
2229
|
+
// the delayed automatic retry (the topic notice below still goes out).
|
|
2230
|
+
this.scheduleStartupRetry(name, 0);
|
|
1971
2231
|
const errorMsg = err instanceof Error ? err.message : String(err);
|
|
1972
2232
|
const topicId = cfg.topic_id ? String(cfg.topic_id) : undefined;
|
|
1973
2233
|
if (this.adapter && topicId) {
|
|
@@ -2087,14 +2347,8 @@ export class FleetManager {
|
|
|
2087
2347
|
while (idx < channels.length) {
|
|
2088
2348
|
const batch = channels.slice(idx, idx + concurrency);
|
|
2089
2349
|
await Promise.allSettled(batch.map(async (ch) => {
|
|
2090
|
-
|
|
2091
|
-
|
|
2092
|
-
if (this.daemons.has(ch.instanceName))
|
|
2093
|
-
startupProgress.markReady();
|
|
2094
|
-
}
|
|
2095
|
-
catch (err) {
|
|
2096
|
-
this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
|
|
2097
|
-
}
|
|
2350
|
+
if (await this.startClassicInstanceUnattended(ch, "classic instance"))
|
|
2351
|
+
startupProgress.markReady();
|
|
2098
2352
|
}));
|
|
2099
2353
|
idx += concurrency;
|
|
2100
2354
|
}
|
|
@@ -8672,6 +8926,12 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8672
8926
|
// make `agend stop` wait forever for its own queue.
|
|
8673
8927
|
this.stormWindow.shutdown();
|
|
8674
8928
|
this.spawnGate.shutdown();
|
|
8929
|
+
for (const pending of this.startupRetries.values())
|
|
8930
|
+
clearTimeout(pending.timer);
|
|
8931
|
+
this.startupRetries.clear();
|
|
8932
|
+
for (const pending of this.startupRetryNotices.values())
|
|
8933
|
+
clearTimeout(pending.timer);
|
|
8934
|
+
this.startupRetryNotices.clear();
|
|
8675
8935
|
if (this.stormOpenNotifyTimer) {
|
|
8676
8936
|
clearTimeout(this.stormOpenNotifyTimer);
|
|
8677
8937
|
this.stormOpenNotifyTimer = null;
|
|
@@ -8997,7 +9257,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
8997
9257
|
if (!this.daemons.has(name)) {
|
|
8998
9258
|
// New instance — startInstance already calls connectIpcToInstance
|
|
8999
9259
|
this.logger.info({ name }, "New instance in config — starting");
|
|
9000
|
-
await this.
|
|
9260
|
+
await this.startInstanceUnattended(name, config, topicMode, "new instance");
|
|
9001
9261
|
}
|
|
9002
9262
|
else if (oldConfig?.instances[name]) {
|
|
9003
9263
|
const daemon = this.daemons.get(name);
|
|
@@ -9008,7 +9268,7 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9008
9268
|
if (!isDeepStrictEqual(oldParts.cold, newParts.cold)) {
|
|
9009
9269
|
this.logger.info({ name }, "Instance config changed — restarting");
|
|
9010
9270
|
await this.stopInstance(name).catch(() => { });
|
|
9011
|
-
await this.
|
|
9271
|
+
await this.startInstanceUnattended(name, config, topicMode, "modified instance");
|
|
9012
9272
|
}
|
|
9013
9273
|
else if (!isDeepStrictEqual(oldParts.hot, newParts.hot)) {
|
|
9014
9274
|
const update = hotConfigUpdate(config);
|
|
@@ -9107,14 +9367,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9107
9367
|
const restartGenerals = restartEntries.filter(([_, cfg]) => cfg.general_topic);
|
|
9108
9368
|
const restartOthers = restartEntries.filter(([_, cfg]) => !cfg.general_topic);
|
|
9109
9369
|
for (const [name, cfg] of restartGenerals) {
|
|
9110
|
-
|
|
9111
|
-
|
|
9112
|
-
if (this.daemons.has(name))
|
|
9113
|
-
restartProgress.markReady();
|
|
9114
|
-
}
|
|
9115
|
-
catch (err) {
|
|
9116
|
-
this.logger.error({ err, name }, "Failed to start general instance");
|
|
9117
|
-
}
|
|
9370
|
+
if (await this.startInstanceUnattended(name, cfg, topicMode, "general instance"))
|
|
9371
|
+
restartProgress.markReady();
|
|
9118
9372
|
}
|
|
9119
9373
|
// General is ready again; now its topic can own the live progress message.
|
|
9120
9374
|
await restartProgress.start(progressTarget);
|
|
@@ -9135,14 +9389,8 @@ Plus the operational skills (fleet-health, instance-lifecycle, scheduling, sessi
|
|
|
9135
9389
|
while (idx < channels.length) {
|
|
9136
9390
|
const batch = channels.slice(idx, idx + concurrency);
|
|
9137
9391
|
await Promise.allSettled(batch.map(async (ch) => {
|
|
9138
|
-
|
|
9139
|
-
|
|
9140
|
-
if (this.daemons.has(ch.instanceName))
|
|
9141
|
-
restartProgress.markReady();
|
|
9142
|
-
}
|
|
9143
|
-
catch (err) {
|
|
9144
|
-
this.logger.warn({ err, instanceName: ch.instanceName }, "Failed to start classic instance");
|
|
9145
|
-
}
|
|
9392
|
+
if (await this.startClassicInstanceUnattended(ch, "classic instance"))
|
|
9393
|
+
restartProgress.markReady();
|
|
9146
9394
|
}));
|
|
9147
9395
|
idx += concurrency;
|
|
9148
9396
|
}
|