@songsid/agend 2.1.4-beta.47 → 2.1.4-beta.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -93,6 +93,13 @@ export interface FleetContext {
93
93
  promptTip?(generalName: string, adapter: import("./channel/types.js").ChannelAdapter, chatId: string, threadId?: string): Promise<"posted" | "empty" | "unavailable">;
94
94
  /** Persistently enable advanced tips without waiting for the dismissal threshold. */
95
95
  unlockAdvancedTips?(userId: string): boolean;
96
+ /** `/install-cli` backend chooser buttons. Caller enforces fleet-admin. */
97
+ promptInstallBackends?(chat: {
98
+ adapter: import("./channel/types.js").ChannelAdapter;
99
+ adapterId: string;
100
+ chatId: string;
101
+ threadId?: string;
102
+ }): Promise<void>;
96
103
  /** `/login` backend chooser buttons. Caller enforces fleet-admin. */
97
104
  promptLoginBackends?(chat: {
98
105
  adapter: import("./channel/types.js").ChannelAdapter;
@@ -22,6 +22,7 @@ import { ClassicChannelManager } from "./classic-channel-manager.js";
22
22
  import type { InstanceState } from "./backend/types.js";
23
23
  import { StormWindow } from "./storm-window.js";
24
24
  import { SpawnGate } from "./spawn-gate.js";
25
+ import { BackendOutageTracker } from "./backend-outage.js";
25
26
  export declare function resolveReplyThreadId(argsThreadId: unknown, instanceConfig?: InstanceConfig): string | undefined;
26
27
  /**
27
28
  * Pure warm-cap victim selection (extracted for testability). Given the current
@@ -73,6 +74,8 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
73
74
  readonly lifecycle: InstanceLifecycle;
74
75
  readonly stormWindow: StormWindow;
75
76
  readonly spawnGate: SpawnGate;
77
+ /** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
78
+ readonly backendOutage: BackendOutageTracker;
76
79
  /** Live view of lifecycle.daemons — used throughout; not deprecated. */
77
80
  get daemons(): Map<string, import("./daemon.js").Daemon>;
78
81
  fleetConfig: FleetConfig | null;
@@ -143,6 +146,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
143
146
  private ipcWaitTails;
144
147
  /** instanceName → restart currently executing; concurrent callers join it. */
145
148
  private restartsInFlight;
149
+ /**
150
+ * Delayed automatic retries for instances whose startup failed. Before this a
151
+ * failed start was logged once and the instance stayed `stopped` until an
152
+ * operator noticed (2026-09-03: 4 kiro instances, all victims of the same
153
+ * backend outage during a post-update herd). instanceName → pending retry.
154
+ */
155
+ private startupRetries;
156
+ /** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
157
+ private explicitStopGeneration;
158
+ /** instanceName → outage hand-off currently executing; a repeat joins it. */
159
+ private handOffsInFlight;
160
+ /** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
161
+ private stopsInFlight;
162
+ /** Aggregation window for the "N instances failed to start" notice. */
163
+ private startupRetryNotices;
164
+ /** Backoff between automatic startup retries; the last step repeats while the backend is down. */
165
+ static readonly STARTUP_RETRY_BACKOFF_MS: number[];
166
+ /** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
167
+ static readonly STARTUP_RETRY_MAX_ATTEMPTS = 6;
168
+ /** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
169
+ static readonly STARTUP_RETRY_STORM_DEFER_MS = 60000;
170
+ static readonly STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1000;
146
171
  private lastInboundMsg;
147
172
  private topicArchiver;
148
173
  controlClient: TmuxControlClient | null;
@@ -368,6 +393,65 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
368
393
  * TODO: per-instance startup timeout (existing issue, not introduced here)
369
394
  */
370
395
  private startInstancesWithConcurrency;
396
+ /**
397
+ * Start from an UNATTENDED path (fleet startup, full restart, config
398
+ * reconcile): nobody is watching the result, so a failure is logged and
399
+ * handed to the delayed automatic retry instead of leaving the instance
400
+ * `stopped` forever. Explicit operator/API starts call startInstance directly
401
+ * and keep their synchronous error. Returns whether the instance is up.
402
+ */
403
+ private startInstanceUnattended;
404
+ /**
405
+ * Schedule attempt `attempt` (0-based) of the automatic startup retry for an
406
+ * instance whose start just failed. Backoff 1 → 5 → 15 min; while the
407
+ * instance's backend is known to be unreachable the 15-min step repeats up to
408
+ * STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
409
+ * retry re-checks the world (still configured, not running, not paused, no
410
+ * tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
411
+ * failed instances comes back at the gate's concurrency, never all at once.
412
+ */
413
+ scheduleStartupRetry(name: string, attempt: number): void;
414
+ /** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
415
+ cancelStartupRetry(name: string): void;
416
+ /** Pending automatic retry, if any (status display / tests). */
417
+ pendingStartupRetry(name: string): {
418
+ attempt: number;
419
+ } | null;
420
+ /**
421
+ * A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
422
+ * stop THAT daemon and schedule the delayed retry. Serialized against the
423
+ * operator paths: an in-flight restart is awaited first (it replaces the
424
+ * daemon itself), the stop is identity-checked so a fresh daemon registered
425
+ * meanwhile is never deleted, and an explicit stop/restart that began during
426
+ * the hand-off (generation bump) owns the outcome — no retry is scheduled
427
+ * behind an operator's back.
428
+ */
429
+ handOffToStartupRetry(name: string, daemon: unknown): Promise<void>;
430
+ private runStartupRetry;
431
+ /** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
432
+ private isConfiguredInstance;
433
+ /**
434
+ * Kind-aware start for the automatic retry: fleet-topic instances come from
435
+ * fleet.yaml, ClassicBot instances exist only in the classic channel manager
436
+ * and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
437
+ * silently dropped them — the retry timer fired and nothing happened).
438
+ */
439
+ private startConfiguredInstance;
440
+ /**
441
+ * Unattended ClassicBot start (fleet startup batch, full-restart batch, the
442
+ * classicBot.yaml reconcile): same contract as startInstanceUnattended —
443
+ * failures are logged and handed to the delayed automatic retry, whose
444
+ * kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
445
+ * whether the instance is up.
446
+ */
447
+ private startClassicInstanceUnattended;
448
+ private backendNameOf;
449
+ /**
450
+ * One fleet-level notice per burst, not one per instance: a post-update herd
451
+ * fails many instances within the same second. Two notices per incident at
452
+ * most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
453
+ */
454
+ private queueStartupRetryNotice;
371
455
  private runnableStartupCount;
372
456
  private configuredStartupInstanceNames;
373
457
  private restartProgressTarget;
@@ -832,6 +916,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
832
916
  chatId: string;
833
917
  threadId?: string;
834
918
  }): Promise<void>;
919
+ /**
920
+ * Backend chooser for a bare `/install-cli`, mirroring promptLoginBackends so
921
+ * both commands feel the same. Built on postNonceButtonPrompt rather than the
922
+ * `/model` selection coordinator: that is the mechanism `/login` already uses,
923
+ * and the one whose canonical-address binding (#682) makes the buttons answer
924
+ * in a Telegram General topic.
925
+ *
926
+ * Unlike the login chooser this does NOT filter to backends the fleet already
927
+ * runs. Installing is how you get a backend you do not have yet, so filtering
928
+ * by configured backends would hide the only entry the admin came for.
929
+ *
930
+ * gemini-cli is omitted: it is deprecated (see backend/factory.ts). Typing
931
+ * `/install-cli gemini-cli` still works — this only stops recommending it.
932
+ */
933
+ promptInstallBackends(chat: {
934
+ adapter: ChannelAdapter;
935
+ adapterId: string;
936
+ chatId: string;
937
+ threadId?: string;
938
+ }): Promise<void>;
939
+ /** Backend chooser button → start that backend's install session. */
940
+ private handleInstallBackendSelect;
835
941
  /**
836
942
  * Start a login session for one backend. Caller enforces admin.
837
943
  * Returns a status line to post, or null when a confirmation prompt was