@songsid/agend 2.1.4-beta.47 → 2.1.4-beta.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/kiro.d.ts +14 -0
- package/dist/backend/kiro.js +48 -0
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/types.d.ts +26 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/backend-outage.d.ts +61 -0
- package/dist/backend-outage.js +71 -0
- package/dist/backend-outage.js.map +1 -0
- package/dist/channel/types.d.ts +1 -1
- package/dist/daemon.d.ts +58 -1
- package/dist/daemon.js +177 -8
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-context.d.ts +7 -0
- package/dist/fleet-manager.d.ts +106 -0
- package/dist/fleet-manager.js +355 -43
- package/dist/fleet-manager.js.map +1 -1
- package/dist/instance-lifecycle.d.ts +32 -0
- package/dist/instance-lifecycle.js +67 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +14 -0
- package/dist/locale.js.map +1 -1
- package/dist/topic-commands.js +11 -0
- package/dist/topic-commands.js.map +1 -1
- package/package.json +1 -1
package/dist/fleet-context.d.ts
CHANGED
|
@@ -93,6 +93,13 @@ export interface FleetContext {
|
|
|
93
93
|
promptTip?(generalName: string, adapter: import("./channel/types.js").ChannelAdapter, chatId: string, threadId?: string): Promise<"posted" | "empty" | "unavailable">;
|
|
94
94
|
/** Persistently enable advanced tips without waiting for the dismissal threshold. */
|
|
95
95
|
unlockAdvancedTips?(userId: string): boolean;
|
|
96
|
+
/** `/install-cli` backend chooser buttons. Caller enforces fleet-admin. */
|
|
97
|
+
promptInstallBackends?(chat: {
|
|
98
|
+
adapter: import("./channel/types.js").ChannelAdapter;
|
|
99
|
+
adapterId: string;
|
|
100
|
+
chatId: string;
|
|
101
|
+
threadId?: string;
|
|
102
|
+
}): Promise<void>;
|
|
96
103
|
/** `/login` backend chooser buttons. Caller enforces fleet-admin. */
|
|
97
104
|
promptLoginBackends?(chat: {
|
|
98
105
|
adapter: import("./channel/types.js").ChannelAdapter;
|
package/dist/fleet-manager.d.ts
CHANGED
|
@@ -22,6 +22,7 @@ import { ClassicChannelManager } from "./classic-channel-manager.js";
|
|
|
22
22
|
import type { InstanceState } from "./backend/types.js";
|
|
23
23
|
import { StormWindow } from "./storm-window.js";
|
|
24
24
|
import { SpawnGate } from "./spawn-gate.js";
|
|
25
|
+
import { BackendOutageTracker } from "./backend-outage.js";
|
|
25
26
|
export declare function resolveReplyThreadId(argsThreadId: unknown, instanceConfig?: InstanceConfig): string | undefined;
|
|
26
27
|
/**
|
|
27
28
|
* Pure warm-cap victim selection (extracted for testability). Given the current
|
|
@@ -73,6 +74,8 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
73
74
|
readonly lifecycle: InstanceLifecycle;
|
|
74
75
|
readonly stormWindow: StormWindow;
|
|
75
76
|
readonly spawnGate: SpawnGate;
|
|
77
|
+
/** Fleet-level backend reachability memory (fed by pty_error / startup panes). */
|
|
78
|
+
readonly backendOutage: BackendOutageTracker;
|
|
76
79
|
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
77
80
|
get daemons(): Map<string, import("./daemon.js").Daemon>;
|
|
78
81
|
fleetConfig: FleetConfig | null;
|
|
@@ -143,6 +146,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
143
146
|
private ipcWaitTails;
|
|
144
147
|
/** instanceName → restart currently executing; concurrent callers join it. */
|
|
145
148
|
private restartsInFlight;
|
|
149
|
+
/**
|
|
150
|
+
* Delayed automatic retries for instances whose startup failed. Before this a
|
|
151
|
+
* failed start was logged once and the instance stayed `stopped` until an
|
|
152
|
+
* operator noticed (2026-09-03: 4 kiro instances, all victims of the same
|
|
153
|
+
* backend outage during a post-update herd). instanceName → pending retry.
|
|
154
|
+
*/
|
|
155
|
+
private startupRetries;
|
|
156
|
+
/** Bumped by every explicit stop (incl. the stop half of a restart); fences the outage hand-off. */
|
|
157
|
+
private explicitStopGeneration;
|
|
158
|
+
/** instanceName → outage hand-off currently executing; a repeat joins it. */
|
|
159
|
+
private handOffsInFlight;
|
|
160
|
+
/** instanceName → explicit stop currently executing; the outage hand-off waits for it. */
|
|
161
|
+
private stopsInFlight;
|
|
162
|
+
/** Aggregation window for the "N instances failed to start" notice. */
|
|
163
|
+
private startupRetryNotices;
|
|
164
|
+
/** Backoff between automatic startup retries; the last step repeats while the backend is down. */
|
|
165
|
+
static readonly STARTUP_RETRY_BACKOFF_MS: number[];
|
|
166
|
+
/** Hard cap on automatic startup retries (3 backoff steps + up to 3 more during a backend outage). */
|
|
167
|
+
static readonly STARTUP_RETRY_MAX_ATTEMPTS = 6;
|
|
168
|
+
/** Re-check interval when a retry is due but the tmux storm window still blocks spawns. */
|
|
169
|
+
static readonly STARTUP_RETRY_STORM_DEFER_MS = 60000;
|
|
170
|
+
static readonly STARTUP_RETRY_NOTICE_AGGREGATE_MS = 1000;
|
|
146
171
|
private lastInboundMsg;
|
|
147
172
|
private topicArchiver;
|
|
148
173
|
controlClient: TmuxControlClient | null;
|
|
@@ -368,6 +393,65 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
368
393
|
* TODO: per-instance startup timeout (existing issue, not introduced here)
|
|
369
394
|
*/
|
|
370
395
|
private startInstancesWithConcurrency;
|
|
396
|
+
/**
|
|
397
|
+
* Start from an UNATTENDED path (fleet startup, full restart, config
|
|
398
|
+
* reconcile): nobody is watching the result, so a failure is logged and
|
|
399
|
+
* handed to the delayed automatic retry instead of leaving the instance
|
|
400
|
+
* `stopped` forever. Explicit operator/API starts call startInstance directly
|
|
401
|
+
* and keep their synchronous error. Returns whether the instance is up.
|
|
402
|
+
*/
|
|
403
|
+
private startInstanceUnattended;
|
|
404
|
+
/**
|
|
405
|
+
* Schedule attempt `attempt` (0-based) of the automatic startup retry for an
|
|
406
|
+
* instance whose start just failed. Backoff 1 → 5 → 15 min; while the
|
|
407
|
+
* instance's backend is known to be unreachable the 15-min step repeats up to
|
|
408
|
+
* STARTUP_RETRY_MAX_ATTEMPTS, after which we give up with one notice. Each
|
|
409
|
+
* retry re-checks the world (still configured, not running, not paused, no
|
|
410
|
+
* tmux storm) and runs through the SpawnGate as "recovery" — so a herd of
|
|
411
|
+
* failed instances comes back at the gate's concurrency, never all at once.
|
|
412
|
+
*/
|
|
413
|
+
scheduleStartupRetry(name: string, attempt: number): void;
|
|
414
|
+
/** Drop a pending automatic retry (an operator start/stop/restart supersedes it). */
|
|
415
|
+
cancelStartupRetry(name: string): void;
|
|
416
|
+
/** Pending automatic retry, if any (status display / tests). */
|
|
417
|
+
pendingStartupRetry(name: string): {
|
|
418
|
+
attempt: number;
|
|
419
|
+
} | null;
|
|
420
|
+
/**
|
|
421
|
+
* A daemon's crash-respawn hit the backend outage (startup_backend_unreachable):
|
|
422
|
+
* stop THAT daemon and schedule the delayed retry. Serialized against the
|
|
423
|
+
* operator paths: an in-flight restart is awaited first (it replaces the
|
|
424
|
+
* daemon itself), the stop is identity-checked so a fresh daemon registered
|
|
425
|
+
* meanwhile is never deleted, and an explicit stop/restart that began during
|
|
426
|
+
* the hand-off (generation bump) owns the outcome — no retry is scheduled
|
|
427
|
+
* behind an operator's back.
|
|
428
|
+
*/
|
|
429
|
+
handOffToStartupRetry(name: string, daemon: unknown): Promise<void>;
|
|
430
|
+
private runStartupRetry;
|
|
431
|
+
/** Fleet-topic (fleet.yaml) or ClassicBot (classic channel) — both are retryable; anything else is gone. */
|
|
432
|
+
private isConfiguredInstance;
|
|
433
|
+
/**
|
|
434
|
+
* Kind-aware start for the automatic retry: fleet-topic instances come from
|
|
435
|
+
* fleet.yaml, ClassicBot instances exist only in the classic channel manager
|
|
436
|
+
* and must be rebuilt through startClassicInstance (a fleet.yaml lookup alone
|
|
437
|
+
* silently dropped them — the retry timer fired and nothing happened).
|
|
438
|
+
*/
|
|
439
|
+
private startConfiguredInstance;
|
|
440
|
+
/**
|
|
441
|
+
* Unattended ClassicBot start (fleet startup batch, full-restart batch, the
|
|
442
|
+
* classicBot.yaml reconcile): same contract as startInstanceUnattended —
|
|
443
|
+
* failures are logged and handed to the delayed automatic retry, whose
|
|
444
|
+
* kind-aware startConfiguredInstance rebuilds the Classic instance. Returns
|
|
445
|
+
* whether the instance is up.
|
|
446
|
+
*/
|
|
447
|
+
private startClassicInstanceUnattended;
|
|
448
|
+
private backendNameOf;
|
|
449
|
+
/**
|
|
450
|
+
* One fleet-level notice per burst, not one per instance: a post-update herd
|
|
451
|
+
* fails many instances within the same second. Two notices per incident at
|
|
452
|
+
* most — "N failed, retrying in X" and, if it comes to that, "gave up on N".
|
|
453
|
+
*/
|
|
454
|
+
private queueStartupRetryNotice;
|
|
371
455
|
private runnableStartupCount;
|
|
372
456
|
private configuredStartupInstanceNames;
|
|
373
457
|
private restartProgressTarget;
|
|
@@ -832,6 +916,28 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
832
916
|
chatId: string;
|
|
833
917
|
threadId?: string;
|
|
834
918
|
}): Promise<void>;
|
|
919
|
+
/**
|
|
920
|
+
* Backend chooser for a bare `/install-cli`, mirroring promptLoginBackends so
|
|
921
|
+
* both commands feel the same. Built on postNonceButtonPrompt rather than the
|
|
922
|
+
* `/model` selection coordinator: that is the mechanism `/login` already uses,
|
|
923
|
+
* and the one whose canonical-address binding (#682) makes the buttons answer
|
|
924
|
+
* in a Telegram General topic.
|
|
925
|
+
*
|
|
926
|
+
* Unlike the login chooser this does NOT filter to backends the fleet already
|
|
927
|
+
* runs. Installing is how you get a backend you do not have yet, so filtering
|
|
928
|
+
* by configured backends would hide the only entry the admin came for.
|
|
929
|
+
*
|
|
930
|
+
* gemini-cli is omitted: it is deprecated (see backend/factory.ts). Typing
|
|
931
|
+
* `/install-cli gemini-cli` still works — this only stops recommending it.
|
|
932
|
+
*/
|
|
933
|
+
promptInstallBackends(chat: {
|
|
934
|
+
adapter: ChannelAdapter;
|
|
935
|
+
adapterId: string;
|
|
936
|
+
chatId: string;
|
|
937
|
+
threadId?: string;
|
|
938
|
+
}): Promise<void>;
|
|
939
|
+
/** Backend chooser button → start that backend's install session. */
|
|
940
|
+
private handleInstallBackendSelect;
|
|
835
941
|
/**
|
|
836
942
|
* Start a login session for one backend. Caller enforces admin.
|
|
837
943
|
* Returns a status line to post, or null when a confirmation prompt was
|