@songsid/agend 2.1.4-beta.48 → 2.1.4-beta.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/kiro.d.ts +14 -0
- package/dist/backend/kiro.js +48 -0
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/types.d.ts +26 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/backend-outage.d.ts +61 -0
- package/dist/backend-outage.js +71 -0
- package/dist/backend-outage.js.map +1 -0
- package/dist/daemon.d.ts +58 -1
- package/dist/daemon.js +177 -8
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-manager.d.ts +84 -0
- package/dist/fleet-manager.js +288 -40
- package/dist/fleet-manager.js.map +1 -1
- package/dist/instance-lifecycle.d.ts +32 -0
- package/dist/instance-lifecycle.js +67 -4
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/locale.js +8 -0
- package/dist/locale.js.map +1 -1
- package/package.json +1 -1
package/dist/daemon.js
CHANGED
|
@@ -361,6 +361,18 @@ export class PendingWorkTracker {
|
|
|
361
361
|
const NORMAL_ENTER_SETTLE_MS = 500;
|
|
362
362
|
/** Bottom-ready re-poll cadence for Enter-dropping TUIs once the pane is quiet. */
|
|
363
363
|
const BOTTOM_READY_POLL_MS = 250;
|
|
364
|
+
/**
|
|
365
|
+
* Startup failed because the CLI's backend is unreachable (see backend-outage.ts).
|
|
366
|
+
* The session-id is deliberately KEPT; the fleet schedules a delayed retry.
|
|
367
|
+
*/
|
|
368
|
+
export class BackendUnreachableStartupError extends Error {
|
|
369
|
+
backend;
|
|
370
|
+
constructor(backend) {
|
|
371
|
+
super(`CLI failed to start: the ${backend} backend is unreachable — session kept for a later retry`);
|
|
372
|
+
this.backend = backend;
|
|
373
|
+
this.name = "BackendUnreachableStartupError";
|
|
374
|
+
}
|
|
375
|
+
}
|
|
364
376
|
/** Bounded wait (under the pane lock) for the prompt to return before retrying a dropped Enter. */
|
|
365
377
|
const STRANDED_RETRY_READY_WAIT_MS = 30_000;
|
|
366
378
|
/** Max "stranded text → submit → wait for prompt → re-check" rounds per delivery; each may send one recovery Enter. */
|
|
@@ -737,6 +749,7 @@ export class Daemon extends EventEmitter {
|
|
|
737
749
|
runtimeIdentity;
|
|
738
750
|
spawnGate;
|
|
739
751
|
stormWindow;
|
|
752
|
+
backendOutage;
|
|
740
753
|
logger;
|
|
741
754
|
tmuxSessionName;
|
|
742
755
|
tmux = null;
|
|
@@ -785,6 +798,7 @@ export class Daemon extends EventEmitter {
|
|
|
785
798
|
resolveSpawnSettled = null;
|
|
786
799
|
spawnDepth = 0;
|
|
787
800
|
skipResume = false;
|
|
801
|
+
startupAborted = false;
|
|
788
802
|
backgroundSessionRecoveryAttempted = false;
|
|
789
803
|
/** Whether the last spawn started a fresh session (not resumed). */
|
|
790
804
|
isNewSession = false;
|
|
@@ -1005,7 +1019,9 @@ export class Daemon extends EventEmitter {
|
|
|
1005
1019
|
this.errorRecoveryDeadlineAt = 0;
|
|
1006
1020
|
this.activeErrorPatternKey = null;
|
|
1007
1021
|
}
|
|
1008
|
-
constructor(name, config, instanceDir, topicMode = false, backend, controlClient, rootLogger, runtimeIdentity, spawnGate, stormWindow
|
|
1022
|
+
constructor(name, config, instanceDir, topicMode = false, backend, controlClient, rootLogger, runtimeIdentity, spawnGate, stormWindow,
|
|
1023
|
+
/** Fleet-level "is this CLI's backend down?" memory — see backend-outage.ts. */
|
|
1024
|
+
backendOutage) {
|
|
1009
1025
|
super();
|
|
1010
1026
|
this.name = name;
|
|
1011
1027
|
this.config = config;
|
|
@@ -1016,6 +1032,7 @@ export class Daemon extends EventEmitter {
|
|
|
1016
1032
|
this.runtimeIdentity = runtimeIdentity;
|
|
1017
1033
|
this.spawnGate = spawnGate;
|
|
1018
1034
|
this.stormWindow = stormWindow;
|
|
1035
|
+
this.backendOutage = backendOutage;
|
|
1019
1036
|
if (!rootLogger)
|
|
1020
1037
|
throw new Error("Daemon requires a shared root logger");
|
|
1021
1038
|
this.runtimeIdentity ??= {
|
|
@@ -1862,7 +1879,9 @@ export class Daemon extends EventEmitter {
|
|
|
1862
1879
|
this.stormWindow?.markRecovered(this.name);
|
|
1863
1880
|
}
|
|
1864
1881
|
catch (err) {
|
|
1865
|
-
this.
|
|
1882
|
+
if (!this.handOffBackendUnreachableRespawn(err)) {
|
|
1883
|
+
this.logger.error({ err }, `Failed to respawn ${cliLabel} window`);
|
|
1884
|
+
}
|
|
1866
1885
|
}
|
|
1867
1886
|
}
|
|
1868
1887
|
catch (err) {
|
|
@@ -2546,15 +2565,20 @@ export class Daemon extends EventEmitter {
|
|
|
2546
2565
|
}
|
|
2547
2566
|
this.pauseWakeState = "waking";
|
|
2548
2567
|
this.beginSpawn();
|
|
2568
|
+
// A wake is a real resume launch: it gets at least the backend's resume
|
|
2569
|
+
// budget (kiro: 60s — the conversation must come back from the backend
|
|
2570
|
+
// before anything paints), never less than the caller's or the configured
|
|
2571
|
+
// startup_timeout_ms.
|
|
2572
|
+
const budgetMs = this.wakeBudgetMs(timeoutMs);
|
|
2549
2573
|
const transition = this.autoPauseController.wakeOnDeliver(async () => {
|
|
2550
2574
|
let timeout;
|
|
2551
2575
|
try {
|
|
2552
2576
|
const ready = await Promise.race([
|
|
2553
|
-
this.trySpawn(true,
|
|
2554
|
-
new Promise(resolve => { timeout = setTimeout(() => resolve(false),
|
|
2577
|
+
this.trySpawn(true, budgetMs),
|
|
2578
|
+
new Promise(resolve => { timeout = setTimeout(() => resolve(false), budgetMs); }),
|
|
2555
2579
|
]);
|
|
2556
2580
|
if (!ready)
|
|
2557
|
-
throw new Error(`Wake timed out before CLI became ready (${
|
|
2581
|
+
throw new Error(`Wake timed out before CLI became ready (${budgetMs}ms)`);
|
|
2558
2582
|
}
|
|
2559
2583
|
finally {
|
|
2560
2584
|
if (timeout)
|
|
@@ -4671,9 +4695,35 @@ export class Daemon extends EventEmitter {
|
|
|
4671
4695
|
throw new Error("No backend configured — cannot spawn CLI window");
|
|
4672
4696
|
}
|
|
4673
4697
|
const attemptedResume = !this.skipResume;
|
|
4674
|
-
|
|
4698
|
+
// A resume launch may get a longer budget than a fresh one (kiro: the
|
|
4699
|
+
// conversation must come back from the backend before anything paints).
|
|
4700
|
+
const resumeBudget = attemptedResume ? this.startupBudgetFor(true) : undefined;
|
|
4701
|
+
let alive = await this.trySpawn(false, resumeBudget);
|
|
4702
|
+
if (!alive && attemptedResume) {
|
|
4703
|
+
// Resume failed. Before abandoning the session:
|
|
4704
|
+
// 1. If the backend is unreachable (fleet-level memory, or this very pane
|
|
4705
|
+
// shows the outage text), clearing the session would trade a
|
|
4706
|
+
// transient outage for permanent conversation loss — and the fresh
|
|
4707
|
+
// start would hang on the same backend. Fail the startup and keep the
|
|
4708
|
+
// session; the fleet's delayed retry comes back once it is up.
|
|
4709
|
+
// 2. Otherwise, for backends that ask for it, retry resume ONCE — the
|
|
4710
|
+
// first miss is usually slowness, not a broken session.
|
|
4711
|
+
await this.noteStartupPaneForBackendOutage();
|
|
4712
|
+
await this.failStartupIfBackendUnreachable();
|
|
4713
|
+
if (this.backend.retriesResumeOnStartupFailure?.() === true) {
|
|
4714
|
+
this.logger.warn("Resume startup failed — retrying resume once before abandoning the session");
|
|
4715
|
+
await this.killProcessTree();
|
|
4716
|
+
await this.tmux.killWindow();
|
|
4717
|
+
alive = await this.trySpawn(false, resumeBudget);
|
|
4718
|
+
if (!alive) {
|
|
4719
|
+
await this.noteStartupPaneForBackendOutage();
|
|
4720
|
+
await this.failStartupIfBackendUnreachable();
|
|
4721
|
+
}
|
|
4722
|
+
}
|
|
4723
|
+
}
|
|
4675
4724
|
if (!alive) {
|
|
4676
|
-
//
|
|
4725
|
+
// Resume (or a fresh start) failed for a reason we do not recognise as an
|
|
4726
|
+
// outage (stale --resume, crash, rate limit, etc.).
|
|
4677
4727
|
// Clean slate: clear session-id, skip resume, and retry once.
|
|
4678
4728
|
this.logger.warn("CLI startup failed — clearing session-id and retrying without resume");
|
|
4679
4729
|
const sidFile = join(this.instanceDir, "session-id");
|
|
@@ -4684,7 +4734,7 @@ export class Daemon extends EventEmitter {
|
|
|
4684
4734
|
this.skipResume = true;
|
|
4685
4735
|
await this.killProcessTree();
|
|
4686
4736
|
await this.tmux.killWindow();
|
|
4687
|
-
const retryAlive = await this.trySpawn();
|
|
4737
|
+
const retryAlive = await this.trySpawn(false, this.startupBudgetFor(false));
|
|
4688
4738
|
if (!retryAlive) {
|
|
4689
4739
|
await this.killProcessTree();
|
|
4690
4740
|
await this.tmux.killWindow();
|
|
@@ -4693,6 +4743,9 @@ export class Daemon extends EventEmitter {
|
|
|
4693
4743
|
}
|
|
4694
4744
|
else if (attemptedResume) {
|
|
4695
4745
|
resumedSuccessfully = true;
|
|
4746
|
+
// A resume needs the backend: its success is positive proof the backend
|
|
4747
|
+
// is reachable again (a fresh prompt is local and proves nothing).
|
|
4748
|
+
this.backendOutage?.clear(this.backendKey());
|
|
4696
4749
|
}
|
|
4697
4750
|
this.lastSpawnAt = Date.now();
|
|
4698
4751
|
this.skipResume = false; // CLI started successfully — reset for next spawn
|
|
@@ -4703,6 +4756,122 @@ export class Daemon extends EventEmitter {
|
|
|
4703
4756
|
}
|
|
4704
4757
|
return resumedSuccessfully;
|
|
4705
4758
|
}
|
|
4759
|
+
/**
|
|
4760
|
+
* Startup budget for this launch: the backend's override (resume-aware), never
|
|
4761
|
+
* below a user-configured startup_timeout_ms; undefined = trySpawn's default.
|
|
4762
|
+
*/
|
|
4763
|
+
startupBudgetFor(resume) {
|
|
4764
|
+
const override = this.backend?.getStartupBudgetMs?.({ resume });
|
|
4765
|
+
if (override == null)
|
|
4766
|
+
return undefined;
|
|
4767
|
+
const configured = this.config.startup_timeout_ms;
|
|
4768
|
+
return configured != null ? Math.max(configured, override) : override;
|
|
4769
|
+
}
|
|
4770
|
+
/** Effective wake budget = max(caller timeout, backend resume override, configured startup_timeout_ms). */
|
|
4771
|
+
wakeBudgetMs(timeoutMs) {
|
|
4772
|
+
return Math.max(timeoutMs, this.startupBudgetFor(!this.skipResume) ?? 0, this.config.startup_timeout_ms ?? 0);
|
|
4773
|
+
}
|
|
4774
|
+
/** The fleet-level backend key this instance runs on (matches the lifecycle's backendOf). */
|
|
4775
|
+
backendKey() {
|
|
4776
|
+
return this.runtimeIdentity?.backend ?? this.config.backend ?? this.backend?.binaryName ?? "unknown";
|
|
4777
|
+
}
|
|
4778
|
+
/**
|
|
4779
|
+
* After a failed launch, read the pane for the backend's fleet-wide outage
|
|
4780
|
+
* text (e.g. kiro's `dispatch failure (timeout) … kiro.dev`) and record it.
|
|
4781
|
+
* The lifecycle only learns about outages from RUNNING instances; during a
|
|
4782
|
+
* post-update herd nothing is running yet, so the startup path must report
|
|
4783
|
+
* what it sees or the first instance in would clear its session for nothing.
|
|
4784
|
+
*/
|
|
4785
|
+
async noteStartupPaneForBackendOutage() {
|
|
4786
|
+
if (!this.backendOutage || !this.backend || !this.tmux)
|
|
4787
|
+
return;
|
|
4788
|
+
const outagePattern = this.backend.getErrorPatterns?.().find(p => p.fleetWide && p.type === "network");
|
|
4789
|
+
if (!outagePattern)
|
|
4790
|
+
return;
|
|
4791
|
+
try {
|
|
4792
|
+
const pane = await this.tmux.capturePane();
|
|
4793
|
+
if (!outagePattern.pattern.test(pane))
|
|
4794
|
+
return;
|
|
4795
|
+
const detail = this.resolveErrorMessage(pane, outagePattern);
|
|
4796
|
+
this.backendOutage.record(this.backendKey(), this.name, detail);
|
|
4797
|
+
this.logger.warn({ detail }, "Startup pane shows the backend outage text");
|
|
4798
|
+
}
|
|
4799
|
+
catch { /* capture failed — nothing to learn */ }
|
|
4800
|
+
}
|
|
4801
|
+
/**
|
|
4802
|
+
* Fail the startup WITHOUT touching the session while the backend is known to
|
|
4803
|
+
* be down. The failed window is torn down first: the alternative — a live but
|
|
4804
|
+
* never-ready CLI left behind while the daemon reports `crashed` — would keep
|
|
4805
|
+
* the health monitor from ever seeing a dead pane and retrying.
|
|
4806
|
+
*/
|
|
4807
|
+
async failStartupIfBackendUnreachable() {
|
|
4808
|
+
if (!this.backendOutage?.isActive(this.backendKey()))
|
|
4809
|
+
return;
|
|
4810
|
+
this.logger.warn("Backend unreachable — keeping the session and failing startup for a delayed retry");
|
|
4811
|
+
await this.killProcessTree();
|
|
4812
|
+
await this.tmux.killWindow();
|
|
4813
|
+
throw new BackendUnreachableStartupError(this.backendKey());
|
|
4814
|
+
}
|
|
4815
|
+
/**
|
|
4816
|
+
* Crash-respawn ran into the backend outage: the ordinary path would leave
|
|
4817
|
+
* `crashed` + a paused monitor + nothing scheduled (and count it as a crash).
|
|
4818
|
+
* Hand the instance to the fleet instead — the lifecycle stops this daemon
|
|
4819
|
+
* and the fleet's delayed startup retry re-creates it later with the session
|
|
4820
|
+
* intact. Returns true when someone took the hand-off.
|
|
4821
|
+
*/
|
|
4822
|
+
handOffBackendUnreachableRespawn(err) {
|
|
4823
|
+
if (!(err instanceof BackendUnreachableStartupError))
|
|
4824
|
+
return false;
|
|
4825
|
+
if (this.listenerCount("startup_backend_unreachable") === 0)
|
|
4826
|
+
return false;
|
|
4827
|
+
this.logger.warn({ backend: err.backend }, "Respawn blocked by backend outage — handing off to the fleet's delayed startup retry");
|
|
4828
|
+
this.healthCheckPaused = true;
|
|
4829
|
+
this.emit("startup_backend_unreachable", { name: this.name, backend: err.backend });
|
|
4830
|
+
return true;
|
|
4831
|
+
}
|
|
4832
|
+
/**
|
|
4833
|
+
* Dispose a Daemon whose start() rejected. The lifecycle registers a daemon
|
|
4834
|
+
* only after start() resolves, so nothing else owns this object — and by the
|
|
4835
|
+
* time the spawn fails it has already bound its IPC server (channel.sock),
|
|
4836
|
+
* written daemon.pid and possibly created a window. Left alone, each failed
|
|
4837
|
+
* start leaks one live-but-unreachable server handle (the next start unlinks
|
|
4838
|
+
* the socket path and binds a new one), multiplied by the automatic startup
|
|
4839
|
+
* retries. Idempotent, and fast: no graceful CLI quit, the CLI never reached
|
|
4840
|
+
* its prompt.
|
|
4841
|
+
*/
|
|
4842
|
+
async abortStartup() {
|
|
4843
|
+
if (this.startupAborted)
|
|
4844
|
+
return;
|
|
4845
|
+
this.startupAborted = true;
|
|
4846
|
+
this.freezeRuntimeMonitors();
|
|
4847
|
+
this.pendingIpcRequests.clear();
|
|
4848
|
+
try {
|
|
4849
|
+
await this.killProcessTree();
|
|
4850
|
+
}
|
|
4851
|
+
catch { /* nothing running */ }
|
|
4852
|
+
if (this.tmux) {
|
|
4853
|
+
const windowId = this.tmux.getWindowId();
|
|
4854
|
+
try {
|
|
4855
|
+
await this.tmux.killWindow();
|
|
4856
|
+
}
|
|
4857
|
+
catch { /* window may not exist */ }
|
|
4858
|
+
if (windowId)
|
|
4859
|
+
this.controlClient?.unregisterWindow(windowId);
|
|
4860
|
+
}
|
|
4861
|
+
try {
|
|
4862
|
+
await this.ipcServer?.close();
|
|
4863
|
+
}
|
|
4864
|
+
catch (err) {
|
|
4865
|
+
this.logger.debug({ err }, "IPC server close failed during startup abort");
|
|
4866
|
+
}
|
|
4867
|
+
this.ipcServer = null;
|
|
4868
|
+
for (const file of ["daemon.pid", "window-id"]) {
|
|
4869
|
+
try {
|
|
4870
|
+
unlinkSync(join(this.instanceDir, file));
|
|
4871
|
+
}
|
|
4872
|
+
catch { /* absent */ }
|
|
4873
|
+
}
|
|
4874
|
+
}
|
|
4706
4875
|
/** Kill the entire process tree of the current tmux pane (CLI + MCP server). */
|
|
4707
4876
|
async killProcessTree(signal = "SIGTERM") {
|
|
4708
4877
|
if (!this.tmux)
|