@songsid/agend 2.1.4-beta.47 → 2.1.4-beta.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/daemon.js CHANGED
@@ -361,6 +361,18 @@ export class PendingWorkTracker {
361
361
  const NORMAL_ENTER_SETTLE_MS = 500;
362
362
  /** Bottom-ready re-poll cadence for Enter-dropping TUIs once the pane is quiet. */
363
363
  const BOTTOM_READY_POLL_MS = 250;
364
+ /**
365
+ * Startup failed because the CLI's backend is unreachable (see backend-outage.ts).
366
+ * The session-id is deliberately KEPT; the fleet schedules a delayed retry.
367
+ */
368
+ export class BackendUnreachableStartupError extends Error {
369
+ backend;
370
+ constructor(backend) {
371
+ super(`CLI failed to start: the ${backend} backend is unreachable — session kept for a later retry`);
372
+ this.backend = backend;
373
+ this.name = "BackendUnreachableStartupError";
374
+ }
375
+ }
364
376
  /** Bounded wait (under the pane lock) for the prompt to return before retrying a dropped Enter. */
365
377
  const STRANDED_RETRY_READY_WAIT_MS = 30_000;
366
378
  /** Max "stranded text → submit → wait for prompt → re-check" rounds per delivery; each may send one recovery Enter. */
@@ -737,6 +749,7 @@ export class Daemon extends EventEmitter {
737
749
  runtimeIdentity;
738
750
  spawnGate;
739
751
  stormWindow;
752
+ backendOutage;
740
753
  logger;
741
754
  tmuxSessionName;
742
755
  tmux = null;
@@ -785,6 +798,7 @@ export class Daemon extends EventEmitter {
785
798
  resolveSpawnSettled = null;
786
799
  spawnDepth = 0;
787
800
  skipResume = false;
801
+ startupAborted = false;
788
802
  backgroundSessionRecoveryAttempted = false;
789
803
  /** Whether the last spawn started a fresh session (not resumed). */
790
804
  isNewSession = false;
@@ -1005,7 +1019,9 @@ export class Daemon extends EventEmitter {
1005
1019
  this.errorRecoveryDeadlineAt = 0;
1006
1020
  this.activeErrorPatternKey = null;
1007
1021
  }
1008
- constructor(name, config, instanceDir, topicMode = false, backend, controlClient, rootLogger, runtimeIdentity, spawnGate, stormWindow) {
1022
+ constructor(name, config, instanceDir, topicMode = false, backend, controlClient, rootLogger, runtimeIdentity, spawnGate, stormWindow,
1023
+ /** Fleet-level "is this CLI's backend down?" memory — see backend-outage.ts. */
1024
+ backendOutage) {
1009
1025
  super();
1010
1026
  this.name = name;
1011
1027
  this.config = config;
@@ -1016,6 +1032,7 @@ export class Daemon extends EventEmitter {
1016
1032
  this.runtimeIdentity = runtimeIdentity;
1017
1033
  this.spawnGate = spawnGate;
1018
1034
  this.stormWindow = stormWindow;
1035
+ this.backendOutage = backendOutage;
1019
1036
  if (!rootLogger)
1020
1037
  throw new Error("Daemon requires a shared root logger");
1021
1038
  this.runtimeIdentity ??= {
@@ -1862,7 +1879,9 @@ export class Daemon extends EventEmitter {
1862
1879
  this.stormWindow?.markRecovered(this.name);
1863
1880
  }
1864
1881
  catch (err) {
1865
- this.logger.error({ err }, `Failed to respawn ${cliLabel} window`);
1882
+ if (!this.handOffBackendUnreachableRespawn(err)) {
1883
+ this.logger.error({ err }, `Failed to respawn ${cliLabel} window`);
1884
+ }
1866
1885
  }
1867
1886
  }
1868
1887
  catch (err) {
@@ -2546,15 +2565,20 @@ export class Daemon extends EventEmitter {
2546
2565
  }
2547
2566
  this.pauseWakeState = "waking";
2548
2567
  this.beginSpawn();
2568
+ // A wake is a real resume launch: it gets at least the backend's resume
2569
+ // budget (kiro: 60s — the conversation must come back from the backend
2570
+ // before anything paints), never less than the caller's or the configured
2571
+ // startup_timeout_ms.
2572
+ const budgetMs = this.wakeBudgetMs(timeoutMs);
2549
2573
  const transition = this.autoPauseController.wakeOnDeliver(async () => {
2550
2574
  let timeout;
2551
2575
  try {
2552
2576
  const ready = await Promise.race([
2553
- this.trySpawn(true, timeoutMs),
2554
- new Promise(resolve => { timeout = setTimeout(() => resolve(false), timeoutMs); }),
2577
+ this.trySpawn(true, budgetMs),
2578
+ new Promise(resolve => { timeout = setTimeout(() => resolve(false), budgetMs); }),
2555
2579
  ]);
2556
2580
  if (!ready)
2557
- throw new Error(`Wake timed out before CLI became ready (${timeoutMs}ms)`);
2581
+ throw new Error(`Wake timed out before CLI became ready (${budgetMs}ms)`);
2558
2582
  }
2559
2583
  finally {
2560
2584
  if (timeout)
@@ -4671,9 +4695,35 @@ export class Daemon extends EventEmitter {
4671
4695
  throw new Error("No backend configured — cannot spawn CLI window");
4672
4696
  }
4673
4697
  const attemptedResume = !this.skipResume;
4674
- const alive = await this.trySpawn();
4698
+ // A resume launch may get a longer budget than a fresh one (kiro: the
4699
+ // conversation must come back from the backend before anything paints).
4700
+ const resumeBudget = attemptedResume ? this.startupBudgetFor(true) : undefined;
4701
+ let alive = await this.trySpawn(false, resumeBudget);
4702
+ if (!alive && attemptedResume) {
4703
+ // Resume failed. Before abandoning the session:
4704
+ // 1. If the backend is unreachable (fleet-level memory, or this very pane
4705
+ // shows the outage text), clearing the session would trade a
4706
+ // transient outage for permanent conversation loss — and the fresh
4707
+ // start would hang on the same backend. Fail the startup and keep the
4708
+ // session; the fleet's delayed retry comes back once it is up.
4709
+ // 2. Otherwise, for backends that ask for it, retry resume ONCE — the
4710
+ // first miss is usually slowness, not a broken session.
4711
+ await this.noteStartupPaneForBackendOutage();
4712
+ await this.failStartupIfBackendUnreachable();
4713
+ if (this.backend.retriesResumeOnStartupFailure?.() === true) {
4714
+ this.logger.warn("Resume startup failed — retrying resume once before abandoning the session");
4715
+ await this.killProcessTree();
4716
+ await this.tmux.killWindow();
4717
+ alive = await this.trySpawn(false, resumeBudget);
4718
+ if (!alive) {
4719
+ await this.noteStartupPaneForBackendOutage();
4720
+ await this.failStartupIfBackendUnreachable();
4721
+ }
4722
+ }
4723
+ }
4675
4724
  if (!alive) {
4676
- // First attempt failed (stale --resume, crash, rate limit, etc.)
4725
+ // Resume (or a fresh start) failed for a reason we do not recognise as an
4726
+ // outage (stale --resume, crash, rate limit, etc.).
4677
4727
  // Clean slate: clear session-id, skip resume, and retry once.
4678
4728
  this.logger.warn("CLI startup failed — clearing session-id and retrying without resume");
4679
4729
  const sidFile = join(this.instanceDir, "session-id");
@@ -4684,7 +4734,7 @@ export class Daemon extends EventEmitter {
4684
4734
  this.skipResume = true;
4685
4735
  await this.killProcessTree();
4686
4736
  await this.tmux.killWindow();
4687
- const retryAlive = await this.trySpawn();
4737
+ const retryAlive = await this.trySpawn(false, this.startupBudgetFor(false));
4688
4738
  if (!retryAlive) {
4689
4739
  await this.killProcessTree();
4690
4740
  await this.tmux.killWindow();
@@ -4693,6 +4743,9 @@ export class Daemon extends EventEmitter {
4693
4743
  }
4694
4744
  else if (attemptedResume) {
4695
4745
  resumedSuccessfully = true;
4746
+ // A resume needs the backend: its success is positive proof the backend
4747
+ // is reachable again (a fresh prompt is local and proves nothing).
4748
+ this.backendOutage?.clear(this.backendKey());
4696
4749
  }
4697
4750
  this.lastSpawnAt = Date.now();
4698
4751
  this.skipResume = false; // CLI started successfully — reset for next spawn
@@ -4703,6 +4756,122 @@ export class Daemon extends EventEmitter {
4703
4756
  }
4704
4757
  return resumedSuccessfully;
4705
4758
  }
4759
+ /**
4760
+ * Startup budget for this launch: the backend's override (resume-aware), never
4761
+ * below a user-configured startup_timeout_ms; undefined = trySpawn's default.
4762
+ */
4763
+ startupBudgetFor(resume) {
4764
+ const override = this.backend?.getStartupBudgetMs?.({ resume });
4765
+ if (override == null)
4766
+ return undefined;
4767
+ const configured = this.config.startup_timeout_ms;
4768
+ return configured != null ? Math.max(configured, override) : override;
4769
+ }
4770
+ /** Effective wake budget = max(caller timeout, backend resume override, configured startup_timeout_ms). */
4771
+ wakeBudgetMs(timeoutMs) {
4772
+ return Math.max(timeoutMs, this.startupBudgetFor(!this.skipResume) ?? 0, this.config.startup_timeout_ms ?? 0);
4773
+ }
4774
+ /** The fleet-level backend key this instance runs on (matches the lifecycle's backendOf). */
4775
+ backendKey() {
4776
+ return this.runtimeIdentity?.backend ?? this.config.backend ?? this.backend?.binaryName ?? "unknown";
4777
+ }
4778
+ /**
4779
+ * After a failed launch, read the pane for the backend's fleet-wide outage
4780
+ * text (e.g. kiro's `dispatch failure (timeout) … kiro.dev`) and record it.
4781
+ * The lifecycle only learns about outages from RUNNING instances; during a
4782
+ * post-update herd nothing is running yet, so the startup path must report
4783
+ * what it sees or the first instance in would clear its session for nothing.
4784
+ */
4785
+ async noteStartupPaneForBackendOutage() {
4786
+ if (!this.backendOutage || !this.backend || !this.tmux)
4787
+ return;
4788
+ const outagePattern = this.backend.getErrorPatterns?.().find(p => p.fleetWide && p.type === "network");
4789
+ if (!outagePattern)
4790
+ return;
4791
+ try {
4792
+ const pane = await this.tmux.capturePane();
4793
+ if (!outagePattern.pattern.test(pane))
4794
+ return;
4795
+ const detail = this.resolveErrorMessage(pane, outagePattern);
4796
+ this.backendOutage.record(this.backendKey(), this.name, detail);
4797
+ this.logger.warn({ detail }, "Startup pane shows the backend outage text");
4798
+ }
4799
+ catch { /* capture failed — nothing to learn */ }
4800
+ }
4801
+ /**
4802
+ * Fail the startup WITHOUT touching the session while the backend is known to
4803
+ * be down. The failed window is torn down first: the alternative — a live but
4804
+ * never-ready CLI left behind while the daemon reports `crashed` — would keep
4805
+ * the health monitor from ever seeing a dead pane and retrying.
4806
+ */
4807
+ async failStartupIfBackendUnreachable() {
4808
+ if (!this.backendOutage?.isActive(this.backendKey()))
4809
+ return;
4810
+ this.logger.warn("Backend unreachable — keeping the session and failing startup for a delayed retry");
4811
+ await this.killProcessTree();
4812
+ await this.tmux.killWindow();
4813
+ throw new BackendUnreachableStartupError(this.backendKey());
4814
+ }
4815
+ /**
4816
+ * Crash-respawn ran into the backend outage: the ordinary path would leave
4817
+ * `crashed` + a paused monitor + nothing scheduled (and count it as a crash).
4818
+ * Hand the instance to the fleet instead — the lifecycle stops this daemon
4819
+ * and the fleet's delayed startup retry re-creates it later with the session
4820
+ * intact. Returns true when someone took the hand-off.
4821
+ */
4822
+ handOffBackendUnreachableRespawn(err) {
4823
+ if (!(err instanceof BackendUnreachableStartupError))
4824
+ return false;
4825
+ if (this.listenerCount("startup_backend_unreachable") === 0)
4826
+ return false;
4827
+ this.logger.warn({ backend: err.backend }, "Respawn blocked by backend outage — handing off to the fleet's delayed startup retry");
4828
+ this.healthCheckPaused = true;
4829
+ this.emit("startup_backend_unreachable", { name: this.name, backend: err.backend });
4830
+ return true;
4831
+ }
4832
+ /**
4833
+ * Dispose a Daemon whose start() rejected. The lifecycle registers a daemon
4834
+ * only after start() resolves, so nothing else owns this object — and by the
4835
+ * time the spawn fails it has already bound its IPC server (channel.sock),
4836
+ * written daemon.pid and possibly created a window. Left alone, each failed
4837
+ * start leaks one live-but-unreachable server handle (the next start unlinks
4838
+ * the socket path and binds a new one), multiplied by the automatic startup
4839
+ * retries. Idempotent, and fast: no graceful CLI quit, the CLI never reached
4840
+ * its prompt.
4841
+ */
4842
+ async abortStartup() {
4843
+ if (this.startupAborted)
4844
+ return;
4845
+ this.startupAborted = true;
4846
+ this.freezeRuntimeMonitors();
4847
+ this.pendingIpcRequests.clear();
4848
+ try {
4849
+ await this.killProcessTree();
4850
+ }
4851
+ catch { /* nothing running */ }
4852
+ if (this.tmux) {
4853
+ const windowId = this.tmux.getWindowId();
4854
+ try {
4855
+ await this.tmux.killWindow();
4856
+ }
4857
+ catch { /* window may not exist */ }
4858
+ if (windowId)
4859
+ this.controlClient?.unregisterWindow(windowId);
4860
+ }
4861
+ try {
4862
+ await this.ipcServer?.close();
4863
+ }
4864
+ catch (err) {
4865
+ this.logger.debug({ err }, "IPC server close failed during startup abort");
4866
+ }
4867
+ this.ipcServer = null;
4868
+ for (const file of ["daemon.pid", "window-id"]) {
4869
+ try {
4870
+ unlinkSync(join(this.instanceDir, file));
4871
+ }
4872
+ catch { /* absent */ }
4873
+ }
4874
+ }
4706
4875
  /** Kill the entire process tree of the current tmux pane (CLI + MCP server). */
4707
4876
  async killProcessTree(signal = "SIGTERM") {
4708
4877
  if (!this.tmux)