talon-agent 5.2.0 → 5.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.2.0",
3
+ "version": "5.2.2",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -110,7 +110,7 @@
110
110
  "@openai/agents": "^0.18.0",
111
111
  "@openai/codex-sdk": "^0.154.0",
112
112
  "@opencode-ai/sdk": "^1.17.4",
113
- "@playwright/mcp": "0.0.81",
113
+ "@playwright/mcp": "0.0.56",
114
114
  "@types/cross-spawn": "^6.0.6",
115
115
  "@types/qrcode": "^1.5.6",
116
116
  "baileys": "^7.0.0-rc14",
package/src/app.ts CHANGED
@@ -29,7 +29,16 @@ import { shutdownTriggers } from "./core/background/triggers/index.js";
29
29
  import { shutdownAgents } from "./core/agents/index.js";
30
30
  import { pruneSettledTriggers } from "./storage/triggers.js";
31
31
  import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
32
- import { spawnSuccessor } from "./core/daemon/respawn.js";
32
+ import {
33
+ BOOT_SMOKE_FLAG,
34
+ BOOT_SMOKE_OK,
35
+ spawnSuccessor,
36
+ } from "./core/daemon/respawn.js";
37
+ import {
38
+ crashCleanup,
39
+ crashStep,
40
+ handleUncaughtException,
41
+ } from "./core/daemon/crash.js";
33
42
  import { log, logError, logWarn } from "./util/log.js";
34
43
  import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
35
44
  import {
@@ -62,6 +71,17 @@ import {
62
71
  stopResourceSampler,
63
72
  } from "./core/daemon/resource-sampler.js";
64
73
 
74
+ // `/update` runs the freshly installed tree with --boot-smoke before it
75
+ // hands off. Every static import above has just been resolved against
76
+ // the new node_modules — reaching this line is the proof that the tree
77
+ // the successor will run is importable at all. Nothing has booted yet,
78
+ // so this is both the strongest and the last harmless place to say so.
79
+ // See core/update/self-update.ts for what a failure does instead.
80
+ if (process.argv.includes(BOOT_SMOKE_FLAG)) {
81
+ console.log(BOOT_SMOKE_OK);
82
+ process.exit(0);
83
+ }
84
+
65
85
  const { config } = await bootPhase("bootstrap", () => bootstrap());
66
86
 
67
87
  // Record this process as the daemon. The gateway port is appended once
@@ -114,6 +134,10 @@ onBackendChange((holder, newBackend, info) => {
114
134
  let shuttingDown = false;
115
135
  let triggerPruneTimer: ReturnType<typeof setInterval> | null = null;
116
136
 
137
+ // The composition root owns the SQLite handle, so it hands the crash
138
+ // path its checkpoint (see core/daemon/crash.ts).
139
+ const crashHooks = { flushDatabase };
140
+
117
141
  const SHUTDOWN_TIMEOUT_MS = 15_000;
118
142
  const DRAIN_TIMEOUT_MS = 5_000;
119
143
 
@@ -127,7 +151,11 @@ async function shutdownStep(name: string, fn: () => unknown): Promise<void> {
127
151
  try {
128
152
  await fn();
129
153
  } catch (err) {
130
- logError("shutdown", `${name} failed`, err);
154
+ // Even the report is best-effort: a shutdown triggered by a full
155
+ // disk must not die inside its own error path.
156
+ crashStep("shutdown report", () =>
157
+ logError("shutdown", `${name} failed`, err),
158
+ );
131
159
  }
132
160
  }
133
161
 
@@ -138,15 +166,18 @@ async function gracefulShutdown(signal: string): Promise<void> {
138
166
 
139
167
  const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
140
168
  const forceTimer = setTimeout(() => {
141
- logError("shutdown", "Timeout exceeded, forcing exit");
142
- // Hand off even on the forced path. A restart must survive a
143
- // subsystem that won't stop (a wedged FUSE unmount, an MCP server
144
- // ignoring SIGTERM, a backend child that never acks): without this
145
- // the timeout exits without a successor and `/restart` silently
146
- // takes the daemon down for good. The successor may briefly race
147
- // the long-poll we failed to release, but grammy retries the 409 —
148
- // a few seconds of overlap beats staying down.
149
- spawnSuccessor();
169
+ // Cleanup first, report second. Handing off matters most here: a
170
+ // restart must survive a subsystem that won't stop (a wedged FUSE
171
+ // unmount, an MCP server ignoring SIGTERM, a backend child that
172
+ // never acks), and it must also survive a logger that can't write —
173
+ // logging first is what cost us the successor on 2026-09-18. The
174
+ // successor may briefly race the long-poll we failed to release,
175
+ // but grammy retries the 409 — a few seconds of overlap beats
176
+ // staying down.
177
+ crashCleanup(crashHooks);
178
+ crashStep("timeout report", () =>
179
+ logError("shutdown", "Timeout exceeded, forcing exit"),
180
+ );
150
181
  process.exit(1);
151
182
  }, SHUTDOWN_TIMEOUT_MS);
152
183
  forceTimer.unref();
@@ -225,37 +256,29 @@ async function gracefulShutdown(signal: string): Promise<void> {
225
256
  const { shutdownHub } = await import("./core/mcp-hub/index.js");
226
257
  await shutdownHub();
227
258
  });
228
- flushDatabase();
259
+ // Each tail step stands alone: a full disk can make any of them throw,
260
+ // and none of them may cost us the ones that follow.
261
+ crashStep("database flush", () => flushDatabase());
229
262
  // Guarded removal: only clear the record if it still names us. A
230
263
  // successor that raced ahead and wrote its own pid here must not be
231
264
  // orphaned (the bug that made `talon restart` spawn duplicate daemons).
232
- removePidRecordIfOwnedBy(process.pid);
233
- log("shutdown", "State saved");
265
+ crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
266
+ crashStep("shutdown report", () => log("shutdown", "State saved"));
234
267
  // Hand off last: the frontends are stopped, so the successor binds
235
268
  // Telegram's long-poll only after we have released it. No-op unless
236
269
  // /restart or /update armed a respawn.
237
- spawnSuccessor();
270
+ crashStep("respawn handoff", () => spawnSuccessor());
238
271
  process.exit(0);
239
272
  }
240
273
 
241
274
  process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
242
275
  process.on("SIGINT", () => gracefulShutdown("SIGINT"));
243
276
 
244
- process.on("uncaughtException", (err) => {
245
- // EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
246
- // gramjs will reconnect; crashing the process here is wrong.
247
- if ((err as NodeJS.ErrnoException).code === "EPIPE") {
248
- logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`);
249
- return;
250
- }
251
- logError("bot", "Uncaught exception", err);
252
- flushDatabase();
253
- // Same pid-guarded removal as the graceful path — a crashed daemon
254
- // must not leave a record that makes `talon status` chase a dead or
255
- // recycled pid.
256
- removePidRecordIfOwnedBy(process.pid);
257
- process.exit(1);
258
- });
277
+ // Cleanup runs before the crash is reported (and the EPIPE suppression
278
+ // is unchanged) — see core/daemon/crash.ts for why the order matters.
279
+ process.on("uncaughtException", (err) =>
280
+ handleUncaughtException(err, crashHooks),
281
+ );
259
282
 
260
283
  process.on("unhandledRejection", (reason) => {
261
284
  logWarn(
@@ -333,8 +356,9 @@ async function main(): Promise<void> {
333
356
  }
334
357
 
335
358
  main().catch((err) => {
336
- logError("bot", "Fatal startup error", err);
337
- flushDatabase();
338
- removePidRecordIfOwnedBy(process.pid);
359
+ crashCleanup(crashHooks);
360
+ crashStep("startup report", () =>
361
+ logError("bot", "Fatal startup error", err),
362
+ );
339
363
  process.exit(1);
340
364
  });
@@ -0,0 +1,82 @@
1
+ /**
2
+ * Crash-path cleanup — what has to happen when the daemon goes down
3
+ * abnormally: an uncaught exception, the forced exit after a shutdown
4
+ * timeout, or a fatal startup error.
5
+ *
6
+ * Ordering is the whole point. The old handler logged first and cleaned
7
+ * up afterwards, which works right up until logging is the thing that
8
+ * broke. On 2026-09-18 the root disk filled; the log file's write stream
9
+ * emitted ENOSPC with nobody listening, the uncaught-exception handler
10
+ * called `logError` as its first act, threw inside itself, and Node
11
+ * aborted the process — no database checkpoint, a stale pidfile left
12
+ * behind, and (the expensive part) an armed `/update` handoff that never
13
+ * spawned its successor. The daemon simply vanished.
14
+ *
15
+ * So: the essentials first, each in its own try/catch, logging last and
16
+ * strictly best-effort. `src/util/log.ts` now keeps a broken log file
17
+ * from throwing at all — this is the second line of defence, for every
18
+ * other way logging can fail while the machine is sick.
19
+ */
20
+
21
+ import { logError, logWarn } from "../../util/log.js";
22
+ import { removePidRecordIfOwnedBy } from "./pidfile.js";
23
+ import { spawnSuccessor } from "./respawn.js";
24
+
25
+ /**
26
+ * Steps the crash path can only get from the composition root.
27
+ * `flushDatabase` is injected because the SQLite handle stays inside
28
+ * `storage/` (.dependency-cruiser: db-handle-stays-in-storage).
29
+ */
30
+ export type CrashHooks = {
31
+ flushDatabase: () => void;
32
+ };
33
+
34
+ /**
35
+ * Run one crash-path step. Never throws, and never uses the logger —
36
+ * a broken logger is precisely the case this path exists for, so the
37
+ * console is the fallback of last resort.
38
+ */
39
+ export function crashStep(name: string, fn: () => void): void {
40
+ try {
41
+ fn();
42
+ } catch (err) {
43
+ try {
44
+ console.error(`[talon] crash cleanup: ${name} failed`, err);
45
+ } catch {
46
+ /* stdio is gone too — nothing left to try */
47
+ }
48
+ }
49
+ }
50
+
51
+ /**
52
+ * The non-logging essentials, in the order that matters:
53
+ * 1. drop the pid record, so `talon status` stops chasing a dead pid
54
+ * (before the successor writes its own — the guarded removal then
55
+ * cannot possibly orphan it),
56
+ * 2. hand off to the successor if `/restart` or `/update` armed one
57
+ * (a no-op otherwise, see respawn.ts),
58
+ * 3. checkpoint the database.
59
+ */
60
+ export function crashCleanup(hooks: CrashHooks): void {
61
+ crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
62
+ crashStep("respawn handoff", () => spawnSuccessor());
63
+ crashStep("database flush", hooks.flushDatabase);
64
+ }
65
+
66
+ /**
67
+ * `process.on("uncaughtException")` body. Cleanup happens before the
68
+ * crash is reported, never after.
69
+ */
70
+ export function handleUncaughtException(err: Error, hooks: CrashHooks): void {
71
+ // EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
72
+ // gramjs will reconnect; crashing the process here is wrong.
73
+ if ((err as NodeJS.ErrnoException).code === "EPIPE") {
74
+ crashStep("EPIPE notice", () =>
75
+ logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`),
76
+ );
77
+ return;
78
+ }
79
+ crashCleanup(hooks);
80
+ crashStep("crash report", () => logError("bot", "Uncaught exception", err));
81
+ process.exit(1);
82
+ }
@@ -0,0 +1,192 @@
1
+ /**
2
+ * Handoff watcher — the process that proves a `/restart` or `/update`
3
+ * actually landed.
4
+ *
5
+ * The outgoing daemon spawns its successor and then calls
6
+ * `process.exit()`. Until 2026-09-18 that was the whole handoff, which
7
+ * means nothing in the system knew whether the successor had come up.
8
+ * On that day one didn't: it was spawned, lived about twenty seconds,
9
+ * never bound its gateway, and died — and because the dying parent was
10
+ * the only party to the handoff, Talon simply stayed down until a human
11
+ * noticed forty-five minutes later.
12
+ *
13
+ * So the handoff gets a witness. `spawnSuccessor()` (./respawn.ts)
14
+ * starts this watcher detached, sharing the successor's respawn.log fd.
15
+ * It polls identity-verified discovery (./discovery.ts — `app: "talon"`,
16
+ * `mode: "daemon"`, matching pid) until the successor answers /health or
17
+ * the window closes. Only a /health answer counts: discovery will also
18
+ * report a daemon whose pid is merely alive, and "the process exists" is
19
+ * exactly the claim that was false for 20 seconds on 2026-09-18. If the
20
+ * successor never serves, the watcher starts the daemon exactly
21
+ * the way `talon start` does (./control.ts — same spawn, same boot
22
+ * verification) and says why in the log. The watcher is tiny on purpose:
23
+ * `src/index.ts` dispatches its subcommand before the app graph loads,
24
+ * so it costs a bare runtime and these three modules.
25
+ */
26
+
27
+ import { dirname, resolve } from "node:path";
28
+ import { log, logError, logWarn } from "../../util/log.js";
29
+ import { startDaemon, type StartOutcome } from "./control.js";
30
+ import { findRunningInstance, type RunningInstance } from "./discovery.js";
31
+ import { isProcessAlive } from "./pidfile.js";
32
+
33
+ /** Hidden subcommand: `talon _handoff-watch <successor-pid>`. */
34
+ export const HANDOFF_WATCH_SUBCOMMAND = "_handoff-watch";
35
+
36
+ /**
37
+ * How long a successor gets to answer /health. Generous on purpose: a
38
+ * cold boot with every plugin and MCP server takes ~10s on the reference
39
+ * host, and `startDaemon()` itself waits 30s before calling a boot late.
40
+ */
41
+ const HANDOFF_WINDOW_MS = 90_000;
42
+ const POLL_MS = 500;
43
+
44
+ export type HandoffOutcome =
45
+ | { ok: true; via: "successor" | "restart"; pid: number; port?: number }
46
+ | { ok: false; reason: string };
47
+
48
+ export type WatchHandoffOptions = {
49
+ /** The pid `spawnSuccessor()` created. */
50
+ childPid: number;
51
+ /** Repo/package root, for the `talon start` equivalent. */
52
+ pkgRoot: string;
53
+ windowMs?: number;
54
+ pollMs?: number;
55
+ pidfilePath?: string;
56
+ /** Injection seams for tests. */
57
+ find?: typeof findRunningInstance;
58
+ alive?: (pid: number) => boolean;
59
+ start?: typeof startDaemon;
60
+ sleep?: (ms: number) => Promise<void>;
61
+ };
62
+
63
+ function defaultSleep(ms: number): Promise<void> {
64
+ return new Promise((r) => setTimeout(r, ms));
65
+ }
66
+
67
+ /** Why the wait ended without a live daemon. */
68
+ type WaitFailure = "successor-exited" | "window-expired";
69
+
70
+ /** Discovery found a process; only a /health answer proves it serves. */
71
+ function isServing(instance: RunningInstance | null): boolean {
72
+ return instance?.health !== undefined;
73
+ }
74
+
75
+ /**
76
+ * Poll until a serving daemon answers, the successor process
77
+ * disappears, or the window closes. A daemon that isn't our child still
78
+ * counts: the goal is a live Talon, not a particular pid.
79
+ */
80
+ async function awaitDaemon(
81
+ opts: WatchHandoffOptions,
82
+ ): Promise<RunningInstance | WaitFailure> {
83
+ const find = opts.find ?? findRunningInstance;
84
+ const alive = opts.alive ?? isProcessAlive;
85
+ const sleep = opts.sleep ?? defaultSleep;
86
+ const pollMs = opts.pollMs ?? POLL_MS;
87
+ const deadline = Date.now() + (opts.windowMs ?? HANDOFF_WINDOW_MS);
88
+
89
+ while (Date.now() < deadline) {
90
+ const instance = await find(opts.pidfilePath);
91
+ if (instance && isServing(instance)) return instance;
92
+ if (!alive(opts.childPid)) return "successor-exited";
93
+ await sleep(pollMs);
94
+ }
95
+ return "window-expired";
96
+ }
97
+
98
+ const FAILURE_DETAIL: Record<WaitFailure, string> = {
99
+ "successor-exited": "the successor exited before serving /health",
100
+ "window-expired": "the successor never answered /health in time",
101
+ };
102
+
103
+ function describeStart(outcome: StartOutcome): string {
104
+ if (outcome.ok) return `started (pid ${outcome.pid})`;
105
+ if (outcome.reason === "already-running") {
106
+ return `already running (pid ${outcome.instance.pid})`;
107
+ }
108
+ if (outcome.reason === "boot-timeout") return "spawned but not yet healthy";
109
+ return `${outcome.reason}${outcome.detail ? `: ${outcome.detail}` : ""}`;
110
+ }
111
+
112
+ function toOutcome(started: StartOutcome, why: string): HandoffOutcome {
113
+ if (started.ok) {
114
+ return { ok: true, via: "restart", pid: started.pid, port: started.port };
115
+ }
116
+ if (started.reason === "already-running") {
117
+ const inst = started.instance;
118
+ // `talon start` refuses while a pid is alive — right, since a second
119
+ // daemon would fight the first for Telegram's getUpdates. But an
120
+ // alive pid that has never served /health is the failure, not the
121
+ // recovery, so it is reported as one.
122
+ if (!isServing(inst)) {
123
+ return {
124
+ ok: false,
125
+ reason: `${why}; pid ${inst.pid} is alive but not serving — kill it and run \`talon start\``,
126
+ };
127
+ }
128
+ return { ok: true, via: "restart", pid: inst.pid, port: inst.port };
129
+ }
130
+ return { ok: false, reason: `${why}; restart ${describeStart(started)}` };
131
+ }
132
+
133
+ /**
134
+ * Verify the handoff, and repair it if it failed. Never throws: this
135
+ * process exists only to make the outcome known.
136
+ */
137
+ export async function watchHandoff(
138
+ opts: WatchHandoffOptions,
139
+ ): Promise<HandoffOutcome> {
140
+ const result = await awaitDaemon(opts);
141
+ if (typeof result !== "string") {
142
+ const via = result.pid === opts.childPid ? "successor" : "restart";
143
+ log(
144
+ "shutdown",
145
+ `Handoff verified — daemon pid ${result.pid} serving on ` +
146
+ `:${result.port ?? "?"} (${via})`,
147
+ );
148
+ return { ok: true, via, pid: result.pid, port: result.port };
149
+ }
150
+
151
+ const why = FAILURE_DETAIL[result];
152
+ logWarn(
153
+ "shutdown",
154
+ `Handoff failed — ${why}; starting Talon the way \`talon start\` does`,
155
+ );
156
+ const start = opts.start ?? startDaemon;
157
+ const started = await start({
158
+ pkgRoot: opts.pkgRoot,
159
+ pidfilePath: opts.pidfilePath,
160
+ });
161
+ const outcome = toOutcome(started, why);
162
+ if (outcome.ok)
163
+ log("shutdown", `Handoff recovered — ${describeStart(started)}`);
164
+ else logError("shutdown", `Handoff unrecoverable — ${outcome.reason}`);
165
+ return outcome;
166
+ }
167
+
168
+ /** src/core/daemon/ → the package root. */
169
+ function packageRoot(): string {
170
+ const here = import.meta.dirname ?? process.cwd();
171
+ return here.includes("$bunfs") || here.includes("~BUN")
172
+ ? dirname(process.execPath)
173
+ : resolve(here, "..", "..", "..");
174
+ }
175
+
176
+ /** Entry point for `talon _handoff-watch <pid>` (src/index.ts). */
177
+ export async function runHandoffWatch(argv: readonly string[]): Promise<void> {
178
+ const childPid = Number.parseInt(argv[0] ?? "", 10);
179
+ if (!Number.isInteger(childPid) || childPid <= 0) {
180
+ logError("shutdown", `Handoff watcher got no successor pid (${argv[0]})`);
181
+ process.exitCode = 2;
182
+ return;
183
+ }
184
+ log("shutdown", `Handoff watcher armed for pid ${childPid}`);
185
+ try {
186
+ const outcome = await watchHandoff({ childPid, pkgRoot: packageRoot() });
187
+ process.exitCode = outcome.ok ? 0 : 1;
188
+ } catch (err) {
189
+ logError("shutdown", "Handoff watcher crashed", err);
190
+ process.exitCode = 1;
191
+ }
192
+ }
@@ -1,38 +1,88 @@
1
1
  /**
2
- * Self-respawn helper for /restart commands across frontends.
2
+ * Self-respawn helper for /restart and /update across frontends.
3
3
  *
4
- * Spawns a fresh copy of the current process — same Node binary,
5
- * same `execArgv` (preserving the tsx loader so `.ts` entrypoints
6
- * still resolve), same script + user args, same cwd + env. The new
7
- * child is detached with stdio:"ignore" so it survives the parent's
8
- * exit; calling `unref()` lets the parent exit without waiting on
9
- * it.
4
+ * Spawns a fresh copy of the current process — same runtime binary, same
5
+ * `execArgv` (preserving a loader so `.ts` entrypoints still resolve),
6
+ * same script + user args, same cwd + env. The new child is detached so
7
+ * it survives the parent's exit; `unref()` lets the parent exit without
8
+ * waiting on it.
10
9
  *
11
- * Why not call the daemon's `talon restart` CLI? That path assumed
12
- * the bot was started via the daemon (talon.pid managed by
13
- * `daemonStart()`) and broke for anything else — `npm start`, `npx
14
- * tsx src/index.ts`, systemd, foreman, pm2, or running under a
15
- * debugger. Respawning from our own `process.argv` works regardless
16
- * of launch method.
10
+ * Why not call the daemon's `talon restart` CLI? That path assumed the
11
+ * bot was started via the daemon (talon.pid managed by `daemonStart()`)
12
+ * and broke for anything else — `npm start`, `npx tsx src/index.ts`,
13
+ * systemd, foreman, pm2, or running under a debugger. Respawning from
14
+ * our own `process.argv` works regardless of launch method.
17
15
  *
18
- * Ordering matters. `respawnSelf()` only *arms* the handoff and
19
- * raises SIGTERM; the successor is spawned by `spawnSuccessor()` at
20
- * the tail of graceful shutdown, once the frontends have stopped.
21
- * Spawning up-front (the previous behaviour) left the successor
22
- * long-polling `getUpdates` while the outgoing process was still
23
- * draining in-flight queries — up to DRAIN_TIMEOUT_MS of two live
24
- * pollers. Telegram answers only one of them and re-delivers the
25
- * unconfirmed updates to the other, so a restart mid-turn produced
26
- * a 409 Conflict on the way out and duplicate replies on the way in.
27
- * Releasing the poll before the successor binds it removes the
28
- * overlap rather than relying on grammy's 409 retry to paper over it.
16
+ * Ordering matters. `respawnSelf()` only *arms* the handoff and raises
17
+ * SIGTERM; the successor is spawned by `spawnSuccessor()` at the tail of
18
+ * graceful shutdown, once the frontends have stopped. Spawning up-front
19
+ * (the original behaviour) left the successor long-polling `getUpdates`
20
+ * while the outgoing process was still draining in-flight queries — up
21
+ * to DRAIN_TIMEOUT_MS of two live pollers. Telegram answers only one of
22
+ * them and re-delivers the unconfirmed updates to the other, so a
23
+ * restart mid-turn produced a 409 Conflict on the way out and duplicate
24
+ * replies on the way in. Releasing the poll before the successor binds
25
+ * it removes the overlap rather than relying on grammy's 409 retry to
26
+ * paper over it.
27
+ *
28
+ * Two things the 2026-09-18 outage added, both about the fact that the
29
+ * outgoing process is dying and cannot be the one responsible for the
30
+ * outcome:
31
+ *
32
+ * - The successor's stdout and stderr go to ~/.talon/respawn.log, not
33
+ * to "ignore". A successor that dies before its own logger exists —
34
+ * a broken import after a dependency install, a fatal bind, a
35
+ * runtime that aborts — used to leave no trace in any file, on any
36
+ * process. That is precisely what happened: a successor was spawned,
37
+ * lived ~20s, never bound its gateway, and vanished without a line.
38
+ * - A watcher process is spawned alongside it (./handoff.ts). It
39
+ * outlives us, verifies the successor over identity-checked /health
40
+ * within a bounded window, and starts the daemon the way `talon
41
+ * start` does if the successor never comes up. Nothing in the
42
+ * handoff depends on a process that is about to call process.exit().
29
43
  */
30
44
 
31
45
  import { spawn } from "node:child_process";
32
- import { log, logError } from "../../util/log.js";
46
+ import { log, logError, openRespawnLog } from "../../util/log.js";
47
+ import { HANDOFF_WATCH_SUBCOMMAND } from "./handoff.js";
33
48
 
34
49
  let pendingReason: string | null = null;
35
50
 
51
+ /**
52
+ * Argv flag that makes the daemon entry resolve its whole import graph
53
+ * and exit 0 without booting (src/app.ts acts on it before the first
54
+ * bootstrap step). `/update` runs the freshly installed tree with this
55
+ * flag before handing off — see core/update/self-update.ts.
56
+ */
57
+ export const BOOT_SMOKE_FLAG = "--boot-smoke";
58
+
59
+ /** Printed by a successful smoke run; the update step matches on it. */
60
+ export const BOOT_SMOKE_OK = "talon boot-smoke ok";
61
+
62
+ /**
63
+ * A bun-compiled binary embeds its source tree: `process.argv[1]` points
64
+ * into the virtual FS ($bunfs / ~BUN) and has no path on disk, so the
65
+ * binary is re-invoked with no script argument at all.
66
+ */
67
+ function isEmbeddedEntry(entry: string): boolean {
68
+ return entry.includes("$bunfs") || entry.includes("~BUN");
69
+ }
70
+
71
+ /** The exact command that re-runs this process. */
72
+ export function successorCommand(): { cmd: string; args: string[] } {
73
+ return {
74
+ cmd: process.argv[0],
75
+ args: [...process.execArgv, ...process.argv.slice(1)],
76
+ };
77
+ }
78
+
79
+ /** The same runtime + entry, re-invoked with one of our own subcommands. */
80
+ function selfInvocation(extra: string[]): { cmd: string; args: string[] } {
81
+ const entry = process.argv[1] ?? "";
82
+ const prefix = isEmbeddedEntry(entry) ? [] : [...process.execArgv, entry];
83
+ return { cmd: process.argv[0], args: [...prefix, ...extra] };
84
+ }
85
+
36
86
  /**
37
87
  * Arm a respawn and raise SIGTERM on ourselves so the existing
38
88
  * graceful-shutdown path cleanly stops the frontends, flushes state,
@@ -57,46 +107,71 @@ export function respawnRequested(): boolean {
57
107
  return pendingReason !== null;
58
108
  }
59
109
 
110
+ export type SpawnFn = typeof spawn;
111
+
112
+ /**
113
+ * Start the watcher that outlives this process and answers the only
114
+ * question that matters: did the successor actually come up? Never
115
+ * throws — a missing watcher must not cost us the successor itself.
116
+ */
117
+ function spawnHandoffWatcher(
118
+ childPid: number,
119
+ fd: number | null,
120
+ spawnFn: SpawnFn,
121
+ ): void {
122
+ try {
123
+ const { cmd, args } = selfInvocation([
124
+ HANDOFF_WATCH_SUBCOMMAND,
125
+ String(childPid),
126
+ ]);
127
+ const watcher = spawnFn(cmd, args, {
128
+ cwd: process.cwd(),
129
+ detached: true,
130
+ stdio: ["ignore", fd ?? "ignore", fd ?? "ignore"],
131
+ env: { ...process.env },
132
+ });
133
+ watcher.once("error", (err) => {
134
+ logError("shutdown", "Handoff watcher failed to start", err);
135
+ });
136
+ watcher.unref();
137
+ log("shutdown", `Handoff watcher started (pid ${watcher.pid})`);
138
+ } catch (err) {
139
+ logError("shutdown", "Handoff watcher failed to start", err);
140
+ }
141
+ }
142
+
60
143
  /**
61
- * Spawn the successor process. Called at the end of graceful
62
- * shutdown, after the frontends have stopped — so the incoming
63
- * process binds Telegram's long-poll only once this one has let go
64
- * of it. No-op unless `respawnSelf()` armed a handoff.
144
+ * Spawn the successor process. Called at the end of graceful shutdown,
145
+ * after the frontends have stopped — so the incoming process binds
146
+ * Telegram's long-poll only once this one has let go of it. No-op unless
147
+ * `respawnSelf()` armed a handoff.
65
148
  *
66
149
  * Never throws: a failed handoff must not prevent this process from
67
- * exiting. An external supervisor (systemd, pm2, the user's
68
- * terminal) can pick things up.
150
+ * exiting. The watcher is the safety net, not an external supervisor.
69
151
  */
70
- export function spawnSuccessor(): void {
152
+ export function spawnSuccessor(spawnFn: SpawnFn = spawn): void {
71
153
  if (pendingReason === null) return;
72
154
  const reason = pendingReason;
73
155
  pendingReason = null;
74
156
 
157
+ // One fd for both streams, shared with the watcher: the successor's
158
+ // boot output and the watcher's verdict land in the same file, in
159
+ // order, even when the successor never gets far enough to log.
160
+ const fd = openRespawnLog();
75
161
  try {
76
- const child = spawn(
77
- process.argv[0],
78
- [...process.execArgv, ...process.argv.slice(1)],
79
- {
80
- cwd: process.cwd(),
81
- detached: true,
82
- stdio: "ignore",
83
- env: { ...process.env },
84
- },
85
- );
162
+ const child = spawnFn(process.argv[0], successorCommand().args, {
163
+ cwd: process.cwd(),
164
+ detached: true,
165
+ stdio: ["ignore", fd ?? "ignore", fd ?? "ignore"],
166
+ env: { ...process.env },
167
+ });
86
168
  child.once("error", (err) => {
87
- logError(
88
- "shutdown",
89
- `Respawn failed; exiting without a successor — restart manually`,
90
- err,
91
- );
169
+ logError("shutdown", `Respawn failed (${reason}) — see respawn.log`, err);
92
170
  });
93
171
  child.unref();
94
172
  log("shutdown", `Respawn child started (pid ${child.pid}) — ${reason}`);
173
+ if (child.pid) spawnHandoffWatcher(child.pid, fd, spawnFn);
95
174
  } catch (err) {
96
- logError(
97
- "shutdown",
98
- `Respawn failed; exiting without a successor — restart manually`,
99
- err,
100
- );
175
+ logError("shutdown", `Respawn failed (${reason}) — see respawn.log`, err);
101
176
  }
102
177
  }