talon-agent 5.2.0 → 5.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/src/app.ts +57 -33
- package/src/core/daemon/crash.ts +82 -0
- package/src/core/daemon/handoff.ts +192 -0
- package/src/core/daemon/respawn.ts +127 -52
- package/src/core/update/self-update.ts +44 -0
- package/src/index.ts +13 -5
- package/src/plugins/playwright/index.ts +39 -3
- package/src/plugins/playwright/provision.ts +21 -0
- package/src/plugins/playwright/version-coupling.ts +195 -0
- package/src/util/log.ts +318 -26
- package/src/util/paths.ts +5 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "talon-agent",
|
|
3
|
-
"version": "5.2.
|
|
3
|
+
"version": "5.2.2",
|
|
4
4
|
"description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
|
|
5
5
|
"author": "Dylan Neve",
|
|
6
6
|
"license": "MIT",
|
|
@@ -110,7 +110,7 @@
|
|
|
110
110
|
"@openai/agents": "^0.18.0",
|
|
111
111
|
"@openai/codex-sdk": "^0.154.0",
|
|
112
112
|
"@opencode-ai/sdk": "^1.17.4",
|
|
113
|
-
"@playwright/mcp": "0.0.
|
|
113
|
+
"@playwright/mcp": "0.0.56",
|
|
114
114
|
"@types/cross-spawn": "^6.0.6",
|
|
115
115
|
"@types/qrcode": "^1.5.6",
|
|
116
116
|
"baileys": "^7.0.0-rc14",
|
package/src/app.ts
CHANGED
|
@@ -29,7 +29,16 @@ import { shutdownTriggers } from "./core/background/triggers/index.js";
|
|
|
29
29
|
import { shutdownAgents } from "./core/agents/index.js";
|
|
30
30
|
import { pruneSettledTriggers } from "./storage/triggers.js";
|
|
31
31
|
import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
|
|
32
|
-
import {
|
|
32
|
+
import {
|
|
33
|
+
BOOT_SMOKE_FLAG,
|
|
34
|
+
BOOT_SMOKE_OK,
|
|
35
|
+
spawnSuccessor,
|
|
36
|
+
} from "./core/daemon/respawn.js";
|
|
37
|
+
import {
|
|
38
|
+
crashCleanup,
|
|
39
|
+
crashStep,
|
|
40
|
+
handleUncaughtException,
|
|
41
|
+
} from "./core/daemon/crash.js";
|
|
33
42
|
import { log, logError, logWarn } from "./util/log.js";
|
|
34
43
|
import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
|
|
35
44
|
import {
|
|
@@ -62,6 +71,17 @@ import {
|
|
|
62
71
|
stopResourceSampler,
|
|
63
72
|
} from "./core/daemon/resource-sampler.js";
|
|
64
73
|
|
|
74
|
+
// `/update` runs the freshly installed tree with --boot-smoke before it
|
|
75
|
+
// hands off. Every static import above has just been resolved against
|
|
76
|
+
// the new node_modules — reaching this line is the proof that the tree
|
|
77
|
+
// the successor will run is importable at all. Nothing has booted yet,
|
|
78
|
+
// so this is both the strongest and the last harmless place to say so.
|
|
79
|
+
// See core/update/self-update.ts for what a failure does instead.
|
|
80
|
+
if (process.argv.includes(BOOT_SMOKE_FLAG)) {
|
|
81
|
+
console.log(BOOT_SMOKE_OK);
|
|
82
|
+
process.exit(0);
|
|
83
|
+
}
|
|
84
|
+
|
|
65
85
|
const { config } = await bootPhase("bootstrap", () => bootstrap());
|
|
66
86
|
|
|
67
87
|
// Record this process as the daemon. The gateway port is appended once
|
|
@@ -114,6 +134,10 @@ onBackendChange((holder, newBackend, info) => {
|
|
|
114
134
|
let shuttingDown = false;
|
|
115
135
|
let triggerPruneTimer: ReturnType<typeof setInterval> | null = null;
|
|
116
136
|
|
|
137
|
+
// The composition root owns the SQLite handle, so it hands the crash
|
|
138
|
+
// path its checkpoint (see core/daemon/crash.ts).
|
|
139
|
+
const crashHooks = { flushDatabase };
|
|
140
|
+
|
|
117
141
|
const SHUTDOWN_TIMEOUT_MS = 15_000;
|
|
118
142
|
const DRAIN_TIMEOUT_MS = 5_000;
|
|
119
143
|
|
|
@@ -127,7 +151,11 @@ async function shutdownStep(name: string, fn: () => unknown): Promise<void> {
|
|
|
127
151
|
try {
|
|
128
152
|
await fn();
|
|
129
153
|
} catch (err) {
|
|
130
|
-
|
|
154
|
+
// Even the report is best-effort: a shutdown triggered by a full
|
|
155
|
+
// disk must not die inside its own error path.
|
|
156
|
+
crashStep("shutdown report", () =>
|
|
157
|
+
logError("shutdown", `${name} failed`, err),
|
|
158
|
+
);
|
|
131
159
|
}
|
|
132
160
|
}
|
|
133
161
|
|
|
@@ -138,15 +166,18 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
138
166
|
|
|
139
167
|
const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
|
|
140
168
|
const forceTimer = setTimeout(() => {
|
|
141
|
-
|
|
142
|
-
//
|
|
143
|
-
//
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
//
|
|
147
|
-
//
|
|
148
|
-
//
|
|
149
|
-
|
|
169
|
+
// Cleanup first, report second. Handing off matters most here: a
|
|
170
|
+
// restart must survive a subsystem that won't stop (a wedged FUSE
|
|
171
|
+
// unmount, an MCP server ignoring SIGTERM, a backend child that
|
|
172
|
+
// never acks), and it must also survive a logger that can't write —
|
|
173
|
+
// logging first is what cost us the successor on 2026-09-18. The
|
|
174
|
+
// successor may briefly race the long-poll we failed to release,
|
|
175
|
+
// but grammy retries the 409 — a few seconds of overlap beats
|
|
176
|
+
// staying down.
|
|
177
|
+
crashCleanup(crashHooks);
|
|
178
|
+
crashStep("timeout report", () =>
|
|
179
|
+
logError("shutdown", "Timeout exceeded, forcing exit"),
|
|
180
|
+
);
|
|
150
181
|
process.exit(1);
|
|
151
182
|
}, SHUTDOWN_TIMEOUT_MS);
|
|
152
183
|
forceTimer.unref();
|
|
@@ -225,37 +256,29 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
225
256
|
const { shutdownHub } = await import("./core/mcp-hub/index.js");
|
|
226
257
|
await shutdownHub();
|
|
227
258
|
});
|
|
228
|
-
|
|
259
|
+
// Each tail step stands alone: a full disk can make any of them throw,
|
|
260
|
+
// and none of them may cost us the ones that follow.
|
|
261
|
+
crashStep("database flush", () => flushDatabase());
|
|
229
262
|
// Guarded removal: only clear the record if it still names us. A
|
|
230
263
|
// successor that raced ahead and wrote its own pid here must not be
|
|
231
264
|
// orphaned (the bug that made `talon restart` spawn duplicate daemons).
|
|
232
|
-
removePidRecordIfOwnedBy(process.pid);
|
|
233
|
-
log("shutdown", "State saved");
|
|
265
|
+
crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
|
|
266
|
+
crashStep("shutdown report", () => log("shutdown", "State saved"));
|
|
234
267
|
// Hand off last: the frontends are stopped, so the successor binds
|
|
235
268
|
// Telegram's long-poll only after we have released it. No-op unless
|
|
236
269
|
// /restart or /update armed a respawn.
|
|
237
|
-
spawnSuccessor();
|
|
270
|
+
crashStep("respawn handoff", () => spawnSuccessor());
|
|
238
271
|
process.exit(0);
|
|
239
272
|
}
|
|
240
273
|
|
|
241
274
|
process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
|
|
242
275
|
process.on("SIGINT", () => gracefulShutdown("SIGINT"));
|
|
243
276
|
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
return;
|
|
250
|
-
}
|
|
251
|
-
logError("bot", "Uncaught exception", err);
|
|
252
|
-
flushDatabase();
|
|
253
|
-
// Same pid-guarded removal as the graceful path — a crashed daemon
|
|
254
|
-
// must not leave a record that makes `talon status` chase a dead or
|
|
255
|
-
// recycled pid.
|
|
256
|
-
removePidRecordIfOwnedBy(process.pid);
|
|
257
|
-
process.exit(1);
|
|
258
|
-
});
|
|
277
|
+
// Cleanup runs before the crash is reported (and the EPIPE suppression
|
|
278
|
+
// is unchanged) — see core/daemon/crash.ts for why the order matters.
|
|
279
|
+
process.on("uncaughtException", (err) =>
|
|
280
|
+
handleUncaughtException(err, crashHooks),
|
|
281
|
+
);
|
|
259
282
|
|
|
260
283
|
process.on("unhandledRejection", (reason) => {
|
|
261
284
|
logWarn(
|
|
@@ -333,8 +356,9 @@ async function main(): Promise<void> {
|
|
|
333
356
|
}
|
|
334
357
|
|
|
335
358
|
main().catch((err) => {
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
359
|
+
crashCleanup(crashHooks);
|
|
360
|
+
crashStep("startup report", () =>
|
|
361
|
+
logError("bot", "Fatal startup error", err),
|
|
362
|
+
);
|
|
339
363
|
process.exit(1);
|
|
340
364
|
});
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Crash-path cleanup — what has to happen when the daemon goes down
|
|
3
|
+
* abnormally: an uncaught exception, the forced exit after a shutdown
|
|
4
|
+
* timeout, or a fatal startup error.
|
|
5
|
+
*
|
|
6
|
+
* Ordering is the whole point. The old handler logged first and cleaned
|
|
7
|
+
* up afterwards, which works right up until logging is the thing that
|
|
8
|
+
* broke. On 2026-09-18 the root disk filled; the log file's write stream
|
|
9
|
+
* emitted ENOSPC with nobody listening, the uncaught-exception handler
|
|
10
|
+
* called `logError` as its first act, threw inside itself, and Node
|
|
11
|
+
* aborted the process — no database checkpoint, a stale pidfile left
|
|
12
|
+
* behind, and (the expensive part) an armed `/update` handoff that never
|
|
13
|
+
* spawned its successor. The daemon simply vanished.
|
|
14
|
+
*
|
|
15
|
+
* So: the essentials first, each in its own try/catch, logging last and
|
|
16
|
+
* strictly best-effort. `src/util/log.ts` now keeps a broken log file
|
|
17
|
+
* from throwing at all — this is the second line of defence, for every
|
|
18
|
+
* other way logging can fail while the machine is sick.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { logError, logWarn } from "../../util/log.js";
|
|
22
|
+
import { removePidRecordIfOwnedBy } from "./pidfile.js";
|
|
23
|
+
import { spawnSuccessor } from "./respawn.js";
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Steps the crash path can only get from the composition root.
|
|
27
|
+
* `flushDatabase` is injected because the SQLite handle stays inside
|
|
28
|
+
* `storage/` (.dependency-cruiser: db-handle-stays-in-storage).
|
|
29
|
+
*/
|
|
30
|
+
export type CrashHooks = {
|
|
31
|
+
flushDatabase: () => void;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Run one crash-path step. Never throws, and never uses the logger —
|
|
36
|
+
* a broken logger is precisely the case this path exists for, so the
|
|
37
|
+
* console is the fallback of last resort.
|
|
38
|
+
*/
|
|
39
|
+
export function crashStep(name: string, fn: () => void): void {
|
|
40
|
+
try {
|
|
41
|
+
fn();
|
|
42
|
+
} catch (err) {
|
|
43
|
+
try {
|
|
44
|
+
console.error(`[talon] crash cleanup: ${name} failed`, err);
|
|
45
|
+
} catch {
|
|
46
|
+
/* stdio is gone too — nothing left to try */
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* The non-logging essentials, in the order that matters:
|
|
53
|
+
* 1. drop the pid record, so `talon status` stops chasing a dead pid
|
|
54
|
+
* (before the successor writes its own — the guarded removal then
|
|
55
|
+
* cannot possibly orphan it),
|
|
56
|
+
* 2. hand off to the successor if `/restart` or `/update` armed one
|
|
57
|
+
* (a no-op otherwise, see respawn.ts),
|
|
58
|
+
* 3. checkpoint the database.
|
|
59
|
+
*/
|
|
60
|
+
export function crashCleanup(hooks: CrashHooks): void {
|
|
61
|
+
crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
|
|
62
|
+
crashStep("respawn handoff", () => spawnSuccessor());
|
|
63
|
+
crashStep("database flush", hooks.flushDatabase);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* `process.on("uncaughtException")` body. Cleanup happens before the
|
|
68
|
+
* crash is reported, never after.
|
|
69
|
+
*/
|
|
70
|
+
export function handleUncaughtException(err: Error, hooks: CrashHooks): void {
|
|
71
|
+
// EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
|
|
72
|
+
// gramjs will reconnect; crashing the process here is wrong.
|
|
73
|
+
if ((err as NodeJS.ErrnoException).code === "EPIPE") {
|
|
74
|
+
crashStep("EPIPE notice", () =>
|
|
75
|
+
logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`),
|
|
76
|
+
);
|
|
77
|
+
return;
|
|
78
|
+
}
|
|
79
|
+
crashCleanup(hooks);
|
|
80
|
+
crashStep("crash report", () => logError("bot", "Uncaught exception", err));
|
|
81
|
+
process.exit(1);
|
|
82
|
+
}
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Handoff watcher — the process that proves a `/restart` or `/update`
|
|
3
|
+
* actually landed.
|
|
4
|
+
*
|
|
5
|
+
* The outgoing daemon spawns its successor and then calls
|
|
6
|
+
* `process.exit()`. Until 2026-09-18 that was the whole handoff, which
|
|
7
|
+
* means nothing in the system knew whether the successor had come up.
|
|
8
|
+
* On that day one didn't: it was spawned, lived about twenty seconds,
|
|
9
|
+
* never bound its gateway, and died — and because the dying parent was
|
|
10
|
+
* the only party to the handoff, Talon simply stayed down until a human
|
|
11
|
+
* noticed forty-five minutes later.
|
|
12
|
+
*
|
|
13
|
+
* So the handoff gets a witness. `spawnSuccessor()` (./respawn.ts)
|
|
14
|
+
* starts this watcher detached, sharing the successor's respawn.log fd.
|
|
15
|
+
* It polls identity-verified discovery (./discovery.ts — `app: "talon"`,
|
|
16
|
+
* `mode: "daemon"`, matching pid) until the successor answers /health or
|
|
17
|
+
* the window closes. Only a /health answer counts: discovery will also
|
|
18
|
+
* report a daemon whose pid is merely alive, and "the process exists" is
|
|
19
|
+
* exactly the claim that was false for 20 seconds on 2026-09-18. If the
|
|
20
|
+
* successor never serves, the watcher starts the daemon exactly
|
|
21
|
+
* the way `talon start` does (./control.ts — same spawn, same boot
|
|
22
|
+
* verification) and says why in the log. The watcher is tiny on purpose:
|
|
23
|
+
* `src/index.ts` dispatches its subcommand before the app graph loads,
|
|
24
|
+
* so it costs a bare runtime and these three modules.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { dirname, resolve } from "node:path";
|
|
28
|
+
import { log, logError, logWarn } from "../../util/log.js";
|
|
29
|
+
import { startDaemon, type StartOutcome } from "./control.js";
|
|
30
|
+
import { findRunningInstance, type RunningInstance } from "./discovery.js";
|
|
31
|
+
import { isProcessAlive } from "./pidfile.js";
|
|
32
|
+
|
|
33
|
+
/** Hidden subcommand: `talon _handoff-watch <successor-pid>`. */
|
|
34
|
+
export const HANDOFF_WATCH_SUBCOMMAND = "_handoff-watch";
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* How long a successor gets to answer /health. Generous on purpose: a
|
|
38
|
+
* cold boot with every plugin and MCP server takes ~10s on the reference
|
|
39
|
+
* host, and `startDaemon()` itself waits 30s before calling a boot late.
|
|
40
|
+
*/
|
|
41
|
+
const HANDOFF_WINDOW_MS = 90_000;
|
|
42
|
+
const POLL_MS = 500;
|
|
43
|
+
|
|
44
|
+
export type HandoffOutcome =
|
|
45
|
+
| { ok: true; via: "successor" | "restart"; pid: number; port?: number }
|
|
46
|
+
| { ok: false; reason: string };
|
|
47
|
+
|
|
48
|
+
export type WatchHandoffOptions = {
|
|
49
|
+
/** The pid `spawnSuccessor()` created. */
|
|
50
|
+
childPid: number;
|
|
51
|
+
/** Repo/package root, for the `talon start` equivalent. */
|
|
52
|
+
pkgRoot: string;
|
|
53
|
+
windowMs?: number;
|
|
54
|
+
pollMs?: number;
|
|
55
|
+
pidfilePath?: string;
|
|
56
|
+
/** Injection seams for tests. */
|
|
57
|
+
find?: typeof findRunningInstance;
|
|
58
|
+
alive?: (pid: number) => boolean;
|
|
59
|
+
start?: typeof startDaemon;
|
|
60
|
+
sleep?: (ms: number) => Promise<void>;
|
|
61
|
+
};
|
|
62
|
+
|
|
63
|
+
function defaultSleep(ms: number): Promise<void> {
|
|
64
|
+
return new Promise((r) => setTimeout(r, ms));
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Why the wait ended without a live daemon. */
|
|
68
|
+
type WaitFailure = "successor-exited" | "window-expired";
|
|
69
|
+
|
|
70
|
+
/** Discovery found a process; only a /health answer proves it serves. */
|
|
71
|
+
function isServing(instance: RunningInstance | null): boolean {
|
|
72
|
+
return instance?.health !== undefined;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Poll until a serving daemon answers, the successor process
|
|
77
|
+
* disappears, or the window closes. A daemon that isn't our child still
|
|
78
|
+
* counts: the goal is a live Talon, not a particular pid.
|
|
79
|
+
*/
|
|
80
|
+
async function awaitDaemon(
|
|
81
|
+
opts: WatchHandoffOptions,
|
|
82
|
+
): Promise<RunningInstance | WaitFailure> {
|
|
83
|
+
const find = opts.find ?? findRunningInstance;
|
|
84
|
+
const alive = opts.alive ?? isProcessAlive;
|
|
85
|
+
const sleep = opts.sleep ?? defaultSleep;
|
|
86
|
+
const pollMs = opts.pollMs ?? POLL_MS;
|
|
87
|
+
const deadline = Date.now() + (opts.windowMs ?? HANDOFF_WINDOW_MS);
|
|
88
|
+
|
|
89
|
+
while (Date.now() < deadline) {
|
|
90
|
+
const instance = await find(opts.pidfilePath);
|
|
91
|
+
if (instance && isServing(instance)) return instance;
|
|
92
|
+
if (!alive(opts.childPid)) return "successor-exited";
|
|
93
|
+
await sleep(pollMs);
|
|
94
|
+
}
|
|
95
|
+
return "window-expired";
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const FAILURE_DETAIL: Record<WaitFailure, string> = {
|
|
99
|
+
"successor-exited": "the successor exited before serving /health",
|
|
100
|
+
"window-expired": "the successor never answered /health in time",
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
function describeStart(outcome: StartOutcome): string {
|
|
104
|
+
if (outcome.ok) return `started (pid ${outcome.pid})`;
|
|
105
|
+
if (outcome.reason === "already-running") {
|
|
106
|
+
return `already running (pid ${outcome.instance.pid})`;
|
|
107
|
+
}
|
|
108
|
+
if (outcome.reason === "boot-timeout") return "spawned but not yet healthy";
|
|
109
|
+
return `${outcome.reason}${outcome.detail ? `: ${outcome.detail}` : ""}`;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function toOutcome(started: StartOutcome, why: string): HandoffOutcome {
|
|
113
|
+
if (started.ok) {
|
|
114
|
+
return { ok: true, via: "restart", pid: started.pid, port: started.port };
|
|
115
|
+
}
|
|
116
|
+
if (started.reason === "already-running") {
|
|
117
|
+
const inst = started.instance;
|
|
118
|
+
// `talon start` refuses while a pid is alive — right, since a second
|
|
119
|
+
// daemon would fight the first for Telegram's getUpdates. But an
|
|
120
|
+
// alive pid that has never served /health is the failure, not the
|
|
121
|
+
// recovery, so it is reported as one.
|
|
122
|
+
if (!isServing(inst)) {
|
|
123
|
+
return {
|
|
124
|
+
ok: false,
|
|
125
|
+
reason: `${why}; pid ${inst.pid} is alive but not serving — kill it and run \`talon start\``,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
return { ok: true, via: "restart", pid: inst.pid, port: inst.port };
|
|
129
|
+
}
|
|
130
|
+
return { ok: false, reason: `${why}; restart ${describeStart(started)}` };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Verify the handoff, and repair it if it failed. Never throws: this
|
|
135
|
+
* process exists only to make the outcome known.
|
|
136
|
+
*/
|
|
137
|
+
export async function watchHandoff(
|
|
138
|
+
opts: WatchHandoffOptions,
|
|
139
|
+
): Promise<HandoffOutcome> {
|
|
140
|
+
const result = await awaitDaemon(opts);
|
|
141
|
+
if (typeof result !== "string") {
|
|
142
|
+
const via = result.pid === opts.childPid ? "successor" : "restart";
|
|
143
|
+
log(
|
|
144
|
+
"shutdown",
|
|
145
|
+
`Handoff verified — daemon pid ${result.pid} serving on ` +
|
|
146
|
+
`:${result.port ?? "?"} (${via})`,
|
|
147
|
+
);
|
|
148
|
+
return { ok: true, via, pid: result.pid, port: result.port };
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const why = FAILURE_DETAIL[result];
|
|
152
|
+
logWarn(
|
|
153
|
+
"shutdown",
|
|
154
|
+
`Handoff failed — ${why}; starting Talon the way \`talon start\` does`,
|
|
155
|
+
);
|
|
156
|
+
const start = opts.start ?? startDaemon;
|
|
157
|
+
const started = await start({
|
|
158
|
+
pkgRoot: opts.pkgRoot,
|
|
159
|
+
pidfilePath: opts.pidfilePath,
|
|
160
|
+
});
|
|
161
|
+
const outcome = toOutcome(started, why);
|
|
162
|
+
if (outcome.ok)
|
|
163
|
+
log("shutdown", `Handoff recovered — ${describeStart(started)}`);
|
|
164
|
+
else logError("shutdown", `Handoff unrecoverable — ${outcome.reason}`);
|
|
165
|
+
return outcome;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** src/core/daemon/ → the package root. */
|
|
169
|
+
function packageRoot(): string {
|
|
170
|
+
const here = import.meta.dirname ?? process.cwd();
|
|
171
|
+
return here.includes("$bunfs") || here.includes("~BUN")
|
|
172
|
+
? dirname(process.execPath)
|
|
173
|
+
: resolve(here, "..", "..", "..");
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/** Entry point for `talon _handoff-watch <pid>` (src/index.ts). */
|
|
177
|
+
export async function runHandoffWatch(argv: readonly string[]): Promise<void> {
|
|
178
|
+
const childPid = Number.parseInt(argv[0] ?? "", 10);
|
|
179
|
+
if (!Number.isInteger(childPid) || childPid <= 0) {
|
|
180
|
+
logError("shutdown", `Handoff watcher got no successor pid (${argv[0]})`);
|
|
181
|
+
process.exitCode = 2;
|
|
182
|
+
return;
|
|
183
|
+
}
|
|
184
|
+
log("shutdown", `Handoff watcher armed for pid ${childPid}`);
|
|
185
|
+
try {
|
|
186
|
+
const outcome = await watchHandoff({ childPid, pkgRoot: packageRoot() });
|
|
187
|
+
process.exitCode = outcome.ok ? 0 : 1;
|
|
188
|
+
} catch (err) {
|
|
189
|
+
logError("shutdown", "Handoff watcher crashed", err);
|
|
190
|
+
process.exitCode = 1;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
@@ -1,38 +1,88 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Self-respawn helper for /restart
|
|
2
|
+
* Self-respawn helper for /restart and /update across frontends.
|
|
3
3
|
*
|
|
4
|
-
* Spawns a fresh copy of the current process — same
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* it.
|
|
4
|
+
* Spawns a fresh copy of the current process — same runtime binary, same
|
|
5
|
+
* `execArgv` (preserving a loader so `.ts` entrypoints still resolve),
|
|
6
|
+
* same script + user args, same cwd + env. The new child is detached so
|
|
7
|
+
* it survives the parent's exit; `unref()` lets the parent exit without
|
|
8
|
+
* waiting on it.
|
|
10
9
|
*
|
|
11
|
-
* Why not call the daemon's `talon restart` CLI? That path assumed
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* of launch method.
|
|
10
|
+
* Why not call the daemon's `talon restart` CLI? That path assumed the
|
|
11
|
+
* bot was started via the daemon (talon.pid managed by `daemonStart()`)
|
|
12
|
+
* and broke for anything else — `npm start`, `npx tsx src/index.ts`,
|
|
13
|
+
* systemd, foreman, pm2, or running under a debugger. Respawning from
|
|
14
|
+
* our own `process.argv` works regardless of launch method.
|
|
17
15
|
*
|
|
18
|
-
* Ordering matters. `respawnSelf()` only *arms* the handoff and
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
16
|
+
* Ordering matters. `respawnSelf()` only *arms* the handoff and raises
|
|
17
|
+
* SIGTERM; the successor is spawned by `spawnSuccessor()` at the tail of
|
|
18
|
+
* graceful shutdown, once the frontends have stopped. Spawning up-front
|
|
19
|
+
* (the original behaviour) left the successor long-polling `getUpdates`
|
|
20
|
+
* while the outgoing process was still draining in-flight queries — up
|
|
21
|
+
* to DRAIN_TIMEOUT_MS of two live pollers. Telegram answers only one of
|
|
22
|
+
* them and re-delivers the unconfirmed updates to the other, so a
|
|
23
|
+
* restart mid-turn produced a 409 Conflict on the way out and duplicate
|
|
24
|
+
* replies on the way in. Releasing the poll before the successor binds
|
|
25
|
+
* it removes the overlap rather than relying on grammy's 409 retry to
|
|
26
|
+
* paper over it.
|
|
27
|
+
*
|
|
28
|
+
* Two things the 2026-09-18 outage added, both about the fact that the
|
|
29
|
+
* outgoing process is dying and cannot be the one responsible for the
|
|
30
|
+
* outcome:
|
|
31
|
+
*
|
|
32
|
+
* - The successor's stdout and stderr go to ~/.talon/respawn.log, not
|
|
33
|
+
* to "ignore". A successor that dies before its own logger exists —
|
|
34
|
+
* a broken import after a dependency install, a fatal bind, a
|
|
35
|
+
* runtime that aborts — used to leave no trace in any file, on any
|
|
36
|
+
* process. That is precisely what happened: a successor was spawned,
|
|
37
|
+
* lived ~20s, never bound its gateway, and vanished without a line.
|
|
38
|
+
* - A watcher process is spawned alongside it (./handoff.ts). It
|
|
39
|
+
* outlives us, verifies the successor over identity-checked /health
|
|
40
|
+
* within a bounded window, and starts the daemon the way `talon
|
|
41
|
+
* start` does if the successor never comes up. Nothing in the
|
|
42
|
+
* handoff depends on a process that is about to call process.exit().
|
|
29
43
|
*/
|
|
30
44
|
|
|
31
45
|
import { spawn } from "node:child_process";
|
|
32
|
-
import { log, logError } from "../../util/log.js";
|
|
46
|
+
import { log, logError, openRespawnLog } from "../../util/log.js";
|
|
47
|
+
import { HANDOFF_WATCH_SUBCOMMAND } from "./handoff.js";
|
|
33
48
|
|
|
34
49
|
let pendingReason: string | null = null;
|
|
35
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Argv flag that makes the daemon entry resolve its whole import graph
|
|
53
|
+
* and exit 0 without booting (src/app.ts acts on it before the first
|
|
54
|
+
* bootstrap step). `/update` runs the freshly installed tree with this
|
|
55
|
+
* flag before handing off — see core/update/self-update.ts.
|
|
56
|
+
*/
|
|
57
|
+
export const BOOT_SMOKE_FLAG = "--boot-smoke";
|
|
58
|
+
|
|
59
|
+
/** Printed by a successful smoke run; the update step matches on it. */
|
|
60
|
+
export const BOOT_SMOKE_OK = "talon boot-smoke ok";
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* A bun-compiled binary embeds its source tree: `process.argv[1]` points
|
|
64
|
+
* into the virtual FS ($bunfs / ~BUN) and has no path on disk, so the
|
|
65
|
+
* binary is re-invoked with no script argument at all.
|
|
66
|
+
*/
|
|
67
|
+
function isEmbeddedEntry(entry: string): boolean {
|
|
68
|
+
return entry.includes("$bunfs") || entry.includes("~BUN");
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** The exact command that re-runs this process. */
|
|
72
|
+
export function successorCommand(): { cmd: string; args: string[] } {
|
|
73
|
+
return {
|
|
74
|
+
cmd: process.argv[0],
|
|
75
|
+
args: [...process.execArgv, ...process.argv.slice(1)],
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** The same runtime + entry, re-invoked with one of our own subcommands. */
|
|
80
|
+
function selfInvocation(extra: string[]): { cmd: string; args: string[] } {
|
|
81
|
+
const entry = process.argv[1] ?? "";
|
|
82
|
+
const prefix = isEmbeddedEntry(entry) ? [] : [...process.execArgv, entry];
|
|
83
|
+
return { cmd: process.argv[0], args: [...prefix, ...extra] };
|
|
84
|
+
}
|
|
85
|
+
|
|
36
86
|
/**
|
|
37
87
|
* Arm a respawn and raise SIGTERM on ourselves so the existing
|
|
38
88
|
* graceful-shutdown path cleanly stops the frontends, flushes state,
|
|
@@ -57,46 +107,71 @@ export function respawnRequested(): boolean {
|
|
|
57
107
|
return pendingReason !== null;
|
|
58
108
|
}
|
|
59
109
|
|
|
110
|
+
export type SpawnFn = typeof spawn;
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Start the watcher that outlives this process and answers the only
|
|
114
|
+
* question that matters: did the successor actually come up? Never
|
|
115
|
+
* throws — a missing watcher must not cost us the successor itself.
|
|
116
|
+
*/
|
|
117
|
+
function spawnHandoffWatcher(
|
|
118
|
+
childPid: number,
|
|
119
|
+
fd: number | null,
|
|
120
|
+
spawnFn: SpawnFn,
|
|
121
|
+
): void {
|
|
122
|
+
try {
|
|
123
|
+
const { cmd, args } = selfInvocation([
|
|
124
|
+
HANDOFF_WATCH_SUBCOMMAND,
|
|
125
|
+
String(childPid),
|
|
126
|
+
]);
|
|
127
|
+
const watcher = spawnFn(cmd, args, {
|
|
128
|
+
cwd: process.cwd(),
|
|
129
|
+
detached: true,
|
|
130
|
+
stdio: ["ignore", fd ?? "ignore", fd ?? "ignore"],
|
|
131
|
+
env: { ...process.env },
|
|
132
|
+
});
|
|
133
|
+
watcher.once("error", (err) => {
|
|
134
|
+
logError("shutdown", "Handoff watcher failed to start", err);
|
|
135
|
+
});
|
|
136
|
+
watcher.unref();
|
|
137
|
+
log("shutdown", `Handoff watcher started (pid ${watcher.pid})`);
|
|
138
|
+
} catch (err) {
|
|
139
|
+
logError("shutdown", "Handoff watcher failed to start", err);
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
60
143
|
/**
|
|
61
|
-
* Spawn the successor process. Called at the end of graceful
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
144
|
+
* Spawn the successor process. Called at the end of graceful shutdown,
|
|
145
|
+
* after the frontends have stopped — so the incoming process binds
|
|
146
|
+
* Telegram's long-poll only once this one has let go of it. No-op unless
|
|
147
|
+
* `respawnSelf()` armed a handoff.
|
|
65
148
|
*
|
|
66
149
|
* Never throws: a failed handoff must not prevent this process from
|
|
67
|
-
* exiting.
|
|
68
|
-
* terminal) can pick things up.
|
|
150
|
+
* exiting. The watcher is the safety net, not an external supervisor.
|
|
69
151
|
*/
|
|
70
|
-
export function spawnSuccessor(): void {
|
|
152
|
+
export function spawnSuccessor(spawnFn: SpawnFn = spawn): void {
|
|
71
153
|
if (pendingReason === null) return;
|
|
72
154
|
const reason = pendingReason;
|
|
73
155
|
pendingReason = null;
|
|
74
156
|
|
|
157
|
+
// One fd for both streams, shared with the watcher: the successor's
|
|
158
|
+
// boot output and the watcher's verdict land in the same file, in
|
|
159
|
+
// order, even when the successor never gets far enough to log.
|
|
160
|
+
const fd = openRespawnLog();
|
|
75
161
|
try {
|
|
76
|
-
const child =
|
|
77
|
-
process.
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
stdio: "ignore",
|
|
83
|
-
env: { ...process.env },
|
|
84
|
-
},
|
|
85
|
-
);
|
|
162
|
+
const child = spawnFn(process.argv[0], successorCommand().args, {
|
|
163
|
+
cwd: process.cwd(),
|
|
164
|
+
detached: true,
|
|
165
|
+
stdio: ["ignore", fd ?? "ignore", fd ?? "ignore"],
|
|
166
|
+
env: { ...process.env },
|
|
167
|
+
});
|
|
86
168
|
child.once("error", (err) => {
|
|
87
|
-
logError(
|
|
88
|
-
"shutdown",
|
|
89
|
-
`Respawn failed; exiting without a successor — restart manually`,
|
|
90
|
-
err,
|
|
91
|
-
);
|
|
169
|
+
logError("shutdown", `Respawn failed (${reason}) — see respawn.log`, err);
|
|
92
170
|
});
|
|
93
171
|
child.unref();
|
|
94
172
|
log("shutdown", `Respawn child started (pid ${child.pid}) — ${reason}`);
|
|
173
|
+
if (child.pid) spawnHandoffWatcher(child.pid, fd, spawnFn);
|
|
95
174
|
} catch (err) {
|
|
96
|
-
logError(
|
|
97
|
-
"shutdown",
|
|
98
|
-
`Respawn failed; exiting without a successor — restart manually`,
|
|
99
|
-
err,
|
|
100
|
-
);
|
|
175
|
+
logError("shutdown", `Respawn failed (${reason}) — see respawn.log`, err);
|
|
101
176
|
}
|
|
102
177
|
}
|