talon-agent 5.2.0 → 5.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.2.0",
3
+ "version": "5.2.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
package/src/app.ts CHANGED
@@ -30,6 +30,11 @@ import { shutdownAgents } from "./core/agents/index.js";
30
30
  import { pruneSettledTriggers } from "./storage/triggers.js";
31
31
  import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
32
32
  import { spawnSuccessor } from "./core/daemon/respawn.js";
33
+ import {
34
+ crashCleanup,
35
+ crashStep,
36
+ handleUncaughtException,
37
+ } from "./core/daemon/crash.js";
33
38
  import { log, logError, logWarn } from "./util/log.js";
34
39
  import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
35
40
  import {
@@ -114,6 +119,10 @@ onBackendChange((holder, newBackend, info) => {
114
119
  let shuttingDown = false;
115
120
  let triggerPruneTimer: ReturnType<typeof setInterval> | null = null;
116
121
 
122
+ // The composition root owns the SQLite handle, so it hands the crash
123
+ // path its checkpoint (see core/daemon/crash.ts).
124
+ const crashHooks = { flushDatabase };
125
+
117
126
  const SHUTDOWN_TIMEOUT_MS = 15_000;
118
127
  const DRAIN_TIMEOUT_MS = 5_000;
119
128
 
@@ -127,7 +136,11 @@ async function shutdownStep(name: string, fn: () => unknown): Promise<void> {
127
136
  try {
128
137
  await fn();
129
138
  } catch (err) {
130
- logError("shutdown", `${name} failed`, err);
139
+ // Even the report is best-effort: a shutdown triggered by a full
140
+ // disk must not die inside its own error path.
141
+ crashStep("shutdown report", () =>
142
+ logError("shutdown", `${name} failed`, err),
143
+ );
131
144
  }
132
145
  }
133
146
 
@@ -138,15 +151,18 @@ async function gracefulShutdown(signal: string): Promise<void> {
138
151
 
139
152
  const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
140
153
  const forceTimer = setTimeout(() => {
141
- logError("shutdown", "Timeout exceeded, forcing exit");
142
- // Hand off even on the forced path. A restart must survive a
143
- // subsystem that won't stop (a wedged FUSE unmount, an MCP server
144
- // ignoring SIGTERM, a backend child that never acks): without this
145
- // the timeout exits without a successor and `/restart` silently
146
- // takes the daemon down for good. The successor may briefly race
147
- // the long-poll we failed to release, but grammy retries the 409 —
148
- // a few seconds of overlap beats staying down.
149
- spawnSuccessor();
154
+ // Cleanup first, report second. Handing off matters most here: a
155
+ // restart must survive a subsystem that won't stop (a wedged FUSE
156
+ // unmount, an MCP server ignoring SIGTERM, a backend child that
157
+ // never acks), and it must also survive a logger that can't write —
158
+ // logging first is what cost us the successor on 2026-09-18. The
159
+ // successor may briefly race the long-poll we failed to release,
160
+ // but grammy retries the 409 — a few seconds of overlap beats
161
+ // staying down.
162
+ crashCleanup(crashHooks);
163
+ crashStep("timeout report", () =>
164
+ logError("shutdown", "Timeout exceeded, forcing exit"),
165
+ );
150
166
  process.exit(1);
151
167
  }, SHUTDOWN_TIMEOUT_MS);
152
168
  forceTimer.unref();
@@ -225,37 +241,29 @@ async function gracefulShutdown(signal: string): Promise<void> {
225
241
  const { shutdownHub } = await import("./core/mcp-hub/index.js");
226
242
  await shutdownHub();
227
243
  });
228
- flushDatabase();
244
+ // Each tail step stands alone: a full disk can make any of them throw,
245
+ // and none of them may cost us the ones that follow.
246
+ crashStep("database flush", () => flushDatabase());
229
247
  // Guarded removal: only clear the record if it still names us. A
230
248
  // successor that raced ahead and wrote its own pid here must not be
231
249
  // orphaned (the bug that made `talon restart` spawn duplicate daemons).
232
- removePidRecordIfOwnedBy(process.pid);
233
- log("shutdown", "State saved");
250
+ crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
251
+ crashStep("shutdown report", () => log("shutdown", "State saved"));
234
252
  // Hand off last: the frontends are stopped, so the successor binds
235
253
  // Telegram's long-poll only after we have released it. No-op unless
236
254
  // /restart or /update armed a respawn.
237
- spawnSuccessor();
255
+ crashStep("respawn handoff", () => spawnSuccessor());
238
256
  process.exit(0);
239
257
  }
240
258
 
241
259
  process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
242
260
  process.on("SIGINT", () => gracefulShutdown("SIGINT"));
243
261
 
244
- process.on("uncaughtException", (err) => {
245
- // EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
246
- // gramjs will reconnect; crashing the process here is wrong.
247
- if ((err as NodeJS.ErrnoException).code === "EPIPE") {
248
- logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`);
249
- return;
250
- }
251
- logError("bot", "Uncaught exception", err);
252
- flushDatabase();
253
- // Same pid-guarded removal as the graceful path — a crashed daemon
254
- // must not leave a record that makes `talon status` chase a dead or
255
- // recycled pid.
256
- removePidRecordIfOwnedBy(process.pid);
257
- process.exit(1);
258
- });
262
+ // Cleanup runs before the crash is reported (and the EPIPE suppression
263
+ // is unchanged) — see core/daemon/crash.ts for why the order matters.
264
+ process.on("uncaughtException", (err) =>
265
+ handleUncaughtException(err, crashHooks),
266
+ );
259
267
 
260
268
  process.on("unhandledRejection", (reason) => {
261
269
  logWarn(
@@ -333,8 +341,9 @@ async function main(): Promise<void> {
333
341
  }
334
342
 
335
343
  main().catch((err) => {
336
- logError("bot", "Fatal startup error", err);
337
- flushDatabase();
338
- removePidRecordIfOwnedBy(process.pid);
344
+ crashCleanup(crashHooks);
345
+ crashStep("startup report", () =>
346
+ logError("bot", "Fatal startup error", err),
347
+ );
339
348
  process.exit(1);
340
349
  });
@@ -0,0 +1,82 @@
1
+ /**
2
+ * Crash-path cleanup — what has to happen when the daemon goes down
3
+ * abnormally: an uncaught exception, the forced exit after a shutdown
4
+ * timeout, or a fatal startup error.
5
+ *
6
+ * Ordering is the whole point. The old handler logged first and cleaned
7
+ * up afterwards, which works right up until logging is the thing that
8
+ * broke. On 2026-09-18 the root disk filled; the log file's write stream
9
+ * emitted ENOSPC with nobody listening, the uncaught-exception handler
10
+ * called `logError` as its first act, threw inside itself, and Node
11
+ * aborted the process — no database checkpoint, a stale pidfile left
12
+ * behind, and (the expensive part) an armed `/update` handoff that never
13
+ * spawned its successor. The daemon simply vanished.
14
+ *
15
+ * So: the essentials first, each in its own try/catch, logging last and
16
+ * strictly best-effort. `src/util/log.ts` now keeps a broken log file
17
+ * from throwing at all — this is the second line of defence, for every
18
+ * other way logging can fail while the machine is sick.
19
+ */
20
+
21
+ import { logError, logWarn } from "../../util/log.js";
22
+ import { removePidRecordIfOwnedBy } from "./pidfile.js";
23
+ import { spawnSuccessor } from "./respawn.js";
24
+
25
+ /**
26
+ * Steps the crash path can only get from the composition root.
27
+ * `flushDatabase` is injected because the SQLite handle stays inside
28
+ * `storage/` (.dependency-cruiser: db-handle-stays-in-storage).
29
+ */
30
+ export type CrashHooks = {
31
+ flushDatabase: () => void;
32
+ };
33
+
34
+ /**
35
+ * Run one crash-path step. Never throws, and never uses the logger —
36
+ * a broken logger is precisely the case this path exists for, so the
37
+ * console is the fallback of last resort.
38
+ */
39
+ export function crashStep(name: string, fn: () => void): void {
40
+ try {
41
+ fn();
42
+ } catch (err) {
43
+ try {
44
+ console.error(`[talon] crash cleanup: ${name} failed`, err);
45
+ } catch {
46
+ /* stdio is gone too — nothing left to try */
47
+ }
48
+ }
49
+ }
50
+
51
+ /**
52
+ * The non-logging essentials, in the order that matters:
53
+ * 1. drop the pid record, so `talon status` stops chasing a dead pid
54
+ * (before the successor writes its own — the guarded removal then
55
+ * cannot possibly orphan it),
56
+ * 2. hand off to the successor if `/restart` or `/update` armed one
57
+ * (a no-op otherwise, see respawn.ts),
58
+ * 3. checkpoint the database.
59
+ */
60
+ export function crashCleanup(hooks: CrashHooks): void {
61
+ crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
62
+ crashStep("respawn handoff", () => spawnSuccessor());
63
+ crashStep("database flush", hooks.flushDatabase);
64
+ }
65
+
66
+ /**
67
+ * `process.on("uncaughtException")` body. Cleanup happens before the
68
+ * crash is reported, never after.
69
+ */
70
+ export function handleUncaughtException(err: Error, hooks: CrashHooks): void {
71
+ // EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
72
+ // gramjs will reconnect; crashing the process here is wrong.
73
+ if ((err as NodeJS.ErrnoException).code === "EPIPE") {
74
+ crashStep("EPIPE notice", () =>
75
+ logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`),
76
+ );
77
+ return;
78
+ }
79
+ crashCleanup(hooks);
80
+ crashStep("crash report", () => logError("bot", "Uncaught exception", err));
81
+ process.exit(1);
82
+ }
package/src/util/log.ts CHANGED
@@ -18,6 +18,7 @@ import {
18
18
  unlinkSync,
19
19
  createWriteStream,
20
20
  } from "node:fs";
21
+ import { Writable } from "node:stream";
21
22
  import { dirs, files } from "./paths.js";
22
23
 
23
24
  export type LogComponent =
@@ -127,23 +128,249 @@ const streams: pino.StreamEntry[] = [];
127
128
 
128
129
  // Console output (disabled in quiet mode), pretty-printed.
129
130
  if (!quiet) {
130
- streams.push({
131
- level: "trace",
132
- stream: prettyStream({
133
- colorize: true,
134
- ignore: "pid,hostname",
135
- translateTime: "HH:MM:ss",
136
- }),
131
+ const consoleStream = prettyStream({
132
+ colorize: true,
133
+ ignore: "pid,hostname",
134
+ translateTime: "HH:MM:ss",
137
135
  });
136
+ // pino-pretty writes through a SonicBoom on fd 1 whose own error
137
+ // handler removes itself after the first non-EPIPE failure
138
+ // (pino-pretty/lib/utils/build-safe-sonic-boom.js). With stdout
139
+ // redirected to a file on a full disk that leaves the pipeline's next
140
+ // error unhandled — i.e. an uncaught exception raised from inside a
141
+ // log call. One permanent listener closes that door; a dead console
142
+ // is not worth a dead daemon.
143
+ consoleStream.on("error", () => {});
144
+ streams.push({ level: "trace", stream: consoleStream });
145
+ }
146
+
147
+ /** How long to wait before the first attempt to reopen a failed log file. */
148
+ const SINK_RETRY_MS = 30_000;
149
+ /** Ceiling for the doubling backoff between reopen attempts. */
150
+ const SINK_MAX_RETRY_MS = 5 * 60_000;
151
+
152
+ export type ResilientFileSinkOptions = {
153
+ /** Opens the underlying file stream. Injection seam for tests. */
154
+ open?: (path: string) => Writable;
155
+ /** Where the pause/resume notices go. Defaults to the console sink. */
156
+ notify?: (level: "warn" | "info", message: string) => void;
157
+ /** First backoff step (default 30s). */
158
+ retryMs?: number;
159
+ /** Backoff ceiling (default 5 min). */
160
+ maxRetryMs?: number;
161
+ };
162
+
163
+ /**
164
+ * A log file destination that cannot take the process down.
165
+ *
166
+ * A bare `createWriteStream` handed to `pino.multistream` is a loaded
167
+ * gun: when the disk fills (ENOSPC), or the file is unlinked under a
168
+ * rotation (EBADF), or the fd goes bad (EIO), the stream emits `error`.
169
+ * With no listener that is an uncaught exception — and it fires from
170
+ * inside a log call, so the crash handler's own `logError` runs on a
171
+ * logger that is already broken. That is how a full disk killed the
172
+ * daemon on 2026-09-18: the process aborted mid-`/update` with no
173
+ * shutdown, no pidfile cleanup, and no successor.
174
+ *
175
+ * This sink owns the file stream instead of exposing it. pino only ever
176
+ * sees this object, which never emits `error` and never blocks:
177
+ * - write failures pause file logging and destroy the broken stream,
178
+ * - lines written while paused are DROPPED and counted (never
179
+ * buffered — the failure mode here is "no space", so growing a
180
+ * buffer is the last thing to do),
181
+ * - an unref'd timer retries the open on a 30s → 5min backoff,
182
+ * - the first write to land again resumes logging and reports how
183
+ * many lines were lost.
184
+ * The console sink keeps working throughout, and carries the two
185
+ * notices.
186
+ */
187
+ export class ResilientFileSink extends Writable {
188
+ private readonly path: string;
189
+ private readonly openStream: (path: string) => Writable;
190
+ private readonly notify: (level: "warn" | "info", message: string) => void;
191
+ private readonly baseRetryMs: number;
192
+ private readonly maxRetryMs: number;
193
+ private inner: Writable | null = null;
194
+ private retryMs: number;
195
+ private retryTimer: ReturnType<typeof setTimeout> | null = null;
196
+ private droppedWhileDown = 0;
197
+ private down = false;
198
+
199
+ constructor(path: string, opts: ResilientFileSinkOptions = {}) {
200
+ // decodeStrings:false keeps pino's serialized lines as strings;
201
+ // autoDestroy:false means a downstream failure can never tear this
202
+ // object down — it is the only thing standing between a broken file
203
+ // and the process.
204
+ super({ decodeStrings: false, autoDestroy: false });
205
+ this.path = path;
206
+ this.openStream = opts.open ?? openLogFile;
207
+ this.notify = opts.notify ?? notifyViaConsoleSink;
208
+ this.baseRetryMs = opts.retryMs ?? SINK_RETRY_MS;
209
+ this.maxRetryMs = opts.maxRetryMs ?? SINK_MAX_RETRY_MS;
210
+ this.retryMs = this.baseRetryMs;
211
+ this.openInner();
212
+ }
213
+
214
+ /** Lines discarded since the sink went down; cleared on recovery. */
215
+ get dropped(): number {
216
+ return this.droppedWhileDown;
217
+ }
218
+
219
+ /** True while file logging is paused (the console sink still runs). */
220
+ get isDown(): boolean {
221
+ return this.down;
222
+ }
223
+
224
+ override _write(
225
+ chunk: unknown,
226
+ _encoding: BufferEncoding,
227
+ callback: (error?: Error | null) => void,
228
+ ): void {
229
+ const inner = this.inner;
230
+ if (inner === null) {
231
+ this.droppedWhileDown++;
232
+ } else {
233
+ try {
234
+ inner.write(chunk as string, (err) => {
235
+ if (!err) this.markHealthy();
236
+ });
237
+ } catch (err) {
238
+ // Synchronous throw (write-after-destroy on a stream we have not
239
+ // been told about yet) — same handling as an `error` event.
240
+ this.fail(err);
241
+ }
242
+ }
243
+ // Always report success: pino must never see this sink fail, and
244
+ // backpressure here would stall whoever called log().
245
+ callback();
246
+ }
247
+
248
+ override _final(callback: (error?: Error | null) => void): void {
249
+ this.clearRetry();
250
+ try {
251
+ this.inner?.end();
252
+ } catch {
253
+ /* going away anyway */
254
+ }
255
+ callback();
256
+ }
257
+
258
+ override _destroy(
259
+ _err: Error | null,
260
+ callback: (error?: Error | null) => void,
261
+ ): void {
262
+ this.clearRetry();
263
+ this.detachInner();
264
+ callback(null);
265
+ }
266
+
267
+ private openInner(): void {
268
+ try {
269
+ const inner = this.openStream(this.path);
270
+ inner.on("error", (err: Error) => {
271
+ // Ignore errors from a stream we have already given up on.
272
+ if (this.inner === inner) this.fail(err);
273
+ });
274
+ this.inner = inner;
275
+ } catch (err) {
276
+ this.fail(err);
277
+ }
278
+ }
279
+
280
+ /** Give up on the current stream and arm a reopen. Never throws. */
281
+ private fail(err: unknown): void {
282
+ const firstFailure = !this.down;
283
+ this.detachInner();
284
+ this.down = true;
285
+ const delay = this.retryMs;
286
+ this.retryMs = Math.min(this.retryMs * 2, this.maxRetryMs);
287
+ this.scheduleRetry(delay);
288
+ if (!firstFailure) return; // a failed reopen is not news
289
+ const code =
290
+ (err as NodeJS.ErrnoException | undefined)?.code ??
291
+ (err instanceof Error ? err.message : String(err));
292
+ this.emitNotice(
293
+ "warn",
294
+ `Log file sink failed (${code}) — file logging paused, ` +
295
+ `retrying in ${Math.round(delay / 1000)}s`,
296
+ );
297
+ }
298
+
299
+ /** A write landed: the file is usable again. */
300
+ private markHealthy(): void {
301
+ if (!this.down) return;
302
+ this.down = false;
303
+ this.retryMs = this.baseRetryMs;
304
+ const dropped = this.droppedWhileDown;
305
+ this.droppedWhileDown = 0;
306
+ this.emitNotice(
307
+ "info",
308
+ `Log file sink recovered — file logging resumed ` +
309
+ `(${dropped} line(s) dropped while it was down)`,
310
+ );
311
+ }
312
+
313
+ private detachInner(): void {
314
+ const inner = this.inner;
315
+ this.inner = null;
316
+ if (inner === null) return;
317
+ try {
318
+ inner.removeAllListeners("error");
319
+ // destroy() can surface one last error — swallow it here rather
320
+ // than let it reach process-level uncaughtException.
321
+ inner.on("error", () => {});
322
+ inner.destroy();
323
+ } catch {
324
+ /* best effort */
325
+ }
326
+ }
327
+
328
+ private scheduleRetry(delay: number): void {
329
+ this.clearRetry();
330
+ const timer = setTimeout(() => {
331
+ this.retryTimer = null;
332
+ this.openInner();
333
+ }, delay);
334
+ // A paused log sink must not hold the event loop open.
335
+ timer.unref();
336
+ this.retryTimer = timer;
337
+ }
338
+
339
+ private clearRetry(): void {
340
+ if (this.retryTimer === null) return;
341
+ clearTimeout(this.retryTimer);
342
+ this.retryTimer = null;
343
+ }
344
+
345
+ /**
346
+ * Notices are deferred: `fail()` can run inside pino's multistream
347
+ * write loop, and logging re-entrantly from there would scramble the
348
+ * metadata pino hangs off the stream for the rest of that loop.
349
+ */
350
+ private emitNotice(level: "warn" | "info", message: string): void {
351
+ queueMicrotask(() => {
352
+ try {
353
+ this.notify(level, message);
354
+ } catch {
355
+ /* the notice is the least important thing here */
356
+ }
357
+ });
358
+ }
359
+ }
360
+
361
+ function openLogFile(path: string): Writable {
362
+ // 0600: turns and tool output land here — same sensitivity as history.
363
+ return createWriteStream(path, { flags: "a", mode: 0o600 });
364
+ }
365
+
366
+ function notifyViaConsoleSink(level: "warn" | "info", message: string): void {
367
+ if (level === "warn") logWarn("file", message);
368
+ else log("file", message);
138
369
  }
139
370
 
140
371
  // JSON file output (always active outside test runs).
141
372
  if (!IS_VITEST) {
142
- streams.push({
143
- level: "trace",
144
- // 0600: turns and tool output land here — same sensitivity as history.
145
- stream: createWriteStream(LOG_FILE, { flags: "a", mode: 0o600 }),
146
- });
373
+ streams.push({ level: "trace", stream: new ResilientFileSink(LOG_FILE) });
147
374
  }
148
375
 
149
376
  const logger =
@@ -151,8 +378,24 @@ const logger =
151
378
  ? pino({ level: "trace" }, pino.multistream(streams))
152
379
  : pino({ level: "silent" });
153
380
 
381
+ /**
382
+ * Emit one record. A logger that throws is worse than a silent one: the
383
+ * throw lands on whoever called log(), which at shutdown is a signal
384
+ * handler or a crash handler, and a throw there takes the daemon down —
385
+ * pino-pretty's SonicBoom, for one, throws "SonicBoom destroyed"
386
+ * synchronously on every write once a failure has destroyed it. Nothing
387
+ * a sink does may escape this module.
388
+ */
389
+ function emit(write: () => void): void {
390
+ try {
391
+ write();
392
+ } catch {
393
+ /* a broken logger must never become a broken daemon */
394
+ }
395
+ }
396
+
154
397
  export function log(component: LogComponent, message: string): void {
155
- logger.info({ component }, message);
398
+ emit(() => logger.info({ component }, message));
156
399
  }
157
400
 
158
401
  export function logError(
@@ -164,20 +407,22 @@ export function logError(
164
407
  // Capture both the concise message (for log consumers that look at `err`)
165
408
  // and the full stack (for diagnostics). pino-pretty renders the `stack`
166
409
  // field on its own line; JSON consumers can read either field.
167
- logger.error({ component, err: err.message, stack: err.stack }, message);
410
+ emit(() =>
411
+ logger.error({ component, err: err.message, stack: err.stack }, message),
412
+ );
168
413
  } else if (err !== undefined) {
169
- logger.error({ component, err: String(err) }, message);
414
+ emit(() => logger.error({ component, err: String(err) }, message));
170
415
  } else {
171
- logger.error({ component }, message);
416
+ emit(() => logger.error({ component }, message));
172
417
  }
173
418
  }
174
419
 
175
420
  export function logWarn(component: LogComponent, message: string): void {
176
- logger.warn({ component }, message);
421
+ emit(() => logger.warn({ component }, message));
177
422
  }
178
423
 
179
424
  export function logDebug(component: LogComponent, message: string): void {
180
- logger.debug({ component }, message);
425
+ emit(() => logger.debug({ component }, message));
181
426
  }
182
427
 
183
428
  // Expose logger to plugins running in the same process