talon-agent 5.2.0 → 5.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app.ts +41 -32
- package/src/core/daemon/crash.ts +82 -0
- package/src/util/log.ts +263 -18
package/package.json
CHANGED
package/src/app.ts
CHANGED
|
@@ -30,6 +30,11 @@ import { shutdownAgents } from "./core/agents/index.js";
|
|
|
30
30
|
import { pruneSettledTriggers } from "./storage/triggers.js";
|
|
31
31
|
import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
|
|
32
32
|
import { spawnSuccessor } from "./core/daemon/respawn.js";
|
|
33
|
+
import {
|
|
34
|
+
crashCleanup,
|
|
35
|
+
crashStep,
|
|
36
|
+
handleUncaughtException,
|
|
37
|
+
} from "./core/daemon/crash.js";
|
|
33
38
|
import { log, logError, logWarn } from "./util/log.js";
|
|
34
39
|
import { bootPhase, bootReport } from "./core/daemon/boot-timer.js";
|
|
35
40
|
import {
|
|
@@ -114,6 +119,10 @@ onBackendChange((holder, newBackend, info) => {
|
|
|
114
119
|
let shuttingDown = false;
|
|
115
120
|
let triggerPruneTimer: ReturnType<typeof setInterval> | null = null;
|
|
116
121
|
|
|
122
|
+
// The composition root owns the SQLite handle, so it hands the crash
|
|
123
|
+
// path its checkpoint (see core/daemon/crash.ts).
|
|
124
|
+
const crashHooks = { flushDatabase };
|
|
125
|
+
|
|
117
126
|
const SHUTDOWN_TIMEOUT_MS = 15_000;
|
|
118
127
|
const DRAIN_TIMEOUT_MS = 5_000;
|
|
119
128
|
|
|
@@ -127,7 +136,11 @@ async function shutdownStep(name: string, fn: () => unknown): Promise<void> {
|
|
|
127
136
|
try {
|
|
128
137
|
await fn();
|
|
129
138
|
} catch (err) {
|
|
130
|
-
|
|
139
|
+
// Even the report is best-effort: a shutdown triggered by a full
|
|
140
|
+
// disk must not die inside its own error path.
|
|
141
|
+
crashStep("shutdown report", () =>
|
|
142
|
+
logError("shutdown", `${name} failed`, err),
|
|
143
|
+
);
|
|
131
144
|
}
|
|
132
145
|
}
|
|
133
146
|
|
|
@@ -138,15 +151,18 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
138
151
|
|
|
139
152
|
const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
|
|
140
153
|
const forceTimer = setTimeout(() => {
|
|
141
|
-
|
|
142
|
-
//
|
|
143
|
-
//
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
//
|
|
147
|
-
//
|
|
148
|
-
//
|
|
149
|
-
|
|
154
|
+
// Cleanup first, report second. Handing off matters most here: a
|
|
155
|
+
// restart must survive a subsystem that won't stop (a wedged FUSE
|
|
156
|
+
// unmount, an MCP server ignoring SIGTERM, a backend child that
|
|
157
|
+
// never acks), and it must also survive a logger that can't write —
|
|
158
|
+
// logging first is what cost us the successor on 2026-09-18. The
|
|
159
|
+
// successor may briefly race the long-poll we failed to release,
|
|
160
|
+
// but grammy retries the 409 — a few seconds of overlap beats
|
|
161
|
+
// staying down.
|
|
162
|
+
crashCleanup(crashHooks);
|
|
163
|
+
crashStep("timeout report", () =>
|
|
164
|
+
logError("shutdown", "Timeout exceeded, forcing exit"),
|
|
165
|
+
);
|
|
150
166
|
process.exit(1);
|
|
151
167
|
}, SHUTDOWN_TIMEOUT_MS);
|
|
152
168
|
forceTimer.unref();
|
|
@@ -225,37 +241,29 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
225
241
|
const { shutdownHub } = await import("./core/mcp-hub/index.js");
|
|
226
242
|
await shutdownHub();
|
|
227
243
|
});
|
|
228
|
-
|
|
244
|
+
// Each tail step stands alone: a full disk can make any of them throw,
|
|
245
|
+
// and none of them may cost us the ones that follow.
|
|
246
|
+
crashStep("database flush", () => flushDatabase());
|
|
229
247
|
// Guarded removal: only clear the record if it still names us. A
|
|
230
248
|
// successor that raced ahead and wrote its own pid here must not be
|
|
231
249
|
// orphaned (the bug that made `talon restart` spawn duplicate daemons).
|
|
232
|
-
removePidRecordIfOwnedBy(process.pid);
|
|
233
|
-
log("shutdown", "State saved");
|
|
250
|
+
crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
|
|
251
|
+
crashStep("shutdown report", () => log("shutdown", "State saved"));
|
|
234
252
|
// Hand off last: the frontends are stopped, so the successor binds
|
|
235
253
|
// Telegram's long-poll only after we have released it. No-op unless
|
|
236
254
|
// /restart or /update armed a respawn.
|
|
237
|
-
spawnSuccessor();
|
|
255
|
+
crashStep("respawn handoff", () => spawnSuccessor());
|
|
238
256
|
process.exit(0);
|
|
239
257
|
}
|
|
240
258
|
|
|
241
259
|
process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
|
|
242
260
|
process.on("SIGINT", () => gracefulShutdown("SIGINT"));
|
|
243
261
|
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
return;
|
|
250
|
-
}
|
|
251
|
-
logError("bot", "Uncaught exception", err);
|
|
252
|
-
flushDatabase();
|
|
253
|
-
// Same pid-guarded removal as the graceful path — a crashed daemon
|
|
254
|
-
// must not leave a record that makes `talon status` chase a dead or
|
|
255
|
-
// recycled pid.
|
|
256
|
-
removePidRecordIfOwnedBy(process.pid);
|
|
257
|
-
process.exit(1);
|
|
258
|
-
});
|
|
262
|
+
// Cleanup runs before the crash is reported (and the EPIPE suppression
|
|
263
|
+
// is unchanged) — see core/daemon/crash.ts for why the order matters.
|
|
264
|
+
process.on("uncaughtException", (err) =>
|
|
265
|
+
handleUncaughtException(err, crashHooks),
|
|
266
|
+
);
|
|
259
267
|
|
|
260
268
|
process.on("unhandledRejection", (reason) => {
|
|
261
269
|
logWarn(
|
|
@@ -333,8 +341,9 @@ async function main(): Promise<void> {
|
|
|
333
341
|
}
|
|
334
342
|
|
|
335
343
|
main().catch((err) => {
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
344
|
+
crashCleanup(crashHooks);
|
|
345
|
+
crashStep("startup report", () =>
|
|
346
|
+
logError("bot", "Fatal startup error", err),
|
|
347
|
+
);
|
|
339
348
|
process.exit(1);
|
|
340
349
|
});
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Crash-path cleanup — what has to happen when the daemon goes down
|
|
3
|
+
* abnormally: an uncaught exception, the forced exit after a shutdown
|
|
4
|
+
* timeout, or a fatal startup error.
|
|
5
|
+
*
|
|
6
|
+
* Ordering is the whole point. The old handler logged first and cleaned
|
|
7
|
+
* up afterwards, which works right up until logging is the thing that
|
|
8
|
+
* broke. On 2026-09-18 the root disk filled; the log file's write stream
|
|
9
|
+
* emitted ENOSPC with nobody listening, the uncaught-exception handler
|
|
10
|
+
* called `logError` as its first act, threw inside itself, and Node
|
|
11
|
+
* aborted the process — no database checkpoint, a stale pidfile left
|
|
12
|
+
* behind, and (the expensive part) an armed `/update` handoff that never
|
|
13
|
+
* spawned its successor. The daemon simply vanished.
|
|
14
|
+
*
|
|
15
|
+
* So: the essentials first, each in its own try/catch, logging last and
|
|
16
|
+
* strictly best-effort. `src/util/log.ts` now keeps a broken log file
|
|
17
|
+
* from throwing at all — this is the second line of defence, for every
|
|
18
|
+
* other way logging can fail while the machine is sick.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { logError, logWarn } from "../../util/log.js";
|
|
22
|
+
import { removePidRecordIfOwnedBy } from "./pidfile.js";
|
|
23
|
+
import { spawnSuccessor } from "./respawn.js";
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Steps the crash path can only get from the composition root.
|
|
27
|
+
* `flushDatabase` is injected because the SQLite handle stays inside
|
|
28
|
+
* `storage/` (.dependency-cruiser: db-handle-stays-in-storage).
|
|
29
|
+
*/
|
|
30
|
+
export type CrashHooks = {
|
|
31
|
+
flushDatabase: () => void;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Run one crash-path step. Never throws, and never uses the logger —
|
|
36
|
+
* a broken logger is precisely the case this path exists for, so the
|
|
37
|
+
* console is the fallback of last resort.
|
|
38
|
+
*/
|
|
39
|
+
export function crashStep(name: string, fn: () => void): void {
|
|
40
|
+
try {
|
|
41
|
+
fn();
|
|
42
|
+
} catch (err) {
|
|
43
|
+
try {
|
|
44
|
+
console.error(`[talon] crash cleanup: ${name} failed`, err);
|
|
45
|
+
} catch {
|
|
46
|
+
/* stdio is gone too — nothing left to try */
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* The non-logging essentials, in the order that matters:
|
|
53
|
+
* 1. drop the pid record, so `talon status` stops chasing a dead pid
|
|
54
|
+
* (before the successor writes its own — the guarded removal then
|
|
55
|
+
* cannot possibly orphan it),
|
|
56
|
+
* 2. hand off to the successor if `/restart` or `/update` armed one
|
|
57
|
+
* (a no-op otherwise, see respawn.ts),
|
|
58
|
+
* 3. checkpoint the database.
|
|
59
|
+
*/
|
|
60
|
+
export function crashCleanup(hooks: CrashHooks): void {
|
|
61
|
+
crashStep("pid record removal", () => removePidRecordIfOwnedBy(process.pid));
|
|
62
|
+
crashStep("respawn handoff", () => spawnSuccessor());
|
|
63
|
+
crashStep("database flush", hooks.flushDatabase);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* `process.on("uncaughtException")` body. Cleanup happens before the
|
|
68
|
+
* crash is reported, never after.
|
|
69
|
+
*/
|
|
70
|
+
export function handleUncaughtException(err: Error, hooks: CrashHooks): void {
|
|
71
|
+
// EPIPE errors from network sockets (e.g. Telegram MTProto) are transient —
|
|
72
|
+
// gramjs will reconnect; crashing the process here is wrong.
|
|
73
|
+
if ((err as NodeJS.ErrnoException).code === "EPIPE") {
|
|
74
|
+
crashStep("EPIPE notice", () =>
|
|
75
|
+
logWarn("bot", `Suppressed transient EPIPE error: ${err.message}`),
|
|
76
|
+
);
|
|
77
|
+
return;
|
|
78
|
+
}
|
|
79
|
+
crashCleanup(hooks);
|
|
80
|
+
crashStep("crash report", () => logError("bot", "Uncaught exception", err));
|
|
81
|
+
process.exit(1);
|
|
82
|
+
}
|
package/src/util/log.ts
CHANGED
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
unlinkSync,
|
|
19
19
|
createWriteStream,
|
|
20
20
|
} from "node:fs";
|
|
21
|
+
import { Writable } from "node:stream";
|
|
21
22
|
import { dirs, files } from "./paths.js";
|
|
22
23
|
|
|
23
24
|
export type LogComponent =
|
|
@@ -127,23 +128,249 @@ const streams: pino.StreamEntry[] = [];
|
|
|
127
128
|
|
|
128
129
|
// Console output (disabled in quiet mode), pretty-printed.
|
|
129
130
|
if (!quiet) {
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
ignore: "pid,hostname",
|
|
135
|
-
translateTime: "HH:MM:ss",
|
|
136
|
-
}),
|
|
131
|
+
const consoleStream = prettyStream({
|
|
132
|
+
colorize: true,
|
|
133
|
+
ignore: "pid,hostname",
|
|
134
|
+
translateTime: "HH:MM:ss",
|
|
137
135
|
});
|
|
136
|
+
// pino-pretty writes through a SonicBoom on fd 1 whose own error
|
|
137
|
+
// handler removes itself after the first non-EPIPE failure
|
|
138
|
+
// (pino-pretty/lib/utils/build-safe-sonic-boom.js). With stdout
|
|
139
|
+
// redirected to a file on a full disk that leaves the pipeline's next
|
|
140
|
+
// error unhandled — i.e. an uncaught exception raised from inside a
|
|
141
|
+
// log call. One permanent listener closes that door; a dead console
|
|
142
|
+
// is not worth a dead daemon.
|
|
143
|
+
consoleStream.on("error", () => {});
|
|
144
|
+
streams.push({ level: "trace", stream: consoleStream });
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** How long to wait before the first attempt to reopen a failed log file. */
|
|
148
|
+
const SINK_RETRY_MS = 30_000;
|
|
149
|
+
/** Ceiling for the doubling backoff between reopen attempts. */
|
|
150
|
+
const SINK_MAX_RETRY_MS = 5 * 60_000;
|
|
151
|
+
|
|
152
|
+
export type ResilientFileSinkOptions = {
|
|
153
|
+
/** Opens the underlying file stream. Injection seam for tests. */
|
|
154
|
+
open?: (path: string) => Writable;
|
|
155
|
+
/** Where the pause/resume notices go. Defaults to the console sink. */
|
|
156
|
+
notify?: (level: "warn" | "info", message: string) => void;
|
|
157
|
+
/** First backoff step (default 30s). */
|
|
158
|
+
retryMs?: number;
|
|
159
|
+
/** Backoff ceiling (default 5 min). */
|
|
160
|
+
maxRetryMs?: number;
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* A log file destination that cannot take the process down.
|
|
165
|
+
*
|
|
166
|
+
* A bare `createWriteStream` handed to `pino.multistream` is a loaded
|
|
167
|
+
* gun: when the disk fills (ENOSPC), or the file is unlinked under a
|
|
168
|
+
* rotation (EBADF), or the fd goes bad (EIO), the stream emits `error`.
|
|
169
|
+
* With no listener that is an uncaught exception — and it fires from
|
|
170
|
+
* inside a log call, so the crash handler's own `logError` runs on a
|
|
171
|
+
* logger that is already broken. That is how a full disk killed the
|
|
172
|
+
* daemon on 2026-09-18: the process aborted mid-`/update` with no
|
|
173
|
+
* shutdown, no pidfile cleanup, and no successor.
|
|
174
|
+
*
|
|
175
|
+
* This sink owns the file stream instead of exposing it. pino only ever
|
|
176
|
+
* sees this object, which never emits `error` and never blocks:
|
|
177
|
+
* - write failures pause file logging and destroy the broken stream,
|
|
178
|
+
* - lines written while paused are DROPPED and counted (never
|
|
179
|
+
* buffered — the failure mode here is "no space", so growing a
|
|
180
|
+
* buffer is the last thing to do),
|
|
181
|
+
* - an unref'd timer retries the open on a 30s → 5min backoff,
|
|
182
|
+
* - the first write to land again resumes logging and reports how
|
|
183
|
+
* many lines were lost.
|
|
184
|
+
* The console sink keeps working throughout, and carries the two
|
|
185
|
+
* notices.
|
|
186
|
+
*/
|
|
187
|
+
export class ResilientFileSink extends Writable {
|
|
188
|
+
private readonly path: string;
|
|
189
|
+
private readonly openStream: (path: string) => Writable;
|
|
190
|
+
private readonly notify: (level: "warn" | "info", message: string) => void;
|
|
191
|
+
private readonly baseRetryMs: number;
|
|
192
|
+
private readonly maxRetryMs: number;
|
|
193
|
+
private inner: Writable | null = null;
|
|
194
|
+
private retryMs: number;
|
|
195
|
+
private retryTimer: ReturnType<typeof setTimeout> | null = null;
|
|
196
|
+
private droppedWhileDown = 0;
|
|
197
|
+
private down = false;
|
|
198
|
+
|
|
199
|
+
constructor(path: string, opts: ResilientFileSinkOptions = {}) {
|
|
200
|
+
// decodeStrings:false keeps pino's serialized lines as strings;
|
|
201
|
+
// autoDestroy:false means a downstream failure can never tear this
|
|
202
|
+
// object down — it is the only thing standing between a broken file
|
|
203
|
+
// and the process.
|
|
204
|
+
super({ decodeStrings: false, autoDestroy: false });
|
|
205
|
+
this.path = path;
|
|
206
|
+
this.openStream = opts.open ?? openLogFile;
|
|
207
|
+
this.notify = opts.notify ?? notifyViaConsoleSink;
|
|
208
|
+
this.baseRetryMs = opts.retryMs ?? SINK_RETRY_MS;
|
|
209
|
+
this.maxRetryMs = opts.maxRetryMs ?? SINK_MAX_RETRY_MS;
|
|
210
|
+
this.retryMs = this.baseRetryMs;
|
|
211
|
+
this.openInner();
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Lines discarded since the sink went down; cleared on recovery. */
|
|
215
|
+
get dropped(): number {
|
|
216
|
+
return this.droppedWhileDown;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/** True while file logging is paused (the console sink still runs). */
|
|
220
|
+
get isDown(): boolean {
|
|
221
|
+
return this.down;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
override _write(
|
|
225
|
+
chunk: unknown,
|
|
226
|
+
_encoding: BufferEncoding,
|
|
227
|
+
callback: (error?: Error | null) => void,
|
|
228
|
+
): void {
|
|
229
|
+
const inner = this.inner;
|
|
230
|
+
if (inner === null) {
|
|
231
|
+
this.droppedWhileDown++;
|
|
232
|
+
} else {
|
|
233
|
+
try {
|
|
234
|
+
inner.write(chunk as string, (err) => {
|
|
235
|
+
if (!err) this.markHealthy();
|
|
236
|
+
});
|
|
237
|
+
} catch (err) {
|
|
238
|
+
// Synchronous throw (write-after-destroy on a stream we have not
|
|
239
|
+
// been told about yet) — same handling as an `error` event.
|
|
240
|
+
this.fail(err);
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
// Always report success: pino must never see this sink fail, and
|
|
244
|
+
// backpressure here would stall whoever called log().
|
|
245
|
+
callback();
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
override _final(callback: (error?: Error | null) => void): void {
|
|
249
|
+
this.clearRetry();
|
|
250
|
+
try {
|
|
251
|
+
this.inner?.end();
|
|
252
|
+
} catch {
|
|
253
|
+
/* going away anyway */
|
|
254
|
+
}
|
|
255
|
+
callback();
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
override _destroy(
|
|
259
|
+
_err: Error | null,
|
|
260
|
+
callback: (error?: Error | null) => void,
|
|
261
|
+
): void {
|
|
262
|
+
this.clearRetry();
|
|
263
|
+
this.detachInner();
|
|
264
|
+
callback(null);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
private openInner(): void {
|
|
268
|
+
try {
|
|
269
|
+
const inner = this.openStream(this.path);
|
|
270
|
+
inner.on("error", (err: Error) => {
|
|
271
|
+
// Ignore errors from a stream we have already given up on.
|
|
272
|
+
if (this.inner === inner) this.fail(err);
|
|
273
|
+
});
|
|
274
|
+
this.inner = inner;
|
|
275
|
+
} catch (err) {
|
|
276
|
+
this.fail(err);
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/** Give up on the current stream and arm a reopen. Never throws. */
|
|
281
|
+
private fail(err: unknown): void {
|
|
282
|
+
const firstFailure = !this.down;
|
|
283
|
+
this.detachInner();
|
|
284
|
+
this.down = true;
|
|
285
|
+
const delay = this.retryMs;
|
|
286
|
+
this.retryMs = Math.min(this.retryMs * 2, this.maxRetryMs);
|
|
287
|
+
this.scheduleRetry(delay);
|
|
288
|
+
if (!firstFailure) return; // a failed reopen is not news
|
|
289
|
+
const code =
|
|
290
|
+
(err as NodeJS.ErrnoException | undefined)?.code ??
|
|
291
|
+
(err instanceof Error ? err.message : String(err));
|
|
292
|
+
this.emitNotice(
|
|
293
|
+
"warn",
|
|
294
|
+
`Log file sink failed (${code}) — file logging paused, ` +
|
|
295
|
+
`retrying in ${Math.round(delay / 1000)}s`,
|
|
296
|
+
);
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
/** A write landed: the file is usable again. */
|
|
300
|
+
private markHealthy(): void {
|
|
301
|
+
if (!this.down) return;
|
|
302
|
+
this.down = false;
|
|
303
|
+
this.retryMs = this.baseRetryMs;
|
|
304
|
+
const dropped = this.droppedWhileDown;
|
|
305
|
+
this.droppedWhileDown = 0;
|
|
306
|
+
this.emitNotice(
|
|
307
|
+
"info",
|
|
308
|
+
`Log file sink recovered — file logging resumed ` +
|
|
309
|
+
`(${dropped} line(s) dropped while it was down)`,
|
|
310
|
+
);
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
private detachInner(): void {
|
|
314
|
+
const inner = this.inner;
|
|
315
|
+
this.inner = null;
|
|
316
|
+
if (inner === null) return;
|
|
317
|
+
try {
|
|
318
|
+
inner.removeAllListeners("error");
|
|
319
|
+
// destroy() can surface one last error — swallow it here rather
|
|
320
|
+
// than let it reach process-level uncaughtException.
|
|
321
|
+
inner.on("error", () => {});
|
|
322
|
+
inner.destroy();
|
|
323
|
+
} catch {
|
|
324
|
+
/* best effort */
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
private scheduleRetry(delay: number): void {
|
|
329
|
+
this.clearRetry();
|
|
330
|
+
const timer = setTimeout(() => {
|
|
331
|
+
this.retryTimer = null;
|
|
332
|
+
this.openInner();
|
|
333
|
+
}, delay);
|
|
334
|
+
// A paused log sink must not hold the event loop open.
|
|
335
|
+
timer.unref();
|
|
336
|
+
this.retryTimer = timer;
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
private clearRetry(): void {
|
|
340
|
+
if (this.retryTimer === null) return;
|
|
341
|
+
clearTimeout(this.retryTimer);
|
|
342
|
+
this.retryTimer = null;
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* Notices are deferred: `fail()` can run inside pino's multistream
|
|
347
|
+
* write loop, and logging re-entrantly from there would scramble the
|
|
348
|
+
* metadata pino hangs off the stream for the rest of that loop.
|
|
349
|
+
*/
|
|
350
|
+
private emitNotice(level: "warn" | "info", message: string): void {
|
|
351
|
+
queueMicrotask(() => {
|
|
352
|
+
try {
|
|
353
|
+
this.notify(level, message);
|
|
354
|
+
} catch {
|
|
355
|
+
/* the notice is the least important thing here */
|
|
356
|
+
}
|
|
357
|
+
});
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
function openLogFile(path: string): Writable {
|
|
362
|
+
// 0600: turns and tool output land here — same sensitivity as history.
|
|
363
|
+
return createWriteStream(path, { flags: "a", mode: 0o600 });
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
function notifyViaConsoleSink(level: "warn" | "info", message: string): void {
|
|
367
|
+
if (level === "warn") logWarn("file", message);
|
|
368
|
+
else log("file", message);
|
|
138
369
|
}
|
|
139
370
|
|
|
140
371
|
// JSON file output (always active outside test runs).
|
|
141
372
|
if (!IS_VITEST) {
|
|
142
|
-
streams.push({
|
|
143
|
-
level: "trace",
|
|
144
|
-
// 0600: turns and tool output land here — same sensitivity as history.
|
|
145
|
-
stream: createWriteStream(LOG_FILE, { flags: "a", mode: 0o600 }),
|
|
146
|
-
});
|
|
373
|
+
streams.push({ level: "trace", stream: new ResilientFileSink(LOG_FILE) });
|
|
147
374
|
}
|
|
148
375
|
|
|
149
376
|
const logger =
|
|
@@ -151,8 +378,24 @@ const logger =
|
|
|
151
378
|
? pino({ level: "trace" }, pino.multistream(streams))
|
|
152
379
|
: pino({ level: "silent" });
|
|
153
380
|
|
|
381
|
+
/**
|
|
382
|
+
* Emit one record. A logger that throws is worse than a silent one: the
|
|
383
|
+
* throw lands on whoever called log(), which at shutdown is a signal
|
|
384
|
+
* handler or a crash handler, and a throw there takes the daemon down —
|
|
385
|
+
* pino-pretty's SonicBoom, for one, throws "SonicBoom destroyed"
|
|
386
|
+
* synchronously on every write once a failure has destroyed it. Nothing
|
|
387
|
+
* a sink does may escape this module.
|
|
388
|
+
*/
|
|
389
|
+
function emit(write: () => void): void {
|
|
390
|
+
try {
|
|
391
|
+
write();
|
|
392
|
+
} catch {
|
|
393
|
+
/* a broken logger must never become a broken daemon */
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
|
|
154
397
|
export function log(component: LogComponent, message: string): void {
|
|
155
|
-
logger.info({ component }, message);
|
|
398
|
+
emit(() => logger.info({ component }, message));
|
|
156
399
|
}
|
|
157
400
|
|
|
158
401
|
export function logError(
|
|
@@ -164,20 +407,22 @@ export function logError(
|
|
|
164
407
|
// Capture both the concise message (for log consumers that look at `err`)
|
|
165
408
|
// and the full stack (for diagnostics). pino-pretty renders the `stack`
|
|
166
409
|
// field on its own line; JSON consumers can read either field.
|
|
167
|
-
|
|
410
|
+
emit(() =>
|
|
411
|
+
logger.error({ component, err: err.message, stack: err.stack }, message),
|
|
412
|
+
);
|
|
168
413
|
} else if (err !== undefined) {
|
|
169
|
-
logger.error({ component, err: String(err) }, message);
|
|
414
|
+
emit(() => logger.error({ component, err: String(err) }, message));
|
|
170
415
|
} else {
|
|
171
|
-
logger.error({ component }, message);
|
|
416
|
+
emit(() => logger.error({ component }, message));
|
|
172
417
|
}
|
|
173
418
|
}
|
|
174
419
|
|
|
175
420
|
export function logWarn(component: LogComponent, message: string): void {
|
|
176
|
-
logger.warn({ component }, message);
|
|
421
|
+
emit(() => logger.warn({ component }, message));
|
|
177
422
|
}
|
|
178
423
|
|
|
179
424
|
export function logDebug(component: LogComponent, message: string): void {
|
|
180
|
-
logger.debug({ component }, message);
|
|
425
|
+
emit(() => logger.debug({ component }, message));
|
|
181
426
|
}
|
|
182
427
|
|
|
183
428
|
// Expose logger to plugins running in the same process
|