@edgehero/pi-dispatch-receiver 1.5.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/start.mjs CHANGED
@@ -30,11 +30,13 @@
30
30
  */
31
31
 
32
32
  import http from "node:http";
33
- import { watch } from "node:fs";
33
+ import { readFileSync, watch } from "node:fs";
34
34
  import { dirname, basename } from "node:path";
35
35
  import { loadReceiverConfig, triggersFilePath, reloadTriggers } from "./config.mjs";
36
36
  import { makeReceiver } from "./receiver.mjs";
37
37
  import { entryExitCode } from "./cli.mjs";
38
+ import { installRejectionPrinter } from "@edgehero/pi-dispatch/exit-code";
39
+ import { isEntryModule } from "@edgehero/pi-dispatch/entry";
38
40
  import { makeGitHubAuth } from "@edgehero/pi-dispatch/get-token";
39
41
  import { resolveGitLabSelfId } from "@edgehero/pi-dispatch/gitlab-identity";
40
42
  import { resolveForgejoSelfId } from "@edgehero/pi-dispatch/forgejo-identity";
@@ -45,7 +47,9 @@ import { makeResolveAzureAuthority } from "./azure-members.mjs";
45
47
  import { makeResolveGitHubAuthority } from "./github-members.mjs";
46
48
  import { makeQueue } from "@edgehero/pi-dispatch/queue";
47
49
  import { makeForgeRouter } from "./route.mjs";
48
- import { parseConnection } from "@edgehero/pi-dispatch/connection";
50
+ import { judgeValkeyAtStart, parseConnection, valkeyClientContext } from "@edgehero/pi-dispatch/connection";
51
+ import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "@edgehero/pi-dispatch/watch-closer";
52
+ import { retryIdentity } from "./boot-retry.mjs";
49
53
 
50
54
  /**
51
55
  * Boot the receiver. Collaborators are injected (defaulting to the real ones) so the whole wiring is
@@ -54,6 +58,10 @@ import { parseConnection } from "@edgehero/pi-dispatch/connection";
54
58
  export async function startReceiver(
55
59
  env = process.env,
56
60
  {
61
+ // Where the receiver's JSON log lines go. Defaults to the real stdout; a test injects a collector
62
+ // rather than reassigning `process.stdout.write`, which under `node --test` is the same channel the
63
+ // child process reports its own results on (issue #266).
64
+ write = (chunk) => process.stdout.write(chunk),
57
65
  makeAuth = makeGitHubAuth,
58
66
  makeQueueFn = makeQueue,
59
67
  makeForgeRouterFn = makeForgeRouter,
@@ -65,12 +73,47 @@ export async function startReceiver(
65
73
  resolveAzureSelfId: resolveAzureSelfIdFn = resolveAzureSelfId,
66
74
  makeResolveAzureAuthority: makeResolveAzureAuthorityFn = makeResolveAzureAuthority,
67
75
  makeResolveGitHubAuthority: makeResolveGitHubAuthorityFn = makeResolveGitHubAuthority,
76
+ // The boot retry's clock and sleep (issue #318). `now` is injected for the reason the worker gives
77
+ // (issue #284): a window read off the default `Date.now` beside a fixed test instant is a fuse that
78
+ // fails in CI on a tree nobody touched. `sleep` is left undefined on the real path so
79
+ // `retryIdentity`'s own default runs; a test injects a recorder and never waits real time.
80
+ now = Date.now,
81
+ sleep,
82
+ // The worker's `extraClosers` shape (issue #301): the triggers watch pushes its stop handle here, the
83
+ // real shutdown below drains it, and a test injects its own array so the watch it armed dies with the
84
+ // boot that armed it instead of leaking an FSWatcher across tests.
85
+ closers = [],
86
+ // Issue #464 (gate round 3): whose Valkey VALKEY_URL reaches, judged before the queue is built and before `listen`,
87
+ // by the rule every client of this project applies (`judgeValkeyAtStart`). A seam: the real one probes this
88
+ // host's addresses and reads /proc.
89
+ judgeValkey = (url, opts) => judgeValkeyAtStart(url, valkeyClientContext({ env: opts.env }), { now: opts.now, ...(opts.sleep ? { sleep: opts.sleep } : {}) }),
68
90
  } = {},
69
91
  ) {
70
92
  // Single-object log line: `makeReceiver` calls `log?.({ event, ... })`, so the sink takes ONE object.
71
- const log = (obj) => process.stdout.write(`${JSON.stringify(obj)}\n`);
93
+ const log = (obj) => write(`${JSON.stringify(obj)}\n`);
72
94
 
73
- const cfg = loadReceiverConfig(env);
95
+ // THE BOOT LOAD'S OWN READ is the baseline for the watch's boot-race check (issue #386), captured here
96
+ // rather than inside `watchTriggers`. That is the whole correction of a first attempt: a baseline taken
97
+ // where the watch arms measures the microseconds around the arming, while the window this race lives in
98
+ // is the one between THIS read and that arming -- which holds identity resolution, retried for up to
99
+ // `RECEIVER_IDENTITY_RETRY_SECONDS`. Measured on the real boot: a baseline at the watch closed 0.1 to 0.4
100
+ // milliseconds and left 50 to 300 milliseconds of a slow identity open, seconds under retries.
101
+ //
102
+ // Taken from the loader's OWN read, through the seam it already has, so there is no second read that an
103
+ // edit could land between: what the comparison is against is exactly what the receiver is running.
104
+ const triggersPath = triggersFilePath(env);
105
+ let triggersAtBoot = null;
106
+ const cfg = loadReceiverConfig(env, {
107
+ readFile: (file, enc) => {
108
+ const text = readFileSync(file, enc);
109
+ if (file === triggersPath) triggersAtBoot = text;
110
+ return text;
111
+ },
112
+ });
113
+
114
+ // One options bag for the four identity arms below (issue #318): the SAME window, clock and sleep for
115
+ // every forge, so the bound the operator configured is the bound every arm obeys.
116
+ const retryOpts = { windowMs: cfg.identityRetryWindowMs, log, now, sleep };
74
117
 
75
118
  // The GitHub arm, when the deployment actually serves GitHub -- now conditional, exactly like the three
76
119
  // sibling arms below (issue #99). It was unconditional, and since GITHUB_AUTH_SOURCE defaults to `gh` and
@@ -86,16 +129,23 @@ export async function startReceiver(
86
129
  let selfId;
87
130
  let resolveAuthority;
88
131
  if (cfg.servesGithub) {
89
- // HARD-FAIL identity resolution -- NO try/catch. A throw here (absent/bad github auth, unresolvable
90
- // id) propagates and the server below is never created: without selfId the bot-loop guard cannot
91
- // run, so refusing to boot is the only safe outcome.
132
+ // HARD-FAIL identity resolution -- still NO try/catch here. A DETERMINATE throw (absent/bad github
133
+ // auth, an unresolvable id, anything tagged piDispatchConfig) propagates same-tick and the server
134
+ // below is never created: without selfId the bot-loop guard cannot run, so refusing to boot is the
135
+ // only safe outcome. What changed (issue #318) is the TRANSIENT case: retryIdentity keeps
136
+ // re-resolving inside cfg.identityRetryWindowMs before letting the throw propagate, so a forge that
137
+ // is merely restarting no longer costs the deployment its receiver (the ~25 seconds the systemd
138
+ // unit's start limit used to bound recovery at, and the nothing launchd and nssm bounded it at).
139
+ // The retry wraps EACH arm, never the whole boot: a whole-body retry would rebuild the queue and
140
+ // router below once per attempt and leak their connections, and hoisting the four resolutions above
141
+ // the queue would reorder the first failure an operator sees.
92
142
  //
93
143
  // The WHOLE auth object is kept, not just selfId: the closer resolver below mints its per-delivery
94
144
  // metadata-read token through this same object (issue #231), so identity and mint capability stay
95
145
  // one credential decision -- an arm that resolved its identity is exactly the arm that can answer
96
146
  // a permission question. This is also why the github handler's missing-resolver 503 is unreachable
97
147
  // in a wired receiver: a boot that fails here mounts no `/` at all.
98
- const auth = await makeAuth(cfg.github);
148
+ const auth = await retryIdentity(() => makeAuth(cfg.github), { forge: "github", ...retryOpts });
99
149
  selfId = auth.selfId;
100
150
  log({ event: "self_identity", id: selfId, source: cfg.github.source });
101
151
  // The lookup token asks the mint to narrow to metadata:read -- the App path honors it GitHub-side,
@@ -111,6 +161,10 @@ export async function startReceiver(
111
161
 
112
162
  // Ride-out connection (no failFast): the receiver is long-running and should survive a Valkey
113
163
  // restart, not give up on a transient disconnect.
164
+ // Issue #464 (gate round 3): a refused Valkey stops the receiver here (a configError, exit 2, as the worker's boot),
165
+ // and one that answers nothing for 20 s ends it with exit 1 so the service manager starts it again. It used to keep
166
+ // running and listening with a queue that could never connect.
167
+ await judgeValkey(cfg.valkeyUrl, { env, now, sleep });
114
168
  const queue = makeQueueFn(parseConnection(cfg.valkeyUrl));
115
169
 
116
170
  // Multi-host routing for deliveries that bind a host-local resource (issue #57, `OQ-032`). A trigger
@@ -122,76 +176,140 @@ export async function startReceiver(
122
176
  // or whose deployment has no named hosts, behaves exactly as it did before this existed.
123
177
  const router = makeForgeRouterFn({ valkeyUrl: cfg.valkeyUrl, shared: queue, log });
124
178
 
125
- // The GitLab arm, when configured. Its identity resolution is HARD-FAIL for the same reason github's
126
- // is: without a selfId the bot-loop guard cannot run, and a receiver that listens without it turns the
127
- // harness's own status comment into another paid job.
128
- let gitlab = null;
129
- if (cfg.gitlab) {
130
- const gitlabSelfId = await resolveSelfIdFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token });
131
- log({ event: "self_identity", forge: "gitlab", id: gitlabSelfId, mode: cfg.gitlab.mode });
132
- gitlab = {
133
- mode: cfg.gitlab.mode,
134
- secret: cfg.gitlab.secret,
135
- selfId: gitlabSelfId,
136
- resolveAuthority: makeResolveAuthorityFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }),
137
- };
138
- }
179
+ // EVERY refusal below this line lands after the queue and the router exist, so it must stop
180
+ // what the boot built (issue #299's rule, the receiver's copy; measured for #318 on main AND
181
+ // this branch, against a live Valkey): with the shared clients left open, the entry assigns
182
+ // process.exitCode and the loop never drains -- the failure line printed at 0.4s and the
183
+ // process was still alive at 90s. Under Type=simple that reads as "active (running)" with
184
+ // nothing listening, Restart= never fires, RestartPreventExitStatus=2 is never consulted, and
185
+ // the retry window's exit-1-into-a-fresh-window contract above never engages. The github arm
186
+ // and the config load sit ABOVE the queue and always exited cleanly; these three arms did not.
187
+ try {
188
+ // The GitLab arm, when configured. Its identity resolution is HARD-FAIL for the same reason github's
189
+ // is: without a selfId the bot-loop guard cannot run, and a receiver that listens without it turns the
190
+ // harness's own status comment into another paid job.
191
+ let gitlab = null;
192
+ if (cfg.gitlab) {
193
+ const gitlabSelfId = await retryIdentity(() => resolveSelfIdFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }), { forge: "gitlab", ...retryOpts });
194
+ log({ event: "self_identity", forge: "gitlab", id: gitlabSelfId, mode: cfg.gitlab.mode });
195
+ gitlab = {
196
+ mode: cfg.gitlab.mode,
197
+ secret: cfg.gitlab.secret,
198
+ selfId: gitlabSelfId,
199
+ resolveAuthority: makeResolveAuthorityFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }),
200
+ };
201
+ }
139
202
 
140
- // The Forgejo arm, when configured. Identity resolution is HARD-FAIL here too, and it is the arm where
141
- // that matters most: a repo-scoped Forgejo token cannot call GET /user, so an operator who follows the
142
- // scoping advice without setting FORGEJO_BOT_ID lands exactly here -- and a receiver that shrugged and
143
- // continued would run with selfId undefined, which never equals a sender id and silently turns the
144
- // harness's own comments into more paid jobs.
145
- let forgejo = null;
146
- if (cfg.forgejo) {
147
- const forgejoSelfId = await resolveForgejoSelfIdFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token, botId: cfg.forgejo.botId });
148
- log({ event: "self_identity", forge: "forgejo", id: forgejoSelfId, source: cfg.forgejo.botId ? "FORGEJO_BOT_ID" : "api" });
149
- forgejo = {
150
- secret: cfg.forgejo.secret,
151
- selfId: forgejoSelfId,
152
- resolveAuthority: makeResolveForgejoAuthorityFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token }),
153
- };
154
- }
203
+ // The Forgejo arm, when configured. Identity resolution is HARD-FAIL here too, and it is the arm where
204
+ // that matters most: a repo-scoped Forgejo token cannot call GET /user, so an operator who follows the
205
+ // scoping advice without setting FORGEJO_BOT_ID lands exactly here -- and a receiver that shrugged and
206
+ // continued would run with selfId undefined, which never equals a sender id and silently turns the
207
+ // harness's own comments into more paid jobs.
208
+ let forgejo = null;
209
+ if (cfg.forgejo) {
210
+ const forgejoSelfId = await retryIdentity(() => resolveForgejoSelfIdFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token, botId: cfg.forgejo.botId }), { forge: "forgejo", ...retryOpts });
211
+ log({ event: "self_identity", forge: "forgejo", id: forgejoSelfId, source: cfg.forgejo.botId ? "FORGEJO_BOT_ID" : "api" });
212
+ forgejo = {
213
+ secret: cfg.forgejo.secret,
214
+ selfId: forgejoSelfId,
215
+ resolveAuthority: makeResolveForgejoAuthorityFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token }),
216
+ };
217
+ }
155
218
 
156
- // The Azure arm, when configured. Identity resolution is HARD-FAIL here too, and it resolves BOTH forms
157
- // of the harness's identity in one call: a pull-request delivery names an actor by GUID and a work item
158
- // names them only by email address, so a guard that knew one form would be blind on half the events.
159
- let azure = null;
160
- if (cfg.azure) {
161
- const azureSelfId = await resolveAzureSelfIdFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token });
162
- log({ event: "self_identity", forge: "azure", id: azureSelfId.id, hasAccountName: azureSelfId.email !== null, mode: cfg.azure.mode });
163
- azure = {
164
- mode: cfg.azure.mode,
165
- secret: cfg.azure.secret,
166
- headerName: cfg.azure.headerName,
167
- selfId: azureSelfId,
168
- resolveAuthority: makeResolveAzureAuthorityFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token }),
169
- };
170
- }
219
+ // The Azure arm, when configured. Identity resolution is HARD-FAIL here too, and it resolves BOTH forms
220
+ // of the harness's identity in one call: a pull-request delivery names an actor by GUID and a work item
221
+ // names them only by email address, so a guard that knew one form would be blind on half the events.
222
+ let azure = null;
223
+ if (cfg.azure) {
224
+ const azureSelfId = await retryIdentity(() => resolveAzureSelfIdFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token }), { forge: "azure", ...retryOpts });
225
+ log({ event: "self_identity", forge: "azure", id: azureSelfId.id, hasAccountName: azureSelfId.email !== null, mode: cfg.azure.mode });
226
+ azure = {
227
+ mode: cfg.azure.mode,
228
+ secret: cfg.azure.secret,
229
+ headerName: cfg.azure.headerName,
230
+ selfId: azureSelfId,
231
+ resolveAuthority: makeResolveAzureAuthorityFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token }),
232
+ };
233
+ }
234
+
235
+ const handler = makeReceiver({ queue, router, selfId, cfg, log, gitlab, forgejo, azure, resolveAuthority });
236
+ const server = createServer(handler);
237
+ server.listen(cfg.port, cfg.bind, () =>
238
+ log({ event: "receiver_started", port: cfg.port, bind: cfg.bind, valkey: cfg.valkeyUrl }),
239
+ );
171
240
 
172
- const handler = makeReceiver({ queue, router, selfId, cfg, log, gitlab, forgejo, azure, resolveAuthority });
173
- const server = createServer(handler);
174
- server.listen(cfg.port, cfg.bind, () =>
175
- log({ event: "receiver_started", port: cfg.port, bind: cfg.bind, valkey: cfg.valkeyUrl }),
176
- );
177
-
178
- // Graceful shutdown AND the live-trigger watcher only on the real entry (default createServer). Under
179
- // test injection the fakes are per-test, so a process-wide signal handler or an fs watcher would leak
180
- // across tests; the reload LOGIC (`reloadTriggers`) is unit-tested directly instead.
181
- if (createServer === http.createServer) {
182
- watchTriggers(env, cfg, log);
183
- const shutdown = async (signal) => {
184
- log({ event: "receiver_stopping", signal });
185
- await new Promise((resolve) => server.close(resolve));
186
- await queue.close();
241
+ // The watch arms UNCONDITIONALLY now (issue #301). Armed only on the real entry, the lifecycle defect
242
+ // was muted under test rather than closed, and this file had no coverage that the watch arms at all --
243
+ // `DES-WATCHERS-CLOSE-WITH-THE-WORKER` rejected exactly that posture for the worker. The closer rides
244
+ // the injected `closers` array, so a test drains what its boot armed and the real shutdown closes it.
245
+ //
246
+ // LAST FALLIBLE STEP, deliberately, and it must stay last: every refusal this boot can produce -- the
247
+ // config load, each hard-fail identity resolution, the router build, even a throwing `listen` -- sits
248
+ // ABOVE this line, so a refused boot has armed nothing and there is never a closer with no one left to
249
+ // drain it. The worker states the same invariant where its watchers arm. A step added BELOW that can
250
+ // throw reopens issue #301 on the refusal path; the HARD-FAIL test pins the refusals that exist today.
251
+ closers.push(watchTriggers(env, cfg, log, triggersAtBoot));
252
+
253
+ // Graceful shutdown only on the real entry (default createServer). Under test injection the fakes are
254
+ // per-test, so a process-wide SIGNAL HANDLER would still leak across tests -- and unlike the watch it
255
+ // has no seam to ride: the closers array cannot un-register a `process.once`. The shutdown's own steps
256
+ // are one call per handle, each covered through its seam.
257
+ if (createServer === http.createServer) {
258
+ const shutdown = async (signal) => {
259
+ log({ event: "receiver_stopping", signal });
260
+ await new Promise((resolve) => server.close(resolve));
261
+ await queue.close();
262
+ await router.close();
263
+ // The watch closer, and whatever joins it later. Per-item try, because a throw here would strand
264
+ // the `process.exit(0)` that the unit's stop depends on -- `index.mjs`'s closer loop states the
265
+ // same rule for the worker.
266
+ for (const c of closers) {
267
+ try {
268
+ c?.close?.();
269
+ } catch {
270
+ // A closer that failed has already stopped mattering.
271
+ }
272
+ }
273
+ process.exit(0);
274
+ };
275
+ process.once("SIGTERM", () => void shutdown("SIGTERM"));
276
+ process.once("SIGINT", () => void shutdown("SIGINT"));
277
+ }
278
+
279
+ return server;
280
+ } catch (err) {
281
+ // Best-effort, per handle, and the refusal rethrows unchanged: the cleanup must never
282
+ // replace the story. BullMQ's close() is idempotent and the router closes its own pool.
283
+ //
284
+ // The error swallow first, and it is measured, not defensive: a queue built moments ago can
285
+ // still be mid-handshake, and every client error in that window is forwarded to the Queue
286
+ // object, where makeQueue attaches no listener -- unlistened, one such event crashed the
287
+ // process at exit 1 and stomped the exit 2 this catch exists to deliver. Scoped to the
288
+ // refusal path: the serving path's error surface is unchanged.
289
+ queue.on?.("error", () => {});
290
+ try {
291
+ // The RAW client, deliberately NOT queue.close(): close() on a maybe-still-connecting
292
+ // client strips every listener in RedisConnection.close()'s finally (removeAllListeners),
293
+ // and the constructor's `initializing.catch` then re-emits the flushed handshake as an
294
+ // 'error' on a LISTENERLESS emitter -- an uncaught crash whose exit 1 stomped this exit
295
+ // code on 3 of 5 measured refusals, and a settle-first close() variant still crashed
296
+ // against a DEAD Valkey. A direct disconnect() leaves the connection's listeners alone,
297
+ // so the same late rejection flows connection -> queue -> the swallow above, and ioredis
298
+ // disconnect() releases the socket AND stops the ride-out reconnects, in every client
299
+ // state (the #300 measurement). `_client` is a BullMQ private, reached on the host-pi
300
+ // precedent for pinned internals: the real-BullMQ integration test is the pin, so a bump
301
+ // that moves it goes red there instead of sliding silently back to the wedge.
302
+ queue.connection?._client?.disconnect?.();
303
+ } catch {
304
+ // A handle that failed to close has stopped mattering to a process about to exit.
305
+ }
306
+ try {
187
307
  await router.close();
188
- process.exit(0);
189
- };
190
- process.once("SIGTERM", () => void shutdown("SIGTERM"));
191
- process.once("SIGINT", () => void shutdown("SIGINT"));
308
+ } catch {
309
+ // Same.
310
+ }
311
+ throw err;
192
312
  }
193
-
194
- return server;
195
313
  }
196
314
 
197
315
  /**
@@ -199,31 +317,60 @@ export async function startReceiver(
199
317
  * admin writes with, which swaps the inode a file-watch would lose), debounce, and re-read on change. A bad
200
318
  * edit keeps the running triggers (reloadTriggers never throws) and logs a kept-old notice. Best-effort: a
201
319
  * platform without `fs.watch` logs and the receiver simply keeps its boot-time triggers.
320
+ *
321
+ * Returns the worker's stop handle (issue #301), registered in `closers` so the shutdown -- or the test
322
+ * that injected the array -- closes the FSWatcher and cancels the debounce it armed. A closer is returned
323
+ * even when the watch could not arm: closing a never-armed handle is a no-op by construction, and a
324
+ * caller that has to ask "did I get one" is how a handle goes unregistered.
202
325
  */
203
- function watchTriggers(env, cfg, log) {
326
+ // `triggersAtBoot` has NO DEFAULT on purpose (issue #386). A default of `null` would silently DISABLE the
327
+ // boot-race check for a caller that forgot it, which is the failure this whole change exists to stop; with
328
+ // none, forgetting it is `undefined`, and `readBeforeArming` reads that as "no boot read to hand over" and
329
+ // says so by falling back to its own read rather than to nothing.
330
+ function watchTriggers(env, cfg, log, triggersAtBoot) {
204
331
  const path = triggersFilePath(env);
205
332
  const dir = dirname(path) || ".";
206
333
  const file = basename(path);
207
- let timer = null;
334
+ const handles = { watcher: null, timer: null, closed: false };
335
+ // The worker's closer, reused rather than re-derived (issue #301). Its `log` takes `(event, fields)`
336
+ // where this file's takes one object, so it is handed an adapter here instead of the module growing a
337
+ // second signature. The reload lines go through `closer.reloadLog` -- the reload's VOICE, gated once
338
+ // the handle closes; the arming lines keep the real `log`, because they run before any close exists.
339
+ const closer = makeWatchCloser(handles, (event, fields) => log({ event, ...fields }));
340
+ const readFile = () => readFileSync(path, "utf8");
341
+ const reload = () => {
342
+ const res = reloadTriggers(env, cfg);
343
+ if (res.ok) closer.reloadLog("triggers_reloaded");
344
+ else closer.reloadLog("triggers_reload_invalid", { reason: res.invalid, kept: true });
345
+ };
346
+ // The baseline is the BOOT LOAD's own read, handed in (issue #386). Comparing against a read taken here
347
+ // would measure the arming instead of the window: this service arms LAST, behind identity resolution that
348
+ // retries for up to RECEIVER_IDENTITY_RETRY_SECONDS, and that is the window an edit is lost in.
349
+ readBeforeArming(handles, readFile, triggersAtBoot);
208
350
  try {
209
- watch(dir, (_event, changed) => {
351
+ handles.watcher = watch(dir, (_event, changed) => {
352
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
210
353
  if (changed && changed !== file) return; // only our file (null changed name -> reload to be safe)
211
- clearTimeout(timer);
212
- timer = setTimeout(() => {
213
- const res = reloadTriggers(env, cfg);
214
- if (res.ok) log({ event: "triggers_reloaded" });
215
- else log({ event: "triggers_reload_invalid", reason: res.invalid, kept: true });
216
- }, 150);
217
- }).unref?.();
354
+ clearTimeout(handles.timer);
355
+ handles.timer = setTimeout(reload, WATCH_DEBOUNCE_MS);
356
+ });
357
+ handles.watcher.unref?.();
218
358
  log({ event: "triggers_watching", path });
219
359
  } catch (err) {
220
360
  log({ event: "triggers_watch_unavailable", reason: err?.message });
221
361
  }
362
+ if (changedWhileArming(handles, readFile)) {
363
+ log({ event: "triggers_reread_after_arming", path });
364
+ reload();
365
+ }
366
+ return closer;
222
367
  }
223
368
 
224
369
  // Entry point when run directly (main: src/start.mjs, no bin). Kept out of startReceiver so tests call
225
370
  // it directly. The error line carries only `err.message` -- never a secret or PII.
226
- if (import.meta.url === `file://${process.argv[1]}` || process.argv[1]?.endsWith("start.mjs")) {
371
+ if (isEntryModule(import.meta.url)) {
372
+ // An unhandled rejection is printed as its message alone (PR #475's review), never Node's print of the whole reason.
373
+ installRejectionPrinter();
227
374
  startReceiver(process.env).catch((err) => {
228
375
  process.stderr.write(`${JSON.stringify({ event: "receiver_start_failed", reason: err?.message })}\n`);
229
376
  // entryExitCode, NOT a bare 1. This file is what `receiver.service` execs -- cli.mjs is not on that