@edgehero/pi-dispatch-receiver 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -0
- package/package.json +7 -4
- package/src/boot-retry.mjs +141 -0
- package/src/cli.mjs +12 -5
- package/src/config.mjs +30 -5
- package/src/filter-azure.mjs +12 -4
- package/src/filter-forgejo.mjs +14 -6
- package/src/filter-gitlab.mjs +14 -6
- package/src/filter.mjs +16 -5
- package/src/poller-config.mjs +6 -1
- package/src/poller.mjs +83 -14
- package/src/receiver.mjs +57 -28
- package/src/route.mjs +126 -0
- package/src/start.mjs +243 -84
package/src/start.mjs
CHANGED
|
@@ -30,11 +30,13 @@
|
|
|
30
30
|
*/
|
|
31
31
|
|
|
32
32
|
import http from "node:http";
|
|
33
|
-
import { watch } from "node:fs";
|
|
33
|
+
import { readFileSync, watch } from "node:fs";
|
|
34
34
|
import { dirname, basename } from "node:path";
|
|
35
35
|
import { loadReceiverConfig, triggersFilePath, reloadTriggers } from "./config.mjs";
|
|
36
36
|
import { makeReceiver } from "./receiver.mjs";
|
|
37
37
|
import { entryExitCode } from "./cli.mjs";
|
|
38
|
+
import { installRejectionPrinter } from "@edgehero/pi-dispatch/exit-code";
|
|
39
|
+
import { isEntryModule } from "@edgehero/pi-dispatch/entry";
|
|
38
40
|
import { makeGitHubAuth } from "@edgehero/pi-dispatch/get-token";
|
|
39
41
|
import { resolveGitLabSelfId } from "@edgehero/pi-dispatch/gitlab-identity";
|
|
40
42
|
import { resolveForgejoSelfId } from "@edgehero/pi-dispatch/forgejo-identity";
|
|
@@ -44,7 +46,10 @@ import { makeResolveForgejoAuthority } from "./forgejo-members.mjs";
|
|
|
44
46
|
import { makeResolveAzureAuthority } from "./azure-members.mjs";
|
|
45
47
|
import { makeResolveGitHubAuthority } from "./github-members.mjs";
|
|
46
48
|
import { makeQueue } from "@edgehero/pi-dispatch/queue";
|
|
47
|
-
import {
|
|
49
|
+
import { makeForgeRouter } from "./route.mjs";
|
|
50
|
+
import { judgeValkeyAtStart, parseConnection, valkeyClientContext } from "@edgehero/pi-dispatch/connection";
|
|
51
|
+
import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "@edgehero/pi-dispatch/watch-closer";
|
|
52
|
+
import { retryIdentity } from "./boot-retry.mjs";
|
|
48
53
|
|
|
49
54
|
/**
|
|
50
55
|
* Boot the receiver. Collaborators are injected (defaulting to the real ones) so the whole wiring is
|
|
@@ -53,8 +58,13 @@ import { parseConnection } from "@edgehero/pi-dispatch/connection";
|
|
|
53
58
|
export async function startReceiver(
|
|
54
59
|
env = process.env,
|
|
55
60
|
{
|
|
61
|
+
// Where the receiver's JSON log lines go. Defaults to the real stdout; a test injects a collector
|
|
62
|
+
// rather than reassigning `process.stdout.write`, which under `node --test` is the same channel the
|
|
63
|
+
// child process reports its own results on (issue #266).
|
|
64
|
+
write = (chunk) => process.stdout.write(chunk),
|
|
56
65
|
makeAuth = makeGitHubAuth,
|
|
57
66
|
makeQueueFn = makeQueue,
|
|
67
|
+
makeForgeRouterFn = makeForgeRouter,
|
|
58
68
|
createServer = http.createServer,
|
|
59
69
|
resolveGitLabSelfId: resolveSelfIdFn = resolveGitLabSelfId,
|
|
60
70
|
makeResolveAuthority: makeResolveAuthorityFn = makeResolveAuthority,
|
|
@@ -63,12 +73,47 @@ export async function startReceiver(
|
|
|
63
73
|
resolveAzureSelfId: resolveAzureSelfIdFn = resolveAzureSelfId,
|
|
64
74
|
makeResolveAzureAuthority: makeResolveAzureAuthorityFn = makeResolveAzureAuthority,
|
|
65
75
|
makeResolveGitHubAuthority: makeResolveGitHubAuthorityFn = makeResolveGitHubAuthority,
|
|
76
|
+
// The boot retry's clock and sleep (issue #318). `now` is injected for the reason the worker gives
|
|
77
|
+
// (issue #284): a window read off the default `Date.now` beside a fixed test instant is a fuse that
|
|
78
|
+
// fails in CI on a tree nobody touched. `sleep` is left undefined on the real path so
|
|
79
|
+
// `retryIdentity`'s own default runs; a test injects a recorder and never waits real time.
|
|
80
|
+
now = Date.now,
|
|
81
|
+
sleep,
|
|
82
|
+
// The worker's `extraClosers` shape (issue #301): the triggers watch pushes its stop handle here, the
|
|
83
|
+
// real shutdown below drains it, and a test injects its own array so the watch it armed dies with the
|
|
84
|
+
// boot that armed it instead of leaking an FSWatcher across tests.
|
|
85
|
+
closers = [],
|
|
86
|
+
// Issue #464 (gate round 3): whose Valkey VALKEY_URL reaches, judged before the queue is built and before `listen`,
|
|
87
|
+
// by the rule every client of this project applies (`judgeValkeyAtStart`). A seam: the real one probes this
|
|
88
|
+
// host's addresses and reads /proc.
|
|
89
|
+
judgeValkey = (url, opts) => judgeValkeyAtStart(url, valkeyClientContext({ env: opts.env }), { now: opts.now, ...(opts.sleep ? { sleep: opts.sleep } : {}) }),
|
|
66
90
|
} = {},
|
|
67
91
|
) {
|
|
68
92
|
// Single-object log line: `makeReceiver` calls `log?.({ event, ... })`, so the sink takes ONE object.
|
|
69
|
-
const log = (obj) =>
|
|
93
|
+
const log = (obj) => write(`${JSON.stringify(obj)}\n`);
|
|
70
94
|
|
|
71
|
-
|
|
95
|
+
// THE BOOT LOAD'S OWN READ is the baseline for the watch's boot-race check (issue #386), captured here
|
|
96
|
+
// rather than inside `watchTriggers`. That is the whole correction of a first attempt: a baseline taken
|
|
97
|
+
// where the watch arms measures the microseconds around the arming, while the window this race lives in
|
|
98
|
+
// is the one between THIS read and that arming -- which holds identity resolution, retried for up to
|
|
99
|
+
// `RECEIVER_IDENTITY_RETRY_SECONDS`. Measured on the real boot: a baseline at the watch closed 0.1 to 0.4
|
|
100
|
+
// milliseconds and left 50 to 300 milliseconds of a slow identity open, seconds under retries.
|
|
101
|
+
//
|
|
102
|
+
// Taken from the loader's OWN read, through the seam it already has, so there is no second read that an
|
|
103
|
+
// edit could land between: what the comparison is against is exactly what the receiver is running.
|
|
104
|
+
const triggersPath = triggersFilePath(env);
|
|
105
|
+
let triggersAtBoot = null;
|
|
106
|
+
const cfg = loadReceiverConfig(env, {
|
|
107
|
+
readFile: (file, enc) => {
|
|
108
|
+
const text = readFileSync(file, enc);
|
|
109
|
+
if (file === triggersPath) triggersAtBoot = text;
|
|
110
|
+
return text;
|
|
111
|
+
},
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
// One options bag for the four identity arms below (issue #318): the SAME window, clock and sleep for
|
|
115
|
+
// every forge, so the bound the operator configured is the bound every arm obeys.
|
|
116
|
+
const retryOpts = { windowMs: cfg.identityRetryWindowMs, log, now, sleep };
|
|
72
117
|
|
|
73
118
|
// The GitHub arm, when the deployment actually serves GitHub -- now conditional, exactly like the three
|
|
74
119
|
// sibling arms below (issue #99). It was unconditional, and since GITHUB_AUTH_SOURCE defaults to `gh` and
|
|
@@ -84,16 +129,23 @@ export async function startReceiver(
|
|
|
84
129
|
let selfId;
|
|
85
130
|
let resolveAuthority;
|
|
86
131
|
if (cfg.servesGithub) {
|
|
87
|
-
// HARD-FAIL identity resolution -- NO try/catch. A throw
|
|
88
|
-
//
|
|
89
|
-
// run, so refusing to boot is the
|
|
132
|
+
// HARD-FAIL identity resolution -- still NO try/catch here. A DETERMINATE throw (absent/bad github
|
|
133
|
+
// auth, an unresolvable id, anything tagged piDispatchConfig) propagates same-tick and the server
|
|
134
|
+
// below is never created: without selfId the bot-loop guard cannot run, so refusing to boot is the
|
|
135
|
+
// only safe outcome. What changed (issue #318) is the TRANSIENT case: retryIdentity keeps
|
|
136
|
+
// re-resolving inside cfg.identityRetryWindowMs before letting the throw propagate, so a forge that
|
|
137
|
+
// is merely restarting no longer costs the deployment its receiver (the ~25 seconds the systemd
|
|
138
|
+
// unit's start limit used to bound recovery at, and the nothing launchd and nssm bounded it at).
|
|
139
|
+
// The retry wraps EACH arm, never the whole boot: a whole-body retry would rebuild the queue and
|
|
140
|
+
// router below once per attempt and leak their connections, and hoisting the four resolutions above
|
|
141
|
+
// the queue would reorder the first failure an operator sees.
|
|
90
142
|
//
|
|
91
143
|
// The WHOLE auth object is kept, not just selfId: the closer resolver below mints its per-delivery
|
|
92
144
|
// metadata-read token through this same object (issue #231), so identity and mint capability stay
|
|
93
145
|
// one credential decision -- an arm that resolved its identity is exactly the arm that can answer
|
|
94
146
|
// a permission question. This is also why the github handler's missing-resolver 503 is unreachable
|
|
95
147
|
// in a wired receiver: a boot that fails here mounts no `/` at all.
|
|
96
|
-
const auth = await makeAuth(cfg.github);
|
|
148
|
+
const auth = await retryIdentity(() => makeAuth(cfg.github), { forge: "github", ...retryOpts });
|
|
97
149
|
selfId = auth.selfId;
|
|
98
150
|
log({ event: "self_identity", id: selfId, source: cfg.github.source });
|
|
99
151
|
// The lookup token asks the mint to narrow to metadata:read -- the App path honors it GitHub-side,
|
|
@@ -109,77 +161,155 @@ export async function startReceiver(
|
|
|
109
161
|
|
|
110
162
|
// Ride-out connection (no failFast): the receiver is long-running and should survive a Valkey
|
|
111
163
|
// restart, not give up on a transient disconnect.
|
|
164
|
+
// Issue #464 (gate round 3): a refused Valkey stops the receiver here (a configError, exit 2, as the worker's boot),
|
|
165
|
+
// and one that answers nothing for 20 s ends it with exit 1 so the service manager starts it again. It used to keep
|
|
166
|
+
// running and listening with a queue that could never connect.
|
|
167
|
+
await judgeValkey(cfg.valkeyUrl, { env, now, sleep });
|
|
112
168
|
const queue = makeQueueFn(parseConnection(cfg.valkeyUrl));
|
|
113
169
|
|
|
114
|
-
//
|
|
115
|
-
//
|
|
116
|
-
//
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
mode: cfg.gitlab.mode,
|
|
123
|
-
secret: cfg.gitlab.secret,
|
|
124
|
-
selfId: gitlabSelfId,
|
|
125
|
-
resolveAuthority: makeResolveAuthorityFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }),
|
|
126
|
-
};
|
|
127
|
-
}
|
|
170
|
+
// Multi-host routing for deliveries that bind a host-local resource (issue #57, `OQ-032`). A trigger
|
|
171
|
+
// naming `run.secretsProfile` or a `run.waitFor` profile can only run where that profile is declared, and
|
|
172
|
+
// which worker pops a shared-queue job is a coin flip -- so the same trigger succeeded or failed by
|
|
173
|
+
// chance, permanently, and read like a configuration error rather than a placement one.
|
|
174
|
+
//
|
|
175
|
+
// Every failure path inside the router returns the shared queue, so a receiver whose Valkey read fails,
|
|
176
|
+
// or whose deployment has no named hosts, behaves exactly as it did before this existed.
|
|
177
|
+
const router = makeForgeRouterFn({ valkeyUrl: cfg.valkeyUrl, shared: queue, log });
|
|
128
178
|
|
|
129
|
-
//
|
|
130
|
-
//
|
|
131
|
-
//
|
|
132
|
-
//
|
|
133
|
-
//
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
179
|
+
// EVERY refusal below this line lands after the queue and the router exist, so it must stop
|
|
180
|
+
// what the boot built (issue #299's rule, the receiver's copy; measured for #318 on main AND
|
|
181
|
+
// this branch, against a live Valkey): with the shared clients left open, the entry assigns
|
|
182
|
+
// process.exitCode and the loop never drains -- the failure line printed at 0.4s and the
|
|
183
|
+
// process was still alive at 90s. Under Type=simple that reads as "active (running)" with
|
|
184
|
+
// nothing listening, Restart= never fires, RestartPreventExitStatus=2 is never consulted, and
|
|
185
|
+
// the retry window's exit-1-into-a-fresh-window contract above never engages. The github arm
|
|
186
|
+
// and the config load sit ABOVE the queue and always exited cleanly; these three arms did not.
|
|
187
|
+
try {
|
|
188
|
+
// The GitLab arm, when configured. Its identity resolution is HARD-FAIL for the same reason github's
|
|
189
|
+
// is: without a selfId the bot-loop guard cannot run, and a receiver that listens without it turns the
|
|
190
|
+
// harness's own status comment into another paid job.
|
|
191
|
+
let gitlab = null;
|
|
192
|
+
if (cfg.gitlab) {
|
|
193
|
+
const gitlabSelfId = await retryIdentity(() => resolveSelfIdFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }), { forge: "gitlab", ...retryOpts });
|
|
194
|
+
log({ event: "self_identity", forge: "gitlab", id: gitlabSelfId, mode: cfg.gitlab.mode });
|
|
195
|
+
gitlab = {
|
|
196
|
+
mode: cfg.gitlab.mode,
|
|
197
|
+
secret: cfg.gitlab.secret,
|
|
198
|
+
selfId: gitlabSelfId,
|
|
199
|
+
resolveAuthority: makeResolveAuthorityFn({ apiUrl: cfg.gitlab.apiUrl, token: cfg.gitlab.token }),
|
|
200
|
+
};
|
|
201
|
+
}
|
|
144
202
|
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
203
|
+
// The Forgejo arm, when configured. Identity resolution is HARD-FAIL here too, and it is the arm where
|
|
204
|
+
// that matters most: a repo-scoped Forgejo token cannot call GET /user, so an operator who follows the
|
|
205
|
+
// scoping advice without setting FORGEJO_BOT_ID lands exactly here -- and a receiver that shrugged and
|
|
206
|
+
// continued would run with selfId undefined, which never equals a sender id and silently turns the
|
|
207
|
+
// harness's own comments into more paid jobs.
|
|
208
|
+
let forgejo = null;
|
|
209
|
+
if (cfg.forgejo) {
|
|
210
|
+
const forgejoSelfId = await retryIdentity(() => resolveForgejoSelfIdFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token, botId: cfg.forgejo.botId }), { forge: "forgejo", ...retryOpts });
|
|
211
|
+
log({ event: "self_identity", forge: "forgejo", id: forgejoSelfId, source: cfg.forgejo.botId ? "FORGEJO_BOT_ID" : "api" });
|
|
212
|
+
forgejo = {
|
|
213
|
+
secret: cfg.forgejo.secret,
|
|
214
|
+
selfId: forgejoSelfId,
|
|
215
|
+
resolveAuthority: makeResolveForgejoAuthorityFn({ apiUrl: cfg.forgejo.apiUrl, token: cfg.forgejo.token }),
|
|
216
|
+
};
|
|
217
|
+
}
|
|
160
218
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
process.exit(0);
|
|
177
|
-
};
|
|
178
|
-
process.once("SIGTERM", () => void shutdown("SIGTERM"));
|
|
179
|
-
process.once("SIGINT", () => void shutdown("SIGINT"));
|
|
180
|
-
}
|
|
219
|
+
// The Azure arm, when configured. Identity resolution is HARD-FAIL here too, and it resolves BOTH forms
|
|
220
|
+
// of the harness's identity in one call: a pull-request delivery names an actor by GUID and a work item
|
|
221
|
+
// names them only by email address, so a guard that knew one form would be blind on half the events.
|
|
222
|
+
let azure = null;
|
|
223
|
+
if (cfg.azure) {
|
|
224
|
+
const azureSelfId = await retryIdentity(() => resolveAzureSelfIdFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token }), { forge: "azure", ...retryOpts });
|
|
225
|
+
log({ event: "self_identity", forge: "azure", id: azureSelfId.id, hasAccountName: azureSelfId.email !== null, mode: cfg.azure.mode });
|
|
226
|
+
azure = {
|
|
227
|
+
mode: cfg.azure.mode,
|
|
228
|
+
secret: cfg.azure.secret,
|
|
229
|
+
headerName: cfg.azure.headerName,
|
|
230
|
+
selfId: azureSelfId,
|
|
231
|
+
resolveAuthority: makeResolveAzureAuthorityFn({ orgUrl: cfg.azure.orgUrl, token: cfg.azure.token }),
|
|
232
|
+
};
|
|
233
|
+
}
|
|
181
234
|
|
|
182
|
-
|
|
235
|
+
const handler = makeReceiver({ queue, router, selfId, cfg, log, gitlab, forgejo, azure, resolveAuthority });
|
|
236
|
+
const server = createServer(handler);
|
|
237
|
+
server.listen(cfg.port, cfg.bind, () =>
|
|
238
|
+
log({ event: "receiver_started", port: cfg.port, bind: cfg.bind, valkey: cfg.valkeyUrl }),
|
|
239
|
+
);
|
|
240
|
+
|
|
241
|
+
// The watch arms UNCONDITIONALLY now (issue #301). Armed only on the real entry, the lifecycle defect
|
|
242
|
+
// was muted under test rather than closed, and this file had no coverage that the watch arms at all --
|
|
243
|
+
// `DES-WATCHERS-CLOSE-WITH-THE-WORKER` rejected exactly that posture for the worker. The closer rides
|
|
244
|
+
// the injected `closers` array, so a test drains what its boot armed and the real shutdown closes it.
|
|
245
|
+
//
|
|
246
|
+
// LAST FALLIBLE STEP, deliberately, and it must stay last: every refusal this boot can produce -- the
|
|
247
|
+
// config load, each hard-fail identity resolution, the router build, even a throwing `listen` -- sits
|
|
248
|
+
// ABOVE this line, so a refused boot has armed nothing and there is never a closer with no one left to
|
|
249
|
+
// drain it. The worker states the same invariant where its watchers arm. A step added BELOW that can
|
|
250
|
+
// throw reopens issue #301 on the refusal path; the HARD-FAIL test pins the refusals that exist today.
|
|
251
|
+
closers.push(watchTriggers(env, cfg, log, triggersAtBoot));
|
|
252
|
+
|
|
253
|
+
// Graceful shutdown only on the real entry (default createServer). Under test injection the fakes are
|
|
254
|
+
// per-test, so a process-wide SIGNAL HANDLER would still leak across tests -- and unlike the watch it
|
|
255
|
+
// has no seam to ride: the closers array cannot un-register a `process.once`. The shutdown's own steps
|
|
256
|
+
// are one call per handle, each covered through its seam.
|
|
257
|
+
if (createServer === http.createServer) {
|
|
258
|
+
const shutdown = async (signal) => {
|
|
259
|
+
log({ event: "receiver_stopping", signal });
|
|
260
|
+
await new Promise((resolve) => server.close(resolve));
|
|
261
|
+
await queue.close();
|
|
262
|
+
await router.close();
|
|
263
|
+
// The watch closer, and whatever joins it later. Per-item try, because a throw here would strand
|
|
264
|
+
// the `process.exit(0)` that the unit's stop depends on -- `index.mjs`'s closer loop states the
|
|
265
|
+
// same rule for the worker.
|
|
266
|
+
for (const c of closers) {
|
|
267
|
+
try {
|
|
268
|
+
c?.close?.();
|
|
269
|
+
} catch {
|
|
270
|
+
// A closer that failed has already stopped mattering.
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
process.exit(0);
|
|
274
|
+
};
|
|
275
|
+
process.once("SIGTERM", () => void shutdown("SIGTERM"));
|
|
276
|
+
process.once("SIGINT", () => void shutdown("SIGINT"));
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
return server;
|
|
280
|
+
} catch (err) {
|
|
281
|
+
// Best-effort, per handle, and the refusal rethrows unchanged: the cleanup must never
|
|
282
|
+
// replace the story. BullMQ's close() is idempotent and the router closes its own pool.
|
|
283
|
+
//
|
|
284
|
+
// The error swallow first, and it is measured, not defensive: a queue built moments ago can
|
|
285
|
+
// still be mid-handshake, and every client error in that window is forwarded to the Queue
|
|
286
|
+
// object, where makeQueue attaches no listener -- unlistened, one such event crashed the
|
|
287
|
+
// process at exit 1 and stomped the exit 2 this catch exists to deliver. Scoped to the
|
|
288
|
+
// refusal path: the serving path's error surface is unchanged.
|
|
289
|
+
queue.on?.("error", () => {});
|
|
290
|
+
try {
|
|
291
|
+
// The RAW client, deliberately NOT queue.close(): close() on a maybe-still-connecting
|
|
292
|
+
// client strips every listener in RedisConnection.close()'s finally (removeAllListeners),
|
|
293
|
+
// and the constructor's `initializing.catch` then re-emits the flushed handshake as an
|
|
294
|
+
// 'error' on a LISTENERLESS emitter -- an uncaught crash whose exit 1 stomped this exit
|
|
295
|
+
// code on 3 of 5 measured refusals, and a settle-first close() variant still crashed
|
|
296
|
+
// against a DEAD Valkey. A direct disconnect() leaves the connection's listeners alone,
|
|
297
|
+
// so the same late rejection flows connection -> queue -> the swallow above, and ioredis
|
|
298
|
+
// disconnect() releases the socket AND stops the ride-out reconnects, in every client
|
|
299
|
+
// state (the #300 measurement). `_client` is a BullMQ private, reached on the host-pi
|
|
300
|
+
// precedent for pinned internals: the real-BullMQ integration test is the pin, so a bump
|
|
301
|
+
// that moves it goes red there instead of sliding silently back to the wedge.
|
|
302
|
+
queue.connection?._client?.disconnect?.();
|
|
303
|
+
} catch {
|
|
304
|
+
// A handle that failed to close has stopped mattering to a process about to exit.
|
|
305
|
+
}
|
|
306
|
+
try {
|
|
307
|
+
await router.close();
|
|
308
|
+
} catch {
|
|
309
|
+
// Same.
|
|
310
|
+
}
|
|
311
|
+
throw err;
|
|
312
|
+
}
|
|
183
313
|
}
|
|
184
314
|
|
|
185
315
|
/**
|
|
@@ -187,31 +317,60 @@ export async function startReceiver(
|
|
|
187
317
|
* admin writes with, which swaps the inode a file-watch would lose), debounce, and re-read on change. A bad
|
|
188
318
|
* edit keeps the running triggers (reloadTriggers never throws) and logs a kept-old notice. Best-effort: a
|
|
189
319
|
* platform without `fs.watch` logs and the receiver simply keeps its boot-time triggers.
|
|
320
|
+
*
|
|
321
|
+
* Returns the worker's stop handle (issue #301), registered in `closers` so the shutdown -- or the test
|
|
322
|
+
* that injected the array -- closes the FSWatcher and cancels the debounce it armed. A closer is returned
|
|
323
|
+
* even when the watch could not arm: closing a never-armed handle is a no-op by construction, and a
|
|
324
|
+
* caller that has to ask "did I get one" is how a handle goes unregistered.
|
|
190
325
|
*/
|
|
191
|
-
|
|
326
|
+
// `triggersAtBoot` has NO DEFAULT on purpose (issue #386). A default of `null` would silently DISABLE the
|
|
327
|
+
// boot-race check for a caller that forgot it, which is the failure this whole change exists to stop; with
|
|
328
|
+
// none, forgetting it is `undefined`, and `readBeforeArming` reads that as "no boot read to hand over" and
|
|
329
|
+
// says so by falling back to its own read rather than to nothing.
|
|
330
|
+
function watchTriggers(env, cfg, log, triggersAtBoot) {
|
|
192
331
|
const path = triggersFilePath(env);
|
|
193
332
|
const dir = dirname(path) || ".";
|
|
194
333
|
const file = basename(path);
|
|
195
|
-
|
|
334
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
335
|
+
// The worker's closer, reused rather than re-derived (issue #301). Its `log` takes `(event, fields)`
|
|
336
|
+
// where this file's takes one object, so it is handed an adapter here instead of the module growing a
|
|
337
|
+
// second signature. The reload lines go through `closer.reloadLog` -- the reload's VOICE, gated once
|
|
338
|
+
// the handle closes; the arming lines keep the real `log`, because they run before any close exists.
|
|
339
|
+
const closer = makeWatchCloser(handles, (event, fields) => log({ event, ...fields }));
|
|
340
|
+
const readFile = () => readFileSync(path, "utf8");
|
|
341
|
+
const reload = () => {
|
|
342
|
+
const res = reloadTriggers(env, cfg);
|
|
343
|
+
if (res.ok) closer.reloadLog("triggers_reloaded");
|
|
344
|
+
else closer.reloadLog("triggers_reload_invalid", { reason: res.invalid, kept: true });
|
|
345
|
+
};
|
|
346
|
+
// The baseline is the BOOT LOAD's own read, handed in (issue #386). Comparing against a read taken here
|
|
347
|
+
// would measure the arming instead of the window: this service arms LAST, behind identity resolution that
|
|
348
|
+
// retries for up to RECEIVER_IDENTITY_RETRY_SECONDS, and that is the window an edit is lost in.
|
|
349
|
+
readBeforeArming(handles, readFile, triggersAtBoot);
|
|
196
350
|
try {
|
|
197
|
-
watch(dir, (_event, changed) => {
|
|
351
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
352
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
198
353
|
if (changed && changed !== file) return; // only our file (null changed name -> reload to be safe)
|
|
199
|
-
clearTimeout(timer);
|
|
200
|
-
timer = setTimeout(
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
else log({ event: "triggers_reload_invalid", reason: res.invalid, kept: true });
|
|
204
|
-
}, 150);
|
|
205
|
-
}).unref?.();
|
|
354
|
+
clearTimeout(handles.timer);
|
|
355
|
+
handles.timer = setTimeout(reload, WATCH_DEBOUNCE_MS);
|
|
356
|
+
});
|
|
357
|
+
handles.watcher.unref?.();
|
|
206
358
|
log({ event: "triggers_watching", path });
|
|
207
359
|
} catch (err) {
|
|
208
360
|
log({ event: "triggers_watch_unavailable", reason: err?.message });
|
|
209
361
|
}
|
|
362
|
+
if (changedWhileArming(handles, readFile)) {
|
|
363
|
+
log({ event: "triggers_reread_after_arming", path });
|
|
364
|
+
reload();
|
|
365
|
+
}
|
|
366
|
+
return closer;
|
|
210
367
|
}
|
|
211
368
|
|
|
212
369
|
// Entry point when run directly (main: src/start.mjs, no bin). Kept out of startReceiver so tests call
|
|
213
370
|
// it directly. The error line carries only `err.message` -- never a secret or PII.
|
|
214
|
-
if (import.meta.url
|
|
371
|
+
if (isEntryModule(import.meta.url)) {
|
|
372
|
+
// An unhandled rejection is printed as its message alone (PR #475's review), never Node's print of the whole reason.
|
|
373
|
+
installRejectionPrinter();
|
|
215
374
|
startReceiver(process.env).catch((err) => {
|
|
216
375
|
process.stderr.write(`${JSON.stringify({ event: "receiver_start_failed", reason: err?.message })}\n`);
|
|
217
376
|
// entryExitCode, NOT a bare 1. This file is what `receiver.service` execs -- cli.mjs is not on that
|