@edgehero/pi-dispatch-receiver 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch-receiver",
3
- "version": "1.4.0",
3
+ "version": "1.5.0",
4
4
  "type": "module",
5
5
  "description": "Webhook receiver for pi-dispatch: the always-on edge that verifies GitHub, GitLab, Forgejo and Azure DevOps deliveries and enqueues (at most) one job per event for the worker.",
6
6
  "keywords": [
package/src/receiver.mjs CHANGED
@@ -134,7 +134,11 @@ export function parseSubset(payload) {
134
134
  * undefined, i.e. never drops the harness's own comments. Absent property therefore means no route; the
135
135
  * failure of a forgotten property is a 404 an operator sees, never a paid recursion they get billed for.
136
136
  */
137
- export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority }) {
137
+ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority, router = null }) {
138
+ // Which queue a delivery is enqueued onto (issue #57, `OQ-032`). The default is the shared queue for
139
+ // EVERY job, which is what this receiver did before multi-host existed -- so a deployment that wires no
140
+ // router, and every test that constructs one without it, is byte-identical.
141
+ const routeTo = router ? (kind, job) => router.queueFor(kind, job) : () => queue;
138
142
  // Built only when the deployment serves GitHub. Construction is not free of the secret either: the
139
143
  // `new Webhooks({ secret })` inside makeVerifiedHandler throws "options.secret required" on an absent
140
144
  // one, so not building the arm is what lets a github-free deployment legitimately have no secret --
@@ -143,7 +147,7 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
143
147
  // `resolveAuthority` is the GITHUB closer resolver (issue #231), riding top-level beside `selfId`
144
148
  // because github's dependencies always have -- the other forges bundle theirs in per-forge objects.
145
149
  // Optional, because only a close delivery an armed close rule matches ever consults it.
146
- const github = cfg.servesGithub ? makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) : null;
150
+ const github = cfg.servesGithub ? makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) : null;
147
151
 
148
152
  // A TABLE, built once, rather than one `if` per forge. Two forges made that a single branch; four make
149
153
  // it a chain, and a chain is where one arm quietly ends up checked after the fallthrough. A path present
@@ -151,9 +155,9 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
151
155
  // falls through to GitHub, which is what keeps `/` working -- and a configured-off GitHub answers the
152
156
  // same 404 from the fallthrough itself, below.
153
157
  const routes = {
154
- "/gitlab": gitlab ? makeGitLabHandler({ queue, cfg, log, ...gitlab }) : null,
155
- "/forgejo": forgejo ? makeForgejoHandler({ queue, cfg, log, ...forgejo }) : null,
156
- "/azure": azure ? makeAzureHandler({ queue, cfg, log, ...azure }) : null,
158
+ "/gitlab": gitlab ? makeGitLabHandler({ queue, routeTo, cfg, log, ...gitlab }) : null,
159
+ "/forgejo": forgejo ? makeForgejoHandler({ queue, routeTo, cfg, log, ...forgejo }) : null,
160
+ "/azure": azure ? makeAzureHandler({ queue, routeTo, cfg, log, ...azure }) : null,
157
161
  };
158
162
 
159
163
  return async function receiverHandler(req, res) {
@@ -223,7 +227,7 @@ async function fanout(job, enqueue) {
223
227
  * uses, so a lookup is never spent on a delivery the gate then ignores -- every label, comment, PR and
224
228
  * review delivery, and every close nothing wants, stays payload-only and byte-identical to before.
225
229
  */
226
- function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
230
+ function makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) {
227
231
  return makeVerifiedHandler({ secret: cfg.webhookSecret }, async ({ rawBody, event, delivery }, res) => {
228
232
  let subset;
229
233
  try {
@@ -264,7 +268,7 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
264
268
  // Fanout (REQ-REPLICA-RUNS) lives in `fanout` above; the 202/503 decision stays here, where it always was.
265
269
  let replicas;
266
270
  try {
267
- replicas = await fanout(result.job, (j) => enqueueGitHubJob(queue, j));
271
+ replicas = await fanout(result.job, async (j) => await enqueueGitHubJob(await routeTo("github", j), j));
268
272
  } catch (err) {
269
273
  // Own try/catch so a Valkey-down enqueue is a 503 (retryable), not verify's outer 500.
270
274
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
@@ -289,7 +293,7 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
289
293
  * The lookup runs only for events that could still fire -- after verification, and after the payload has
290
294
  * been projected -- so an unauthenticated flood cannot make this project call GitLab at all.
291
295
  */
292
- function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
296
+ function makeGitLabHandler({ routeTo, queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
293
297
  return makeGitLabVerifiedHandler({ mode, secret, ...(now ? { now } : {}) }, async ({ rawBody, delivery }, res) => {
294
298
  let subset;
295
299
  try {
@@ -314,7 +318,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
314
318
 
315
319
  let replicas;
316
320
  try {
317
- replicas = await fanout(result.job, (j) => enqueueGitLabJob(queue, j));
321
+ replicas = await fanout(result.job, async (j) => await enqueueGitLabJob(await routeTo("gitlab", j), j));
318
322
  } catch (err) {
319
323
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
320
324
  return respond(res, 503, { error: "enqueue-failed" }); // GitLab redelivers; dedup by webhook-id coalesces
@@ -341,7 +345,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
341
345
  * per construction, and serving two sources from one would mean either sharing a secret between forges or
342
346
  * letting the request choose which one it was checked against.
343
347
  */
344
- function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority }) {
348
+ function makeForgejoHandler({ routeTo, queue, cfg, log, secret, selfId, resolveAuthority }) {
345
349
  return makeVerifiedHandler({ secret }, async ({ rawBody, event, delivery }, res) => {
346
350
  let subset;
347
351
  try {
@@ -366,7 +370,7 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
366
370
 
367
371
  let replicas;
368
372
  try {
369
- replicas = await fanout(result.job, (j) => enqueueForgeJob(queue, "forgejo", j));
373
+ replicas = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("forgejo", j), "forgejo", j));
370
374
  } catch (err) {
371
375
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
372
376
  return respond(res, 503, { error: "enqueue-failed" }); // Forgejo redelivers; dedup by GUID coalesces
@@ -391,7 +395,7 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
391
395
  * `"Display Name <email>"`. Both the resolver and the bot-loop guard handle both forms -- see
392
396
  * filter-azure.mjs, where the ordering constraint is stated in full.
393
397
  */
394
- function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
398
+ function makeAzureHandler({ routeTo, queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
395
399
  return makeAzureVerifiedHandler({ mode, secret, headerName }, async ({ rawBody }, res) => {
396
400
  let subset;
397
401
  try {
@@ -424,7 +428,7 @@ function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, r
424
428
 
425
429
  let replicas;
426
430
  try {
427
- replicas = await fanout(result.job, (j) => enqueueForgeJob(queue, "azure", j));
431
+ replicas = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("azure", j), "azure", j));
428
432
  } catch (err) {
429
433
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
430
434
  return respond(res, 503, { error: "enqueue-failed" });
package/src/route.mjs ADDED
@@ -0,0 +1,126 @@
1
+ /**
2
+ * Which queue a forge delivery is enqueued onto, when the deployment runs on more than one machine
3
+ * (issue #57, `OQ-032`).
4
+ *
5
+ * The receiver is the only process that can make this decision. Routing has to happen at ENQUEUE -- a
6
+ * delayed job is promoted on each worker's own clock, so the fastest clock wins every attempt and a job
7
+ * cannot reliably be handed from a host that will not serve it to one that will -- and forge deliveries are
8
+ * enqueued here. `capabilities.mjs` owns the RULE and is shared with the worker so the two cannot drift;
9
+ * this module owns the plumbing: one cached registry read, and a pool of queue handles.
10
+ *
11
+ * EVERYTHING HERE FAILS OPEN ONTO THE SHARED QUEUE, which is what this receiver did before any of it
12
+ * existed. An unreadable registry, a timed-out read, a malformed row, a name that is not routable: every
13
+ * one of them returns the shared queue, so the worst this can do is fail to improve on a coin flip. A
14
+ * webhook handler is the wrong place to invent a new way to drop work.
15
+ */
16
+
17
+ import { readLiveHosts } from "@edgehero/pi-dispatch/host-registry";
18
+ import { makeQueue, hostQueueName } from "@edgehero/pi-dispatch/queue";
19
+ import { parseConnection, makeRedisClient } from "@edgehero/pi-dispatch/connection";
20
+ import { forgeDeliveryJobId } from "@edgehero/pi-dispatch/job-id";
21
+ import { jobNeeds, routeForgeJob } from "@edgehero/pi-dispatch/capabilities";
22
+
23
+ /**
24
+ * How long a registry read is reused.
25
+ *
26
+ * The registry beats every 15s, so anything below that mostly re-reads a value that cannot have changed.
27
+ * Five seconds is well inside one beat and bounds the extra load a busy receiver puts on Valkey at one read
28
+ * per five seconds rather than one per delivery -- and a delivery burst, which is exactly when this must
29
+ * not add latency, is served entirely from cache.
30
+ *
31
+ * Staleness costs nothing that freshness would have saved: a host that appears within the window is missed
32
+ * and its work goes to the shared queue, which is today's behaviour, and a host that vanishes is already
33
+ * guarded by `ROUTE_FRESH_MS` on the row's own beat timestamp.
34
+ */
35
+ const HOSTS_TTL_MS = 5_000;
36
+
37
+ /** Bound on the registry read itself. A routing hint is never worth making a webhook wait. */
38
+ const READ_TIMEOUT_MS = 1_500;
39
+
40
+ /**
41
+ * @returns {{queueFor: (kind: string, job: object) => Promise<object>, close: () => Promise<void>}}
42
+ */
43
+ export function makeForgeRouter({
44
+ valkeyUrl,
45
+ shared,
46
+ log = () => {},
47
+ makeQueueFn = makeQueue,
48
+ parseConnectionFn = parseConnection,
49
+ redisFn = makeRedisClient,
50
+ readLiveHostsFn = readLiveHosts,
51
+ now = () => Date.now(),
52
+ ttlMs = HOSTS_TTL_MS,
53
+ } = {}) {
54
+ const pool = new Map(); // queue name -> Queue, opened lazily and only for hosts actually routed to
55
+ let redis = null;
56
+ let cached = { at: -Infinity, hosts: [] };
57
+ let inFlight = null;
58
+
59
+ // One read at a time. Without this a burst of deliveries arriving on a cold cache each start their own
60
+ // registry read, which is the moment the deployment can least afford N of them.
61
+ const hosts = async () => {
62
+ if (now() - cached.at < ttlMs) return cached.hosts;
63
+ if (inFlight) return await inFlight;
64
+ inFlight = (async () => {
65
+ try {
66
+ if (!redis) {
67
+ redis = redisFn(valkeyUrl);
68
+ // A receiver must never print ioredis reconnect noise into its own log: this client is an
69
+ // optimisation, and its failures are already handled by falling back to the shared queue.
70
+ redis.on?.("error", () => {});
71
+ }
72
+ const res = await readLiveHostsFn(redis, { timeoutMs: READ_TIMEOUT_MS });
73
+ cached = { at: now(), hosts: Array.isArray(res?.hosts) ? res.hosts : [] };
74
+ } catch {
75
+ // Cache the FAILURE too, so an unreachable Valkey costs one read per window rather than one
76
+ // per delivery. The empty list routes everything to the shared queue.
77
+ cached = { at: now(), hosts: [] };
78
+ } finally {
79
+ inFlight = null;
80
+ }
81
+ return cached.hosts;
82
+ })();
83
+ return await inFlight;
84
+ };
85
+
86
+ return {
87
+ async queueFor(kind, job) {
88
+ const needs = jobNeeds(job);
89
+ // The overwhelming majority of deliveries bind neither a secret profile nor a wait profile, and
90
+ // they must not pay even a cache lookup for a decision that cannot apply to them.
91
+ if (needs.length === 0) return shared;
92
+
93
+ let name = null;
94
+ try {
95
+ name = routeForgeJob({ hosts: await hosts(), needs, jobId: forgeDeliveryJobId(kind, job?.trigger?.deliveryId, job?.replica) });
96
+ } catch {
97
+ return shared; // an unroutable id, a bad row: the shared queue is always a correct answer
98
+ }
99
+ if (!name) return shared;
100
+
101
+ const queueName = hostQueueName(name);
102
+ if (!pool.has(queueName)) pool.set(queueName, makeQueueFn(parseConnectionFn(valkeyUrl), { name: queueName }));
103
+ // Logged because a routed job is the one case where "which host ran it" was DECIDED rather than
104
+ // observed, and an operator debugging a job that never started needs to know it was sent
105
+ // somewhere specific. Host names are deployment topology: this reaches the log, never a forge
106
+ // comment (`triggers.mjs` sets that rule for refusal messages and it holds here).
107
+ log({ event: "forge_job_routed", kind, host: name, needs: needs.join(",") });
108
+ return pool.get(queueName);
109
+ },
110
+ async close() {
111
+ for (const q of pool.values()) {
112
+ try {
113
+ await q.close();
114
+ } catch {
115
+ // best-effort teardown
116
+ }
117
+ }
118
+ pool.clear();
119
+ try {
120
+ redis?.disconnect?.();
121
+ } catch {
122
+ // best-effort teardown
123
+ }
124
+ },
125
+ };
126
+ }
package/src/start.mjs CHANGED
@@ -44,6 +44,7 @@ import { makeResolveForgejoAuthority } from "./forgejo-members.mjs";
44
44
  import { makeResolveAzureAuthority } from "./azure-members.mjs";
45
45
  import { makeResolveGitHubAuthority } from "./github-members.mjs";
46
46
  import { makeQueue } from "@edgehero/pi-dispatch/queue";
47
+ import { makeForgeRouter } from "./route.mjs";
47
48
  import { parseConnection } from "@edgehero/pi-dispatch/connection";
48
49
 
49
50
  /**
@@ -55,6 +56,7 @@ export async function startReceiver(
55
56
  {
56
57
  makeAuth = makeGitHubAuth,
57
58
  makeQueueFn = makeQueue,
59
+ makeForgeRouterFn = makeForgeRouter,
58
60
  createServer = http.createServer,
59
61
  resolveGitLabSelfId: resolveSelfIdFn = resolveGitLabSelfId,
60
62
  makeResolveAuthority: makeResolveAuthorityFn = makeResolveAuthority,
@@ -111,6 +113,15 @@ export async function startReceiver(
111
113
  // restart, not give up on a transient disconnect.
112
114
  const queue = makeQueueFn(parseConnection(cfg.valkeyUrl));
113
115
 
116
+ // Multi-host routing for deliveries that bind a host-local resource (issue #57, `OQ-032`). A trigger
117
+ // naming `run.secretsProfile` or a `run.waitFor` profile can only run where that profile is declared, and
118
+ // which worker pops a shared-queue job is a coin flip -- so the same trigger succeeded or failed by
119
+ // chance, permanently, and read like a configuration error rather than a placement one.
120
+ //
121
+ // Every failure path inside the router returns the shared queue, so a receiver whose Valkey read fails,
122
+ // or whose deployment has no named hosts, behaves exactly as it did before this existed.
123
+ const router = makeForgeRouterFn({ valkeyUrl: cfg.valkeyUrl, shared: queue, log });
124
+
114
125
  // The GitLab arm, when configured. Its identity resolution is HARD-FAIL for the same reason github's
115
126
  // is: without a selfId the bot-loop guard cannot run, and a receiver that listens without it turns the
116
127
  // harness's own status comment into another paid job.
@@ -158,7 +169,7 @@ export async function startReceiver(
158
169
  };
159
170
  }
160
171
 
161
- const handler = makeReceiver({ queue, selfId, cfg, log, gitlab, forgejo, azure, resolveAuthority });
172
+ const handler = makeReceiver({ queue, router, selfId, cfg, log, gitlab, forgejo, azure, resolveAuthority });
162
173
  const server = createServer(handler);
163
174
  server.listen(cfg.port, cfg.bind, () =>
164
175
  log({ event: "receiver_started", port: cfg.port, bind: cfg.bind, valkey: cfg.valkeyUrl }),
@@ -173,6 +184,7 @@ export async function startReceiver(
173
184
  log({ event: "receiver_stopping", signal });
174
185
  await new Promise((resolve) => server.close(resolve));
175
186
  await queue.close();
187
+ await router.close();
176
188
  process.exit(0);
177
189
  };
178
190
  process.once("SIGTERM", () => void shutdown("SIGTERM"));