@edgehero/pi-dispatch-receiver 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/poller.mjs CHANGED
@@ -108,13 +108,14 @@
108
108
  import { createSign } from "node:crypto";
109
109
  import { readFile as fsReadFile } from "node:fs/promises";
110
110
  import { configError } from "@edgehero/pi-dispatch/config";
111
- import { parseConnection } from "@edgehero/pi-dispatch/connection";
111
+ import { judgeValkeyAtStart, makeRedisClient, parseConnection, valkeyClientContext } from "@edgehero/pi-dispatch/connection";
112
112
  import { makeGitHubAuth } from "@edgehero/pi-dispatch/get-token";
113
113
  import { enqueueGitHubJob, makeQueue } from "@edgehero/pi-dispatch/queue";
114
114
  import { filter, hasCloseTriggers, wantsCloserAuthority } from "./filter.mjs";
115
115
  import { makeResolveGitHubAuthority } from "./github-members.mjs";
116
116
  import { parseSubset } from "./receiver.mjs";
117
117
  import { loadPollerConfig } from "./poller-config.mjs";
118
+ import { retryIdentity } from "./boot-retry.mjs";
118
119
 
119
120
  const API_URL = "https://api.github.com";
120
121
  // 35 days: strictly outlives the 31-day gh-* job retention -- see the header's cursor/TTL section.
@@ -153,9 +154,10 @@ class RateLimited extends Error {
153
154
  * Boot the poller: config, HARD-FAIL identity, repo set, then the cycle loop. Collaborators are
154
155
  * injected with real defaults (start.mjs's convention), so the whole producer is testable offline
155
156
  * with no GitHub, no Valkey, and no timers:
156
- * { fetchFn, redis, queueFn, out, now, random, sleep, selfIdFn, tokenFn, fsDeps }
157
+ * { fetchFn, redis, queueFn, out, now, random, sleep, signals, selfIdFn, tokenFn, fsDeps }
157
158
  * Returns `{ stop, done }`: `stop()` ends the loop after the in-flight work; `done` resolves once
158
- * owned connections are closed.
159
+ * owned connections are closed, with the inter-cycle default-sleep timer cleared by then (issue
160
+ * #325) -- after `done`, no timer armed by the poller remains.
159
161
  */
160
162
  export async function startPoller(env = process.env, deps = {}) {
161
163
  const {
@@ -164,6 +166,12 @@ export async function startPoller(env = process.env, deps = {}) {
164
166
  now = Date.now,
165
167
  random = Math.random,
166
168
  sleep,
169
+ // The signal gate, decoupled from the sleep default (issue #325): `sleep === undefined` used to
170
+ // double as the "real run" flag, which made the REAL default timer unarmable under test without
171
+ // also registering process-wide SIGTERM/SIGINT handlers -- the exact leak the gate exists to
172
+ // prevent. A test that arms the real sleep passes `signals: false`; the one production caller
173
+ // (cli.mjs, no deps) keeps handlers exactly as before through this default.
174
+ signals = sleep === undefined,
167
175
  redis,
168
176
  queueFn,
169
177
  selfIdFn,
@@ -172,7 +180,11 @@ export async function startPoller(env = process.env, deps = {}) {
172
180
  fsDeps = {},
173
181
  makeAuth = makeGitHubAuth,
174
182
  makeQueueFn = makeQueue,
183
+ // Issue #464 (gate round 3): how the poller builds its own Valkey client; a seam, so a test sees none built.
184
+ makeRedisFn = makeRedisClient,
175
185
  makeResolveGitHubAuthority: makeResolveGitHubAuthorityFn = makeResolveGitHubAuthority,
186
+ // Issue #464 (gate round 3): the start-time judgement of VALKEY_URL, as the receiver's (start.mjs).
187
+ judgeValkey = (url, opts) => judgeValkeyAtStart(url, valkeyClientContext({ env: opts.env }), { now: opts.now, ...(opts.sleep ? { sleep: opts.sleep } : {}) }),
176
188
  } = deps;
177
189
 
178
190
  const cfg = loadPollerConfig(env, fsDeps);
@@ -181,9 +193,21 @@ export async function startPoller(env = process.env, deps = {}) {
181
193
  // the bot-loop guard's sole input, and a poller running without it would read the harness's own
182
194
  // completion comment next cycle and enqueue it -- an unbounded paid recursion, only slower than the
183
195
  // webhook version. No try/catch: an unresolvable identity must prevent the loop from ever starting.
196
+ // A TRANSIENT failure retries in-process under the same window serve uses (issue #318): a
197
+ // pure-polling deployment loses work to a stopped process just as surely as a webhook one.
198
+ //
199
+ // The default sleep sits above the gate because the gate sleeps too (issue #318); the boot retry
200
+ // simply awaits its cancellable promise to completion, so `cancel` is the LOOP's concern alone.
201
+ const sleepFn = sleep ?? cancellableSleep;
184
202
  let auth = null;
203
+ // `auth ??= await ...` assigns only when the mint RESOLVES, so a retried getAuth re-invokes
204
+ // makeAuth. Caching the PROMISE instead would memoize the first failure and turn the retry loop
205
+ // into a rethrow spinner -- pinned by the poller's retry test.
185
206
  const getAuth = async () => (auth ??= await makeAuth(cfg.github));
186
- const selfId = selfIdFn ? await selfIdFn(cfg.github) : (await getAuth()).selfId;
207
+ const selfId = await retryIdentity(
208
+ () => (selfIdFn ? selfIdFn(cfg.github) : getAuth().then((a) => a.selfId)),
209
+ { forge: "github", windowMs: cfg.identityRetryWindowMs, log: out, now, sleep: sleepFn },
210
+ );
187
211
  out({ event: "self_identity", id: selfId, source: cfg.github.source });
188
212
 
189
213
  // The polling credential comes from the same auth config the worker validates. pat/gh hand back
@@ -210,9 +234,12 @@ export async function startPoller(env = process.env, deps = {}) {
210
234
  // queue connection: a long-running producer should survive a Valkey restart.
211
235
  let redisClient = redis ?? null;
212
236
  let ownRedis = false;
237
+ // Issue #464 (gate round 3): judged before this process builds its own Valkey clients: a refusal exits 2, nothing
238
+ // answering for 20 s exits 1, never a poller running on a queue that cannot connect.
239
+ if (redisClient === null || queueFn === undefined || queueFn === null) await judgeValkey(cfg.valkeyUrl, { env, now, sleep });
213
240
  if (redisClient === null) {
214
- const { default: Redis } = await import("ioredis");
215
- redisClient = new Redis(cfg.valkeyUrl);
241
+ // Through connection.mjs, as every Valkey client of this project (issue #464): it judges and pins the address.
242
+ redisClient = makeRedisFn(cfg.valkeyUrl);
216
243
  ownRedis = true;
217
244
  }
218
245
 
@@ -263,7 +290,6 @@ export async function startPoller(env = process.env, deps = {}) {
263
290
  stopped = true;
264
291
  wake();
265
292
  };
266
- const sleepFn = sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
267
293
 
268
294
  const done = (async () => {
269
295
  let cycleNo = 0;
@@ -328,14 +354,27 @@ export async function startPoller(env = process.env, deps = {}) {
328
354
  // in one synchronized stampede).
329
355
  let delayMs = Math.max(cfg.intervalSeconds, stats.minDelaySeconds) * 1000;
330
356
  if (rateResetMs !== null) delayMs = Math.max(delayMs, rateResetMs - now() + jitterMs(random));
331
- await Promise.race([sleepFn(delayMs), stopWaker]);
357
+ // Hold the sleep so the LOSER can be cleared: when stop() wins this race, an uncancelled
358
+ // default timer stays armed for up to a full poll interval after `done` resolves (issue
359
+ // #325) -- masked in production by the signal handler's process.exit(0) below, and real for
360
+ // every caller that stops the poller without exiting. Optional-chained because injected
361
+ // test sleeps are plain promises carrying no cancel; their call count and arguments are
362
+ // untouched, and a rejecting one still propagates through the finally.
363
+ const nap = sleepFn(delayMs);
364
+ try {
365
+ await Promise.race([nap, stopWaker]);
366
+ } finally {
367
+ nap.cancel?.();
368
+ }
332
369
  }
333
370
  await closeOwned();
334
371
  })();
335
372
 
336
- // Signal handlers only on a real run (no injected sleep) -- start.mjs's rule, same reason: under
337
- // test injection the fakes are per-test, and a process-wide handler would leak across tests.
338
- if (sleep === undefined) {
373
+ // Signal handlers only on a real run (`signals`, defaulting from the sleep seam) -- start.mjs's
374
+ // rule, same reason: under test injection the fakes are per-test, and a process-wide handler would
375
+ // leak across tests. Its own seam since issue #325, so a test can arm the REAL default sleep
376
+ // without inheriting the handlers.
377
+ if (signals) {
339
378
  const shutdown = async (signal) => {
340
379
  out({ event: "poller_stopping", signal });
341
380
  stop();
@@ -358,6 +397,27 @@ function jitterMs(random) {
358
397
  return 1_000 + Math.floor(random() * 29_000);
359
398
  }
360
399
 
400
+ /**
401
+ * The default inter-cycle sleep: a REF'D setTimeout whose promise carries its own `cancel` (issue
402
+ * #325). Ref'd deliberately -- the poller IS its process's main loop and holding the loop through the
403
+ * delay is the point (`DES-RETENTION-SWEEPS-ON-A-TIMER`'s rejected list records the contrast; between
404
+ * cycles this timer is the only thing keeping a pure-poll process alive, cli.mjs's "awaiting done").
405
+ * The defect was never the ref, only survival past stop(): "unref'd is not cleaned up" has a ref'd
406
+ * twin, a cleared-nothing timer that holds the loop for up to a full interval after `done` resolved.
407
+ * A cancelled sleep never resolves, which is safe here because its only awaiter is a race `stopWaker`
408
+ * has already settled, and cancel on an already-fired timer is a no-op, so the winning side's clear
409
+ * costs nothing. EXPORTED for `settleWithin`'s reason (worker/src/start.mjs): the cleared-timer
410
+ * guarantee deserves a deterministic async_hooks pin on the helper, not a census of a full boot.
411
+ */
412
+ export function cancellableSleep(ms) {
413
+ let timer;
414
+ const p = new Promise((resolve) => {
415
+ timer = setTimeout(resolve, ms);
416
+ });
417
+ p.cancel = () => clearTimeout(timer);
418
+ return p;
419
+ }
420
+
361
421
  /** The redis key family for one repo -- see the schema table in the module header. */
362
422
  function keyNames(repo) {
363
423
  const p = `poll:${repo}`;
@@ -916,11 +976,20 @@ async function gate(ctx, eventName, payload, deliveryId, stats) {
916
976
  return;
917
977
  }
918
978
  const replicas = result.job.replicas ?? 1;
979
+ let created = 0;
919
980
  for (let i = 1; i <= replicas; i++) {
920
- await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
981
+ const r = await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
982
+ // Issue #289, the receiver arms' honesty rule at the poller's own seam: a semantic-window swallow
983
+ // gets its own line with the surviving id, `enqueued` counts only what was CREATED, and a bare-id
984
+ // return from an enqueue seam predating the shape counts as created -- the old behaviour exactly.
985
+ if (r && typeof r === "object" && r.deduplicated === true) {
986
+ ctx.out({ event: "deduplicated", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, jobId: r.jobId, survivingJobId: r.survivingJobId });
987
+ } else {
988
+ created += 1;
989
+ }
921
990
  }
922
- stats.enqueued += replicas;
923
- ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
991
+ stats.enqueued += created;
992
+ if (created > 0) ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas: created });
924
993
  }
925
994
 
926
995
  /**
package/src/receiver.mjs CHANGED
@@ -134,7 +134,11 @@ export function parseSubset(payload) {
134
134
  * undefined, i.e. never drops the harness's own comments. Absent property therefore means no route; the
135
135
  * failure of a forgotten property is a 404 an operator sees, never a paid recursion they get billed for.
136
136
  */
137
- export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority }) {
137
+ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority, router = null }) {
138
+ // Which queue a delivery is enqueued onto (issue #57, `OQ-032`). The default is the shared queue for
139
+ // EVERY job, which is what this receiver did before multi-host existed -- so a deployment that wires no
140
+ // router, and every test that constructs one without it, is byte-identical.
141
+ const routeTo = router ? (kind, job) => router.queueFor(kind, job) : () => queue;
138
142
  // Built only when the deployment serves GitHub. Construction is not free of the secret either: the
139
143
  // `new Webhooks({ secret })` inside makeVerifiedHandler throws "options.secret required" on an absent
140
144
  // one, so not building the arm is what lets a github-free deployment legitimately have no secret --
@@ -143,7 +147,7 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
143
147
  // `resolveAuthority` is the GITHUB closer resolver (issue #231), riding top-level beside `selfId`
144
148
  // because github's dependencies always have -- the other forges bundle theirs in per-forge objects.
145
149
  // Optional, because only a close delivery an armed close rule matches ever consults it.
146
- const github = cfg.servesGithub ? makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) : null;
150
+ const github = cfg.servesGithub ? makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) : null;
147
151
 
148
152
  // A TABLE, built once, rather than one `if` per forge. Two forges made that a single branch; four make
149
153
  // it a chain, and a chain is where one arm quietly ends up checked after the fallthrough. A path present
@@ -151,9 +155,9 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
151
155
  // falls through to GitHub, which is what keeps `/` working -- and a configured-off GitHub answers the
152
156
  // same 404 from the fallthrough itself, below.
153
157
  const routes = {
154
- "/gitlab": gitlab ? makeGitLabHandler({ queue, cfg, log, ...gitlab }) : null,
155
- "/forgejo": forgejo ? makeForgejoHandler({ queue, cfg, log, ...forgejo }) : null,
156
- "/azure": azure ? makeAzureHandler({ queue, cfg, log, ...azure }) : null,
158
+ "/gitlab": gitlab ? makeGitLabHandler({ queue, routeTo, cfg, log, ...gitlab }) : null,
159
+ "/forgejo": forgejo ? makeForgejoHandler({ queue, routeTo, cfg, log, ...forgejo }) : null,
160
+ "/azure": azure ? makeAzureHandler({ queue, routeTo, cfg, log, ...azure }) : null,
157
161
  };
158
162
 
159
163
  return async function receiverHandler(req, res) {
@@ -201,14 +205,43 @@ function pathOf(url) {
201
205
  * `enqueue` is a callback because the four arms spell their enqueue differently (a named github/gitlab
202
206
  * wrapper, or `enqueueForgeJob` with an explicit kind); the fanout itself is forge-blind.
203
207
  *
204
- * @returns {Promise<number>} how many jobs were enqueued, for the caller's `enqueued` log line.
208
+ * @returns {Promise<{replicas: number, created: number, deduplicated: Array<{jobId, survivingJobId}>}>}
209
+ * what actually happened, per replica (issue #289): `created` counts jobs that now EXIST because of
210
+ * this delivery, and `deduplicated` carries each swallow the semantic window made, with the id that
211
+ * survived. A bare-id return from an enqueue that predates the shape counts as created -- the old
212
+ * behaviour exactly.
205
213
  */
206
214
  async function fanout(job, enqueue) {
207
215
  const replicas = job.replicas ?? 1;
216
+ let created = 0;
217
+ const deduplicated = [];
208
218
  for (let i = 1; i <= replicas; i++) {
209
- await enqueue(replicas > 1 ? { ...job, replica: i } : job);
219
+ const r = await enqueue(replicas > 1 ? { ...job, replica: i } : job);
220
+ if (r && typeof r === "object" && r.deduplicated === true) deduplicated.push({ jobId: r.jobId, survivingJobId: r.survivingJobId });
221
+ else created += 1;
210
222
  }
211
- return replicas;
223
+ return { replicas, created, deduplicated };
224
+ }
225
+
226
+ /**
227
+ * One log-and-answer for a fanout outcome, shared by all four arms so a weakened copy cannot hide (the
228
+ * four-copies doctrine above). Issue #289's honesty rule: `enqueued` is logged only when something was
229
+ * CREATED, with `replicas` meaning jobs that now exist; every semantic-window swallow gets its own
230
+ * `deduplicated` line naming the surviving job's id (every field forge- or worker-minted, no payload
231
+ * text); and a delivery that created NOTHING answers `202 {status:"deduplicated"}` -- still a 2xx,
232
+ * because a non-2xx would trigger a redelivery storm for a delivery that was handled, but no longer
233
+ * "queued", because a receiver that logs the swallow while answering success on the wire would be the
234
+ * honest-log/lying-wire split this project refuses. Each arm passes its own target grammar.
235
+ */
236
+ function respondEnqueueOutcome({ log, res, delivery, repo, target, flow, outcome }) {
237
+ for (const d of outcome.deduplicated) {
238
+ log?.({ event: "deduplicated", delivery, repo, target, flow, jobId: d.jobId, survivingJobId: d.survivingJobId });
239
+ }
240
+ if (outcome.created > 0) {
241
+ log?.({ event: "enqueued", delivery, repo, target, flow, replicas: outcome.created });
242
+ return respond(res, 202, { status: "queued" });
243
+ }
244
+ return respond(res, 202, { status: "deduplicated" });
212
245
  }
213
246
 
214
247
  /**
@@ -223,7 +256,7 @@ async function fanout(job, enqueue) {
223
256
  * uses, so a lookup is never spent on a delivery the gate then ignores -- every label, comment, PR and
224
257
  * review delivery, and every close nothing wants, stays payload-only and byte-identical to before.
225
258
  */
226
- function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
259
+ function makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) {
227
260
  return makeVerifiedHandler({ secret: cfg.webhookSecret }, async ({ rawBody, event, delivery }, res) => {
228
261
  let subset;
229
262
  try {
@@ -262,17 +295,16 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
262
295
  }
263
296
 
264
297
  // Fanout (REQ-REPLICA-RUNS) lives in `fanout` above; the 202/503 decision stays here, where it always was.
265
- let replicas;
298
+ let outcome;
266
299
  try {
267
- replicas = await fanout(result.job, (j) => enqueueGitHubJob(queue, j));
300
+ outcome = await fanout(result.job, async (j) => await enqueueGitHubJob(await routeTo("github", j), j));
268
301
  } catch (err) {
269
302
  // Own try/catch so a Valkey-down enqueue is a 503 (retryable), not verify's outer 500.
270
303
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
271
304
  return respond(res, 503, { error: "enqueue-failed" }); // GitHub redelivers; dedup by GUID coalesces
272
305
  }
273
306
 
274
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
275
- return respond(res, 202, { status: "queued" });
307
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
276
308
  });
277
309
  }
278
310
 
@@ -289,7 +321,7 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
289
321
  * The lookup runs only for events that could still fire -- after verification, and after the payload has
290
322
  * been projected -- so an unauthenticated flood cannot make this project call GitLab at all.
291
323
  */
292
- function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
324
+ function makeGitLabHandler({ routeTo, queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
293
325
  return makeGitLabVerifiedHandler({ mode, secret, ...(now ? { now } : {}) }, async ({ rawBody, delivery }, res) => {
294
326
  let subset;
295
327
  try {
@@ -312,9 +344,9 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
312
344
  return respond(res, 204);
313
345
  }
314
346
 
315
- let replicas;
347
+ let outcome;
316
348
  try {
317
- replicas = await fanout(result.job, (j) => enqueueGitLabJob(queue, j));
349
+ outcome = await fanout(result.job, async (j) => await enqueueGitLabJob(await routeTo("gitlab", j), j));
318
350
  } catch (err) {
319
351
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
320
352
  return respond(res, 503, { error: "enqueue-failed" }); // GitLab redelivers; dedup by webhook-id coalesces
@@ -323,8 +355,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
323
355
  // `!` for a merge request, `#` for an issue -- GitLab's own notation, and the same discrimination
324
356
  // the semantic dedup key makes, because the two are separate number sequences.
325
357
  const sep = result.job.target.type === "pull_request" ? "!" : "#";
326
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, replicas });
327
- return respond(res, 202, { status: "queued" });
358
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
328
359
  });
329
360
  }
330
361
 
@@ -341,7 +372,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
341
372
  * per construction, and serving two sources from one would mean either sharing a secret between forges or
342
373
  * letting the request choose which one it was checked against.
343
374
  */
344
- function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority }) {
375
+ function makeForgejoHandler({ routeTo, queue, cfg, log, secret, selfId, resolveAuthority }) {
345
376
  return makeVerifiedHandler({ secret }, async ({ rawBody, event, delivery }, res) => {
346
377
  let subset;
347
378
  try {
@@ -364,16 +395,15 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
364
395
  return respond(res, 204);
365
396
  }
366
397
 
367
- let replicas;
398
+ let outcome;
368
399
  try {
369
- replicas = await fanout(result.job, (j) => enqueueForgeJob(queue, "forgejo", j));
400
+ outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("forgejo", j), "forgejo", j));
370
401
  } catch (err) {
371
402
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
372
403
  return respond(res, 503, { error: "enqueue-failed" }); // Forgejo redelivers; dedup by GUID coalesces
373
404
  }
374
405
 
375
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
376
- return respond(res, 202, { status: "queued" });
406
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
377
407
  });
378
408
  }
379
409
 
@@ -391,7 +421,7 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
391
421
  * `"Display Name <email>"`. Both the resolver and the bot-loop guard handle both forms -- see
392
422
  * filter-azure.mjs, where the ordering constraint is stated in full.
393
423
  */
394
- function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
424
+ function makeAzureHandler({ routeTo, queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
395
425
  return makeAzureVerifiedHandler({ mode, secret, headerName }, async ({ rawBody }, res) => {
396
426
  let subset;
397
427
  try {
@@ -422,9 +452,9 @@ function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, r
422
452
  return respond(res, 204);
423
453
  }
424
454
 
425
- let replicas;
455
+ let outcome;
426
456
  try {
427
- replicas = await fanout(result.job, (j) => enqueueForgeJob(queue, "azure", j));
457
+ outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("azure", j), "azure", j));
428
458
  } catch (err) {
429
459
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
430
460
  return respond(res, 503, { error: "enqueue-failed" });
@@ -433,7 +463,6 @@ function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, r
433
463
  // `!` for a pull request, `#` for a work item -- Azure numbers them separately, and this is the same
434
464
  // discrimination the semantic dedup key makes.
435
465
  const sep = result.job.target.type === "pull_request" ? "!" : "#";
436
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, replicas });
437
- return respond(res, 202, { status: "queued" });
466
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
438
467
  });
439
468
  }
package/src/route.mjs ADDED
@@ -0,0 +1,126 @@
1
+ /**
2
+ * Which queue a forge delivery is enqueued onto, when the deployment runs on more than one machine
3
+ * (issue #57, `OQ-032`).
4
+ *
5
+ * The receiver is the only process that can make this decision. Routing has to happen at ENQUEUE -- a
6
+ * delayed job is promoted on each worker's own clock, so the fastest clock wins every attempt and a job
7
+ * cannot reliably be handed from a host that will not serve it to one that will -- and forge deliveries are
8
+ * enqueued here. `capabilities.mjs` owns the RULE and is shared with the worker so the two cannot drift;
9
+ * this module owns the plumbing: one cached registry read, and a pool of queue handles.
10
+ *
11
+ * EVERYTHING HERE FAILS OPEN ONTO THE SHARED QUEUE, which is what this receiver did before any of it
12
+ * existed. An unreadable registry, a timed-out read, a malformed row, a name that is not routable: every
13
+ * one of them returns the shared queue, so the worst this can do is fail to improve on a coin flip. A
14
+ * webhook handler is the wrong place to invent a new way to drop work.
15
+ */
16
+
17
+ import { readLiveHosts } from "@edgehero/pi-dispatch/host-registry";
18
+ import { makeQueue, hostQueueName } from "@edgehero/pi-dispatch/queue";
19
+ import { parseConnection, makeRedisClient } from "@edgehero/pi-dispatch/connection";
20
+ import { forgeDeliveryJobId } from "@edgehero/pi-dispatch/job-id";
21
+ import { jobNeeds, routeForgeJob } from "@edgehero/pi-dispatch/capabilities";
22
+
23
+ /**
24
+ * How long a registry read is reused.
25
+ *
26
+ * The registry beats every 15s, so anything below that mostly re-reads a value that cannot have changed.
27
+ * Five seconds is well inside one beat and bounds the extra load a busy receiver puts on Valkey at one read
28
+ * per five seconds rather than one per delivery -- and a delivery burst, which is exactly when this must
29
+ * not add latency, is served entirely from cache.
30
+ *
31
+ * Staleness costs nothing that freshness would have saved: a host that appears within the window is missed
32
+ * and its work goes to the shared queue, which is today's behaviour, and a host that vanishes is already
33
+ * guarded by `ROUTE_FRESH_MS` on the row's own beat timestamp.
34
+ */
35
+ const HOSTS_TTL_MS = 5_000;
36
+
37
+ /** Bound on the registry read itself. A routing hint is never worth making a webhook wait. */
38
+ const READ_TIMEOUT_MS = 1_500;
39
+
40
+ /**
41
+ * @returns {{queueFor: (kind: string, job: object) => Promise<object>, close: () => Promise<void>}}
42
+ */
43
+ export function makeForgeRouter({
44
+ valkeyUrl,
45
+ shared,
46
+ log = () => {},
47
+ makeQueueFn = makeQueue,
48
+ parseConnectionFn = parseConnection,
49
+ redisFn = makeRedisClient,
50
+ readLiveHostsFn = readLiveHosts,
51
+ now = () => Date.now(),
52
+ ttlMs = HOSTS_TTL_MS,
53
+ } = {}) {
54
+ const pool = new Map(); // queue name -> Queue, opened lazily and only for hosts actually routed to
55
+ let redis = null;
56
+ let cached = { at: -Infinity, hosts: [] };
57
+ let inFlight = null;
58
+
59
+ // One read at a time. Without this a burst of deliveries arriving on a cold cache each start their own
60
+ // registry read, which is the moment the deployment can least afford N of them.
61
+ const hosts = async () => {
62
+ if (now() - cached.at < ttlMs) return cached.hosts;
63
+ if (inFlight) return await inFlight;
64
+ inFlight = (async () => {
65
+ try {
66
+ if (!redis) {
67
+ redis = redisFn(valkeyUrl);
68
+ // A receiver must never print ioredis reconnect noise into its own log: this client is an
69
+ // optimisation, and its failures are already handled by falling back to the shared queue.
70
+ redis.on?.("error", () => {});
71
+ }
72
+ const res = await readLiveHostsFn(redis, { timeoutMs: READ_TIMEOUT_MS });
73
+ cached = { at: now(), hosts: Array.isArray(res?.hosts) ? res.hosts : [] };
74
+ } catch {
75
+ // Cache the FAILURE too, so an unreachable Valkey costs one read per window rather than one
76
+ // per delivery. The empty list routes everything to the shared queue.
77
+ cached = { at: now(), hosts: [] };
78
+ } finally {
79
+ inFlight = null;
80
+ }
81
+ return cached.hosts;
82
+ })();
83
+ return await inFlight;
84
+ };
85
+
86
+ return {
87
+ async queueFor(kind, job) {
88
+ const needs = jobNeeds(job);
89
+ // The overwhelming majority of deliveries bind neither a secret profile nor a wait profile, and
90
+ // they must not pay even a cache lookup for a decision that cannot apply to them.
91
+ if (needs.length === 0) return shared;
92
+
93
+ let name = null;
94
+ try {
95
+ name = routeForgeJob({ hosts: await hosts(), needs, jobId: forgeDeliveryJobId(kind, job?.trigger?.deliveryId, job?.replica) });
96
+ } catch {
97
+ return shared; // an unroutable id, a bad row: the shared queue is always a correct answer
98
+ }
99
+ if (!name) return shared;
100
+
101
+ const queueName = hostQueueName(name);
102
+ if (!pool.has(queueName)) pool.set(queueName, makeQueueFn(parseConnectionFn(valkeyUrl), { name: queueName }));
103
+ // Logged because a routed job is the one case where "which host ran it" was DECIDED rather than
104
+ // observed, and an operator debugging a job that never started needs to know it was sent
105
+ // somewhere specific. Host names are deployment topology: this reaches the log, never a forge
106
+ // comment (`triggers.mjs` sets that rule for refusal messages and it holds here).
107
+ log({ event: "forge_job_routed", kind, host: name, needs: needs.join(",") });
108
+ return pool.get(queueName);
109
+ },
110
+ async close() {
111
+ for (const q of pool.values()) {
112
+ try {
113
+ await q.close();
114
+ } catch {
115
+ // best-effort teardown
116
+ }
117
+ }
118
+ pool.clear();
119
+ try {
120
+ redis?.disconnect?.();
121
+ } catch {
122
+ // best-effort teardown
123
+ }
124
+ },
125
+ };
126
+ }