@edgehero/pi-dispatch-receiver 1.5.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/poller.mjs CHANGED
@@ -108,13 +108,14 @@
108
108
  import { createSign } from "node:crypto";
109
109
  import { readFile as fsReadFile } from "node:fs/promises";
110
110
  import { configError } from "@edgehero/pi-dispatch/config";
111
- import { parseConnection } from "@edgehero/pi-dispatch/connection";
111
+ import { judgeValkeyAtStart, makeRedisClient, parseConnection, valkeyClientContext } from "@edgehero/pi-dispatch/connection";
112
112
  import { makeGitHubAuth } from "@edgehero/pi-dispatch/get-token";
113
113
  import { enqueueGitHubJob, makeQueue } from "@edgehero/pi-dispatch/queue";
114
114
  import { filter, hasCloseTriggers, wantsCloserAuthority } from "./filter.mjs";
115
115
  import { makeResolveGitHubAuthority } from "./github-members.mjs";
116
116
  import { parseSubset } from "./receiver.mjs";
117
117
  import { loadPollerConfig } from "./poller-config.mjs";
118
+ import { retryIdentity } from "./boot-retry.mjs";
118
119
 
119
120
  const API_URL = "https://api.github.com";
120
121
  // 35 days: strictly outlives the 31-day gh-* job retention -- see the header's cursor/TTL section.
@@ -153,9 +154,10 @@ class RateLimited extends Error {
153
154
  * Boot the poller: config, HARD-FAIL identity, repo set, then the cycle loop. Collaborators are
154
155
  * injected with real defaults (start.mjs's convention), so the whole producer is testable offline
155
156
  * with no GitHub, no Valkey, and no timers:
156
- * { fetchFn, redis, queueFn, out, now, random, sleep, selfIdFn, tokenFn, fsDeps }
157
+ * { fetchFn, redis, queueFn, out, now, random, sleep, signals, selfIdFn, tokenFn, fsDeps }
157
158
  * Returns `{ stop, done }`: `stop()` ends the loop after the in-flight work; `done` resolves once
158
- * owned connections are closed.
159
+ * owned connections are closed, with the inter-cycle default-sleep timer cleared by then (issue
160
+ * #325) -- after `done`, no timer armed by the poller remains.
159
161
  */
160
162
  export async function startPoller(env = process.env, deps = {}) {
161
163
  const {
@@ -164,6 +166,12 @@ export async function startPoller(env = process.env, deps = {}) {
164
166
  now = Date.now,
165
167
  random = Math.random,
166
168
  sleep,
169
+ // The signal gate, decoupled from the sleep default (issue #325): `sleep === undefined` used to
170
+ // double as the "real run" flag, which made the REAL default timer unarmable under test without
171
+ // also registering process-wide SIGTERM/SIGINT handlers -- the exact leak the gate exists to
172
+ // prevent. A test that arms the real sleep passes `signals: false`; the one production caller
173
+ // (cli.mjs, no deps) keeps handlers exactly as before through this default.
174
+ signals = sleep === undefined,
167
175
  redis,
168
176
  queueFn,
169
177
  selfIdFn,
@@ -172,7 +180,11 @@ export async function startPoller(env = process.env, deps = {}) {
172
180
  fsDeps = {},
173
181
  makeAuth = makeGitHubAuth,
174
182
  makeQueueFn = makeQueue,
183
+ // Issue #464 (gate round 3): how the poller builds its own Valkey client; a seam, so a test sees none built.
184
+ makeRedisFn = makeRedisClient,
175
185
  makeResolveGitHubAuthority: makeResolveGitHubAuthorityFn = makeResolveGitHubAuthority,
186
+ // Issue #464 (gate round 3): the start-time judgement of VALKEY_URL, as the receiver's (start.mjs).
187
+ judgeValkey = (url, opts) => judgeValkeyAtStart(url, valkeyClientContext({ env: opts.env }), { now: opts.now, ...(opts.sleep ? { sleep: opts.sleep } : {}) }),
176
188
  } = deps;
177
189
 
178
190
  const cfg = loadPollerConfig(env, fsDeps);
@@ -181,9 +193,21 @@ export async function startPoller(env = process.env, deps = {}) {
181
193
  // the bot-loop guard's sole input, and a poller running without it would read the harness's own
182
194
  // completion comment next cycle and enqueue it -- an unbounded paid recursion, only slower than the
183
195
  // webhook version. No try/catch: an unresolvable identity must prevent the loop from ever starting.
196
+ // A TRANSIENT failure retries in-process under the same window serve uses (issue #318): a
197
+ // pure-polling deployment loses work to a stopped process just as surely as a webhook one.
198
+ //
199
+ // The default sleep sits above the gate because the gate sleeps too (issue #318); the boot retry
200
+ // simply awaits its cancellable promise to completion, so `cancel` is the LOOP's concern alone.
201
+ const sleepFn = sleep ?? cancellableSleep;
184
202
  let auth = null;
203
+ // `auth ??= await ...` assigns only when the mint RESOLVES, so a retried getAuth re-invokes
204
+ // makeAuth. Caching the PROMISE instead would memoize the first failure and turn the retry loop
205
+ // into a rethrow spinner -- pinned by the poller's retry test.
185
206
  const getAuth = async () => (auth ??= await makeAuth(cfg.github));
186
- const selfId = selfIdFn ? await selfIdFn(cfg.github) : (await getAuth()).selfId;
207
+ const selfId = await retryIdentity(
208
+ () => (selfIdFn ? selfIdFn(cfg.github) : getAuth().then((a) => a.selfId)),
209
+ { forge: "github", windowMs: cfg.identityRetryWindowMs, log: out, now, sleep: sleepFn },
210
+ );
187
211
  out({ event: "self_identity", id: selfId, source: cfg.github.source });
188
212
 
189
213
  // The polling credential comes from the same auth config the worker validates. pat/gh hand back
@@ -210,9 +234,12 @@ export async function startPoller(env = process.env, deps = {}) {
210
234
  // queue connection: a long-running producer should survive a Valkey restart.
211
235
  let redisClient = redis ?? null;
212
236
  let ownRedis = false;
237
+ // Issue #464 (gate round 3): judged before this process builds its own Valkey clients: a refusal exits 2, nothing
238
+ // answering for 20 s exits 1, never a poller running on a queue that cannot connect.
239
+ if (redisClient === null || queueFn === undefined || queueFn === null) await judgeValkey(cfg.valkeyUrl, { env, now, sleep });
213
240
  if (redisClient === null) {
214
- const { default: Redis } = await import("ioredis");
215
- redisClient = new Redis(cfg.valkeyUrl);
241
+ // Through connection.mjs, as every Valkey client of this project (issue #464): it judges and pins the address.
242
+ redisClient = makeRedisFn(cfg.valkeyUrl);
216
243
  ownRedis = true;
217
244
  }
218
245
 
@@ -263,7 +290,6 @@ export async function startPoller(env = process.env, deps = {}) {
263
290
  stopped = true;
264
291
  wake();
265
292
  };
266
- const sleepFn = sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
267
293
 
268
294
  const done = (async () => {
269
295
  let cycleNo = 0;
@@ -328,14 +354,27 @@ export async function startPoller(env = process.env, deps = {}) {
328
354
  // in one synchronized stampede).
329
355
  let delayMs = Math.max(cfg.intervalSeconds, stats.minDelaySeconds) * 1000;
330
356
  if (rateResetMs !== null) delayMs = Math.max(delayMs, rateResetMs - now() + jitterMs(random));
331
- await Promise.race([sleepFn(delayMs), stopWaker]);
357
+ // Hold the sleep so the LOSER can be cleared: when stop() wins this race, an uncancelled
358
+ // default timer stays armed for up to a full poll interval after `done` resolves (issue
359
+ // #325) -- masked in production by the signal handler's process.exit(0) below, and real for
360
+ // every caller that stops the poller without exiting. Optional-chained because injected
361
+ // test sleeps are plain promises carrying no cancel; their call count and arguments are
362
+ // untouched, and a rejecting one still propagates through the finally.
363
+ const nap = sleepFn(delayMs);
364
+ try {
365
+ await Promise.race([nap, stopWaker]);
366
+ } finally {
367
+ nap.cancel?.();
368
+ }
332
369
  }
333
370
  await closeOwned();
334
371
  })();
335
372
 
336
- // Signal handlers only on a real run (no injected sleep) -- start.mjs's rule, same reason: under
337
- // test injection the fakes are per-test, and a process-wide handler would leak across tests.
338
- if (sleep === undefined) {
373
+ // Signal handlers only on a real run (`signals`, defaulting from the sleep seam) -- start.mjs's
374
+ // rule, same reason: under test injection the fakes are per-test, and a process-wide handler would
375
+ // leak across tests. Its own seam since issue #325, so a test can arm the REAL default sleep
376
+ // without inheriting the handlers.
377
+ if (signals) {
339
378
  const shutdown = async (signal) => {
340
379
  out({ event: "poller_stopping", signal });
341
380
  stop();
@@ -358,6 +397,27 @@ function jitterMs(random) {
358
397
  return 1_000 + Math.floor(random() * 29_000);
359
398
  }
360
399
 
400
+ /**
401
+ * The default inter-cycle sleep: a REF'D setTimeout whose promise carries its own `cancel` (issue
402
+ * #325). Ref'd deliberately -- the poller IS its process's main loop and holding the loop through the
403
+ * delay is the point (`DES-RETENTION-SWEEPS-ON-A-TIMER`'s rejected list records the contrast; between
404
+ * cycles this timer is the only thing keeping a pure-poll process alive, cli.mjs's "awaiting done").
405
+ * The defect was never the ref, only survival past stop(): "unref'd is not cleaned up" has a ref'd
406
+ * twin, a cleared-nothing timer that holds the loop for up to a full interval after `done` resolved.
407
+ * A cancelled sleep never resolves, which is safe here because its only awaiter is a race `stopWaker`
408
+ * has already settled, and cancel on an already-fired timer is a no-op, so the winning side's clear
409
+ * costs nothing. EXPORTED for `settleWithin`'s reason (worker/src/start.mjs): the cleared-timer
410
+ * guarantee deserves a deterministic async_hooks pin on the helper, not a census of a full boot.
411
+ */
412
+ export function cancellableSleep(ms) {
413
+ let timer;
414
+ const p = new Promise((resolve) => {
415
+ timer = setTimeout(resolve, ms);
416
+ });
417
+ p.cancel = () => clearTimeout(timer);
418
+ return p;
419
+ }
420
+
361
421
  /** The redis key family for one repo -- see the schema table in the module header. */
362
422
  function keyNames(repo) {
363
423
  const p = `poll:${repo}`;
@@ -916,11 +976,20 @@ async function gate(ctx, eventName, payload, deliveryId, stats) {
916
976
  return;
917
977
  }
918
978
  const replicas = result.job.replicas ?? 1;
979
+ let created = 0;
919
980
  for (let i = 1; i <= replicas; i++) {
920
- await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
981
+ const r = await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
982
+ // Issue #289, the receiver arms' honesty rule at the poller's own seam: a semantic-window swallow
983
+ // gets its own line with the surviving id, `enqueued` counts only what was CREATED, and a bare-id
984
+ // return from an enqueue seam predating the shape counts as created -- the old behaviour exactly.
985
+ if (r && typeof r === "object" && r.deduplicated === true) {
986
+ ctx.out({ event: "deduplicated", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, jobId: r.jobId, survivingJobId: r.survivingJobId });
987
+ } else {
988
+ created += 1;
989
+ }
921
990
  }
922
- stats.enqueued += replicas;
923
- ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
991
+ stats.enqueued += created;
992
+ if (created > 0) ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas: created });
924
993
  }
925
994
 
926
995
  /**
package/src/receiver.mjs CHANGED
@@ -205,14 +205,43 @@ function pathOf(url) {
205
205
  * `enqueue` is a callback because the four arms spell their enqueue differently (a named github/gitlab
206
206
  * wrapper, or `enqueueForgeJob` with an explicit kind); the fanout itself is forge-blind.
207
207
  *
208
- * @returns {Promise<number>} how many jobs were enqueued, for the caller's `enqueued` log line.
208
+ * @returns {Promise<{replicas: number, created: number, deduplicated: Array<{jobId, survivingJobId}>}>}
209
+ * what actually happened, per replica (issue #289): `created` counts jobs that now EXIST because of
210
+ * this delivery, and `deduplicated` carries each swallow the semantic window made, with the id that
211
+ * survived. A bare-id return from an enqueue that predates the shape counts as created -- the old
212
+ * behaviour exactly.
209
213
  */
210
214
  async function fanout(job, enqueue) {
211
215
  const replicas = job.replicas ?? 1;
216
+ let created = 0;
217
+ const deduplicated = [];
212
218
  for (let i = 1; i <= replicas; i++) {
213
- await enqueue(replicas > 1 ? { ...job, replica: i } : job);
219
+ const r = await enqueue(replicas > 1 ? { ...job, replica: i } : job);
220
+ if (r && typeof r === "object" && r.deduplicated === true) deduplicated.push({ jobId: r.jobId, survivingJobId: r.survivingJobId });
221
+ else created += 1;
214
222
  }
215
- return replicas;
223
+ return { replicas, created, deduplicated };
224
+ }
225
+
226
+ /**
227
+ * One log-and-answer for a fanout outcome, shared by all four arms so a weakened copy cannot hide (the
228
+ * four-copies doctrine above). Issue #289's honesty rule: `enqueued` is logged only when something was
229
+ * CREATED, with `replicas` meaning jobs that now exist; every semantic-window swallow gets its own
230
+ * `deduplicated` line naming the surviving job's id (every field forge- or worker-minted, no payload
231
+ * text); and a delivery that created NOTHING answers `202 {status:"deduplicated"}` -- still a 2xx,
232
+ * because a non-2xx would trigger a redelivery storm for a delivery that was handled, but no longer
233
+ * "queued", because a receiver that logs the swallow while answering success on the wire would be the
234
+ * honest-log/lying-wire split this project refuses. Each arm passes its own target grammar.
235
+ */
236
+ function respondEnqueueOutcome({ log, res, delivery, repo, target, flow, outcome }) {
237
+ for (const d of outcome.deduplicated) {
238
+ log?.({ event: "deduplicated", delivery, repo, target, flow, jobId: d.jobId, survivingJobId: d.survivingJobId });
239
+ }
240
+ if (outcome.created > 0) {
241
+ log?.({ event: "enqueued", delivery, repo, target, flow, replicas: outcome.created });
242
+ return respond(res, 202, { status: "queued" });
243
+ }
244
+ return respond(res, 202, { status: "deduplicated" });
216
245
  }
217
246
 
218
247
  /**
@@ -266,17 +295,16 @@ function makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority
266
295
  }
267
296
 
268
297
  // Fanout (REQ-REPLICA-RUNS) lives in `fanout` above; the 202/503 decision stays here, where it always was.
269
- let replicas;
298
+ let outcome;
270
299
  try {
271
- replicas = await fanout(result.job, async (j) => await enqueueGitHubJob(await routeTo("github", j), j));
300
+ outcome = await fanout(result.job, async (j) => await enqueueGitHubJob(await routeTo("github", j), j));
272
301
  } catch (err) {
273
302
  // Own try/catch so a Valkey-down enqueue is a 503 (retryable), not verify's outer 500.
274
303
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
275
304
  return respond(res, 503, { error: "enqueue-failed" }); // GitHub redelivers; dedup by GUID coalesces
276
305
  }
277
306
 
278
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
279
- return respond(res, 202, { status: "queued" });
307
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
280
308
  });
281
309
  }
282
310
 
@@ -316,9 +344,9 @@ function makeGitLabHandler({ routeTo, queue, cfg, log, mode, secret, selfId, res
316
344
  return respond(res, 204);
317
345
  }
318
346
 
319
- let replicas;
347
+ let outcome;
320
348
  try {
321
- replicas = await fanout(result.job, async (j) => await enqueueGitLabJob(await routeTo("gitlab", j), j));
349
+ outcome = await fanout(result.job, async (j) => await enqueueGitLabJob(await routeTo("gitlab", j), j));
322
350
  } catch (err) {
323
351
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
324
352
  return respond(res, 503, { error: "enqueue-failed" }); // GitLab redelivers; dedup by webhook-id coalesces
@@ -327,8 +355,7 @@ function makeGitLabHandler({ routeTo, queue, cfg, log, mode, secret, selfId, res
327
355
  // `!` for a merge request, `#` for an issue -- GitLab's own notation, and the same discrimination
328
356
  // the semantic dedup key makes, because the two are separate number sequences.
329
357
  const sep = result.job.target.type === "pull_request" ? "!" : "#";
330
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, replicas });
331
- return respond(res, 202, { status: "queued" });
358
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
332
359
  });
333
360
  }
334
361
 
@@ -368,16 +395,15 @@ function makeForgejoHandler({ routeTo, queue, cfg, log, secret, selfId, resolveA
368
395
  return respond(res, 204);
369
396
  }
370
397
 
371
- let replicas;
398
+ let outcome;
372
399
  try {
373
- replicas = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("forgejo", j), "forgejo", j));
400
+ outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("forgejo", j), "forgejo", j));
374
401
  } catch (err) {
375
402
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
376
403
  return respond(res, 503, { error: "enqueue-failed" }); // Forgejo redelivers; dedup by GUID coalesces
377
404
  }
378
405
 
379
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
380
- return respond(res, 202, { status: "queued" });
406
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
381
407
  });
382
408
  }
383
409
 
@@ -426,9 +452,9 @@ function makeAzureHandler({ routeTo, queue, cfg, log, mode, secret, headerName,
426
452
  return respond(res, 204);
427
453
  }
428
454
 
429
- let replicas;
455
+ let outcome;
430
456
  try {
431
- replicas = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("azure", j), "azure", j));
457
+ outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("azure", j), "azure", j));
432
458
  } catch (err) {
433
459
  log?.({ event: "enqueue_failed", delivery, reason: err?.message });
434
460
  return respond(res, 503, { error: "enqueue-failed" });
@@ -437,7 +463,6 @@ function makeAzureHandler({ routeTo, queue, cfg, log, mode, secret, headerName,
437
463
  // `!` for a pull request, `#` for a work item -- Azure numbers them separately, and this is the same
438
464
  // discrimination the semantic dedup key makes.
439
465
  const sep = result.job.target.type === "pull_request" ? "!" : "#";
440
- log?.({ event: "enqueued", delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, replicas });
441
- return respond(res, 202, { status: "queued" });
466
+ return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
442
467
  });
443
468
  }