@edgehero/pi-dispatch-receiver 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -0
- package/package.json +7 -4
- package/src/boot-retry.mjs +141 -0
- package/src/cli.mjs +12 -5
- package/src/config.mjs +30 -5
- package/src/filter-azure.mjs +12 -4
- package/src/filter-forgejo.mjs +14 -6
- package/src/filter-gitlab.mjs +14 -6
- package/src/filter.mjs +16 -5
- package/src/poller-config.mjs +6 -1
- package/src/poller.mjs +83 -14
- package/src/receiver.mjs +57 -28
- package/src/route.mjs +126 -0
- package/src/start.mjs +243 -84
package/src/poller.mjs
CHANGED
|
@@ -108,13 +108,14 @@
|
|
|
108
108
|
import { createSign } from "node:crypto";
|
|
109
109
|
import { readFile as fsReadFile } from "node:fs/promises";
|
|
110
110
|
import { configError } from "@edgehero/pi-dispatch/config";
|
|
111
|
-
import { parseConnection } from "@edgehero/pi-dispatch/connection";
|
|
111
|
+
import { judgeValkeyAtStart, makeRedisClient, parseConnection, valkeyClientContext } from "@edgehero/pi-dispatch/connection";
|
|
112
112
|
import { makeGitHubAuth } from "@edgehero/pi-dispatch/get-token";
|
|
113
113
|
import { enqueueGitHubJob, makeQueue } from "@edgehero/pi-dispatch/queue";
|
|
114
114
|
import { filter, hasCloseTriggers, wantsCloserAuthority } from "./filter.mjs";
|
|
115
115
|
import { makeResolveGitHubAuthority } from "./github-members.mjs";
|
|
116
116
|
import { parseSubset } from "./receiver.mjs";
|
|
117
117
|
import { loadPollerConfig } from "./poller-config.mjs";
|
|
118
|
+
import { retryIdentity } from "./boot-retry.mjs";
|
|
118
119
|
|
|
119
120
|
const API_URL = "https://api.github.com";
|
|
120
121
|
// 35 days: strictly outlives the 31-day gh-* job retention -- see the header's cursor/TTL section.
|
|
@@ -153,9 +154,10 @@ class RateLimited extends Error {
|
|
|
153
154
|
* Boot the poller: config, HARD-FAIL identity, repo set, then the cycle loop. Collaborators are
|
|
154
155
|
* injected with real defaults (start.mjs's convention), so the whole producer is testable offline
|
|
155
156
|
* with no GitHub, no Valkey, and no timers:
|
|
156
|
-
* { fetchFn, redis, queueFn, out, now, random, sleep, selfIdFn, tokenFn, fsDeps }
|
|
157
|
+
* { fetchFn, redis, queueFn, out, now, random, sleep, signals, selfIdFn, tokenFn, fsDeps }
|
|
157
158
|
* Returns `{ stop, done }`: `stop()` ends the loop after the in-flight work; `done` resolves once
|
|
158
|
-
* owned connections are closed
|
|
159
|
+
* owned connections are closed, with the inter-cycle default-sleep timer cleared by then (issue
|
|
160
|
+
* #325) -- after `done`, no timer armed by the poller remains.
|
|
159
161
|
*/
|
|
160
162
|
export async function startPoller(env = process.env, deps = {}) {
|
|
161
163
|
const {
|
|
@@ -164,6 +166,12 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
164
166
|
now = Date.now,
|
|
165
167
|
random = Math.random,
|
|
166
168
|
sleep,
|
|
169
|
+
// The signal gate, decoupled from the sleep default (issue #325): `sleep === undefined` used to
|
|
170
|
+
// double as the "real run" flag, which made the REAL default timer unarmable under test without
|
|
171
|
+
// also registering process-wide SIGTERM/SIGINT handlers -- the exact leak the gate exists to
|
|
172
|
+
// prevent. A test that arms the real sleep passes `signals: false`; the one production caller
|
|
173
|
+
// (cli.mjs, no deps) keeps handlers exactly as before through this default.
|
|
174
|
+
signals = sleep === undefined,
|
|
167
175
|
redis,
|
|
168
176
|
queueFn,
|
|
169
177
|
selfIdFn,
|
|
@@ -172,7 +180,11 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
172
180
|
fsDeps = {},
|
|
173
181
|
makeAuth = makeGitHubAuth,
|
|
174
182
|
makeQueueFn = makeQueue,
|
|
183
|
+
// Issue #464 (gate round 3): how the poller builds its own Valkey client; a seam, so a test sees none built.
|
|
184
|
+
makeRedisFn = makeRedisClient,
|
|
175
185
|
makeResolveGitHubAuthority: makeResolveGitHubAuthorityFn = makeResolveGitHubAuthority,
|
|
186
|
+
// Issue #464 (gate round 3): the start-time judgement of VALKEY_URL, as the receiver's (start.mjs).
|
|
187
|
+
judgeValkey = (url, opts) => judgeValkeyAtStart(url, valkeyClientContext({ env: opts.env }), { now: opts.now, ...(opts.sleep ? { sleep: opts.sleep } : {}) }),
|
|
176
188
|
} = deps;
|
|
177
189
|
|
|
178
190
|
const cfg = loadPollerConfig(env, fsDeps);
|
|
@@ -181,9 +193,21 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
181
193
|
// the bot-loop guard's sole input, and a poller running without it would read the harness's own
|
|
182
194
|
// completion comment next cycle and enqueue it -- an unbounded paid recursion, only slower than the
|
|
183
195
|
// webhook version. No try/catch: an unresolvable identity must prevent the loop from ever starting.
|
|
196
|
+
// A TRANSIENT failure retries in-process under the same window serve uses (issue #318): a
|
|
197
|
+
// pure-polling deployment loses work to a stopped process just as surely as a webhook one.
|
|
198
|
+
//
|
|
199
|
+
// The default sleep sits above the gate because the gate sleeps too (issue #318); the boot retry
|
|
200
|
+
// simply awaits its cancellable promise to completion, so `cancel` is the LOOP's concern alone.
|
|
201
|
+
const sleepFn = sleep ?? cancellableSleep;
|
|
184
202
|
let auth = null;
|
|
203
|
+
// `auth ??= await ...` assigns only when the mint RESOLVES, so a retried getAuth re-invokes
|
|
204
|
+
// makeAuth. Caching the PROMISE instead would memoize the first failure and turn the retry loop
|
|
205
|
+
// into a rethrow spinner -- pinned by the poller's retry test.
|
|
185
206
|
const getAuth = async () => (auth ??= await makeAuth(cfg.github));
|
|
186
|
-
const selfId =
|
|
207
|
+
const selfId = await retryIdentity(
|
|
208
|
+
() => (selfIdFn ? selfIdFn(cfg.github) : getAuth().then((a) => a.selfId)),
|
|
209
|
+
{ forge: "github", windowMs: cfg.identityRetryWindowMs, log: out, now, sleep: sleepFn },
|
|
210
|
+
);
|
|
187
211
|
out({ event: "self_identity", id: selfId, source: cfg.github.source });
|
|
188
212
|
|
|
189
213
|
// The polling credential comes from the same auth config the worker validates. pat/gh hand back
|
|
@@ -210,9 +234,12 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
210
234
|
// queue connection: a long-running producer should survive a Valkey restart.
|
|
211
235
|
let redisClient = redis ?? null;
|
|
212
236
|
let ownRedis = false;
|
|
237
|
+
// Issue #464 (gate round 3): judged before this process builds its own Valkey clients: a refusal exits 2, nothing
|
|
238
|
+
// answering for 20 s exits 1, never a poller running on a queue that cannot connect.
|
|
239
|
+
if (redisClient === null || queueFn === undefined || queueFn === null) await judgeValkey(cfg.valkeyUrl, { env, now, sleep });
|
|
213
240
|
if (redisClient === null) {
|
|
214
|
-
|
|
215
|
-
redisClient =
|
|
241
|
+
// Through connection.mjs, as every Valkey client of this project (issue #464): it judges and pins the address.
|
|
242
|
+
redisClient = makeRedisFn(cfg.valkeyUrl);
|
|
216
243
|
ownRedis = true;
|
|
217
244
|
}
|
|
218
245
|
|
|
@@ -263,7 +290,6 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
263
290
|
stopped = true;
|
|
264
291
|
wake();
|
|
265
292
|
};
|
|
266
|
-
const sleepFn = sleep ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
|
|
267
293
|
|
|
268
294
|
const done = (async () => {
|
|
269
295
|
let cycleNo = 0;
|
|
@@ -328,14 +354,27 @@ export async function startPoller(env = process.env, deps = {}) {
|
|
|
328
354
|
// in one synchronized stampede).
|
|
329
355
|
let delayMs = Math.max(cfg.intervalSeconds, stats.minDelaySeconds) * 1000;
|
|
330
356
|
if (rateResetMs !== null) delayMs = Math.max(delayMs, rateResetMs - now() + jitterMs(random));
|
|
331
|
-
|
|
357
|
+
// Hold the sleep so the LOSER can be cleared: when stop() wins this race, an uncancelled
|
|
358
|
+
// default timer stays armed for up to a full poll interval after `done` resolves (issue
|
|
359
|
+
// #325) -- masked in production by the signal handler's process.exit(0) below, and real for
|
|
360
|
+
// every caller that stops the poller without exiting. Optional-chained because injected
|
|
361
|
+
// test sleeps are plain promises carrying no cancel; their call count and arguments are
|
|
362
|
+
// untouched, and a rejecting one still propagates through the finally.
|
|
363
|
+
const nap = sleepFn(delayMs);
|
|
364
|
+
try {
|
|
365
|
+
await Promise.race([nap, stopWaker]);
|
|
366
|
+
} finally {
|
|
367
|
+
nap.cancel?.();
|
|
368
|
+
}
|
|
332
369
|
}
|
|
333
370
|
await closeOwned();
|
|
334
371
|
})();
|
|
335
372
|
|
|
336
|
-
// Signal handlers only on a real run (
|
|
337
|
-
// test injection the fakes are per-test, and a process-wide handler would
|
|
338
|
-
|
|
373
|
+
// Signal handlers only on a real run (`signals`, defaulting from the sleep seam) -- start.mjs's
|
|
374
|
+
// rule, same reason: under test injection the fakes are per-test, and a process-wide handler would
|
|
375
|
+
// leak across tests. Its own seam since issue #325, so a test can arm the REAL default sleep
|
|
376
|
+
// without inheriting the handlers.
|
|
377
|
+
if (signals) {
|
|
339
378
|
const shutdown = async (signal) => {
|
|
340
379
|
out({ event: "poller_stopping", signal });
|
|
341
380
|
stop();
|
|
@@ -358,6 +397,27 @@ function jitterMs(random) {
|
|
|
358
397
|
return 1_000 + Math.floor(random() * 29_000);
|
|
359
398
|
}
|
|
360
399
|
|
|
400
|
+
/**
|
|
401
|
+
* The default inter-cycle sleep: a REF'D setTimeout whose promise carries its own `cancel` (issue
|
|
402
|
+
* #325). Ref'd deliberately -- the poller IS its process's main loop and holding the loop through the
|
|
403
|
+
* delay is the point (`DES-RETENTION-SWEEPS-ON-A-TIMER`'s rejected list records the contrast; between
|
|
404
|
+
* cycles this timer is the only thing keeping a pure-poll process alive, cli.mjs's "awaiting done").
|
|
405
|
+
* The defect was never the ref, only survival past stop(): "unref'd is not cleaned up" has a ref'd
|
|
406
|
+
* twin, a cleared-nothing timer that holds the loop for up to a full interval after `done` resolved.
|
|
407
|
+
* A cancelled sleep never resolves, which is safe here because its only awaiter is a race `stopWaker`
|
|
408
|
+
* has already settled, and cancel on an already-fired timer is a no-op, so the winning side's clear
|
|
409
|
+
* costs nothing. EXPORTED for `settleWithin`'s reason (worker/src/start.mjs): the cleared-timer
|
|
410
|
+
* guarantee deserves a deterministic async_hooks pin on the helper, not a census of a full boot.
|
|
411
|
+
*/
|
|
412
|
+
export function cancellableSleep(ms) {
|
|
413
|
+
let timer;
|
|
414
|
+
const p = new Promise((resolve) => {
|
|
415
|
+
timer = setTimeout(resolve, ms);
|
|
416
|
+
});
|
|
417
|
+
p.cancel = () => clearTimeout(timer);
|
|
418
|
+
return p;
|
|
419
|
+
}
|
|
420
|
+
|
|
361
421
|
/** The redis key family for one repo -- see the schema table in the module header. */
|
|
362
422
|
function keyNames(repo) {
|
|
363
423
|
const p = `poll:${repo}`;
|
|
@@ -916,11 +976,20 @@ async function gate(ctx, eventName, payload, deliveryId, stats) {
|
|
|
916
976
|
return;
|
|
917
977
|
}
|
|
918
978
|
const replicas = result.job.replicas ?? 1;
|
|
979
|
+
let created = 0;
|
|
919
980
|
for (let i = 1; i <= replicas; i++) {
|
|
920
|
-
await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
|
|
981
|
+
const r = await ctx.enqueue(replicas > 1 ? { ...result.job, replica: i } : result.job);
|
|
982
|
+
// Issue #289, the receiver arms' honesty rule at the poller's own seam: a semantic-window swallow
|
|
983
|
+
// gets its own line with the surviving id, `enqueued` counts only what was CREATED, and a bare-id
|
|
984
|
+
// return from an enqueue seam predating the shape counts as created -- the old behaviour exactly.
|
|
985
|
+
if (r && typeof r === "object" && r.deduplicated === true) {
|
|
986
|
+
ctx.out({ event: "deduplicated", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, jobId: r.jobId, survivingJobId: r.survivingJobId });
|
|
987
|
+
} else {
|
|
988
|
+
created += 1;
|
|
989
|
+
}
|
|
921
990
|
}
|
|
922
|
-
stats.enqueued +=
|
|
923
|
-
ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas });
|
|
991
|
+
stats.enqueued += created;
|
|
992
|
+
if (created > 0) ctx.out({ event: "enqueued", delivery: deliveryId, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, replicas: created });
|
|
924
993
|
}
|
|
925
994
|
|
|
926
995
|
/**
|
package/src/receiver.mjs
CHANGED
|
@@ -134,7 +134,11 @@ export function parseSubset(payload) {
|
|
|
134
134
|
* undefined, i.e. never drops the harness's own comments. Absent property therefore means no route; the
|
|
135
135
|
* failure of a forgotten property is a 404 an operator sees, never a paid recursion they get billed for.
|
|
136
136
|
*/
|
|
137
|
-
export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority }) {
|
|
137
|
+
export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo = null, azure = null, resolveAuthority, router = null }) {
|
|
138
|
+
// Which queue a delivery is enqueued onto (issue #57, `OQ-032`). The default is the shared queue for
|
|
139
|
+
// EVERY job, which is what this receiver did before multi-host existed -- so a deployment that wires no
|
|
140
|
+
// router, and every test that constructs one without it, is byte-identical.
|
|
141
|
+
const routeTo = router ? (kind, job) => router.queueFor(kind, job) : () => queue;
|
|
138
142
|
// Built only when the deployment serves GitHub. Construction is not free of the secret either: the
|
|
139
143
|
// `new Webhooks({ secret })` inside makeVerifiedHandler throws "options.secret required" on an absent
|
|
140
144
|
// one, so not building the arm is what lets a github-free deployment legitimately have no secret --
|
|
@@ -143,7 +147,7 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
|
|
|
143
147
|
// `resolveAuthority` is the GITHUB closer resolver (issue #231), riding top-level beside `selfId`
|
|
144
148
|
// because github's dependencies always have -- the other forges bundle theirs in per-forge objects.
|
|
145
149
|
// Optional, because only a close delivery an armed close rule matches ever consults it.
|
|
146
|
-
const github = cfg.servesGithub ? makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) : null;
|
|
150
|
+
const github = cfg.servesGithub ? makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) : null;
|
|
147
151
|
|
|
148
152
|
// A TABLE, built once, rather than one `if` per forge. Two forges made that a single branch; four make
|
|
149
153
|
// it a chain, and a chain is where one arm quietly ends up checked after the fallthrough. A path present
|
|
@@ -151,9 +155,9 @@ export function makeReceiver({ queue, selfId, cfg, log, gitlab = null, forgejo =
|
|
|
151
155
|
// falls through to GitHub, which is what keeps `/` working -- and a configured-off GitHub answers the
|
|
152
156
|
// same 404 from the fallthrough itself, below.
|
|
153
157
|
const routes = {
|
|
154
|
-
"/gitlab": gitlab ? makeGitLabHandler({ queue, cfg, log, ...gitlab }) : null,
|
|
155
|
-
"/forgejo": forgejo ? makeForgejoHandler({ queue, cfg, log, ...forgejo }) : null,
|
|
156
|
-
"/azure": azure ? makeAzureHandler({ queue, cfg, log, ...azure }) : null,
|
|
158
|
+
"/gitlab": gitlab ? makeGitLabHandler({ queue, routeTo, cfg, log, ...gitlab }) : null,
|
|
159
|
+
"/forgejo": forgejo ? makeForgejoHandler({ queue, routeTo, cfg, log, ...forgejo }) : null,
|
|
160
|
+
"/azure": azure ? makeAzureHandler({ queue, routeTo, cfg, log, ...azure }) : null,
|
|
157
161
|
};
|
|
158
162
|
|
|
159
163
|
return async function receiverHandler(req, res) {
|
|
@@ -201,14 +205,43 @@ function pathOf(url) {
|
|
|
201
205
|
* `enqueue` is a callback because the four arms spell their enqueue differently (a named github/gitlab
|
|
202
206
|
* wrapper, or `enqueueForgeJob` with an explicit kind); the fanout itself is forge-blind.
|
|
203
207
|
*
|
|
204
|
-
* @returns {Promise<number
|
|
208
|
+
* @returns {Promise<{replicas: number, created: number, deduplicated: Array<{jobId, survivingJobId}>}>}
|
|
209
|
+
* what actually happened, per replica (issue #289): `created` counts jobs that now EXIST because of
|
|
210
|
+
* this delivery, and `deduplicated` carries each swallow the semantic window made, with the id that
|
|
211
|
+
* survived. A bare-id return from an enqueue that predates the shape counts as created -- the old
|
|
212
|
+
* behaviour exactly.
|
|
205
213
|
*/
|
|
206
214
|
async function fanout(job, enqueue) {
|
|
207
215
|
const replicas = job.replicas ?? 1;
|
|
216
|
+
let created = 0;
|
|
217
|
+
const deduplicated = [];
|
|
208
218
|
for (let i = 1; i <= replicas; i++) {
|
|
209
|
-
await enqueue(replicas > 1 ? { ...job, replica: i } : job);
|
|
219
|
+
const r = await enqueue(replicas > 1 ? { ...job, replica: i } : job);
|
|
220
|
+
if (r && typeof r === "object" && r.deduplicated === true) deduplicated.push({ jobId: r.jobId, survivingJobId: r.survivingJobId });
|
|
221
|
+
else created += 1;
|
|
210
222
|
}
|
|
211
|
-
return replicas;
|
|
223
|
+
return { replicas, created, deduplicated };
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* One log-and-answer for a fanout outcome, shared by all four arms so a weakened copy cannot hide (the
|
|
228
|
+
* four-copies doctrine above). Issue #289's honesty rule: `enqueued` is logged only when something was
|
|
229
|
+
* CREATED, with `replicas` meaning jobs that now exist; every semantic-window swallow gets its own
|
|
230
|
+
* `deduplicated` line naming the surviving job's id (every field forge- or worker-minted, no payload
|
|
231
|
+
* text); and a delivery that created NOTHING answers `202 {status:"deduplicated"}` -- still a 2xx,
|
|
232
|
+
* because a non-2xx would trigger a redelivery storm for a delivery that was handled, but no longer
|
|
233
|
+
* "queued", because a receiver that logs the swallow while answering success on the wire would be the
|
|
234
|
+
* honest-log/lying-wire split this project refuses. Each arm passes its own target grammar.
|
|
235
|
+
*/
|
|
236
|
+
function respondEnqueueOutcome({ log, res, delivery, repo, target, flow, outcome }) {
|
|
237
|
+
for (const d of outcome.deduplicated) {
|
|
238
|
+
log?.({ event: "deduplicated", delivery, repo, target, flow, jobId: d.jobId, survivingJobId: d.survivingJobId });
|
|
239
|
+
}
|
|
240
|
+
if (outcome.created > 0) {
|
|
241
|
+
log?.({ event: "enqueued", delivery, repo, target, flow, replicas: outcome.created });
|
|
242
|
+
return respond(res, 202, { status: "queued" });
|
|
243
|
+
}
|
|
244
|
+
return respond(res, 202, { status: "deduplicated" });
|
|
212
245
|
}
|
|
213
246
|
|
|
214
247
|
/**
|
|
@@ -223,7 +256,7 @@ async function fanout(job, enqueue) {
|
|
|
223
256
|
* uses, so a lookup is never spent on a delivery the gate then ignores -- every label, comment, PR and
|
|
224
257
|
* review delivery, and every close nothing wants, stays payload-only and byte-identical to before.
|
|
225
258
|
*/
|
|
226
|
-
function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
|
|
259
|
+
function makeGitHubHandler({ queue, routeTo, selfId, cfg, log, resolveAuthority }) {
|
|
227
260
|
return makeVerifiedHandler({ secret: cfg.webhookSecret }, async ({ rawBody, event, delivery }, res) => {
|
|
228
261
|
let subset;
|
|
229
262
|
try {
|
|
@@ -262,17 +295,16 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
|
|
|
262
295
|
}
|
|
263
296
|
|
|
264
297
|
// Fanout (REQ-REPLICA-RUNS) lives in `fanout` above; the 202/503 decision stays here, where it always was.
|
|
265
|
-
let
|
|
298
|
+
let outcome;
|
|
266
299
|
try {
|
|
267
|
-
|
|
300
|
+
outcome = await fanout(result.job, async (j) => await enqueueGitHubJob(await routeTo("github", j), j));
|
|
268
301
|
} catch (err) {
|
|
269
302
|
// Own try/catch so a Valkey-down enqueue is a 503 (retryable), not verify's outer 500.
|
|
270
303
|
log?.({ event: "enqueue_failed", delivery, reason: err?.message });
|
|
271
304
|
return respond(res, 503, { error: "enqueue-failed" }); // GitHub redelivers; dedup by GUID coalesces
|
|
272
305
|
}
|
|
273
306
|
|
|
274
|
-
|
|
275
|
-
return respond(res, 202, { status: "queued" });
|
|
307
|
+
return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
|
|
276
308
|
});
|
|
277
309
|
}
|
|
278
310
|
|
|
@@ -289,7 +321,7 @@ function makeGitHubHandler({ queue, selfId, cfg, log, resolveAuthority }) {
|
|
|
289
321
|
* The lookup runs only for events that could still fire -- after verification, and after the payload has
|
|
290
322
|
* been projected -- so an unauthenticated flood cannot make this project call GitLab at all.
|
|
291
323
|
*/
|
|
292
|
-
function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
|
|
324
|
+
function makeGitLabHandler({ routeTo, queue, cfg, log, mode, secret, selfId, resolveAuthority, now }) {
|
|
293
325
|
return makeGitLabVerifiedHandler({ mode, secret, ...(now ? { now } : {}) }, async ({ rawBody, delivery }, res) => {
|
|
294
326
|
let subset;
|
|
295
327
|
try {
|
|
@@ -312,9 +344,9 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
|
|
|
312
344
|
return respond(res, 204);
|
|
313
345
|
}
|
|
314
346
|
|
|
315
|
-
let
|
|
347
|
+
let outcome;
|
|
316
348
|
try {
|
|
317
|
-
|
|
349
|
+
outcome = await fanout(result.job, async (j) => await enqueueGitLabJob(await routeTo("gitlab", j), j));
|
|
318
350
|
} catch (err) {
|
|
319
351
|
log?.({ event: "enqueue_failed", delivery, reason: err?.message });
|
|
320
352
|
return respond(res, 503, { error: "enqueue-failed" }); // GitLab redelivers; dedup by webhook-id coalesces
|
|
@@ -323,8 +355,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
|
|
|
323
355
|
// `!` for a merge request, `#` for an issue -- GitLab's own notation, and the same discrimination
|
|
324
356
|
// the semantic dedup key makes, because the two are separate number sequences.
|
|
325
357
|
const sep = result.job.target.type === "pull_request" ? "!" : "#";
|
|
326
|
-
|
|
327
|
-
return respond(res, 202, { status: "queued" });
|
|
358
|
+
return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
|
|
328
359
|
});
|
|
329
360
|
}
|
|
330
361
|
|
|
@@ -341,7 +372,7 @@ function makeGitLabHandler({ queue, cfg, log, mode, secret, selfId, resolveAutho
|
|
|
341
372
|
* per construction, and serving two sources from one would mean either sharing a secret between forges or
|
|
342
373
|
* letting the request choose which one it was checked against.
|
|
343
374
|
*/
|
|
344
|
-
function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority }) {
|
|
375
|
+
function makeForgejoHandler({ routeTo, queue, cfg, log, secret, selfId, resolveAuthority }) {
|
|
345
376
|
return makeVerifiedHandler({ secret }, async ({ rawBody, event, delivery }, res) => {
|
|
346
377
|
let subset;
|
|
347
378
|
try {
|
|
@@ -364,16 +395,15 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
|
|
|
364
395
|
return respond(res, 204);
|
|
365
396
|
}
|
|
366
397
|
|
|
367
|
-
let
|
|
398
|
+
let outcome;
|
|
368
399
|
try {
|
|
369
|
-
|
|
400
|
+
outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("forgejo", j), "forgejo", j));
|
|
370
401
|
} catch (err) {
|
|
371
402
|
log?.({ event: "enqueue_failed", delivery, reason: err?.message });
|
|
372
403
|
return respond(res, 503, { error: "enqueue-failed" }); // Forgejo redelivers; dedup by GUID coalesces
|
|
373
404
|
}
|
|
374
405
|
|
|
375
|
-
|
|
376
|
-
return respond(res, 202, { status: "queued" });
|
|
406
|
+
return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.target.type}#${result.job.target.number}`, flow: result.job.flow, outcome });
|
|
377
407
|
});
|
|
378
408
|
}
|
|
379
409
|
|
|
@@ -391,7 +421,7 @@ function makeForgejoHandler({ queue, cfg, log, secret, selfId, resolveAuthority
|
|
|
391
421
|
* `"Display Name <email>"`. Both the resolver and the bot-loop guard handle both forms -- see
|
|
392
422
|
* filter-azure.mjs, where the ordering constraint is stated in full.
|
|
393
423
|
*/
|
|
394
|
-
function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
|
|
424
|
+
function makeAzureHandler({ routeTo, queue, cfg, log, mode, secret, headerName, selfId, resolveAuthority }) {
|
|
395
425
|
return makeAzureVerifiedHandler({ mode, secret, headerName }, async ({ rawBody }, res) => {
|
|
396
426
|
let subset;
|
|
397
427
|
try {
|
|
@@ -422,9 +452,9 @@ function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, r
|
|
|
422
452
|
return respond(res, 204);
|
|
423
453
|
}
|
|
424
454
|
|
|
425
|
-
let
|
|
455
|
+
let outcome;
|
|
426
456
|
try {
|
|
427
|
-
|
|
457
|
+
outcome = await fanout(result.job, async (j) => await enqueueForgeJob(await routeTo("azure", j), "azure", j));
|
|
428
458
|
} catch (err) {
|
|
429
459
|
log?.({ event: "enqueue_failed", delivery, reason: err?.message });
|
|
430
460
|
return respond(res, 503, { error: "enqueue-failed" });
|
|
@@ -433,7 +463,6 @@ function makeAzureHandler({ queue, cfg, log, mode, secret, headerName, selfId, r
|
|
|
433
463
|
// `!` for a pull request, `#` for a work item -- Azure numbers them separately, and this is the same
|
|
434
464
|
// discrimination the semantic dedup key makes.
|
|
435
465
|
const sep = result.job.target.type === "pull_request" ? "!" : "#";
|
|
436
|
-
|
|
437
|
-
return respond(res, 202, { status: "queued" });
|
|
466
|
+
return respondEnqueueOutcome({ log, res, delivery, repo: result.job.repo, target: `${result.job.repo}${sep}${result.job.target.number}`, flow: result.job.flow, outcome });
|
|
438
467
|
});
|
|
439
468
|
}
|
package/src/route.mjs
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which queue a forge delivery is enqueued onto, when the deployment runs on more than one machine
|
|
3
|
+
* (issue #57, `OQ-032`).
|
|
4
|
+
*
|
|
5
|
+
* The receiver is the only process that can make this decision. Routing has to happen at ENQUEUE -- a
|
|
6
|
+
* delayed job is promoted on each worker's own clock, so the fastest clock wins every attempt and a job
|
|
7
|
+
* cannot reliably be handed from a host that will not serve it to one that will -- and forge deliveries are
|
|
8
|
+
* enqueued here. `capabilities.mjs` owns the RULE and is shared with the worker so the two cannot drift;
|
|
9
|
+
* this module owns the plumbing: one cached registry read, and a pool of queue handles.
|
|
10
|
+
*
|
|
11
|
+
* EVERYTHING HERE FAILS OPEN ONTO THE SHARED QUEUE, which is what this receiver did before any of it
|
|
12
|
+
* existed. An unreadable registry, a timed-out read, a malformed row, a name that is not routable: every
|
|
13
|
+
* one of them returns the shared queue, so the worst this can do is fail to improve on a coin flip. A
|
|
14
|
+
* webhook handler is the wrong place to invent a new way to drop work.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { readLiveHosts } from "@edgehero/pi-dispatch/host-registry";
|
|
18
|
+
import { makeQueue, hostQueueName } from "@edgehero/pi-dispatch/queue";
|
|
19
|
+
import { parseConnection, makeRedisClient } from "@edgehero/pi-dispatch/connection";
|
|
20
|
+
import { forgeDeliveryJobId } from "@edgehero/pi-dispatch/job-id";
|
|
21
|
+
import { jobNeeds, routeForgeJob } from "@edgehero/pi-dispatch/capabilities";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* How long a registry read is reused.
|
|
25
|
+
*
|
|
26
|
+
* The registry beats every 15s, so anything below that mostly re-reads a value that cannot have changed.
|
|
27
|
+
* Five seconds is well inside one beat and bounds the extra load a busy receiver puts on Valkey at one read
|
|
28
|
+
* per five seconds rather than one per delivery -- and a delivery burst, which is exactly when this must
|
|
29
|
+
* not add latency, is served entirely from cache.
|
|
30
|
+
*
|
|
31
|
+
* Staleness costs nothing that freshness would have saved: a host that appears within the window is missed
|
|
32
|
+
* and its work goes to the shared queue, which is today's behaviour, and a host that vanishes is already
|
|
33
|
+
* guarded by `ROUTE_FRESH_MS` on the row's own beat timestamp.
|
|
34
|
+
*/
|
|
35
|
+
const HOSTS_TTL_MS = 5_000;
|
|
36
|
+
|
|
37
|
+
/** Bound on the registry read itself. A routing hint is never worth making a webhook wait. */
|
|
38
|
+
const READ_TIMEOUT_MS = 1_500;
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* @returns {{queueFor: (kind: string, job: object) => Promise<object>, close: () => Promise<void>}}
|
|
42
|
+
*/
|
|
43
|
+
export function makeForgeRouter({
|
|
44
|
+
valkeyUrl,
|
|
45
|
+
shared,
|
|
46
|
+
log = () => {},
|
|
47
|
+
makeQueueFn = makeQueue,
|
|
48
|
+
parseConnectionFn = parseConnection,
|
|
49
|
+
redisFn = makeRedisClient,
|
|
50
|
+
readLiveHostsFn = readLiveHosts,
|
|
51
|
+
now = () => Date.now(),
|
|
52
|
+
ttlMs = HOSTS_TTL_MS,
|
|
53
|
+
} = {}) {
|
|
54
|
+
const pool = new Map(); // queue name -> Queue, opened lazily and only for hosts actually routed to
|
|
55
|
+
let redis = null;
|
|
56
|
+
let cached = { at: -Infinity, hosts: [] };
|
|
57
|
+
let inFlight = null;
|
|
58
|
+
|
|
59
|
+
// One read at a time. Without this a burst of deliveries arriving on a cold cache each start their own
|
|
60
|
+
// registry read, which is the moment the deployment can least afford N of them.
|
|
61
|
+
const hosts = async () => {
|
|
62
|
+
if (now() - cached.at < ttlMs) return cached.hosts;
|
|
63
|
+
if (inFlight) return await inFlight;
|
|
64
|
+
inFlight = (async () => {
|
|
65
|
+
try {
|
|
66
|
+
if (!redis) {
|
|
67
|
+
redis = redisFn(valkeyUrl);
|
|
68
|
+
// A receiver must never print ioredis reconnect noise into its own log: this client is an
|
|
69
|
+
// optimisation, and its failures are already handled by falling back to the shared queue.
|
|
70
|
+
redis.on?.("error", () => {});
|
|
71
|
+
}
|
|
72
|
+
const res = await readLiveHostsFn(redis, { timeoutMs: READ_TIMEOUT_MS });
|
|
73
|
+
cached = { at: now(), hosts: Array.isArray(res?.hosts) ? res.hosts : [] };
|
|
74
|
+
} catch {
|
|
75
|
+
// Cache the FAILURE too, so an unreachable Valkey costs one read per window rather than one
|
|
76
|
+
// per delivery. The empty list routes everything to the shared queue.
|
|
77
|
+
cached = { at: now(), hosts: [] };
|
|
78
|
+
} finally {
|
|
79
|
+
inFlight = null;
|
|
80
|
+
}
|
|
81
|
+
return cached.hosts;
|
|
82
|
+
})();
|
|
83
|
+
return await inFlight;
|
|
84
|
+
};
|
|
85
|
+
|
|
86
|
+
return {
|
|
87
|
+
async queueFor(kind, job) {
|
|
88
|
+
const needs = jobNeeds(job);
|
|
89
|
+
// The overwhelming majority of deliveries bind neither a secret profile nor a wait profile, and
|
|
90
|
+
// they must not pay even a cache lookup for a decision that cannot apply to them.
|
|
91
|
+
if (needs.length === 0) return shared;
|
|
92
|
+
|
|
93
|
+
let name = null;
|
|
94
|
+
try {
|
|
95
|
+
name = routeForgeJob({ hosts: await hosts(), needs, jobId: forgeDeliveryJobId(kind, job?.trigger?.deliveryId, job?.replica) });
|
|
96
|
+
} catch {
|
|
97
|
+
return shared; // an unroutable id, a bad row: the shared queue is always a correct answer
|
|
98
|
+
}
|
|
99
|
+
if (!name) return shared;
|
|
100
|
+
|
|
101
|
+
const queueName = hostQueueName(name);
|
|
102
|
+
if (!pool.has(queueName)) pool.set(queueName, makeQueueFn(parseConnectionFn(valkeyUrl), { name: queueName }));
|
|
103
|
+
// Logged because a routed job is the one case where "which host ran it" was DECIDED rather than
|
|
104
|
+
// observed, and an operator debugging a job that never started needs to know it was sent
|
|
105
|
+
// somewhere specific. Host names are deployment topology: this reaches the log, never a forge
|
|
106
|
+
// comment (`triggers.mjs` sets that rule for refusal messages and it holds here).
|
|
107
|
+
log({ event: "forge_job_routed", kind, host: name, needs: needs.join(",") });
|
|
108
|
+
return pool.get(queueName);
|
|
109
|
+
},
|
|
110
|
+
async close() {
|
|
111
|
+
for (const q of pool.values()) {
|
|
112
|
+
try {
|
|
113
|
+
await q.close();
|
|
114
|
+
} catch {
|
|
115
|
+
// best-effort teardown
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
pool.clear();
|
|
119
|
+
try {
|
|
120
|
+
redis?.disconnect?.();
|
|
121
|
+
} catch {
|
|
122
|
+
// best-effort teardown
|
|
123
|
+
}
|
|
124
|
+
},
|
|
125
|
+
};
|
|
126
|
+
}
|