@edgehero/pi-dispatch 1.6.1 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +7 -0
- package/package.json +2 -1
- package/src/cli.mjs +74 -17
- package/src/config.mjs +83 -0
- package/src/cron.mjs +116 -4
- package/src/doctor.mjs +113 -3
- package/src/fingerprint.mjs +78 -0
- package/src/fleet-lease.mjs +179 -0
- package/src/host-registry.mjs +279 -0
- package/src/image-preflight.mjs +8 -3
- package/src/index.mjs +196 -51
- package/src/queue.mjs +130 -2
- package/src/run-history.mjs +23 -1
- package/src/schedules.mjs +67 -4
- package/src/start.mjs +217 -27
package/src/index.mjs
CHANGED
|
@@ -3,13 +3,16 @@ import { promisify } from "node:util";
|
|
|
3
3
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
4
4
|
import { InfraRetry, runJob } from "./processor.mjs";
|
|
5
5
|
import { targetFor } from "./run-history.mjs";
|
|
6
|
-
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
|
|
6
|
+
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
|
|
7
7
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
8
8
|
import { makeWaitState } from "./wait-state.mjs";
|
|
9
9
|
|
|
10
10
|
const exec = promisify(execFile);
|
|
11
11
|
|
|
12
12
|
export const QUEUE = "pi-jobs";
|
|
13
|
+
|
|
14
|
+
/** The key the host-wide in-flight count lives under. One machine, one counter, whatever the queue. */
|
|
15
|
+
export const HOST_SLOT_KEY = "host";
|
|
13
16
|
export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
14
17
|
// The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
|
|
15
18
|
// JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
|
|
@@ -60,7 +63,7 @@ const THROTTLE_FLOOR_MS = 11_000;
|
|
|
60
63
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
61
64
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
62
65
|
*/
|
|
63
|
-
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
66
|
+
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
64
67
|
return async function processor(job, token, signal) {
|
|
65
68
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
66
69
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -230,6 +233,37 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
230
233
|
// Declared outside the try below because the branches AFTER it read both.
|
|
231
234
|
let verdict = null;
|
|
232
235
|
let checked = null;
|
|
236
|
+
// A SECOND LAYER BENEATH THE FIRST, never a replacement (issue #57). The in-process map above
|
|
237
|
+
// stays exactly as it was and remains the correct per-host duty-cycle bound -- slots x timeout
|
|
238
|
+
// is the most wall-clock THIS worker spends answering questions instead of running jobs. What it
|
|
239
|
+
// cannot bound is the fleet: a held job's wakes land on any host, so `PI_WAIT_CHECK_SLOTS`
|
|
240
|
+
// silently multiplied by host count, and the one symptom the bound has gets QUIETER as you
|
|
241
|
+
// scale out, because multiplication produces fewer denials per host.
|
|
242
|
+
//
|
|
243
|
+
// Null when no peer could exist, so a single-host deployment issues no command at all.
|
|
244
|
+
let fleetSlot = null;
|
|
245
|
+
if (checkLease) {
|
|
246
|
+
// The TTL is DERIVED from what the lease actually guards: the gate holds it across every profile
|
|
247
|
+
// in turn, each bounded by `PI_WAIT_CHECK_TIMEOUT_MS`, so it is one timeout per PROFILE plus one
|
|
248
|
+
// for the overhead between them. Deriving it from the SLOT COUNT instead -- an unrelated
|
|
249
|
+
// quantity -- made a three-profile job at the shipped defaults hold 30s against a 20s lease,
|
|
250
|
+
// so the slot expired mid-check and another host took it while this one was still using it.
|
|
251
|
+
fleetSlot = await checkLease.acquire(job.id, { slots: checkSlotCount(), ttlMs: (profiles.length + 1) * checkTimeoutMs() });
|
|
252
|
+
if (!fleetSlot) {
|
|
253
|
+
// The same outcome as a local denial and the same remedy, so the same cadence and the same
|
|
254
|
+
// event -- with one conditional field, which is what keeps an unarmed deployment's log line
|
|
255
|
+
// byte-identical.
|
|
256
|
+
checkSlots.release(WAIT_CHECK_KEY);
|
|
257
|
+
const denials = await waitState.noteThrottle(job.id, { denied: true });
|
|
258
|
+
if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots: checkSlotCount(), where: "fleet", hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
|
|
259
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
|
|
260
|
+
const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
|
|
261
|
+
const delay = wait + Math.floor(wait * 0.1 * random());
|
|
262
|
+
deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots: checkSlotCount(), where: "fleet" });
|
|
263
|
+
await job.moveToDelayed(nowMs + delay, token);
|
|
264
|
+
throw new DelayedError();
|
|
265
|
+
}
|
|
266
|
+
}
|
|
233
267
|
// THE LEASE IS HELD FROM THE `tryAcquire` ABOVE, so every exit from here down must release it.
|
|
234
268
|
// The try opens here and not at the check loop, which is where it used to open: the supersede
|
|
235
269
|
// claim sits between the two, and BOTH of its exits leave -- one returns `wait-superseded`,
|
|
@@ -262,6 +296,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
262
296
|
if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
|
|
263
297
|
}
|
|
264
298
|
} finally {
|
|
299
|
+
await fleetSlot?.release?.();
|
|
265
300
|
checkSlots.release(WAIT_CHECK_KEY);
|
|
266
301
|
}
|
|
267
302
|
|
|
@@ -378,19 +413,78 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
378
413
|
// occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
|
|
379
414
|
// (measured: 301ms of live container overlap through this very processor) unless this gate holds.
|
|
380
415
|
// Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
|
|
416
|
+
// THE HOST-WIDE SLOT (issue #57), taken before the scope slot and released in the same finally.
|
|
417
|
+
//
|
|
418
|
+
// It exists only when this worker drains a second, host-affine queue. BullMQ's concurrency is per
|
|
419
|
+
// Worker, so two queues at `PI_CONCURRENCY` would run twice the containers -- and that knob bounds a
|
|
420
|
+
// MACHINE (its RAM, its share of the provider's concurrent-stream budget), not a queue. Deferral
|
|
421
|
+
// rather than refusal, at the scope gate's own cadence and for its reason: a full host is transient
|
|
422
|
+
// state, never a verdict about the job (CONST-RETRY-INFRA-ONLY).
|
|
423
|
+
//
|
|
424
|
+
// Before the scope acquire, so a job that cannot run on this machine at all never takes a folder
|
|
425
|
+
// mutex it would immediately have to give back, and so the two releases nest rather than interleave.
|
|
426
|
+
let hostHeld = false;
|
|
427
|
+
if (hostBound) {
|
|
428
|
+
if (!hostBound.slots.tryAcquire(HOST_SLOT_KEY, hostBound.limit())) {
|
|
429
|
+
deps?.log?.("host_busy_deferred", { jobId: job.id, delayMs: SCOPE_BUSY_RECHECK_MS });
|
|
430
|
+
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
431
|
+
throw new DelayedError();
|
|
432
|
+
}
|
|
433
|
+
hostHeld = true;
|
|
434
|
+
}
|
|
435
|
+
|
|
381
436
|
const limits = scopedLimits();
|
|
382
437
|
const scope = canonicalScope(job.data);
|
|
383
438
|
let held = false;
|
|
439
|
+
let scopeSlot = null;
|
|
384
440
|
if (scope) {
|
|
385
|
-
|
|
441
|
+
const ceiling = concurrencyFor(job.data, limits);
|
|
442
|
+
if (!inFlight.tryAcquire(scope, ceiling)) {
|
|
386
443
|
// Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
|
|
387
444
|
// The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
|
|
388
445
|
// host path); the delayed count and the job id are what an operator needs to see it.
|
|
389
446
|
deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
|
|
447
|
+
// The host slot goes back before we defer: `makeInFlight().release` is not idempotent, so a slot
|
|
448
|
+
// held across a deferral would be a slot this machine never gets back.
|
|
449
|
+
if (hostHeld) {
|
|
450
|
+
hostBound.slots.release(HOST_SLOT_KEY);
|
|
451
|
+
hostHeld = false;
|
|
452
|
+
}
|
|
390
453
|
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
391
454
|
throw new DelayedError();
|
|
392
455
|
}
|
|
393
456
|
held = true;
|
|
457
|
+
|
|
458
|
+
// THE FLEET-WIDE HALF of a scoped ceiling (issue #57). A `scoped-limits.json` row's day/week/month
|
|
459
|
+
// caps are already atomic INCRs on shared keys; its `concurrent` was a per-process Map, so it
|
|
460
|
+
// multiplied by host count -- a MONEY bound silently widened by the operator's deployment shape,
|
|
461
|
+
// which `INT-SCOPED-LIMITS-FILE-CONTRACT` calls out as the failure its version rule exists for.
|
|
462
|
+
//
|
|
463
|
+
// LOCAL scopes deliberately never claim, and the reason is not economy. The key is a hash of a
|
|
464
|
+
// PATH STRING, which carries no identity: `/srv/site` on two machines is, in the common case, two
|
|
465
|
+
// different repositories that share a layout convention. A shared claim keyed on that would
|
|
466
|
+
// serialise two genuinely independent working trees and break exactly the deployments this feature
|
|
467
|
+
// exists to enable. Local folders are answered by ROUTING instead -- a folder exists on one host,
|
|
468
|
+
// so its in-process mutex already spans everything it needs to.
|
|
469
|
+
//
|
|
470
|
+
// And an unlimited forge scope never claims either: `concurrencyFor` returns Infinity with no
|
|
471
|
+
// matching row, so a deployment with no scoped-limits file issues no command at all.
|
|
472
|
+
if (scopeLease && job.data?.kind !== "local" && Number.isFinite(ceiling)) {
|
|
473
|
+
scopeSlot = await scopeLease.acquire(job.id, { slots: ceiling, keyArgs: [scopeKeyPrefix(scope).slice("budget:s:".length)] });
|
|
474
|
+
if (!scopeSlot) {
|
|
475
|
+
// The local slot goes back BEFORE we defer: `makeInFlight().release` is not idempotent, so a
|
|
476
|
+
// slot held across a deferral is a slot this host never gets back.
|
|
477
|
+
inFlight.release(scope);
|
|
478
|
+
held = false;
|
|
479
|
+
if (hostHeld) {
|
|
480
|
+
hostBound.slots.release(HOST_SLOT_KEY);
|
|
481
|
+
hostHeld = false;
|
|
482
|
+
}
|
|
483
|
+
deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS, where: "fleet" });
|
|
484
|
+
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
485
|
+
throw new DelayedError();
|
|
486
|
+
}
|
|
487
|
+
}
|
|
394
488
|
}
|
|
395
489
|
|
|
396
490
|
let startedAt;
|
|
@@ -423,6 +517,12 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
423
517
|
inFlight.release(scope);
|
|
424
518
|
held = false;
|
|
425
519
|
}
|
|
520
|
+
if (hostHeld) {
|
|
521
|
+
hostBound.slots.release(HOST_SLOT_KEY);
|
|
522
|
+
hostHeld = false;
|
|
523
|
+
}
|
|
524
|
+
void scopeSlot?.release?.();
|
|
525
|
+
scopeSlot = null;
|
|
426
526
|
clearTimeout(timer);
|
|
427
527
|
throw error;
|
|
428
528
|
}
|
|
@@ -517,62 +617,107 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
517
617
|
// Release FIRST and never throw (release clamps at zero by construction): a throw here would
|
|
518
618
|
// mask the job's real error, and a missed release wedges the scope until a worker restart.
|
|
519
619
|
if (held) inFlight.release(scope);
|
|
620
|
+
if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
|
|
621
|
+
// AWAITED, not fire-and-forget. Two reasons, and the second is the one that bites: an unawaited
|
|
622
|
+
// DEL is dropped by `shutdown`'s `process.exit(0)`, stranding the claim for its whole TTL on a
|
|
623
|
+
// restart -- and the next same-scope job would otherwise race the release, be denied, and sit out a
|
|
624
|
+
// full re-check interval while the slot it wanted went free behind it. The finally is already inside
|
|
625
|
+
// an async function, and `release` never throws.
|
|
626
|
+
await scopeSlot?.release?.();
|
|
520
627
|
clearTimeout(timer);
|
|
521
628
|
signal.removeEventListener("abort", onAbort);
|
|
522
629
|
}
|
|
523
630
|
};
|
|
524
631
|
}
|
|
525
632
|
|
|
526
|
-
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
633
|
+
export function createWorker({ connection, name, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
|
|
634
|
+
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
635
|
+
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
636
|
+
// because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
|
|
637
|
+
// the delayed set is gated on each worker's own `Date.now()`, so the fastest clock wins every hop and a
|
|
638
|
+
// job that had to reach another host might never get there.
|
|
639
|
+
const names = hostQueue ? [QUEUE, hostQueue] : [QUEUE];
|
|
640
|
+
const workers = [];
|
|
641
|
+
|
|
642
|
+
// THE HOST-WIDE BOUND, and the reason it has to exist at all. `PI_CONCURRENCY` bounds a HOST -- its RAM
|
|
643
|
+
// and its share of the provider's concurrent-stream budget (DES-CONCURRENCY-3) -- but BullMQ's own
|
|
644
|
+
// concurrency is per Worker, so two Workers at 3 would run six containers. This semaphore restores the
|
|
645
|
+
// bound as a property of the machine. Process memory is still the correct store, for this entry's own
|
|
646
|
+
// unchanged reason: it counts THIS host's containers, and the boot reaper clears survivors before
|
|
647
|
+
// draining. Armed only when a host queue exists, so a single-host deployment builds one Worker and
|
|
648
|
+
// never reaches the acquire.
|
|
649
|
+
const hostBound = hostQueue ? { slots: hostSlots, limit: () => liveConcurrency() } : null;
|
|
650
|
+
const liveConcurrency = () => workers[0]?.concurrency ?? concurrency;
|
|
651
|
+
|
|
652
|
+
for (const queueName of names) {
|
|
653
|
+
let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
|
|
654
|
+
const processor = makeProcessor({
|
|
655
|
+
// Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
|
|
656
|
+
// and the shared handle could not reach it.
|
|
657
|
+
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
658
|
+
stopContainer: (name) => exec("docker", ["stop", "-t", "5", name]),
|
|
659
|
+
redis,
|
|
660
|
+
getSettings,
|
|
661
|
+
// Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
|
|
662
|
+
// next job start, and with two queues both have to move or the host bound and the queue bounds
|
|
663
|
+
// stop agreeing. Guarded so only an integer that actually differs touches the property.
|
|
664
|
+
applyConcurrency: (n) => {
|
|
665
|
+
if (!Number.isInteger(n)) return;
|
|
666
|
+
for (const w of workers) if (w.concurrency !== n) w.concurrency = n;
|
|
667
|
+
},
|
|
668
|
+
pauseUntil,
|
|
669
|
+
// SHARED across both workers, and that sharing is the point rather than an optimisation: the
|
|
670
|
+
// folder mutex, the per-scope ceiling and the wait-check lease all bound the HOST, so two
|
|
671
|
+
// independent maps would double every one of them exactly as two Workers double concurrency.
|
|
672
|
+
scopedLimits,
|
|
673
|
+
inFlight,
|
|
674
|
+
hostBound,
|
|
675
|
+
scopeLease,
|
|
676
|
+
// Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
|
|
677
|
+
// redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
|
|
678
|
+
waitState,
|
|
679
|
+
afterMaxMs,
|
|
680
|
+
// Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
|
|
681
|
+
// because the overlay can lower it through `dispatch_set` and a check must never take the last free
|
|
682
|
+
// slot from a paid job.
|
|
683
|
+
checkSlots,
|
|
684
|
+
checkLease,
|
|
685
|
+
checkSlotCount,
|
|
686
|
+
checkTimeoutMs,
|
|
687
|
+
concurrencyNow: concurrencyNow ?? liveConcurrency,
|
|
688
|
+
intervalMs,
|
|
689
|
+
maxWaitMs,
|
|
690
|
+
maxChecks,
|
|
691
|
+
maxFaults,
|
|
692
|
+
deps,
|
|
693
|
+
recordRun,
|
|
694
|
+
});
|
|
695
|
+
|
|
696
|
+
worker = new Worker(queueName, processor, {
|
|
697
|
+
// maxRetriesPerRequest: null is REQUIRED for BullMQ's blocking connections, or it throws.
|
|
698
|
+
connection: { ...connection, maxRetriesPerRequest: null },
|
|
699
|
+
concurrency,
|
|
700
|
+
maxStalledCount: 0, // a stalled paid job FAILS, never silently re-runs (verified live)
|
|
701
|
+
// Issue #57. Conditional, so a bare createWorker builds a byte-identical options object -- and because
|
|
702
|
+
// bullmq's own matcher accepts both the named and unnamed client-name spellings, naming costs nothing.
|
|
703
|
+
...(name ? { name } : {}),
|
|
704
|
+
...(limiter ? { limiter } : {}),
|
|
705
|
+
});
|
|
706
|
+
workers.push(worker);
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
const primary = workers[0];
|
|
710
|
+
// The host-queue worker, for the caller that must register listeners on both. Attached rather than
|
|
711
|
+
// returned as a pair so every existing caller keeps receiving exactly what it received before.
|
|
712
|
+
primary.hostWorker = workers[1] ?? null;
|
|
570
713
|
|
|
571
714
|
const shutdown = async () => {
|
|
572
715
|
// Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
|
|
573
|
-
// worker.close() would wait up to 30 minutes for the container.
|
|
574
|
-
|
|
575
|
-
|
|
716
|
+
// worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
|
|
717
|
+
// registrations would mean two `process.exit(0)` racing, and the second worker's containers would
|
|
718
|
+
// outlive the handler that was meant to stop them.
|
|
719
|
+
for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
|
|
720
|
+
for (const w of workers) await w.close().catch(() => {});
|
|
576
721
|
// Close auxiliary resources (e.g. a cron scheduler) after the worker drains. Per-item catch
|
|
577
722
|
// so one failing or absent closer never strands the others or blocks exit -- matches the
|
|
578
723
|
// swallow posture on cancelAllJobs above.
|
|
@@ -586,5 +731,5 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
|
|
|
586
731
|
// worker still aborts in-flight jobs and docker-stops their containers rather than orphaning them.
|
|
587
732
|
if (process.platform === "win32") process.once("SIGBREAK", shutdown);
|
|
588
733
|
|
|
589
|
-
return
|
|
734
|
+
return primary;
|
|
590
735
|
}
|
package/src/queue.mjs
CHANGED
|
@@ -8,10 +8,67 @@ import { PR_CLOSE_ACTIONS } from "./triggers.mjs";
|
|
|
8
8
|
const PR_CLOSE_WORDS = new Set(Object.values(PR_CLOSE_ACTIONS));
|
|
9
9
|
|
|
10
10
|
export const QUEUE = "pi-jobs";
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* The queue a HOST-AFFINE job goes to (issue #57): work only one machine can do, because the folder, the
|
|
14
|
+
* secret resolver or the wait-check script lives there.
|
|
15
|
+
*
|
|
16
|
+
* `@` is the separator because it is outside the worker-name charset (`[A-Za-z0-9._-]`), so
|
|
17
|
+
* `pi-jobs@<name>` decomposes unambiguously and a name can never contain one. A SUFFIX rather than a
|
|
18
|
+
* prefix so `KEYS bull:pi-jobs*` still shows an operator the whole deployment.
|
|
19
|
+
*
|
|
20
|
+
* Deliberately a separate queue rather than a field the pickup gate filters on. BullMQ has no selective
|
|
21
|
+
* pop, so filtering would mean taking a job and putting it back -- and promotion out of the delayed set is
|
|
22
|
+
* gated on each worker's OWN `Date.now()` in two places, so the host whose clock runs fastest wins every
|
|
23
|
+
* hop deterministically. A job that had to reach a different host might never get there, and jitter cannot
|
|
24
|
+
* fix it: it randomises WHEN the wake is, not WHO wins it.
|
|
25
|
+
*/
|
|
26
|
+
export const hostQueueName = (worker) => `${QUEUE}@${worker}`;
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Every queue name this deployment drains: the shared one, plus one per named host (issue #57).
|
|
30
|
+
*
|
|
31
|
+
* Derived from the REGISTRY rather than from configuration, because the reader is usually the admin or
|
|
32
|
+
* the CLI, which know their own host at best and the fleet not at all. A deployment with no named worker
|
|
33
|
+
* that declared NO name yields exactly `[QUEUE]`, so every existing caller is unchanged. Note that this
|
|
34
|
+
* is derived from `routes`, not from a row existing: every worker publishes a row, named or not.
|
|
35
|
+
*
|
|
36
|
+
* This exists because a host queue that no reader knows about is worse than no host queue: the panel
|
|
37
|
+
* would show zero schedulers while cron ran, and `pi-dispatch pause` would stop half a deployment while
|
|
38
|
+
* reporting success -- the silent no-op its own comment already warns about for a mistyped name.
|
|
39
|
+
*/
|
|
40
|
+
export function fleetQueueNames(hosts) {
|
|
41
|
+
const seen = new Set();
|
|
42
|
+
for (const h of hosts ?? []) {
|
|
43
|
+
// A host has a queue only when it DECLARED a name. Every worker publishes a registry row -- that is
|
|
44
|
+
// what lets an unnamed fleet be seen at all -- but an undeclared one drains only the shared queue,
|
|
45
|
+
// so deriving queue names from every row would invent `pi-jobs@<hostname>` for a queue nothing
|
|
46
|
+
// reads: pausing it would create a real key for a phantom, and the counts would be a queue that can
|
|
47
|
+
// never have jobs.
|
|
48
|
+
if (h?.routes !== true && h?.routes !== "true") continue;
|
|
49
|
+
const name = h?.name;
|
|
50
|
+
// VALIDATED, because this is peer-written data crossing a trust boundary. `hostQueueName`'s own
|
|
51
|
+
// contract leans on the charset -- `@` is the separator precisely because a name cannot contain one
|
|
52
|
+
// -- and nothing else re-checks it. A name with a `:` makes `new Queue` throw, which would take the
|
|
53
|
+
// kill switch out entirely; one with an `@` would not decompose.
|
|
54
|
+
if (typeof name !== "string" || !WORKER_NAME_RE.test(name)) continue;
|
|
55
|
+
seen.add(name);
|
|
56
|
+
}
|
|
57
|
+
// Deduped: two rows naming one host would double-count its jobs in a summed status.
|
|
58
|
+
return [QUEUE, ...[...seen].sort().map(hostQueueName)];
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** The name charset, duplicated from `config.mjs` deliberately: this module imports nothing. */
|
|
62
|
+
const WORKER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
|
|
63
|
+
|
|
11
64
|
export { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId };
|
|
12
65
|
|
|
13
|
-
|
|
14
|
-
|
|
66
|
+
/**
|
|
67
|
+
* A queue handle. `name` defaults to the shared queue, so every existing caller is unchanged and a
|
|
68
|
+
* single-host deployment never names anything else.
|
|
69
|
+
*/
|
|
70
|
+
export function makeQueue(connection, { name = QUEUE } = {}) {
|
|
71
|
+
return new Queue(name, { connection });
|
|
15
72
|
}
|
|
16
73
|
|
|
17
74
|
/**
|
|
@@ -226,3 +283,74 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
|
|
|
226
283
|
});
|
|
227
284
|
return jobId;
|
|
228
285
|
}
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Every host queue that EXISTS, read from BullMQ's own keyspace rather than from the host registry.
|
|
289
|
+
*
|
|
290
|
+
* The registry answers "who is alive", and for a kill switch that is the wrong question. A host whose
|
|
291
|
+
* registry writes fail for ninety seconds loses its row while its BullMQ worker -- a separate connection,
|
|
292
|
+
* built with `maxRetriesPerRequest: null` precisely to ride out blips -- keeps draining. Pausing "the
|
|
293
|
+
* fleet" would then miss it and report success. The same gap opens for the ~15s before a booting worker's
|
|
294
|
+
* first beat lands, and on every `service restart`, since a clean shutdown DELs the row.
|
|
295
|
+
*
|
|
296
|
+
* Worse is the direction with no recovery path: pause while a host is live durably pauses its queue, and a
|
|
297
|
+
* later resume while that host is DOWN enumerates nothing for it. The queue stays paused permanently, and
|
|
298
|
+
* no surface can see it, because every surface was reading the registry too.
|
|
299
|
+
*
|
|
300
|
+
* A queue's meta key is durable and outlives its worker, so this asks the only authority that cannot go
|
|
301
|
+
* stale: the queues themselves. It also restores what `host-registry.mjs` claims about itself -- delete the
|
|
302
|
+
* whole `host:*` keyspace and nothing decides differently -- which the registry-derived kill switch had
|
|
303
|
+
* quietly made false.
|
|
304
|
+
*
|
|
305
|
+
* SCAN, not KEYS, and it is why this is NOT on the panel's per-tick path: it is for the rare command where
|
|
306
|
+
* being wrong costs money, not for a reader that runs every second. Fails open to `[]`, so an unreadable
|
|
307
|
+
* keyspace degrades to the registry's answer rather than refusing.
|
|
308
|
+
*/
|
|
309
|
+
export async function discoverHostQueues(redis, { timeoutMs = 2_000, count = 500 } = {}) {
|
|
310
|
+
const prefix = `bull:${QUEUE}@`;
|
|
311
|
+
const names = new Set();
|
|
312
|
+
try {
|
|
313
|
+
const deadline = Date.now() + timeoutMs;
|
|
314
|
+
let cursor = "0";
|
|
315
|
+
do {
|
|
316
|
+
// BOUNDED, because BullMQ's connections carry `maxRetriesPerRequest: null` and a command against
|
|
317
|
+
// an unreachable server therefore QUEUES FOREVER rather than rejecting -- so an unguarded await
|
|
318
|
+
// here would hang the kill switch instead of failing it open. The same trap the registry's
|
|
319
|
+
// `bounded` exists for.
|
|
320
|
+
//
|
|
321
|
+
// CLEARED on the way out, and NOT `unref`'d. Leaving it pending held the event loop open for the
|
|
322
|
+
// rest of the budget after the work was done, so `pi-dispatch pause` sat for two seconds having
|
|
323
|
+
// already paused everything; unref'ing instead would stop it firing when the hang is the last
|
|
324
|
+
// thing on the loop, which is the one case it exists for.
|
|
325
|
+
let timer;
|
|
326
|
+
const [next, keys] = await Promise.race([
|
|
327
|
+
redis.scan(cursor, "MATCH", `${prefix}*:meta`, "COUNT", count),
|
|
328
|
+
new Promise((_, reject) => {
|
|
329
|
+
timer = setTimeout(() => reject(new Error("scan timed out")), Math.max(1, deadline - Date.now()));
|
|
330
|
+
}),
|
|
331
|
+
]).finally(() => clearTimeout(timer));
|
|
332
|
+
cursor = next;
|
|
333
|
+
for (const key of keys ?? []) {
|
|
334
|
+
const name = String(key).slice(prefix.length, -":meta".length);
|
|
335
|
+
// Validated like every other peer-derived name: a key an operator hand-created could hold
|
|
336
|
+
// anything, and `new Queue` throws on a `:`, which would take the kill switch out entirely.
|
|
337
|
+
if (WORKER_NAME_RE.test(name)) names.add(name);
|
|
338
|
+
}
|
|
339
|
+
} while (cursor !== "0" && Date.now() < deadline);
|
|
340
|
+
} catch {
|
|
341
|
+
// Fail open: the registry's answer alone is still better than refusing to pause.
|
|
342
|
+
return [];
|
|
343
|
+
}
|
|
344
|
+
return [...names].sort().map(hostQueueName);
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* The union of what is LIVE (the registry) and what EXISTS (the keyspace), which is the set a kill switch
|
|
349
|
+
* must act on: a live host with no queue yet has nothing to pause, and a dead host's queue still holds
|
|
350
|
+
* jobs and still has a paused flag somebody has to be able to clear.
|
|
351
|
+
*/
|
|
352
|
+
export function unionQueueNames(fromRegistry, fromKeyspace) {
|
|
353
|
+
const seen = new Set([...(fromRegistry ?? []), ...(fromKeyspace ?? [])]);
|
|
354
|
+
seen.delete(QUEUE);
|
|
355
|
+
return [QUEUE, ...[...seen].sort()];
|
|
356
|
+
}
|
package/src/run-history.mjs
CHANGED
|
@@ -378,7 +378,7 @@ function rebuildUsage(u) {
|
|
|
378
378
|
* default to `null` when the outcome does not carry them, so the record shape is stable whether or not
|
|
379
379
|
* the source reports those fields.
|
|
380
380
|
*/
|
|
381
|
-
export function buildRecord({ job, result, error, startedAt, endedAt }) {
|
|
381
|
+
export function buildRecord({ job, result, error, startedAt, endedAt, host = null }) {
|
|
382
382
|
const data = job.data ?? {};
|
|
383
383
|
const kind = data.kind ?? job.name;
|
|
384
384
|
const source = result ?? error ?? {};
|
|
@@ -441,6 +441,28 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
|
|
|
441
441
|
// BRANCH NAME ARE DELIBERATELY ABSENT: this record's PII-free-by-construction property rests on it
|
|
442
442
|
// holding no attacker-chosen string, and a branch name is exactly that.
|
|
443
443
|
session: source.session ?? null,
|
|
444
|
+
// Which machine ran this (issue #57). Additive, nullable, an explicit literal, TAIL position: every
|
|
445
|
+
// prior addition took the tail, and "field order is the serialisation order" is this record's
|
|
446
|
+
// contract, so the tail is the only placement that leaves twenty-four existing positions untouched.
|
|
447
|
+
// Unconditional rather than a conditional spread, on the same contract sentence and on the
|
|
448
|
+
// `tokens`/`usage`/`session` precedent that null-with-the-key-present is this record's normal case.
|
|
449
|
+
//
|
|
450
|
+
// ADMISSIBILITY, which has to engage this record's own sentences rather than sidestep them. The
|
|
451
|
+
// PII-free-by-construction property rests on holding NO ATTACKER-CHOSEN STRING, and `host`
|
|
452
|
+
// satisfies that absolutely: there is no path from a webhook payload, an issue body, a branch name
|
|
453
|
+
// or a folder to this value. It is fixed once at boot from one environment variable against a
|
|
454
|
+
// charset that excludes `/` and `\`, so it cannot even be path-shaped -- which is the property
|
|
455
|
+
// `targetFor` drops a local folder to its basename to get.
|
|
456
|
+
//
|
|
457
|
+
// What it is NOT is anonymous, and that is worth writing down rather than glossing. The default is
|
|
458
|
+
// `os.hostname()`, and on a personal machine a hostname is often a person's name. The honest word
|
|
459
|
+
// is OPERATOR-DISCLOSED: the operator names their own machines, this record is written to their own
|
|
460
|
+
// disk, that name is already on every packet the machine sends, and `PI_WORKER_NAME` is the
|
|
461
|
+
// documented answer for anyone who wants something else here.
|
|
462
|
+
//
|
|
463
|
+
// Passed in rather than read from a module-level value, so `buildRecord` stays pure and every
|
|
464
|
+
// existing caller keeps getting `null` without knowing this field exists.
|
|
465
|
+
host,
|
|
444
466
|
};
|
|
445
467
|
}
|
|
446
468
|
|
package/src/schedules.mjs
CHANGED
|
@@ -23,7 +23,7 @@ import { parseTriggers } from "./triggers.mjs";
|
|
|
23
23
|
* valid deployment. `readFileSync`/`existsSync` are injectable so tests exercise the full path with no
|
|
24
24
|
* real filesystem.
|
|
25
25
|
*/
|
|
26
|
-
export function loadSchedules(config, { readFileSync = fsReadFileSync, existsSync = fsExistsSync } = {}) {
|
|
26
|
+
export function loadSchedules(config, { readFileSync = fsReadFileSync, existsSync = fsExistsSync, fleet = false } = {}) {
|
|
27
27
|
const path = config.triggersFile;
|
|
28
28
|
if (path === null || path === undefined) return []; // cron disabled
|
|
29
29
|
|
|
@@ -33,14 +33,77 @@ export function loadSchedules(config, { readFileSync = fsReadFileSync, existsSyn
|
|
|
33
33
|
|
|
34
34
|
const triggers = parseTriggers(readFileSync(path, "utf8"), path);
|
|
35
35
|
|
|
36
|
-
return triggers.filter((t) => t.on.type === "cron").map((t) => normalizeCronSchedule(t, path, existsSync));
|
|
36
|
+
return triggers.filter((t) => t.on.type === "cron").map((t) => normalizeCronSchedule(t, path, existsSync, fleet));
|
|
37
37
|
}
|
|
38
38
|
|
|
39
|
-
|
|
39
|
+
/**
|
|
40
|
+
* The cron set AS AUTHORED, before any placement decision (issue #57).
|
|
41
|
+
*
|
|
42
|
+
* This is the object two hosts have to agree about, and it is deliberately not `loadSchedules`'s output.
|
|
43
|
+
* That function resolves PLACEMENT -- it replaces every trigger whose folder is on another machine with
|
|
44
|
+
* a stub -- so its result differs per host BY CONSTRUCTION. Fingerprinting it would make every correctly
|
|
45
|
+
* configured fleet refuse itself forever: mini1 owns `/a`, mini2 owns `/b`, their sets never match, and
|
|
46
|
+
* neither ever reconciles again. What they share is the FILE, so the file is what gets hashed.
|
|
47
|
+
*
|
|
48
|
+
* Pure and fs-free apart from the read: no `existsSync`, because existence is exactly the question that
|
|
49
|
+
* makes two honest hosts differ.
|
|
50
|
+
*/
|
|
51
|
+
export function authoredCron(config, { readFileSync = fsReadFileSync, existsSync = fsExistsSync } = {}) {
|
|
52
|
+
const path = config.triggersFile;
|
|
53
|
+
if (path === null || path === undefined) return null; // cron disabled: no opinion at all (see cronFingerprint)
|
|
54
|
+
if (!existsSync(path)) return null;
|
|
55
|
+
try {
|
|
56
|
+
return parseTriggers(readFileSync(path, "utf8"), path)
|
|
57
|
+
.filter((t) => t.on.type === "cron")
|
|
58
|
+
.map((t) => ({ schedulerId: t.on.id, pattern: t.on.pattern, run: t.run }));
|
|
59
|
+
} catch {
|
|
60
|
+
// A file this host cannot parse is not an opinion about what should be scheduled. It refuses boot
|
|
61
|
+
// elsewhere and keeps last-good on reload; here it must not become a fingerprint that disagrees
|
|
62
|
+
// with every peer.
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Split a schedule set into the triggers THIS host serves and the ones it does not (issue #57).
|
|
69
|
+
*
|
|
70
|
+
* `loadSchedules` already refused everything a pure validator could refuse and everything the filesystem
|
|
71
|
+
* could answer for a trigger this host owns. What is left is the third question, and it is the one Gap 2
|
|
72
|
+
* is about: a folder that is not here is not necessarily a mistake, it may simply be another machine's.
|
|
73
|
+
*
|
|
74
|
+
* Exported so the split is testable without an fs, and so a caller can report what it will not be running.
|
|
75
|
+
*/
|
|
76
|
+
export function servedSchedules(schedules) {
|
|
77
|
+
const served = [];
|
|
78
|
+
const unserved = [];
|
|
79
|
+
for (const s of schedules) (s.unserved ? unserved : served).push(s);
|
|
80
|
+
return { served, unserved };
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function normalizeCronSchedule({ on, run }, path, existsSync, fleet) {
|
|
40
84
|
// The pure validator already guaranteed a non-empty, `:`-free, charset-valid, unique id and a
|
|
41
85
|
// well-formed pattern; folder existence is the one fs-dependent check it deferred to here.
|
|
86
|
+
//
|
|
87
|
+
// ON A FLEET THAT IS THE WRONG QUESTION. `INT-TRIGGERS-FILE-CONTRACT` splits this as "type here,
|
|
88
|
+
// reality where it can be known", and issue #57 adds a third level: PLACEMENT, where the fleet is
|
|
89
|
+
// known. A folder that is absent on THIS machine may simply belong to another one, and refusing the
|
|
90
|
+
// worker's boot for it takes every unrelated trigger -- every forge job, every other folder -- offline
|
|
91
|
+
// with it. That is the sentence #57's own acceptance forbids.
|
|
92
|
+
//
|
|
93
|
+
// `fleet` is `PI_WORKER_NAME` being DECLARED, deliberately, and not a registry read. Two reasons, and
|
|
94
|
+
// both are failures I would otherwise have shipped. A registry read here would make a fleet-wide
|
|
95
|
+
// restart into a fleet-wide boot refusal, because every host would come up seeing no peers yet. And it
|
|
96
|
+
// would have to happen after the Valkey client exists, which is BELOW the four destructive boot sweeps
|
|
97
|
+
// -- so a single-host deployment with one typo'd folder would reap containers, prune history and delete
|
|
98
|
+
// sandboxes on every restart before refusing. Declaring a name is the operator saying "this is a
|
|
99
|
+
// fleet", it is known before anything runs, and it keeps a single-host deployment byte-identical.
|
|
42
100
|
if (!existsSync(run.folder)) {
|
|
43
|
-
|
|
101
|
+
if (!fleet) {
|
|
102
|
+
throw configError(`cron trigger "${on.id}": run.folder does not exist: ${run.folder} (${path})`);
|
|
103
|
+
}
|
|
104
|
+
// Not mine. Its skillsDir is not my business either: `isAbsolute` is OS-dependent, and judging
|
|
105
|
+
// another host's path on my platform is the exact mistake the shared validator refuses to make.
|
|
106
|
+
return { schedulerId: on.id, unserved: "folder-absent" };
|
|
44
107
|
}
|
|
45
108
|
|
|
46
109
|
// `run.skillsDir` gets the same treatment, and for the same reason (REQ-PER-TRIGGER-SKILLS): the pure
|