@edgehero/pi-dispatch 1.6.1 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,279 @@
1
+ /**
2
+ * The `host:` keyspace: which workers are alive, and what each one is (issue #57).
3
+ *
4
+ * Every gap in the multi-host issue wants the same missing fact -- WHICH HOSTS EXIST AND WHAT CAN EACH
5
+ * ONE DO -- so one structure serves all of them rather than each growing its own. This module is that
6
+ * structure and nothing else. It carries no configuration and no authority: every row is one host's
7
+ * SELF-DESCRIPTION, published by that host, and no reader may write another's.
8
+ *
9
+ * host:live SET of worker names -- an index the READER prunes, never a source of truth
10
+ * host:h:<name> HASH of that host's self-description, PEXPIRE'd on every beat
11
+ *
12
+ * WHY THIS IS NOT THE REDIS STATE `OQ-008` AND `DES-CONCURRENCY-3` REFUSED. That refusal is about a
13
+ * claim whose truth-maker lives on the host while the claim lives in Redis -- an in-flight count
14
+ * asserting a container the boot reaper had just killed, with no way for the two to notice they
15
+ * disagree. Here the truth-maker IS the host process, and the refresh IS the claim: when the process
16
+ * dies the claim stops being renewed and expires on its own. There is no reaper to contradict and no
17
+ * second authority to drift from.
18
+ *
19
+ * The sharper test, and the one to apply to anything added here later: DELETE THE WHOLE `host:*`
20
+ * KEYSPACE WHILE THE FLEET RUNS, and every host must behave exactly as it does today. That holds
21
+ * because nothing here is consulted to decide anything a single-host deployment decides differently --
22
+ * a reader that cannot read treats absence as "no peers", which is v1.6.1's behaviour. A Redis-side
23
+ * toggle fails that test: deleting it loses the operator's edit. Anything that would fail it does not
24
+ * belong in this keyspace.
25
+ *
26
+ * THE CONTENT RULE, which is a contract and not a style note: names, integers and digests. Never a
27
+ * path, never a URL with credentials, never a repository name, never operator free text. This is
28
+ * `targetFor`'s `local:<basename>` discipline applied to a Valkey value, and it earns its strictness
29
+ * from the reader rather than the writer -- the panel is where an operator screenshots, and doctor
30
+ * prints these rows. A value that must be carried but cannot satisfy the rule is HASHED before it gets
31
+ * here (`scopeKeyPrefix`'s idiom), never abbreviated.
32
+ */
33
+
34
+ /** The index. A SET cannot expire its members, so the leak is handled by the reader, as `wait:held` does. */
35
+ export const HOST_SET = "host:live";
36
+
37
+ /** One host's row. `h:` is a sub-namespace letter under one prefix, as `budget:` uses `w:`/`m:`/`t:`/`s:`. */
38
+ export const hostKey = (name) => `host:h:${name}`;
39
+
40
+ /** How often a live worker republishes. Cheap: three writes, off every job path. */
41
+ export const HOST_BEAT_MS = 15_000;
42
+
43
+ /**
44
+ * How long a row outlives its last beat. SIX missed beats, not one or two: a heartbeat competes with the
45
+ * event loop of a process whose whole job is spawning containers, so a single late beat is ordinary and
46
+ * must not evict a healthy host. Ninety seconds is short enough that a crashed host stops being counted
47
+ * within one panel refresh cycle of an operator noticing anything at all.
48
+ */
49
+ export const HOST_TTL_MS = 90_000;
50
+
51
+ /**
52
+ * Build the registry accessor. `redis` is the client the budget and the wait state already share.
53
+ *
54
+ * EVERY METHOD FAILS OPEN, and the direction is the design: a registry fault must be able to cost a
55
+ * panel row and must never be able to invent a refusal. Nothing here is on the paid path, so there is no
56
+ * case where throwing would be more correct than continuing.
57
+ */
58
+ /**
59
+ * How long any single registry command may take before it is treated as a failure.
60
+ *
61
+ * THIS BOUND IS THE WHOLE FAIL-OPEN MECHANISM, and getting it wrong is subtle enough to be worth the
62
+ * paragraph. `makeRedisClient` sets `maxRetriesPerRequest: null`, which BullMQ's blocking connections
63
+ * require and which means a command issued against an unreachable server does NOT reject -- it QUEUES,
64
+ * forever. So a `try/catch` around it catches nothing: the failure mode of this client is a HANG, and a
65
+ * hang is not an exception. Every `await` in this module therefore goes through `bounded`, which converts
66
+ * "never answers" into "answered no" so the catch below can do its job.
67
+ */
68
+ const REGISTRY_OP_TIMEOUT_MS = 2_000;
69
+
70
+ /** Reject rather than wait forever. See REGISTRY_OP_TIMEOUT_MS for why this is not optional here. */
71
+ function bounded(promise, ms = REGISTRY_OP_TIMEOUT_MS) {
72
+ return new Promise((resolve, reject) => {
73
+ // NOT unref'd, deliberately. An unref'd timer does not fire when nothing else is holding the event
74
+ // loop, so the bound would silently stop existing in exactly the situation it is for -- a shutdown
75
+ // where the hung command is the last thing running. The cost is that a pending call can delay exit
76
+ // by at most this timeout, which is the point: bounded, and short.
77
+ const t = setTimeout(() => reject(new Error("registry timeout")), ms);
78
+ Promise.resolve(promise).then(
79
+ (v) => (clearTimeout(t), resolve(v)),
80
+ (e) => (clearTimeout(t), reject(e)),
81
+ );
82
+ });
83
+ }
84
+
85
+ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs = HOST_TTL_MS, log = () => {}, timeoutMs = REGISTRY_OP_TIMEOUT_MS }) {
86
+ let timer = null;
87
+ let facts = {};
88
+ // Set the moment `close` is entered, so a beat already in flight cannot re-HSET the row AFTER the DEL
89
+ // and recreate the very ghost `close` exists to remove.
90
+ let closed = false;
91
+ // Logged once per TRANSITION rather than once per beat: at four beats a minute, a Valkey outage would
92
+ // otherwise drown the log it is meant to serve. `notePackageKey` sets this precedent and states it.
93
+ let reachable = true;
94
+
95
+ const write = async (fields) => {
96
+ const key = hostKey(name);
97
+ const flat = [];
98
+ // A fact may be a VALUE or a THUNK. A thunk is what lets a fact that changes without a restart --
99
+ // the cron fingerprint after a live triggers-file edit, the live concurrency after an overlay
100
+ // change -- be re-read on every beat rather than frozen at the call that started the heartbeat.
101
+ // Without it a peer would compare against what this host believed at boot, and two hosts would see
102
+ // each other's fingerprint oscillate on the beat period after any edit.
103
+ for (const [k, v] of Object.entries(fields)) {
104
+ let resolved;
105
+ try {
106
+ resolved = typeof v === "function" ? v() : v;
107
+ } catch {
108
+ resolved = ""; // a fact that cannot be computed is absent, never a reason to skip the beat
109
+ }
110
+ const value = String(resolved ?? "");
111
+ // THE CONTENT RULE, ENFORCED HERE rather than asserted in a test. A test can only check the fields
112
+ // it happens to publish itself, so it could never catch the day someone adds `logsDir` at a call
113
+ // site. This can: a value that is path-shaped is DROPPED and named, loudly, once.
114
+ //
115
+ // The rule is mechanical rather than a blanket ban on `/`, because one legitimate value contains
116
+ // one: an IANA zone is `Europe/Amsterdam`. What no admissible value has is a filesystem ROOT -- a
117
+ // leading separator, a backslash, or a drive letter.
118
+ if (/^[/\\]|\\|^[A-Za-z]:[/\\]/.test(value)) {
119
+ log("host_registry_field_refused", { host: name, field: k, reason: "path-shaped" });
120
+ continue;
121
+ }
122
+ flat.push(k, value);
123
+ }
124
+ await bounded(redis.hset(key, ...flat), timeoutMs);
125
+ // PEXPIRE ON EVERY BEAT, which REVERSES this project's stated TTL rule ("set the TTL only when the
126
+ // key is first created, so a long window cannot push its expiry forward" -- budget.mjs). The
127
+ // reversal is correct because the OBJECT is different, and the distinction is worth keeping:
128
+ //
129
+ // a COUNTER whose TTL refreshes on every increment stops being a window, which is why the budget
130
+ // keys set it once and why a busy day must not postpone its own reset;
131
+ //
132
+ // a LEASE's expiry IS the liveness claim, so refreshing it is not a leak, it is the mechanism.
133
+ //
134
+ // The in-repo precedent is lease-shaped and two files away: `wait-state.hold` re-PEXPIREs the
135
+ // supersede lease `wait:key:<dedupId>` -- and only after checking the lease is still ours, which is
136
+ // the same ownership check this key gets for free by being named after its only writer.
137
+ await bounded(redis.pexpire(key, ttlMs), timeoutMs);
138
+ await bounded(redis.sadd(HOST_SET, name), timeoutMs);
139
+ };
140
+
141
+ // The beat currently in flight, so `close` can DRAIN before it deletes. A `closed` flag checked at the
142
+ // top of `beat` is not enough on its own: a beat that had already passed that check would land its
143
+ // HSET after the DEL and recreate the very ghost `close` exists to remove.
144
+ let inFlight = null;
145
+
146
+ const beat = async (fields = facts) => {
147
+ if (closed) return;
148
+ facts = { ...facts, ...fields };
149
+ try {
150
+ inFlight = write({ ...facts, name, beatAt: now() });
151
+ await inFlight;
152
+ if (!reachable) {
153
+ reachable = true;
154
+ log("host_registry_restored", { host: name });
155
+ }
156
+ } catch (err) {
157
+ if (reachable) {
158
+ reachable = false;
159
+ log("host_registry_unreachable", { host: name, reason: err?.message });
160
+ }
161
+ } finally {
162
+ inFlight = null;
163
+ }
164
+ };
165
+
166
+ return {
167
+ /** This worker's own name, so a caller never re-derives it from config and gets a different answer. */
168
+ self: () => name,
169
+
170
+ /**
171
+ * Flush this host's current facts NOW, for a caller that must be visible before it reads peers.
172
+ *
173
+ * Takes NO fields, deliberately. It used to merge them, and that quietly destroyed the thunk
174
+ * mechanism it depends on: a caller passing `fpCron` as a computed STRING replaced the closure the
175
+ * heartbeat installed, so every later beat published a frozen value and two hosts' fingerprints
176
+ * could drift apart again. A caller that wants a fact re-read on every beat installs a thunk once,
177
+ * at `start`; a caller that wants it published NOW calls this.
178
+ */
179
+ async publish() {
180
+ await beat();
181
+ },
182
+
183
+ /**
184
+ * Every live host EXCEPT this one. Separate from `readLiveHosts` because "who else is there" is the
185
+ * question every caller actually has, and excluding self at the one place stops each of them doing it
186
+ * differently -- and stops a stale self-row, written by a previous process of this same host, being
187
+ * read as a peer that disagrees with the process that is running now.
188
+ */
189
+ async livePeers() {
190
+ const res = await readLiveHosts(redis, { now, timeoutMs });
191
+ if (res.unreachable) return res;
192
+ return { hosts: res.hosts.filter((h) => h.name !== name) };
193
+ },
194
+
195
+ /**
196
+ * Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
197
+ * `setTimeout` -- and `.unref()`'d so it can never hold the process open, which is the posture the
198
+ * three `fs.watch` watchers already take. `stop` is registered as an extraCloser beside the runtime
199
+ * queue, so a clean shutdown clears it before `process.exit`.
200
+ */
201
+ async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
202
+ if (closed || timer) return; // a second start would leak the first interval
203
+ await beat({ ...fields, startedAt: now() });
204
+ if (closed) return; // close() landed while the first beat was in flight
205
+ timer = setInterval(() => void beat(), intervalMs);
206
+ timer.unref?.();
207
+ },
208
+
209
+ /**
210
+ * Leave. A clean shutdown DELETES the row rather than letting it expire, so the TTL only ever
211
+ * covers a crash and a rolling restart does not leave a ghost peer behind for ninety seconds.
212
+ * `wait-state.release` takes the same posture for the same reason.
213
+ */
214
+ async close() {
215
+ if (closed) return; // idempotent, and the flag is also what stops an in-flight beat resurrecting the row
216
+ closed = true;
217
+ if (timer) clearInterval(timer);
218
+ timer = null;
219
+ // Drain before deleting, bounded like everything else here: an unbounded wait on a beat that is
220
+ // itself hung would be the shutdown hang this module's timeout exists to prevent.
221
+ await bounded(inFlight ?? Promise.resolve(), timeoutMs).catch(() => {});
222
+ try {
223
+ await bounded(redis.del(hostKey(name)), timeoutMs);
224
+ await bounded(redis.srem(HOST_SET, name), timeoutMs);
225
+ } catch {
226
+ // Fail open: the TTL is the backstop, and a ghost row costs a panel line, never a decision.
227
+ }
228
+ },
229
+ };
230
+ }
231
+
232
+ /**
233
+ * Every live host's row, with the index pruned as it is read.
234
+ *
235
+ * THE INDEX LEAK IS HANDLED HERE, BY THE READER, and that is a decision rather than an oversight: a SET
236
+ * cannot expire its members, so a host that dies leaves one behind. The alternative -- no index, and a
237
+ * `SCAN host:h:*` instead -- was measured and refused for the `wait:held` keyspace on exactly this
238
+ * shape: the panel reads every second, and a keyspace scan walks HARDEST on deployments with nothing to
239
+ * show, because the early exit never fires. So the index exists, a member whose hash is gone is stale by
240
+ * definition, and the reader removes it in passing.
241
+ *
242
+ * `{ unreachable }` and `[]` are DIFFERENT ANSWERS and must never be collapsed by a caller, even where
243
+ * one treats them alike: "there are no other hosts" and "I could not find out" differ, and a panel that
244
+ * renders the second as the first tells an operator their fleet is gone when Valkey merely blinked.
245
+ */
246
+ export async function readLiveHosts(redis, { now = () => Date.now(), timeoutMs = REGISTRY_OP_TIMEOUT_MS } = {}) {
247
+ let names;
248
+ try {
249
+ names = await bounded(redis.smembers(HOST_SET), timeoutMs);
250
+ // Inside the try, not after it: a client that answers with something non-iterable would otherwise
251
+ // throw straight out of a function whose contract is that every method fails open.
252
+ if (names !== null && names !== undefined && !Array.isArray(names)) throw new Error("smembers did not return a list");
253
+ } catch (err) {
254
+ return { unreachable: err?.message ?? "registry unreadable" };
255
+ }
256
+ const hosts = [];
257
+ for (const member of names ?? []) {
258
+ try {
259
+ const row = await bounded(redis.hgetall(hostKey(member)), timeoutMs);
260
+ if (!row || Object.keys(row).length === 0) {
261
+ await bounded(redis.srem(HOST_SET, member), timeoutMs).catch(() => {});
262
+ continue;
263
+ }
264
+ const beatAt = typeof row.beatAt === "string" && row.beatAt.trim() !== "" ? Number(row.beatAt) : NaN;
265
+ hosts.push({
266
+ ...row,
267
+ name: row.name || member,
268
+ // Derived rather than stored, so the panel can say "stale 2m" about a row that still lives.
269
+ // A row whose clock is AHEAD of ours reads as 0 rather than negative: the difference is the
270
+ // skew between two machines, which is its own signal and not this field's to report.
271
+ staleMs: Number.isFinite(beatAt) ? Math.max(0, now() - beatAt) : null,
272
+ });
273
+ } catch {
274
+ // One unreadable row degrades that row, never the listing.
275
+ }
276
+ }
277
+ hosts.sort((a, b) => String(a.name).localeCompare(String(b.name)));
278
+ return { hosts };
279
+ }
@@ -70,7 +70,7 @@ export function makeImagePreflight({ image, spawnFn = spawn }) {
70
70
  // become the push race it exists to avoid, with nothing in the run record saying so.
71
71
  const probe = await runDocker(spawnFn, ["image", "inspect", `--format={{.Id}}${FIELD_SEP}${PI_VERSION_TEMPLATE}${FIELD_SEP}${FORGES_TEMPLATE}${FIELD_SEP}${CAPABILITIES_TEMPLATE}`, wanted], true);
72
72
  if (probe.code === 0) {
73
- const [piVersion, forges, capabilities] = parseLabels(probe.stdout);
73
+ const [piVersion, forges, capabilities, imageDigest] = parseLabels(probe.stdout);
74
74
  const kind = job?.kind;
75
75
  // Absent label => ALLOW. The polarity matters and is the opposite of what "declare your
76
76
  // capabilities" suggests: every operator-built image predating this label (OQ-012) declares
@@ -96,7 +96,7 @@ export function makeImagePreflight({ image, spawnFn = spawn }) {
96
96
  if (job?.command !== undefined && !(capabilities ?? []).includes("commands")) {
97
97
  return { commandUnsupported: wanted, declared: capabilities ?? [] };
98
98
  }
99
- return { ok: true, image: wanted, piVersion };
99
+ return { ok: true, image: wanted, piVersion, imageDigest };
100
100
  }
101
101
  if ((await runDocker(spawnFn, ["info"])).code === 0) return { missing: wanted };
102
102
  return { unavailable: wanted };
@@ -135,10 +135,15 @@ function parseLabels(stdout) {
135
135
  // the safe answer on every field. On `capabilities` "safe" means the caller refuses a replica job,
136
136
  // which is the same direction a genuinely unlabelled image goes.
137
137
  const parts = String(stdout ?? "").trim().split(FIELD_SEP);
138
+ // Field 0 is `{{.Id}}`, the image's own digest. It has been fetched on every job since this format
139
+ // string had four fields and was thrown away until issue #57, which needs it to answer "are these two
140
+ // hosts running the same image?" -- a question `OQ-012` records as unanswerable and which turns out to
141
+ // cost nothing to answer, because the inspect that would have asked it already runs.
142
+ const imageDigest = label(parts[0]);
138
143
  const piVersion = label(parts[1]);
139
144
  // A label that is present but parses to nothing usable is treated as ABSENT rather than as an empty
140
145
  // list -- on `forges` the latter would refuse every job on an image whose label was merely malformed.
141
- return [piVersion, list(parts[2]), list(parts[3])];
146
+ return [piVersion, list(parts[2]), list(parts[3]), imageDigest];
142
147
  }
143
148
 
144
149
  /** One comma-separated label value as a non-empty array, or `null` when it declares nothing usable. */
package/src/index.mjs CHANGED
@@ -3,13 +3,16 @@ import { promisify } from "node:util";
3
3
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
4
4
  import { InfraRetry, runJob } from "./processor.mjs";
5
5
  import { targetFor } from "./run-history.mjs";
6
- import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
6
+ import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
7
7
  import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
8
8
  import { makeWaitState } from "./wait-state.mjs";
9
9
 
10
10
  const exec = promisify(execFile);
11
11
 
12
12
  export const QUEUE = "pi-jobs";
13
+
14
+ /** The key the host-wide in-flight count lives under. One machine, one counter, whatever the queue. */
15
+ export const HOST_SLOT_KEY = "host";
13
16
  export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
14
17
  // The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
15
18
  // JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
@@ -60,7 +63,7 @@ const THROTTLE_FLOOR_MS = 11_000;
60
63
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
61
64
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
62
65
  */
63
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
66
+ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
64
67
  return async function processor(job, token, signal) {
65
68
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
66
69
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -230,6 +233,37 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
230
233
  // Declared outside the try below because the branches AFTER it read both.
231
234
  let verdict = null;
232
235
  let checked = null;
236
+ // A SECOND LAYER BENEATH THE FIRST, never a replacement (issue #57). The in-process map above
237
+ // stays exactly as it was and remains the correct per-host duty-cycle bound -- slots x timeout
238
+ // is the most wall-clock THIS worker spends answering questions instead of running jobs. What it
239
+ // cannot bound is the fleet: a held job's wakes land on any host, so `PI_WAIT_CHECK_SLOTS`
240
+ // silently multiplied by host count, and the one symptom the bound has gets QUIETER as you
241
+ // scale out, because multiplication produces fewer denials per host.
242
+ //
243
+ // Null when no peer could exist, so a single-host deployment issues no command at all.
244
+ let fleetSlot = null;
245
+ if (checkLease) {
246
+ // The TTL is DERIVED from what the lease actually guards: the gate holds it across every profile
247
+ // in turn, each bounded by `PI_WAIT_CHECK_TIMEOUT_MS`, so it is one timeout per PROFILE plus one
248
+ // for the overhead between them. Deriving it from the SLOT COUNT instead -- an unrelated
249
+ // quantity -- made a three-profile job at the shipped defaults hold 30s against a 20s lease,
250
+ // so the slot expired mid-check and another host took it while this one was still using it.
251
+ fleetSlot = await checkLease.acquire(job.id, { slots: checkSlotCount(), ttlMs: (profiles.length + 1) * checkTimeoutMs() });
252
+ if (!fleetSlot) {
253
+ // The same outcome as a local denial and the same remedy, so the same cadence and the same
254
+ // event -- with one conditional field, which is what keeps an unarmed deployment's log line
255
+ // byte-identical.
256
+ checkSlots.release(WAIT_CHECK_KEY);
257
+ const denials = await waitState.noteThrottle(job.id, { denied: true });
258
+ if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots: checkSlotCount(), where: "fleet", hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
259
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
260
+ const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
261
+ const delay = wait + Math.floor(wait * 0.1 * random());
262
+ deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots: checkSlotCount(), where: "fleet" });
263
+ await job.moveToDelayed(nowMs + delay, token);
264
+ throw new DelayedError();
265
+ }
266
+ }
233
267
  // THE LEASE IS HELD FROM THE `tryAcquire` ABOVE, so every exit from here down must release it.
234
268
  // The try opens here and not at the check loop, which is where it used to open: the supersede
235
269
  // claim sits between the two, and BOTH of its exits leave -- one returns `wait-superseded`,
@@ -262,6 +296,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
262
296
  if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
263
297
  }
264
298
  } finally {
299
+ await fleetSlot?.release?.();
265
300
  checkSlots.release(WAIT_CHECK_KEY);
266
301
  }
267
302
 
@@ -378,19 +413,78 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
378
413
  // occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
379
414
  // (measured: 301ms of live container overlap through this very processor) unless this gate holds.
380
415
  // Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
416
+ // THE HOST-WIDE SLOT (issue #57), taken before the scope slot and released in the same finally.
417
+ //
418
+ // It exists only when this worker drains a second, host-affine queue. BullMQ's concurrency is per
419
+ // Worker, so two queues at `PI_CONCURRENCY` would run twice the containers -- and that knob bounds a
420
+ // MACHINE (its RAM, its share of the provider's concurrent-stream budget), not a queue. Deferral
421
+ // rather than refusal, at the scope gate's own cadence and for its reason: a full host is transient
422
+ // state, never a verdict about the job (CONST-RETRY-INFRA-ONLY).
423
+ //
424
+ // Before the scope acquire, so a job that cannot run on this machine at all never takes a folder
425
+ // mutex it would immediately have to give back, and so the two releases nest rather than interleave.
426
+ let hostHeld = false;
427
+ if (hostBound) {
428
+ if (!hostBound.slots.tryAcquire(HOST_SLOT_KEY, hostBound.limit())) {
429
+ deps?.log?.("host_busy_deferred", { jobId: job.id, delayMs: SCOPE_BUSY_RECHECK_MS });
430
+ await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
431
+ throw new DelayedError();
432
+ }
433
+ hostHeld = true;
434
+ }
435
+
381
436
  const limits = scopedLimits();
382
437
  const scope = canonicalScope(job.data);
383
438
  let held = false;
439
+ let scopeSlot = null;
384
440
  if (scope) {
385
- if (!inFlight.tryAcquire(scope, concurrencyFor(job.data, limits))) {
441
+ const ceiling = concurrencyFor(job.data, limits);
442
+ if (!inFlight.tryAcquire(scope, ceiling)) {
386
443
  // Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
387
444
  // The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
388
445
  // host path); the delayed count and the job id are what an operator needs to see it.
389
446
  deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
447
+ // The host slot goes back before we defer: `makeInFlight().release` is not idempotent, so a slot
448
+ // held across a deferral would be a slot this machine never gets back.
449
+ if (hostHeld) {
450
+ hostBound.slots.release(HOST_SLOT_KEY);
451
+ hostHeld = false;
452
+ }
390
453
  await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
391
454
  throw new DelayedError();
392
455
  }
393
456
  held = true;
457
+
458
+ // THE FLEET-WIDE HALF of a scoped ceiling (issue #57). A `scoped-limits.json` row's day/week/month
459
+ // caps are already atomic INCRs on shared keys; its `concurrent` was a per-process Map, so it
460
+ // multiplied by host count -- a MONEY bound silently widened by the operator's deployment shape,
461
+ // which `INT-SCOPED-LIMITS-FILE-CONTRACT` calls out as the failure its version rule exists for.
462
+ //
463
+ // LOCAL scopes deliberately never claim, and the reason is not economy. The key is a hash of a
464
+ // PATH STRING, which carries no identity: `/srv/site` on two machines is, in the common case, two
465
+ // different repositories that share a layout convention. A shared claim keyed on that would
466
+ // serialise two genuinely independent working trees and break exactly the deployments this feature
467
+ // exists to enable. Local folders are answered by ROUTING instead -- a folder exists on one host,
468
+ // so its in-process mutex already spans everything it needs to.
469
+ //
470
+ // And an unlimited forge scope never claims either: `concurrencyFor` returns Infinity with no
471
+ // matching row, so a deployment with no scoped-limits file issues no command at all.
472
+ if (scopeLease && job.data?.kind !== "local" && Number.isFinite(ceiling)) {
473
+ scopeSlot = await scopeLease.acquire(job.id, { slots: ceiling, keyArgs: [scopeKeyPrefix(scope).slice("budget:s:".length)] });
474
+ if (!scopeSlot) {
475
+ // The local slot goes back BEFORE we defer: `makeInFlight().release` is not idempotent, so a
476
+ // slot held across a deferral is a slot this host never gets back.
477
+ inFlight.release(scope);
478
+ held = false;
479
+ if (hostHeld) {
480
+ hostBound.slots.release(HOST_SLOT_KEY);
481
+ hostHeld = false;
482
+ }
483
+ deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS, where: "fleet" });
484
+ await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
485
+ throw new DelayedError();
486
+ }
487
+ }
394
488
  }
395
489
 
396
490
  let startedAt;
@@ -423,6 +517,12 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
423
517
  inFlight.release(scope);
424
518
  held = false;
425
519
  }
520
+ if (hostHeld) {
521
+ hostBound.slots.release(HOST_SLOT_KEY);
522
+ hostHeld = false;
523
+ }
524
+ void scopeSlot?.release?.();
525
+ scopeSlot = null;
426
526
  clearTimeout(timer);
427
527
  throw error;
428
528
  }
@@ -517,62 +617,107 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
517
617
  // Release FIRST and never throw (release clamps at zero by construction): a throw here would
518
618
  // mask the job's real error, and a missed release wedges the scope until a worker restart.
519
619
  if (held) inFlight.release(scope);
620
+ if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
621
+ // AWAITED, not fire-and-forget. Two reasons, and the second is the one that bites: an unawaited
622
+ // DEL is dropped by `shutdown`'s `process.exit(0)`, stranding the claim for its whole TTL on a
623
+ // restart -- and the next same-scope job would otherwise race the release, be denied, and sit out a
624
+ // full re-check interval while the slot it wanted went free behind it. The finally is already inside
625
+ // an async function, and `release` never throws.
626
+ await scopeSlot?.release?.();
520
627
  clearTimeout(timer);
521
628
  signal.removeEventListener("abort", onAbort);
522
629
  }
523
630
  };
524
631
  }
525
632
 
526
- export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
527
- let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
528
- const processor = makeProcessor({
529
- cancelJob: (id, reason) => worker.cancelJob(id, reason),
530
- stopContainer: (name) => exec("docker", ["stop", "-t", "5", name]),
531
- redis,
532
- getSettings,
533
- // Late-bound over `worker`: an overlay concurrency change re-binds the live slot count at the next
534
- // job start. Guarded so only an integer that actually differs touches the property.
535
- applyConcurrency: (n) => {
536
- if (Number.isInteger(n) && worker.concurrency !== n) worker.concurrency = n;
537
- },
538
- pauseUntil,
539
- // Undefined pass-throughs take makeProcessor's own defaults (no limits; a fresh per-processor
540
- // in-flight map -- one per worker process, which under DES-CONCURRENCY-3's one-worker-per-daemon
541
- // shape means one per daemon).
542
- scopedLimits,
543
- inFlight,
544
- // Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
545
- // redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
546
- waitState,
547
- afterMaxMs,
548
- // Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
549
- // because the overlay can lower it through `dispatch_set` and a check must never take the last free
550
- // slot from a paid job. Late-bound over `worker` exactly as `applyConcurrency` is, and for the same
551
- // reason: the value it needs does not exist until the Worker is constructed.
552
- checkSlots,
553
- checkSlotCount,
554
- concurrencyNow: concurrencyNow ?? (() => worker?.concurrency ?? concurrency),
555
- intervalMs,
556
- maxWaitMs,
557
- maxChecks,
558
- maxFaults,
559
- deps,
560
- recordRun,
561
- });
562
-
563
- worker = new Worker(QUEUE, processor, {
564
- // maxRetriesPerRequest: null is REQUIRED for BullMQ's blocking connections, or it throws.
565
- connection: { ...connection, maxRetriesPerRequest: null },
566
- concurrency,
567
- maxStalledCount: 0, // a stalled paid job FAILS, never silently re-runs (verified live)
568
- ...(limiter ? { limiter } : {}),
569
- });
633
+ export function createWorker({ connection, name, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
634
+ // One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
635
+ // check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
636
+ // because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
637
+ // the delayed set is gated on each worker's own `Date.now()`, so the fastest clock wins every hop and a
638
+ // job that had to reach another host might never get there.
639
+ const names = hostQueue ? [QUEUE, hostQueue] : [QUEUE];
640
+ const workers = [];
641
+
642
+ // THE HOST-WIDE BOUND, and the reason it has to exist at all. `PI_CONCURRENCY` bounds a HOST -- its RAM
643
+ // and its share of the provider's concurrent-stream budget (DES-CONCURRENCY-3) -- but BullMQ's own
644
+ // concurrency is per Worker, so two Workers at 3 would run six containers. This semaphore restores the
645
+ // bound as a property of the machine. Process memory is still the correct store, for this entry's own
646
+ // unchanged reason: it counts THIS host's containers, and the boot reaper clears survivors before
647
+ // draining. Armed only when a host queue exists, so a single-host deployment builds one Worker and
648
+ // never reaches the acquire.
649
+ const hostBound = hostQueue ? { slots: hostSlots, limit: () => liveConcurrency() } : null;
650
+ const liveConcurrency = () => workers[0]?.concurrency ?? concurrency;
651
+
652
+ for (const queueName of names) {
653
+ let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
654
+ const processor = makeProcessor({
655
+ // Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
656
+ // and the shared handle could not reach it.
657
+ cancelJob: (id, reason) => worker.cancelJob(id, reason),
658
+ stopContainer: (name) => exec("docker", ["stop", "-t", "5", name]),
659
+ redis,
660
+ getSettings,
661
+ // Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
662
+ // next job start, and with two queues both have to move or the host bound and the queue bounds
663
+ // stop agreeing. Guarded so only an integer that actually differs touches the property.
664
+ applyConcurrency: (n) => {
665
+ if (!Number.isInteger(n)) return;
666
+ for (const w of workers) if (w.concurrency !== n) w.concurrency = n;
667
+ },
668
+ pauseUntil,
669
+ // SHARED across both workers, and that sharing is the point rather than an optimisation: the
670
+ // folder mutex, the per-scope ceiling and the wait-check lease all bound the HOST, so two
671
+ // independent maps would double every one of them exactly as two Workers double concurrency.
672
+ scopedLimits,
673
+ inFlight,
674
+ hostBound,
675
+ scopeLease,
676
+ // Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
677
+ // redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
678
+ waitState,
679
+ afterMaxMs,
680
+ // Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
681
+ // because the overlay can lower it through `dispatch_set` and a check must never take the last free
682
+ // slot from a paid job.
683
+ checkSlots,
684
+ checkLease,
685
+ checkSlotCount,
686
+ checkTimeoutMs,
687
+ concurrencyNow: concurrencyNow ?? liveConcurrency,
688
+ intervalMs,
689
+ maxWaitMs,
690
+ maxChecks,
691
+ maxFaults,
692
+ deps,
693
+ recordRun,
694
+ });
695
+
696
+ worker = new Worker(queueName, processor, {
697
+ // maxRetriesPerRequest: null is REQUIRED for BullMQ's blocking connections, or it throws.
698
+ connection: { ...connection, maxRetriesPerRequest: null },
699
+ concurrency,
700
+ maxStalledCount: 0, // a stalled paid job FAILS, never silently re-runs (verified live)
701
+ // Issue #57. Conditional, so a bare createWorker builds a byte-identical options object -- and because
702
+ // bullmq's own matcher accepts both the named and unnamed client-name spellings, naming costs nothing.
703
+ ...(name ? { name } : {}),
704
+ ...(limiter ? { limiter } : {}),
705
+ });
706
+ workers.push(worker);
707
+ }
708
+
709
+ const primary = workers[0];
710
+ // The host-queue worker, for the caller that must register listeners on both. Attached rather than
711
+ // returned as a pair so every existing caller keeps receiving exactly what it received before.
712
+ primary.hostWorker = workers[1] ?? null;
570
713
 
571
714
  const shutdown = async () => {
572
715
  // Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
573
- // worker.close() would wait up to 30 minutes for the container.
574
- await Promise.resolve(worker.cancelAllJobs?.("shutdown")).catch(() => {});
575
- await worker.close();
716
+ // worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
717
+ // registrations would mean two `process.exit(0)` racing, and the second worker's containers would
718
+ // outlive the handler that was meant to stop them.
719
+ for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
720
+ for (const w of workers) await w.close().catch(() => {});
576
721
  // Close auxiliary resources (e.g. a cron scheduler) after the worker drains. Per-item catch
577
722
  // so one failing or absent closer never strands the others or blocks exit -- matches the
578
723
  // swallow posture on cancelAllJobs above.
@@ -586,5 +731,5 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
586
731
  // worker still aborts in-flight jobs and docker-stops their containers rather than orphaning them.
587
732
  if (process.platform === "win32") process.once("SIGBREAK", shutdown);
588
733
 
589
- return worker;
734
+ return primary;
590
735
  }