@edgehero/pi-dispatch 1.6.1 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/start.mjs CHANGED
@@ -1,10 +1,10 @@
1
1
  import { execFile } from "node:child_process";
2
- import { watch } from "node:fs";
2
+ import { readFileSync, watch } from "node:fs";
3
3
  import { dirname, basename, join } from "node:path";
4
4
  import { promisify } from "node:util";
5
5
  import { configError, loadConfig } from "./config.mjs";
6
6
  import { makeRedisClient, parseConnection } from "./connection.mjs";
7
- import { reconcile, reloadSchedules } from "./cron.mjs";
7
+ import { reconcileGated, reloadSchedules } from "./cron.mjs";
8
8
  import { makeGitHubAuth } from "./get-token.mjs";
9
9
  import { makeGitHubHost } from "./github-host.mjs";
10
10
  import { makeGitLabAuth } from "./gitlab-auth.mjs";
@@ -14,8 +14,12 @@ import { makeForgejoHost } from "./forgejo-host.mjs";
14
14
  import { makeAzureAuth } from "./azure-auth.mjs";
15
15
  import { makeAzureHost } from "./azure-host.mjs";
16
16
  import { makeEgressPreflight } from "./egress.mjs";
17
+ import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
18
+ import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
19
+ import { cronFingerprint } from "./fingerprint.mjs";
20
+ import { makeHostRegistry } from "./host-registry.mjs";
17
21
  import { makeImagePreflight } from "./image-preflight.mjs";
18
- import { createWorker } from "./index.mjs";
22
+ import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
19
23
  import { makeCollectChain } from "./outbox.mjs";
20
24
  import { containerPackagePaths, readStageManifest } from "./packages.mjs";
21
25
  import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
@@ -24,19 +28,45 @@ import { makeSandboxReaper } from "./sandbox-store.mjs";
24
28
  import { makeSessionStore } from "./session-store.mjs";
25
29
  import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
26
30
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
27
- import { loadScopedLimits } from "./scoped-limits.mjs";
31
+ import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
28
32
  import { makeWaitChecker } from "./wait-check.mjs";
29
33
  import { makeWaitState } from "./wait-state.mjs";
30
- import { makeQueue } from "./queue.mjs";
34
+ import { hostQueueName, makeQueue } from "./queue.mjs";
31
35
  import { makeRunContainer } from "./run-container.mjs";
32
36
  import { makeSecretsResolver } from "./secrets.mjs";
33
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
37
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
38
+ import { makeRunMirror } from "./run-mirror.mjs";
34
39
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
35
- import { loadSchedules } from "./schedules.mjs";
40
+ import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
36
41
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
37
42
 
38
43
  const exec = promisify(execFile);
39
44
 
45
+ /** How long boot will wait for `docker image inspect` before shipping without a digest. */
46
+ const BOOT_IMAGE_TIMEOUT_MS = 5_000;
47
+
48
+ /**
49
+ * How long a fleet-wide scope claim lives. `JOB_TIMEOUT_MS` plus slack: a container cannot outlive that
50
+ * ceiling, so the claim cannot expire underneath a live job -- which is what makes a refresh unnecessary
51
+ * rather than merely unimplemented. DERIVED from that constant rather than written as a number, so the
52
+ * coupling maintains itself if the timeout ever moves.
53
+ */
54
+ const SCOPE_CLAIM_TTL_MS = JOB_TIMEOUT_MS + 5 * 60 * 1000;
55
+
56
+ /**
57
+ * This worker's own package version, for the registry row (issue #57): a rolling upgrade should be
58
+ * visible as a fact about the fleet rather than as a diff someone has to run. Read once, and never
59
+ * fatal -- an unreadable manifest costs a blank field, not a boot. npm always ships `package.json`
60
+ * whatever `files` says, so this resolves from an installed package as well as from a checkout.
61
+ */
62
+ const WORKER_VERSION = (() => {
63
+ try {
64
+ return JSON.parse(readFileSync(new URL("../package.json", import.meta.url), "utf8")).version ?? "";
65
+ } catch {
66
+ return "";
67
+ }
68
+ })();
69
+
40
70
  /**
41
71
  * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
42
72
  * the new worker starts draining. A leaked container keeps spending, so it must go before any new
@@ -59,7 +89,7 @@ const exec = promisify(execFile);
59
89
  * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
60
90
  * logs and the worker keeps its boot-time schedulers.
61
91
  */
62
- function watchTriggersFile(config, queue, log) {
92
+ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
63
93
  const path = config.triggersFile;
64
94
  const dir = dirname(path) || ".";
65
95
  const file = basename(path);
@@ -68,7 +98,7 @@ function watchTriggersFile(config, queue, log) {
68
98
  watch(dir, (_event, changed) => {
69
99
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
70
100
  clearTimeout(timer);
71
- timer = setTimeout(() => void reloadSchedules(config, queue, { log }), 150);
101
+ timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
72
102
  }).unref?.();
73
103
  log("triggers_watching", { path });
74
104
  } catch (err) {
@@ -170,8 +200,17 @@ export function makeReaper({ log }) {
170
200
  log("reaped_network", { network: net });
171
201
  } catch {} // still in use, or already gone -- either way not this boot's problem
172
202
  }
203
+ // Whether the enumeration HAPPENED, which the scope-claim sweep depends on: it may only delete a
204
+ // claim naming this host once this host has actually established that it holds no containers.
205
+ return { reaped: true };
173
206
  } catch (err) {
207
+ // The `docker ps` is inside this try, so this path CANNOT establish that this host holds no
208
+ // containers -- whether it failed before listing anything or after reaping some and then losing
209
+ // the daemon. Either way the claim "I hold nothing" is unproven, and sweeping on it would free
210
+ // slots for containers that may STILL BE RUNNING, letting another host start more alongside
211
+ // them: a money overrun rather than a tidy-up. Conservative in the only safe direction.
174
212
  log("reaper_skipped", { reason: err?.message });
213
+ return { reaped: false };
175
214
  }
176
215
  };
177
216
  }
@@ -200,11 +239,14 @@ export async function startWorker(
200
239
  makeReaper: makeReaperFn = makeReaper,
201
240
  makeLogSink: makeLogSinkFn = makeLogSink,
202
241
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
242
+ makeRunMirror: makeRunMirrorFn = makeRunMirror,
203
243
  makeLogReaper: makeLogReaperFn = makeLogReaper,
204
244
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
205
245
  makeRunContainer: makeRunContainerFn = makeRunContainer,
206
246
  makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
207
247
  makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
248
+ makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
249
+ makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
208
250
  makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
209
251
  makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
210
252
  makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
@@ -215,12 +257,23 @@ export async function startWorker(
215
257
  } = {},
216
258
  ) {
217
259
  const config = loadConfig(env);
218
- const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields })}\n`);
260
+ // `host` sits AFTER the spread, so it is authoritative rather than overridable (issue #57). No call
261
+ // site can know better than this closure which process wrote a line, and one that passed `host` would
262
+ // be lying by construction -- verified: none does. This is also why the stamp lives ONLY here. Every
263
+ // other module takes `log` injected, and two tests pin the KEY SET of the fields object handed to an
264
+ // injected log (`run_record_failed`, `wait_check`); a `host` added at any call site would break them,
265
+ // while one added inside this closure cannot reach them.
266
+ const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
219
267
 
220
268
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
221
269
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
222
270
  // than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
223
- const schedules = loadSchedules(config);
271
+ // A mutable ref, like its `pauseWindows` and `scopedLimits` siblings and for a reason this one only
272
+ // acquired with issue #57: the heartbeat fingerprints what this host CURRENTLY believes should be
273
+ // scheduled, and a `const` frozen at boot would make it publish the pre-edit set forever -- so two
274
+ // hosts would see each other's fingerprint oscillate on the beat period, refusing or agreeing
275
+ // depending on which half of a beat a reload happened to land in.
276
+ const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared }) };
224
277
 
225
278
  // REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
226
279
  // Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
@@ -290,8 +343,10 @@ export async function startWorker(
290
343
 
291
344
  // Clear strays left by a previous crash before the worker starts draining. Best-effort: the reaper
292
345
  // swallows its own docker errors; this guard keeps any reaper failure from blocking boot.
346
+ // Whether the container reaper actually ENUMERATED, which the scope-claim sweep below depends on.
347
+ let reaped = false;
293
348
  try {
294
- await makeReaperFn({ log })();
349
+ reaped = (await makeReaperFn({ log })())?.reaped === true;
295
350
  } catch (err) {
296
351
  log("reaper_skipped", { reason: err?.message });
297
352
  }
@@ -324,12 +379,42 @@ export async function startWorker(
324
379
  // One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
325
380
  // hoisted out of the createWorkerFn arg object.
326
381
  const redis = makeRedisClient(config.valkeyUrl);
382
+
383
+ // This host's own stale scope claims, gated on the reaper having having enumerated: the
384
+ // reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
385
+ // host is a claim for a container that no longer exists. Deleting it is not a second source of truth --
386
+ // it is the SAME source writing down what it just established, which is what answers `OQ-008` here.
387
+ // Best-effort and double-wrapped like every other boot sweep: an OPTIMISATION over the TTL, never the
388
+ // mechanism, so a fault costs one TTL of a stale claim and never a boot.
389
+ try {
390
+ if (config.workerNameDeclared)
391
+ await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current.map((r) => ({ concurrent: r.concurrent, hash: scopeKeyPrefix(r.scope).slice("budget:s:".length) })), log })({ reaped });
392
+ } catch (err) {
393
+ log("scope_claims_sweep_skipped", { reason: err?.message });
394
+ }
395
+
327
396
  // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
328
397
  // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
329
398
  // Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
330
399
  // drains it after the worker.
331
400
  const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
332
401
 
402
+ // THE HOST QUEUE (issue #57): work only this machine can do, because the folder lives here.
403
+ //
404
+ // Armed by the operator DECLARING a name, not by a peer appearing. Two reasons, and the first is
405
+ // decisive: which queue a job is enqueued to is a routing decision made by whoever enqueues it, so it
406
+ // cannot be allowed to flip underneath a running deployment when a second host happens to register --
407
+ // a cron scheduler upserted on one queue and pruned from another is exactly the mutual teardown this
408
+ // issue exists to stop. And a second BullMQ Worker is a second blocking connection, which a single-host
409
+ // deployment should not pay for silently. Declaring a name IS the multi-host declaration; `doctor`
410
+ // warns when peers exist and nobody has made it.
411
+ const hostQueue = config.workerNameDeclared ? hostQueueName(config.workerName) : null;
412
+ // The long-lived handle the cron watcher reloads through. Its own when a host queue is armed, so a
413
+ // live triggers-file edit lands on the same queue the boot reconcile used; otherwise the shared
414
+ // runtime queue, exactly as before. Registered as an extraCloser only when it is a NEW handle --
415
+ // closing `runtimeQueue` twice would be closing another owner's connection.
416
+ const cronQueue = hostQueue ? makeQueue(parseConnection(config.valkeyUrl), { name: hostQueue }) : runtimeQueue;
417
+
333
418
  // REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
334
419
  // captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
335
420
  // `.json` record via recordRun is ALWAYS on, so every run leaves a stable, non-PII trace regardless.
@@ -364,8 +449,22 @@ export async function startWorker(
364
449
  // that can neither disarm nor pre-spend-check.
365
450
  const onceTriggersFile = env.PI_TRIGGERS_FILE ?? join(process.cwd(), "triggers.json");
366
451
  const disarmOnce = makeDisarmOnce({ triggersPath: onceTriggersFile, log });
452
+ // The fleet-visible copy of the run history (issue #57, Gap 3). Armed only on a deployment that declared
453
+ // a worker name: an unnamed one is a single host, its own files ARE the whole history, and a mirror
454
+ // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
455
+ // no job then issues a single extra Valkey command.
456
+ const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
367
457
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
368
- writeRecord(buildRecord({ job, result, error, startedAt, endedAt }));
458
+ // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
459
+ // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
460
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName });
461
+ writeRecord(record);
462
+ // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
463
+ // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
464
+ // and a view that can outlive its source is a second source of truth. Not awaited, because this is
465
+ // the job's own completion path: a slow Valkey may cost a row in a panel and must never hold up a
466
+ // job that has already finished and already been written to disk. `mirror` never rejects.
467
+ void runMirror?.mirror(record, sanitizeJobId(record.jobId));
369
468
  // Strictly AFTER the durable record: "fired" means "produced a run record", and the crash
370
469
  // direction this ordering buys is the chosen one -- an armed one-shot with a record, never a
371
470
  // disarm before writeRecord RETURNED. Returned, not succeeded: the record writer swallows fs
@@ -404,10 +503,13 @@ export async function startWorker(
404
503
  const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
405
504
 
406
505
  // INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
407
- // enqueues chained children onto runtimeQueue via enqueueLocalJob -- the same pi-jobs queue the stall
408
- // guard tears down through. Never throws, so a chain fault cannot flip a completed parent
506
+ // enqueues chained children onto the CRON queue via enqueueLocalJob -- this host's own when one is
507
+ // armed, since a chained child continues the working tree this machine just used. Never throws, so a chain fault cannot flip a completed parent
409
508
  // (CONST-RETRY-INFRA-ONLY). The processor calls it as the sole COMPLETED-path chain step.
410
- const collectChain = makeCollectChain({ queue: runtimeQueue, config, log });
509
+ // Onto the HOST queue when one is armed. A chained child is same-folder and local-parent-only
510
+ // (`OQ-009`), so the working tree it needs is the one this machine just used: routing it anywhere
511
+ // else would enqueue a job only this host can run onto a queue every host drains.
512
+ const collectChain = makeCollectChain({ queue: cronQueue, config, log });
411
513
 
412
514
  // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest at EACH job start, like
413
515
  // getSettings above and the pause-window ref below.
@@ -455,13 +557,83 @@ export async function startWorker(
455
557
  if (config.globalPiDir && !readStageManifest({ globalPiDir: config.globalPiDir })) log("packages_manifest_absent", { overlay: config.globalPiDir });
456
558
  getPackagePaths();
457
559
 
560
+ // Issue #57. Published before the worker starts draining, so a peer that boots a moment later sees this
561
+ // host rather than an empty fleet. The image identity rides the SAME preflight the job path uses -- one
562
+ // inspect implementation, one format string -- called once here with an empty job, which resolves the
563
+ // deployment default and trips none of the per-job label gates.
564
+ //
565
+ // The boot line and the registry may cache this where the GATE may not, and the distinction is the whole
566
+ // argument: a gate that caches gives a WRONG DECISION when an operator builds or removes an image
567
+ // mid-day, which is why `imagePreflight` is deliberately not memoised below. A heartbeat that caches
568
+ // gives a STALE ROW, and nothing reads a row to decide anything.
569
+ // ONE preflight instance, constructed once and shared: `start-wiring.test.mjs` pins that, and the
570
+ // reason is the module's own -- the tag the preflight checked has to be the tag `docker run` is
571
+ // handed, and two constructions are two chances for that to stop being true.
572
+ const imagePreflight = makeImagePreflightFn({ image: config.jobImage });
573
+ // BOUNDED, because `.catch()` cannot rescue a promise that never settles: `runDocker` resolves only on
574
+ // the child's `close` or `error` and has no timeout of its own, so a wedged daemon would hang boot
575
+ // here. This read is a nicety -- a digest for the boot line and the registry -- and a nicety may
576
+ // never be able to stop a worker starting. The per-JOB preflight keeps its unbounded wait, where a
577
+ // wedged daemon is the job's problem and the 30-minute job timeout already covers it.
578
+ const bootImage = await Promise.race([
579
+ imagePreflight({}).catch(() => ({})),
580
+ new Promise((resolve) => setTimeout(() => resolve({}), BOOT_IMAGE_TIMEOUT_MS).unref?.()),
581
+ ]);
582
+ // Resolved once: `Intl` is not free, and this value cannot change without a restart.
583
+ const hostTz = Intl.DateTimeFormat().resolvedOptions().timeZone ?? "";
584
+ const registry = makeHostRegistryFn({ redis, name: config.workerName, log });
585
+ // NOT awaited, and that is load-bearing rather than an optimisation. `makeRedisClient` sets
586
+ // `maxRetriesPerRequest: null` -- required for BullMQ's blocking connections -- which means a command
587
+ // issued against an unreachable server QUEUES FOREVER instead of rejecting. Awaiting the first beat
588
+ // would therefore hang boot indefinitely on a deployment whose Valkey is down, turning a telemetry
589
+ // keyspace into a boot dependency. The registry is never on a decision path, so a worker that comes
590
+ // up before its own row does is correct: the row appears when Valkey does.
591
+ void registry.start({
592
+ version: WORKER_VERSION,
593
+ image: config.jobImage,
594
+ imageDigest: bootImage.imageDigest ?? "",
595
+ piVersion: bootImage.piVersion ?? "",
596
+ // A thunk, because the spec says this row carries the LIVE slot count and the overlay can lower it
597
+ // mid-run through `dispatch_set`. A literal here would publish the boot value forever.
598
+ concurrency: () => worker?.concurrency ?? bootConcurrency,
599
+ pid: process.pid,
600
+ // Whether this host DRAINS a queue of its own. Every worker publishes a row; only a host that declared
601
+ // a name has somewhere for routed work to go, and a reader must not invent a queue for one that has not.
602
+ routes: config.workerNameDeclared,
603
+ // What this host can serve that another might not (issue #57, `OQ-032`): the secret and wait profiles
604
+ // it has declared. NAMES only, never the resolver paths behind them -- a path is PII on Windows and
605
+ // operator topology everywhere, and the receiver only needs to know WHICH host, not what it runs.
606
+ //
607
+ // Recomputed per beat rather than frozen at boot, for the reason the digest is: a host that gains a
608
+ // profile on restart must start attracting that work within one beat, and one that loses it must stop.
609
+ caps: () => serializeCaps(capabilityTokens(config)),
610
+ // The host's IANA zone, because a cron PATTERN carries none: `triggers.json` has no `tz` field and
611
+ // BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's LOCAL time.
612
+ // On one host that is exactly what an operator means; on two in different zones the same pattern is
613
+ // two different instants, and nothing anywhere says so. Published now so a later slice can refuse.
614
+ tz: hostTz,
615
+ // The cron fingerprint rides every beat from the LIVE ref, so a peer always compares against what
616
+ // this host believes now rather than what it believed at boot. `null` means abstain: cron disabled
617
+ // here is no opinion at all, and such a host must never be able to disagree with one that has one.
618
+ fpCron: () => cronFingerprint(authoredCron(config), { tz: hostTz }) ?? "",
619
+ cronCount: () => schedules.current.length,
620
+ });
621
+
622
+
458
623
  const worker = createWorkerFn({
459
624
  connection: parseConnection(config.valkeyUrl),
625
+ hostQueue,
626
+ // Names the BullMQ Worker, which makes `getWorkers()` rows tell hosts apart -- bullmq appends
627
+ // `:w:<name>` to the client name and `moveToActive` stamps `processedBy` onto each active job's
628
+ // hash, so per-job host attribution arrives for free. A NICETY on top of the registry and never the
629
+ // source of truth: that call rests on CLIENT SETNAME, which bullmq's own doc-comment says some
630
+ // providers do not support, and a host list that silently empties cannot be what a decision reads.
631
+ name: config.workerName,
460
632
  concurrency: bootConcurrency,
461
633
  getSettings,
462
634
  redis,
463
635
  recordRun,
464
- extraClosers: [runtimeQueue],
636
+ extraClosers: [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])],
465
637
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
466
638
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
467
639
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
@@ -475,7 +647,28 @@ export async function startWorker(
475
647
  waitState: makeWaitState({ redis }),
476
648
  // The polled tier's bounds, read per pickup from config so they are one value with one home. The
477
649
  // slot count is a CEILING the gate clamps against the live concurrency, never the final number.
650
+ // The fleet-wide half of the wait-check bound (issue #57), armed on the same predicate as the host
651
+ // queue: declaring a name is declaring a fleet. Its TTL is DERIVED rather than guessed -- the gate
652
+ // holds the lease across every profile in turn, each bounded by the check timeout, so one timeout
653
+ // per profile plus one for the overhead between them.
654
+ // The fleet-wide half of a scoped `concurrent` ceiling. Its TTL is DERIVED rather than guessed, and
655
+ // derived is what makes a heartbeat unnecessary: `JOB_TIMEOUT_MS` is a hard 30-minute ceiling on how
656
+ // long any container can run, so a TTL above it cannot expire underneath a live job -- which is the
657
+ // failure that would matter, because it would let another host start a second container on a scope
658
+ // the operator limited to one. Nothing refreshes this claim, deliberately: a refresher would be a
659
+ // second thing to get wrong for a window that cannot be reached.
660
+ scopeLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: scopeSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
661
+ checkLease: hostQueue
662
+ ? makeFleetLease({
663
+ redis,
664
+ holderPrefix: config.workerName,
665
+ keyFor: checkSlotKey,
666
+ ttlMs: config.waitCheckTimeoutMs, // a floor; the gate passes the real one, derived from the profile count
667
+ log,
668
+ })
669
+ : null,
478
670
  checkSlotCount: () => config.waitCheckSlots,
671
+ checkTimeoutMs: () => config.waitCheckTimeoutMs,
479
672
  intervalMs: () => config.waitIntervalMs,
480
673
  maxWaitMs: () => config.waitMaxMs,
481
674
  maxChecks: () => config.waitMaxChecks,
@@ -523,7 +716,7 @@ export async function startWorker(
523
716
  // cache would be wrong in both directions -- an operator who builds the image mid-day would stay refused,
524
717
  // one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
525
718
  // boot because it is deploy-time state under a :ro mount; the host's image set is not.
526
- imagePreflight: makeImagePreflightFn({ image: config.jobImage }),
719
+ imagePreflight,
527
720
  // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
528
721
  // value, one place, so the gate that checks the proxy and the runner that attaches to its network
529
722
  // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
@@ -636,10 +829,13 @@ export async function startWorker(
636
829
  // job-image-missing), never
637
830
  // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
638
831
  // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
639
- worker.on("completed", (job, result) =>
832
+ // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
833
+ // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
834
+ const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
835
+ for (const w of allWorkers) w.on("completed", (job, result) =>
640
836
  log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
641
837
  );
642
- worker.on("failed", (job, err) =>
838
+ for (const w of allWorkers) w.on("failed", (job, err) =>
643
839
  log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
644
840
  );
645
841
 
@@ -649,31 +845,45 @@ export async function startWorker(
649
845
  const guard = makeStallGuard({
650
846
  redis,
651
847
  threshold: config.schedulerStallMax,
652
- removeJobScheduler: (id) => runtimeQueue.removeJobScheduler(id),
848
+ // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
849
+ // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
850
+ removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
653
851
  log,
654
852
  });
655
- worker.on("stalled", (jobId) => void guard.onStalled(jobId));
853
+ for (const w of allWorkers) w.on("stalled", (jobId) => void guard.onStalled(jobId));
656
854
 
657
855
  // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
658
856
  // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
659
857
  // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
660
- if (schedules.length > 0) {
661
- const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }));
858
+ // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
859
+ // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
860
+ // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
861
+ // right place to catch, because it can ask the registry and this cannot.
862
+ const { served, unserved } = servedSchedules(schedules.current);
863
+ for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
864
+
865
+ if (served.length > 0) {
866
+ // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
867
+ // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
868
+ // config" is correct again by construction and two hosts can no longer prune each other at all. The
869
+ // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
870
+ // disagreement, which no queue split can detect.
871
+ const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
662
872
  try {
663
- const r = await reconcile(rq, schedules, { log });
664
- log("schedules_installed", { installed: r.installed, removed: r.removed });
873
+ const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
874
+ log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
665
875
  } finally {
666
876
  await rq.close().catch(() => {});
667
877
  }
668
878
  } else {
669
- log("schedules_installed", { installed: 0, removed: 0 });
879
+ log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
670
880
  }
671
881
 
672
882
  // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
673
883
  // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
674
884
  // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
675
885
  if (config.triggersFile) {
676
- watchTriggersFile(config, runtimeQueue, log);
886
+ watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
677
887
  }
678
888
 
679
889
  // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
@@ -689,6 +899,8 @@ export async function startWorker(
689
899
 
690
900
  log("worker_started", {
691
901
  queue: "pi-jobs",
902
+ host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
903
+ imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
692
904
  concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
693
905
  dailyCap: config.dailyCap,
694
906
  weeklyCap: config.weeklyCap, // null when the weekly window is disabled