@edgehero/pi-dispatch 1.6.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/start.mjs CHANGED
@@ -1,10 +1,10 @@
1
1
  import { execFile } from "node:child_process";
2
- import { watch } from "node:fs";
2
+ import { readFileSync, watch } from "node:fs";
3
3
  import { dirname, basename, join } from "node:path";
4
4
  import { promisify } from "node:util";
5
5
  import { configError, loadConfig } from "./config.mjs";
6
6
  import { makeRedisClient, parseConnection } from "./connection.mjs";
7
- import { reconcile, reloadSchedules } from "./cron.mjs";
7
+ import { reconcileGated, reloadSchedules } from "./cron.mjs";
8
8
  import { makeGitHubAuth } from "./get-token.mjs";
9
9
  import { makeGitHubHost } from "./github-host.mjs";
10
10
  import { makeGitLabAuth } from "./gitlab-auth.mjs";
@@ -14,8 +14,11 @@ import { makeForgejoHost } from "./forgejo-host.mjs";
14
14
  import { makeAzureAuth } from "./azure-auth.mjs";
15
15
  import { makeAzureHost } from "./azure-host.mjs";
16
16
  import { makeEgressPreflight } from "./egress.mjs";
17
+ import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
18
+ import { cronFingerprint } from "./fingerprint.mjs";
19
+ import { makeHostRegistry } from "./host-registry.mjs";
17
20
  import { makeImagePreflight } from "./image-preflight.mjs";
18
- import { createWorker } from "./index.mjs";
21
+ import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
19
22
  import { makeCollectChain } from "./outbox.mjs";
20
23
  import { containerPackagePaths, readStageManifest } from "./packages.mjs";
21
24
  import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
@@ -24,19 +27,44 @@ import { makeSandboxReaper } from "./sandbox-store.mjs";
24
27
  import { makeSessionStore } from "./session-store.mjs";
25
28
  import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
26
29
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
27
- import { loadScopedLimits } from "./scoped-limits.mjs";
30
+ import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
28
31
  import { makeWaitChecker } from "./wait-check.mjs";
29
32
  import { makeWaitState } from "./wait-state.mjs";
30
- import { makeQueue } from "./queue.mjs";
33
+ import { hostQueueName, makeQueue } from "./queue.mjs";
31
34
  import { makeRunContainer } from "./run-container.mjs";
32
35
  import { makeSecretsResolver } from "./secrets.mjs";
33
36
  import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
34
37
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
35
- import { loadSchedules } from "./schedules.mjs";
38
+ import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
36
39
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
37
40
 
38
41
  const exec = promisify(execFile);
39
42
 
43
+ /** How long boot will wait for `docker image inspect` before shipping without a digest. */
44
+ const BOOT_IMAGE_TIMEOUT_MS = 5_000;
45
+
46
+ /**
47
+ * How long a fleet-wide scope claim lives. `JOB_TIMEOUT_MS` plus slack: a container cannot outlive that
48
+ * ceiling, so the claim cannot expire underneath a live job -- which is what makes a refresh unnecessary
49
+ * rather than merely unimplemented. DERIVED from that constant rather than written as a number, so the
50
+ * coupling maintains itself if the timeout ever moves.
51
+ */
52
+ const SCOPE_CLAIM_TTL_MS = JOB_TIMEOUT_MS + 5 * 60 * 1000;
53
+
54
+ /**
55
+ * This worker's own package version, for the registry row (issue #57): a rolling upgrade should be
56
+ * visible as a fact about the fleet rather than as a diff someone has to run. Read once, and never
57
+ * fatal -- an unreadable manifest costs a blank field, not a boot. npm always ships `package.json`
58
+ * whatever `files` says, so this resolves from an installed package as well as from a checkout.
59
+ */
60
+ const WORKER_VERSION = (() => {
61
+ try {
62
+ return JSON.parse(readFileSync(new URL("../package.json", import.meta.url), "utf8")).version ?? "";
63
+ } catch {
64
+ return "";
65
+ }
66
+ })();
67
+
40
68
  /**
41
69
  * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
42
70
  * the new worker starts draining. A leaked container keeps spending, so it must go before any new
@@ -59,7 +87,7 @@ const exec = promisify(execFile);
59
87
  * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
60
88
  * logs and the worker keeps its boot-time schedulers.
61
89
  */
62
- function watchTriggersFile(config, queue, log) {
90
+ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
63
91
  const path = config.triggersFile;
64
92
  const dir = dirname(path) || ".";
65
93
  const file = basename(path);
@@ -68,7 +96,7 @@ function watchTriggersFile(config, queue, log) {
68
96
  watch(dir, (_event, changed) => {
69
97
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
70
98
  clearTimeout(timer);
71
- timer = setTimeout(() => void reloadSchedules(config, queue, { log }), 150);
99
+ timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
72
100
  }).unref?.();
73
101
  log("triggers_watching", { path });
74
102
  } catch (err) {
@@ -170,8 +198,17 @@ export function makeReaper({ log }) {
170
198
  log("reaped_network", { network: net });
171
199
  } catch {} // still in use, or already gone -- either way not this boot's problem
172
200
  }
201
+ // Whether the enumeration HAPPENED, which the scope-claim sweep depends on: it may only delete a
202
+ // claim naming this host once this host has actually established that it holds no containers.
203
+ return { reaped: true };
173
204
  } catch (err) {
205
+ // The `docker ps` is inside this try, so this path CANNOT establish that this host holds no
206
+ // containers -- whether it failed before listing anything or after reaping some and then losing
207
+ // the daemon. Either way the claim "I hold nothing" is unproven, and sweeping on it would free
208
+ // slots for containers that may STILL BE RUNNING, letting another host start more alongside
209
+ // them: a money overrun rather than a tidy-up. Conservative in the only safe direction.
174
210
  log("reaper_skipped", { reason: err?.message });
211
+ return { reaped: false };
175
212
  }
176
213
  };
177
214
  }
@@ -205,6 +242,8 @@ export async function startWorker(
205
242
  makeRunContainer: makeRunContainerFn = makeRunContainer,
206
243
  makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
207
244
  makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
245
+ makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
246
+ makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
208
247
  makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
209
248
  makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
210
249
  makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
@@ -215,12 +254,23 @@ export async function startWorker(
215
254
  } = {},
216
255
  ) {
217
256
  const config = loadConfig(env);
218
- const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields })}\n`);
257
+ // `host` sits AFTER the spread, so it is authoritative rather than overridable (issue #57). No call
258
+ // site can know better than this closure which process wrote a line, and one that passed `host` would
259
+ // be lying by construction -- verified: none does. This is also why the stamp lives ONLY here. Every
260
+ // other module takes `log` injected, and two tests pin the KEY SET of the fields object handed to an
261
+ // injected log (`run_record_failed`, `wait_check`); a `host` added at any call site would break them,
262
+ // while one added inside this closure cannot reach them.
263
+ const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
219
264
 
220
265
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
221
266
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
222
267
  // than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
223
- const schedules = loadSchedules(config);
268
+ // A mutable ref, like its `pauseWindows` and `scopedLimits` siblings and for a reason this one only
269
+ // acquired with issue #57: the heartbeat fingerprints what this host CURRENTLY believes should be
270
+ // scheduled, and a `const` frozen at boot would make it publish the pre-edit set forever -- so two
271
+ // hosts would see each other's fingerprint oscillate on the beat period, refusing or agreeing
272
+ // depending on which half of a beat a reload happened to land in.
273
+ const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared }) };
224
274
 
225
275
  // REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
226
276
  // Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
@@ -290,8 +340,10 @@ export async function startWorker(
290
340
 
291
341
  // Clear strays left by a previous crash before the worker starts draining. Best-effort: the reaper
292
342
  // swallows its own docker errors; this guard keeps any reaper failure from blocking boot.
343
+ // Whether the container reaper actually ENUMERATED, which the scope-claim sweep below depends on.
344
+ let reaped = false;
293
345
  try {
294
- await makeReaperFn({ log })();
346
+ reaped = (await makeReaperFn({ log })())?.reaped === true;
295
347
  } catch (err) {
296
348
  log("reaper_skipped", { reason: err?.message });
297
349
  }
@@ -324,12 +376,42 @@ export async function startWorker(
324
376
  // One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
325
377
  // hoisted out of the createWorkerFn arg object.
326
378
  const redis = makeRedisClient(config.valkeyUrl);
379
+
380
+ // This host's own stale scope claims, gated on the reaper having having enumerated: the
381
+ // reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
382
+ // host is a claim for a container that no longer exists. Deleting it is not a second source of truth --
383
+ // it is the SAME source writing down what it just established, which is what answers `OQ-008` here.
384
+ // Best-effort and double-wrapped like every other boot sweep: an OPTIMISATION over the TTL, never the
385
+ // mechanism, so a fault costs one TTL of a stale claim and never a boot.
386
+ try {
387
+ if (config.workerNameDeclared)
388
+ await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current.map((r) => ({ concurrent: r.concurrent, hash: scopeKeyPrefix(r.scope).slice("budget:s:".length) })), log })({ reaped });
389
+ } catch (err) {
390
+ log("scope_claims_sweep_skipped", { reason: err?.message });
391
+ }
392
+
327
393
  // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
328
394
  // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
329
395
  // Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
330
396
  // drains it after the worker.
331
397
  const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
332
398
 
399
+ // THE HOST QUEUE (issue #57): work only this machine can do, because the folder lives here.
400
+ //
401
+ // Armed by the operator DECLARING a name, not by a peer appearing. Two reasons, and the first is
402
+ // decisive: which queue a job is enqueued to is a routing decision made by whoever enqueues it, so it
403
+ // cannot be allowed to flip underneath a running deployment when a second host happens to register --
404
+ // a cron scheduler upserted on one queue and pruned from another is exactly the mutual teardown this
405
+ // issue exists to stop. And a second BullMQ Worker is a second blocking connection, which a single-host
406
+ // deployment should not pay for silently. Declaring a name IS the multi-host declaration; `doctor`
407
+ // warns when peers exist and nobody has made it.
408
+ const hostQueue = config.workerNameDeclared ? hostQueueName(config.workerName) : null;
409
+ // The long-lived handle the cron watcher reloads through. Its own when a host queue is armed, so a
410
+ // live triggers-file edit lands on the same queue the boot reconcile used; otherwise the shared
411
+ // runtime queue, exactly as before. Registered as an extraCloser only when it is a NEW handle --
412
+ // closing `runtimeQueue` twice would be closing another owner's connection.
413
+ const cronQueue = hostQueue ? makeQueue(parseConnection(config.valkeyUrl), { name: hostQueue }) : runtimeQueue;
414
+
333
415
  // REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
334
416
  // captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
335
417
  // `.json` record via recordRun is ALWAYS on, so every run leaves a stable, non-PII trace regardless.
@@ -365,7 +447,9 @@ export async function startWorker(
365
447
  const onceTriggersFile = env.PI_TRIGGERS_FILE ?? join(process.cwd(), "triggers.json");
366
448
  const disarmOnce = makeDisarmOnce({ triggersPath: onceTriggersFile, log });
367
449
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
368
- writeRecord(buildRecord({ job, result, error, startedAt, endedAt }));
450
+ // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
451
+ // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
452
+ writeRecord(buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName }));
369
453
  // Strictly AFTER the durable record: "fired" means "produced a run record", and the crash
370
454
  // direction this ordering buys is the chosen one -- an armed one-shot with a record, never a
371
455
  // disarm before writeRecord RETURNED. Returned, not succeeded: the record writer swallows fs
@@ -404,10 +488,13 @@ export async function startWorker(
404
488
  const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
405
489
 
406
490
  // INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
407
- // enqueues chained children onto runtimeQueue via enqueueLocalJob -- the same pi-jobs queue the stall
408
- // guard tears down through. Never throws, so a chain fault cannot flip a completed parent
491
+ // enqueues chained children onto the CRON queue via enqueueLocalJob -- this host's own when one is
492
+ // armed, since a chained child continues the working tree this machine just used. Never throws, so a chain fault cannot flip a completed parent
409
493
  // (CONST-RETRY-INFRA-ONLY). The processor calls it as the sole COMPLETED-path chain step.
410
- const collectChain = makeCollectChain({ queue: runtimeQueue, config, log });
494
+ // Onto the HOST queue when one is armed. A chained child is same-folder and local-parent-only
495
+ // (`OQ-009`), so the working tree it needs is the one this machine just used: routing it anywhere
496
+ // else would enqueue a job only this host can run onto a queue every host drains.
497
+ const collectChain = makeCollectChain({ queue: cronQueue, config, log });
411
498
 
412
499
  // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest at EACH job start, like
413
500
  // getSettings above and the pause-window ref below.
@@ -455,13 +542,76 @@ export async function startWorker(
455
542
  if (config.globalPiDir && !readStageManifest({ globalPiDir: config.globalPiDir })) log("packages_manifest_absent", { overlay: config.globalPiDir });
456
543
  getPackagePaths();
457
544
 
545
+ // Issue #57. Published before the worker starts draining, so a peer that boots a moment later sees this
546
+ // host rather than an empty fleet. The image identity rides the SAME preflight the job path uses -- one
547
+ // inspect implementation, one format string -- called once here with an empty job, which resolves the
548
+ // deployment default and trips none of the per-job label gates.
549
+ //
550
+ // The boot line and the registry may cache this where the GATE may not, and the distinction is the whole
551
+ // argument: a gate that caches gives a WRONG DECISION when an operator builds or removes an image
552
+ // mid-day, which is why `imagePreflight` is deliberately not memoised below. A heartbeat that caches
553
+ // gives a STALE ROW, and nothing reads a row to decide anything.
554
+ // ONE preflight instance, constructed once and shared: `start-wiring.test.mjs` pins that, and the
555
+ // reason is the module's own -- the tag the preflight checked has to be the tag `docker run` is
556
+ // handed, and two constructions are two chances for that to stop being true.
557
+ const imagePreflight = makeImagePreflightFn({ image: config.jobImage });
558
+ // BOUNDED, because `.catch()` cannot rescue a promise that never settles: `runDocker` resolves only on
559
+ // the child's `close` or `error` and has no timeout of its own, so a wedged daemon would hang boot
560
+ // here. This read is a nicety -- a digest for the boot line and the registry -- and a nicety may
561
+ // never be able to stop a worker starting. The per-JOB preflight keeps its unbounded wait, where a
562
+ // wedged daemon is the job's problem and the 30-minute job timeout already covers it.
563
+ const bootImage = await Promise.race([
564
+ imagePreflight({}).catch(() => ({})),
565
+ new Promise((resolve) => setTimeout(() => resolve({}), BOOT_IMAGE_TIMEOUT_MS).unref?.()),
566
+ ]);
567
+ // Resolved once: `Intl` is not free, and this value cannot change without a restart.
568
+ const hostTz = Intl.DateTimeFormat().resolvedOptions().timeZone ?? "";
569
+ const registry = makeHostRegistryFn({ redis, name: config.workerName, log });
570
+ // NOT awaited, and that is load-bearing rather than an optimisation. `makeRedisClient` sets
571
+ // `maxRetriesPerRequest: null` -- required for BullMQ's blocking connections -- which means a command
572
+ // issued against an unreachable server QUEUES FOREVER instead of rejecting. Awaiting the first beat
573
+ // would therefore hang boot indefinitely on a deployment whose Valkey is down, turning a telemetry
574
+ // keyspace into a boot dependency. The registry is never on a decision path, so a worker that comes
575
+ // up before its own row does is correct: the row appears when Valkey does.
576
+ void registry.start({
577
+ version: WORKER_VERSION,
578
+ image: config.jobImage,
579
+ imageDigest: bootImage.imageDigest ?? "",
580
+ piVersion: bootImage.piVersion ?? "",
581
+ // A thunk, because the spec says this row carries the LIVE slot count and the overlay can lower it
582
+ // mid-run through `dispatch_set`. A literal here would publish the boot value forever.
583
+ concurrency: () => worker?.concurrency ?? bootConcurrency,
584
+ pid: process.pid,
585
+ // Whether this host DRAINS a queue of its own. Every worker publishes a row; only a host that declared
586
+ // a name has somewhere for routed work to go, and a reader must not invent a queue for one that has not.
587
+ routes: config.workerNameDeclared,
588
+ // The host's IANA zone, because a cron PATTERN carries none: `triggers.json` has no `tz` field and
589
+ // BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's LOCAL time.
590
+ // On one host that is exactly what an operator means; on two in different zones the same pattern is
591
+ // two different instants, and nothing anywhere says so. Published now so a later slice can refuse.
592
+ tz: hostTz,
593
+ // The cron fingerprint rides every beat from the LIVE ref, so a peer always compares against what
594
+ // this host believes now rather than what it believed at boot. `null` means abstain: cron disabled
595
+ // here is no opinion at all, and such a host must never be able to disagree with one that has one.
596
+ fpCron: () => cronFingerprint(authoredCron(config), { tz: hostTz }) ?? "",
597
+ cronCount: () => schedules.current.length,
598
+ });
599
+
600
+
458
601
  const worker = createWorkerFn({
459
602
  connection: parseConnection(config.valkeyUrl),
603
+ hostQueue,
604
+ // Names the BullMQ Worker, which makes `getWorkers()` rows tell hosts apart -- bullmq appends
605
+ // `:w:<name>` to the client name and `moveToActive` stamps `processedBy` onto each active job's
606
+ // hash, so per-job host attribution arrives for free. A NICETY on top of the registry and never the
607
+ // source of truth: that call rests on CLIENT SETNAME, which bullmq's own doc-comment says some
608
+ // providers do not support, and a host list that silently empties cannot be what a decision reads.
609
+ name: config.workerName,
460
610
  concurrency: bootConcurrency,
461
611
  getSettings,
462
612
  redis,
463
613
  recordRun,
464
- extraClosers: [runtimeQueue],
614
+ extraClosers: [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])],
465
615
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
466
616
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
467
617
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
@@ -475,7 +625,28 @@ export async function startWorker(
475
625
  waitState: makeWaitState({ redis }),
476
626
  // The polled tier's bounds, read per pickup from config so they are one value with one home. The
477
627
  // slot count is a CEILING the gate clamps against the live concurrency, never the final number.
628
+ // The fleet-wide half of the wait-check bound (issue #57), armed on the same predicate as the host
629
+ // queue: declaring a name is declaring a fleet. Its TTL is DERIVED rather than guessed -- the gate
630
+ // holds the lease across every profile in turn, each bounded by the check timeout, so one timeout
631
+ // per profile plus one for the overhead between them.
632
+ // The fleet-wide half of a scoped `concurrent` ceiling. Its TTL is DERIVED rather than guessed, and
633
+ // derived is what makes a heartbeat unnecessary: `JOB_TIMEOUT_MS` is a hard 30-minute ceiling on how
634
+ // long any container can run, so a TTL above it cannot expire underneath a live job -- which is the
635
+ // failure that would matter, because it would let another host start a second container on a scope
636
+ // the operator limited to one. Nothing refreshes this claim, deliberately: a refresher would be a
637
+ // second thing to get wrong for a window that cannot be reached.
638
+ scopeLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: scopeSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
639
+ checkLease: hostQueue
640
+ ? makeFleetLease({
641
+ redis,
642
+ holderPrefix: config.workerName,
643
+ keyFor: checkSlotKey,
644
+ ttlMs: config.waitCheckTimeoutMs, // a floor; the gate passes the real one, derived from the profile count
645
+ log,
646
+ })
647
+ : null,
478
648
  checkSlotCount: () => config.waitCheckSlots,
649
+ checkTimeoutMs: () => config.waitCheckTimeoutMs,
479
650
  intervalMs: () => config.waitIntervalMs,
480
651
  maxWaitMs: () => config.waitMaxMs,
481
652
  maxChecks: () => config.waitMaxChecks,
@@ -523,7 +694,7 @@ export async function startWorker(
523
694
  // cache would be wrong in both directions -- an operator who builds the image mid-day would stay refused,
524
695
  // one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
525
696
  // boot because it is deploy-time state under a :ro mount; the host's image set is not.
526
- imagePreflight: makeImagePreflightFn({ image: config.jobImage }),
697
+ imagePreflight,
527
698
  // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
528
699
  // value, one place, so the gate that checks the proxy and the runner that attaches to its network
529
700
  // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
@@ -636,10 +807,13 @@ export async function startWorker(
636
807
  // job-image-missing), never
637
808
  // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
638
809
  // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
639
- worker.on("completed", (job, result) =>
810
+ // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
811
+ // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
812
+ const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
813
+ for (const w of allWorkers) w.on("completed", (job, result) =>
640
814
  log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
641
815
  );
642
- worker.on("failed", (job, err) =>
816
+ for (const w of allWorkers) w.on("failed", (job, err) =>
643
817
  log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
644
818
  );
645
819
 
@@ -649,31 +823,45 @@ export async function startWorker(
649
823
  const guard = makeStallGuard({
650
824
  redis,
651
825
  threshold: config.schedulerStallMax,
652
- removeJobScheduler: (id) => runtimeQueue.removeJobScheduler(id),
826
+ // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
827
+ // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
828
+ removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
653
829
  log,
654
830
  });
655
- worker.on("stalled", (jobId) => void guard.onStalled(jobId));
831
+ for (const w of allWorkers) w.on("stalled", (jobId) => void guard.onStalled(jobId));
656
832
 
657
833
  // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
658
834
  // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
659
835
  // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
660
- if (schedules.length > 0) {
661
- const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }));
836
+ // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
837
+ // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
838
+ // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
839
+ // right place to catch, because it can ask the registry and this cannot.
840
+ const { served, unserved } = servedSchedules(schedules.current);
841
+ for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
842
+
843
+ if (served.length > 0) {
844
+ // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
845
+ // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
846
+ // config" is correct again by construction and two hosts can no longer prune each other at all. The
847
+ // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
848
+ // disagreement, which no queue split can detect.
849
+ const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
662
850
  try {
663
- const r = await reconcile(rq, schedules, { log });
664
- log("schedules_installed", { installed: r.installed, removed: r.removed });
851
+ const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
852
+ log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
665
853
  } finally {
666
854
  await rq.close().catch(() => {});
667
855
  }
668
856
  } else {
669
- log("schedules_installed", { installed: 0, removed: 0 });
857
+ log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
670
858
  }
671
859
 
672
860
  // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
673
861
  // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
674
862
  // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
675
863
  if (config.triggersFile) {
676
- watchTriggersFile(config, runtimeQueue, log);
864
+ watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
677
865
  }
678
866
 
679
867
  // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
@@ -689,6 +877,8 @@ export async function startWorker(
689
877
 
690
878
  log("worker_started", {
691
879
  queue: "pi-jobs",
880
+ host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
881
+ imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
692
882
  concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
693
883
  dailyCap: config.dailyCap,
694
884
  weeklyCap: config.weeklyCap, // null when the weekly window is disabled