@edgehero/pi-dispatch 1.10.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +32 -5
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +387 -17
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1395 -268
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
package/src/index.mjs CHANGED
@@ -1,6 +1,12 @@
1
1
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
2
+ import { assertJudgedConnection, onValkeyError } from "./connection.mjs";
2
3
  import { jobContainerName } from "./backend-local.mjs";
3
- import { InfraRetry, runJob } from "./processor.mjs";
4
+ import { scrubCredentials } from "./redact.mjs";
5
+ import { BACKEND_NOT_REGISTERED } from "./backend-registry.mjs";
6
+ import { CANCEL_ACK_TTL_MS, cancelAckKey, cancelReqKey } from "./cancel-state.mjs";
7
+ import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING, TERMINAL_COMMENTS, runJob } from "./processor.mjs";
8
+ import { NETNS_KEEPER_YOUNG_HOLD_MAX_MS, netnsKeeperCrashLoopSentence, netnsKeeperLoopAgainSentence } from "./netns-keeper.mjs";
9
+ import { PODMAN_RESTART_HOLD_EXPIRED, PODMAN_RESTART_HOLD_MAX_MS, PODMAN_RESTART_HOLD_RECHECK_MS } from "./runtime-observations.mjs";
4
10
  import { targetFor } from "./run-history.mjs";
5
11
  import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
6
12
  import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
@@ -74,7 +80,7 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
74
80
  if (settled) return;
75
81
  settled = true;
76
82
  log("stop_did_not_take", { job: job.id, graceMs });
77
- resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null });
83
+ resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null });
78
84
  }, graceMs);
79
85
  // A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
80
86
  // has finished, and this timer only matters while a job is still in flight.
@@ -106,7 +112,7 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
106
112
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
107
113
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
108
114
  */
109
- export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
115
+ export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
110
116
  return async function processor(job, token, signal) {
111
117
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
112
118
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -533,14 +539,55 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
533
539
  let startedAt;
534
540
  let name;
535
541
  let timer;
542
+ let cancelPoll;
543
+ let cancelPolling = false;
544
+ // The poll tick in flight, KEPT (gate round 2 of PR #479), so the error handler can stop the poll and wait for it
545
+ // before it asks whether the job was cancelled: a flag says a tick runs, only its promise says when it is done.
546
+ let cancelTick = null;
547
+ // THE ONE STOP (gate rounds 2 to 4 of PR #479), used wherever this job's outcome is about to be decided: at the
548
+ // container's observed exit, when runJob returns, and in the error handler. `pollStopped` is set first and read by
549
+ // the tick synchronously right before it aborts and acknowledges, so after a stop no cancel is ever acknowledged;
550
+ // a tick already past that point has aborted the signal and writes its ack before the stop's await returns. So
551
+ // "acknowledged" and "the signal says operator-cancel" are the same fact by the time anything is decided.
552
+ //
553
+ // BOUNDED (gate round 4 of PR #479): the worker's Valkey client queues commands while offline, so a tick's read can
554
+ // hang for a whole outage, and an unbounded await held the job's outcome (its record, comment and slot) with it: a
555
+ // restart in that window lost the record. The bound is safe because the flag is set FIRST: a read still hanging
556
+ // returns without acknowledging, and a tick already past the flag check aborted the signal synchronously, so the
557
+ // decision sees the cancel even while its ack write waits. Said once per job when the bound is hit.
558
+ let pollStopped = false;
559
+ let stopBoundSaid = false;
560
+ const stopCancelPoll = async () => {
561
+ pollStopped = true;
562
+ clearInterval(cancelPoll);
563
+ if (!cancelTick) return;
564
+ let bound;
565
+ const hit = await Promise.race([
566
+ Promise.resolve(cancelTick).then(
567
+ () => false,
568
+ () => false,
569
+ ),
570
+ new Promise((resolve) => {
571
+ bound = setTimeout(() => resolve(true), cancelStopBoundMs);
572
+ bound.unref?.();
573
+ }),
574
+ ]);
575
+ clearTimeout(bound);
576
+ if (hit && !stopBoundSaid) {
577
+ stopBoundSaid = true;
578
+ deps?.log?.("cancel_poll_stop_bounded", { jobId: job.id, boundMs: cancelStopBoundMs });
579
+ }
580
+ };
536
581
  let onAbort;
537
582
  let venue;
538
583
  try {
539
584
  // Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
540
585
  // finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
541
- // scope until a worker restart. Nothing in this block CAN throw today (setTimeout and
542
- // addEventListener on the bullmq-allocated controller are total at processor arity 3); the
543
- // guard is structural, not observational.
586
+ // scope until a worker restart. Nothing in this block throws past its own guards (setTimeout,
587
+ // setInterval and addEventListener on the bullmq-allocated controller are total at processor arity
588
+ // 3, and the cancel poll's redis reads happen inside its own guarded ticks). `containerName` is the
589
+ // exception, and it is named: the registry's refusal of an unheld venue is caught at the call, and
590
+ // anything else it throws propagates to the catch below, which releases what was acquired.
544
591
  startedAt = new Date().toISOString();
545
592
  // The producer of the name both boot reapers sweep by substring. Built from the shared prefix
546
593
  // rather than typed here, so a rename cannot land in the producer and not in the sweeps (#227).
@@ -558,7 +605,25 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
558
605
  // the trigger's, which lives in `data`. Built once so the name and the stop cannot resolve
559
606
  // different backends -- the whole reason the name moved onto the registry in the first place.
560
607
  venue = { ...job.data, id: job.id };
561
- name = containerName(venue);
608
+ // The registry REFUSES a venue it does not hold (#277), and that one refusal is caught here.
609
+ // Unguarded, it skipped `runJob` and every `recordRun` below, so a job naming a venue this worker
610
+ // never built -- a producer on a newer build that knows a venue this one does not -- left NO
611
+ // record, no refusal comment, and was retried as if the infrastructure had failed. Such a venue is
612
+ // never blessed here (the registry refuses a blessed name it does not hold, at boot), so `runJob`
613
+ // refuses it pre-spend -- as `backend-unblessed`, or by an earlier refusal that happens to win --
614
+ // and records it; nothing starts, so there is no container for this name to stop. A null name
615
+ // reaches the abort path's stop only if an abort lands in that window, and that stop fails into
616
+ // its own log line.
617
+ //
618
+ // ONLY that refusal, by its code. Any other throw is a REGISTERED venue's own `containerName`
619
+ // failing, a broken adapter: swallowing it would reserve the budget and start a container under a
620
+ // null name that the timeout could not stop and no reaper sweeps, so it propagates as before.
621
+ try {
622
+ name = containerName(venue);
623
+ } catch (err) {
624
+ if (err?.code !== BACKEND_NOT_REGISTERED) throw err;
625
+ name = null;
626
+ }
562
627
  timer = setTimeout(() => {
563
628
  // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
564
629
  Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
@@ -580,12 +645,51 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
580
645
  // host. Losing the kill for one job is bad; losing the process is worse.
581
646
  const note = deps.log ?? (() => {});
582
647
  try {
583
- Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: err?.message }));
648
+ Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: scrubCredentials(err?.message) }));
584
649
  } catch (err) {
585
- note("stop_container_failed", { job: job.id, reason: err?.message });
650
+ note("stop_container_failed", { job: job.id, reason: scrubCredentials(err?.message) });
586
651
  }
587
652
  };
588
653
  signal.addEventListener("abort", onAbort, { once: true });
654
+
655
+ // The operator-cancel poll (issue #287). `cancelJob` is reachable only from THIS process, so the
656
+ // CLI and the panel leave a `cancel:req:<jobId>` key instead, and the worker that holds the job
657
+ // answers. Polled beside the kill timer rather than subscribed, for the reasons cancel-state.mjs
658
+ // records; one GET per 2s per active job is invisible against the 30s abort grace. Guarded on
659
+ // `redis.get` because bare test wirings pass a redis with only `incr`/`expire` -- they arm
660
+ // nothing and stay byte-identical. Each tick is re-entrancy-guarded and swallows redis faults:
661
+ // a blip may cost the ack, never the job.
662
+ if (typeof redis?.get === "function") {
663
+ cancelPoll = setInterval(() => {
664
+ if (cancelPolling) return;
665
+ cancelPolling = true;
666
+ cancelTick = (async () => {
667
+ try {
668
+ const req = await redis.get(cancelReqKey(job.id));
669
+ if (req === null || req === undefined) return;
670
+ // Stopped while this read was in flight: the outcome is being decided, so this request is left
671
+ // unanswered (its requester's timeout re-reads the job), never acknowledged after the fact.
672
+ if (pollStopped) return;
673
+ // Bound to this worker (the injection comment in createWorker): `false` means the job
674
+ // left the tracked map in this same instant, in which case the finally below clears
675
+ // this interval anyway and the request's TTL reaps the key.
676
+ const took = await Promise.resolve(cancelJob(job.id, "operator-cancel")).catch(() => false);
677
+ if (took === false) return;
678
+ // Ack first, then consume the request: losing the DEL to a blip costs one redundant
679
+ // re-read, losing the ack costs the operator a false "nobody answered".
680
+ await redis.set(cancelAckKey(job.id), hostName, "PX", CANCEL_ACK_TTL_MS).catch(() => {});
681
+ await redis.del(cancelReqKey(job.id)).catch(() => {});
682
+ clearInterval(cancelPoll);
683
+ } catch {
684
+ // Fail open: the next tick re-asks.
685
+ } finally {
686
+ cancelPolling = false;
687
+ }
688
+ })();
689
+ }, cancelPollMs);
690
+ // Like the abort-grace timer: this must never keep an otherwise-finished worker alive.
691
+ cancelPoll.unref?.();
692
+ }
589
693
  } catch (error) {
590
694
  // Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
591
695
  // scope must never be releasable twice -- a double release frees another holder's slot.
@@ -600,6 +704,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
600
704
  void scopeSlot?.release?.();
601
705
  scopeSlot = null;
602
706
  clearTimeout(timer);
707
+ clearInterval(cancelPoll);
603
708
  throw error;
604
709
  }
605
710
 
@@ -669,7 +774,26 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
669
774
  // next boot reaper sweeps; what is NOT leaked is the slot, the lease and the reservation.
670
775
  // The `stop_did_not_take` line is the loud half, because a host whose daemon ignores a stop
671
776
  // is a fact an operator has to learn from somewhere.
672
- runContainer: (ctx) => boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})),
777
+ // Issue #287: WHO aborted rides the result, read off the signal at the one seam both abort
778
+ // paths flow through (a real stop resolves through here, and boundAfterAbort's synthesised
779
+ // `{ code: 137, aborted: true }` does too). `cancelJob(id, reason)` becomes `signal.reason`,
780
+ // so the processor can classify an operator's cancel apart from the kill timer's without
781
+ // this file growing a second channel. Mapped only when `aborted` -- an unaborted result
782
+ // stays byte-identical -- and only a STRING rides: a non-string reason (AbortError objects,
783
+ // future bullmq surprises) collapses to null, which classifies as worker-abort, the
784
+ // conservative direction.
785
+ // Gate round 4 of PR #479: the poll STOPS at the container's observed exit, since after it there is nothing
786
+ // left for a cancel to stop. A cancel acknowledged before that moment raced the exit and lost nothing it was
787
+ // promised: the run is recorded as that cancel (the existing path, "partial work may exist") even when the
788
+ // container happened to exit on its own, because the operator was told the record would say operator-cancel.
789
+ runContainer: (ctx) =>
790
+ boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})).then(async (r) => {
791
+ await stopCancelPoll();
792
+ const raced = r && r.aborted !== true && signal.aborted === true && signal.reason === "operator-cancel";
793
+ if (raced) deps.log?.("cancel_acked_as_container_exited", { jobId: job.id, exitCode: r.code ?? null });
794
+ const run = raced ? { ...r, aborted: true } : r;
795
+ return run?.aborted ? { ...run, abortReason: typeof signal.reason === "string" ? signal.reason : null } : run;
796
+ }),
673
797
  // REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
674
798
  // to be abortable for the same reason runContainer does: a resolver blocking on an unreachable
675
799
  // vault would otherwise hold its slot until its own timeout, and an abort landing mid-resolution
@@ -711,9 +835,149 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
711
835
  // replacing wrapper would lose it in the same silence.
712
836
  ...(deps.checkOnceSpent ? { checkOnceSpent: (j, opts) => deps.checkOnceSpent(j, { ...opts, queueJobId: job.id }) } : {}),
713
837
  });
714
- recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
715
- return result;
838
+ // Gate round 4 of PR #479: stopped again as runJob returns, before anything is recorded (a result with no
839
+ // container never passed the exit stop above). A cancel that lands after is never acknowledged, and its
840
+ // requester's re-read says the job finished. One acknowledged before, on a result that is not the cancel (a
841
+ // pre-spend refusal that finished after the ack, since a container would have seen the abort), is recorded as
842
+ // the cancel the operator was told of, the result it replaced named in the log.
843
+ await stopCancelPoll();
844
+ let outcome = result;
845
+ if (signal.aborted === true && signal.reason === "operator-cancel" && result?.reason !== "operator-cancel") {
846
+ deps.log?.("cancel_acked_before_result", { jobId: job.id, was: `${result?.outcome}/${result?.reason ?? ""}` });
847
+ outcome = { ...result, outcome: "policy", reason: "operator-cancel" };
848
+ }
849
+ recordRun({ job, result: outcome, startedAt, endedAt: new Date().toISOString() });
850
+ return outcome;
716
851
  } catch (error) {
852
+ // Issue #448 (gate round 2 of PR #473): a local job held until rootful Podman's service restarts goes back to the
853
+ // delayed set WITHOUT spending an attempt, the pause gate's move, since a service held up past one retry backoff
854
+ // failed a job that heals by itself (measured). Bounded by the hold's own start, stored on the job because a
855
+ // deferral leaves `attemptsMade` alone and a worker restart must not reset it; past the bound it fails for good
856
+ // with its own reason token, whose comment names the restart. Nothing is recorded per deferral: the job never
857
+ // started, and one record a minute for an hour would bury the run history.
858
+ //
859
+ // Gate round 3: the hour counts only while the worker kept checking. A check that comes more than two recheck
860
+ // periods after the last one (the queue was paused, `pi-dispatch pause`, or no worker ran) starts the hold
861
+ // afresh, since the time between was no evidence that the service stayed up; the last check's time is stored
862
+ // beside the start. And the expired hold is RECORDED with its own token, the one its comment and the failure
863
+ // hook carry, never the `container-never-started` of the throw it ends.
864
+ // A CANCELLED JOB IS NEVER RUN AGAIN (gate of PR #479). The operator-cancel poll runs from pickup to the finally
865
+ // below, so a cancel can be acknowledged (the signal aborted with "operator-cancel") after the job was picked up
866
+ // and before this handler hands it back to the queue: a hold's `moveToDelayed`, or an `InfraRetry` BullMQ
867
+ // retries after its backoff. Either ran the job later, with the request key already consumed, so nothing
868
+ // cancelled it again. So every error this handler would hand back (every `InfraRetry`, the two holds' included,
869
+ // since both are one) is asked ONE question first, in ONE place, so the three paths cannot drift apart: was this
870
+ // job cancelled? A request not yet polled is read too, and acknowledged and consumed as the poll would. Either
871
+ // ends the job as the operator's cancel, the policy result the aborted-container path already returns, recorded,
872
+ // one fixed comment, never retried. A shutdown's abort is not a cancel: that job is held or retried as before,
873
+ // so it survives the restart.
874
+ // `readPending: false` asks only whether the poll ALREADY acknowledged a cancel (the signal), and leaves a request it
875
+ // had not read unanswered.
876
+ const operatorCancelled = async ({ readPending = true } = {}) => {
877
+ if (signal?.aborted === true) return signal.reason === "operator-cancel";
878
+ if (!readPending) return false;
879
+ if (typeof redis?.get !== "function") return false;
880
+ let req = null;
881
+ try {
882
+ req = await redis.get(cancelReqKey(job.id));
883
+ } catch {
884
+ return false;
885
+ }
886
+ if (req === null || req === undefined) return false;
887
+ await Promise.resolve(redis.set?.(cancelAckKey(job.id), hostName, "PX", CANCEL_ACK_TTL_MS)).catch(() => {});
888
+ await Promise.resolve(redis.del?.(cancelReqKey(job.id))).catch(() => {});
889
+ return true;
890
+ };
891
+ // A retry that already spent (a container that ran and exited as infra) keeps what it spent on the record, and
892
+ // says the processor's own operator-cancel sentence; one that never started says it was cancelled before it did.
893
+ // The session rides along as the aborted-container path keeps it (`mergeSession`, already applied to the throw).
894
+ // A NON-retryable error (gate round 3) is not known to have started nothing, so it says the processor's own
895
+ // operator-cancel sentence, keeps whatever `budgetReserved` it carried, and its own words go to the log, never lost.
896
+ const endCancelled = async ({ retryable }) => {
897
+ const spent = error?.budgetReserved === true;
898
+ const beforeStart = retryable && !spent;
899
+ const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null) };
900
+ deps?.log?.("job_cancelled_instead_of_retry", { jobId: job.id, spent, retryable, ...(retryable ? {} : { failure: scrubCredentials(String(error?.message ?? error)).slice(0, 300) }) });
901
+ if (deps?.comment) await Promise.resolve(deps.comment(job.data, beforeStart ? CANCELLED_BEFORE_START_COMMENT : TERMINAL_COMMENTS["operator-cancel"])).catch(() => {});
902
+ recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
903
+ return result;
904
+ };
905
+ // STOP THE POLL BEFORE ASKING (gate round 2 of PR #479). The poll ran until the finally below, so a request that
906
+ // landed after the question was still acknowledged by it (ack written, request consumed, the CLI saying "cancel
907
+ // accepted") while the job went on to its hold or its retry and ran again. So the poll is stopped first and any
908
+ // tick already in flight awaited, and only then is the question asked: a cancel the poll took is seen on the
909
+ // signal, one it had not read is read here, and one that lands after is never acknowledged, so its requester
910
+ // truthfully says nothing was changed while the job is held or retried, and a second cancel removes it.
911
+ //
912
+ // EVERY ERROR, NOT ONLY A RETRY (gate round 3 of PR #479). On a non-retryable error the still-running poll could
913
+ // acknowledge a cancel, so the CLI said the record would say operator-cancel while it said failed. The poll is
914
+ // stopped the same way before the terminal path, and the question differs by one thing only. A cancel the poll
915
+ // ALREADY acknowledged (the signal) ends the job as operator-cancel: the requester was told that, so the record
916
+ // must agree, and the failure's own words go to the log. A request it had NOT read is left unanswered: the job
917
+ // has failed for its own reason, the record says so, and the requester's re-read then says the job finished
918
+ // before its worker read the cancel and nothing was changed. Consistent either way, never one said and the
919
+ // other recorded. A retryable error reads that request too, since the job would otherwise run again.
920
+ await stopCancelPoll();
921
+ const retryable = error instanceof InfraRetry;
922
+ if (await operatorCancelled({ readPending: retryable })) return await endCancelled({ retryable });
923
+ if (error?.holdUntilRestart === true) {
924
+ const heldAt = now();
925
+ const last = job.data?.podmanRestartHoldLastMs;
926
+ const resumed = !Number.isFinite(last) || heldAt - last > 2 * PODMAN_RESTART_HOLD_RECHECK_MS;
927
+ const since = !resumed && Number.isFinite(job.data?.podmanRestartHoldSinceMs) ? job.data.podmanRestartHoldSinceMs : heldAt;
928
+ if (heldAt - since < PODMAN_RESTART_HOLD_MAX_MS) {
929
+ await job.updateData({ ...job.data, podmanRestartHoldSinceMs: since, podmanRestartHoldLastMs: heldAt });
930
+ deps?.log?.("podman_restart_hold", { jobId: job.id, heldForMs: heldAt - since, delayMs: PODMAN_RESTART_HOLD_RECHECK_MS, reason: String(error.message).slice(0, 300) });
931
+ await job.moveToDelayed(heldAt + PODMAN_RESTART_HOLD_RECHECK_MS, token);
932
+ throw new DelayedError();
933
+ }
934
+ const expired = Object.assign(new UnrecoverableError(`held ${Math.round((heldAt - since) / 60_000)} min for rootful Podman's service to restart, and it did not: ${error.message}`), { reason: PODMAN_RESTART_HOLD_EXPIRED, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
935
+ recordRun({ job, error: expired, startedAt, endedAt: new Date().toISOString() });
936
+ throw expired;
937
+ }
938
+ // Issue #476: a job whose egress preflight found the rootless network keeper running on its own bridge but younger
939
+ // than the minimum age is HELD until it is old enough (one `moveToDelayed`, no attempt), the move the hold above
940
+ // makes, since a keeper merely young is not a keeper that does not hold: every joint start of the stack read it
941
+ // at under a second and spent a queued job's attempt (measured on 4.9.3). A crash loop is young at every start,
942
+ // so the hold is BOUNDED, and ends early on the evidence of one: the keeper the job waited on is gone, or another
943
+ // start has taken its place. Either ends in an ordinary retry (an attempt, as any keeper that does not hold
944
+ // costs) with its own reason token naming the loop. The hold's state is stored on the job, as the podman.service
945
+ // hold's is, and read only while the checks come within the bound of each other: a later one starts afresh.
946
+ const keeperHold = keeperHoldState(job, now());
947
+ if (error?.keeperYoung && typeof error.keeperYoung === "object") {
948
+ const young = error.keeperYoung;
949
+ const restarted = keeperHold.startedMs !== null && keeperHold.startedMs !== young.startedMs;
950
+ const heldMs = keeperHold.at - keeperHold.since;
951
+ if (!restarted && heldMs < NETNS_KEEPER_YOUNG_HOLD_MAX_MS) {
952
+ await job.updateData({ ...job.data, netnsKeeperHoldSinceMs: keeperHold.since, netnsKeeperHoldLastMs: keeperHold.at, netnsKeeperHoldStartedMs: young.startedMs });
953
+ deps?.log?.("netns_keeper_young_hold", { jobId: job.id, keeperAgeMs: young.ageMs, heldForMs: heldMs, delayMs: young.waitMs });
954
+ await job.moveToDelayed(keeperHold.at + young.waitMs, token);
955
+ throw new DelayedError();
956
+ }
957
+ const was = keeperHold.startedMs ?? young.startedMs;
958
+ await markKeeperLoop(job, was === young.startedMs ? [was] : [was, young.startedMs]);
959
+ const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was, now: young.startedMs, heldMs, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
960
+ recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
961
+ throw loop;
962
+ }
963
+ if (error?.reason === NETNS_KEEPER_NOT_HOLDING && keeperHold.startedMs !== null) {
964
+ // The keeper this job was waiting on is no longer running on its bridge: it died young, the loop's other face.
965
+ await markKeeperLoop(job, [keeperHold.startedMs]);
966
+ const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was: keeperHold.startedMs, now: null, problem: error.keeperProblem ?? null, heldMs: keeperHold.at - keeperHold.since, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
967
+ recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
968
+ throw loop;
969
+ }
970
+ // A LATER ATTEMPT OF A JOB THAT SAW THE LOOP (gate of PR #479). The loop's retry comes after the queue's 60 s
971
+ // backoff, when the hold's 30 s window has closed, and it meets a keeper that restarted out of order against the
972
+ // proxy, or none: measured, it failed as `netns-keeper-not-holding`, and that attempt's record and terminal
973
+ // comment replaced the loop's. The marker (the keeper starts seen) is not reset by the window, so any keeper
974
+ // failure of this job after it is still named the loop.
975
+ const loopSeen = job.data?.netnsKeeperLoopSeen;
976
+ if (error?.reason === NETNS_KEEPER_NOT_HOLDING && Array.isArray(loopSeen) && loopSeen.length > 0) {
977
+ const loop = new InfraRetry(netnsKeeperLoopAgainSentence({ seen: loopSeen, problem: error.keeperProblem ?? null, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
978
+ recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
979
+ throw loop;
980
+ }
717
981
  recordRun({ job, error, startedAt, endedAt: new Date().toISOString() });
718
982
  if (error instanceof InfraRetry) throw error; // retryable: BullMQ retries per attempts
719
983
  // A non-retryable, non-infra error (our bug) must not retry forever. UnrecoverableError
@@ -731,11 +995,46 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
731
995
  // an async function, and `release` never throws.
732
996
  await scopeSlot?.release?.();
733
997
  clearTimeout(timer);
998
+ clearInterval(cancelPoll);
734
999
  signal.removeEventListener("abort", onAbort);
735
1000
  }
736
1001
  };
737
1002
  }
738
1003
 
1004
+ /**
1005
+ * How long the stop of a job's cancel poll waits for a tick already in flight (gate round 4 of PR #479). About one Valkey
1006
+ * round trip under load, and far under the 30 s abort grace; past it a read still hanging cannot acknowledge anything.
1007
+ */
1008
+ export const CANCEL_STOP_BOUND_MS = 1_000;
1009
+
1010
+ /** The comment for a job the operator cancelled before it started, where it would have been held or retried (gate of PR #479). */
1011
+ export const CANCELLED_BEFORE_START_COMMENT = "Stopped: the operator cancelled this run before it started. Nothing was spent. Not retried.";
1012
+
1013
+ /**
1014
+ * Store the keeper starts a crash loop was seen at on the job (gate of PR #479), outside the hold's 30 s window, so a
1015
+ * later attempt that fails on the keeper is still named the loop. Fail open: a lost marker costs only the name.
1016
+ */
1017
+ async function markKeeperLoop(job, seen) {
1018
+ try {
1019
+ await job.updateData({ ...job.data, netnsKeeperLoopSeen: seen.filter((ms) => Number.isFinite(ms)) });
1020
+ } catch {
1021
+ // The loop is still named on this attempt; only a later attempt's name depends on the marker.
1022
+ }
1023
+ }
1024
+
1025
+ /**
1026
+ * A job's young-keeper hold as stored on it (issue #476): `{ at, since, startedMs }`, `at` being now. The stored start
1027
+ * and the keeper start it waited on count only while the last check was within `NETNS_KEEPER_YOUNG_HOLD_MAX_MS` of
1028
+ * now; an older one (a retry after its backoff, a paused queue) is a fresh hold with no keeper start seen.
1029
+ */
1030
+ export function keeperHoldState(job, at) {
1031
+ const last = job?.data?.netnsKeeperHoldLastMs;
1032
+ const current = Number.isFinite(last) && at - last >= 0 && at - last <= NETNS_KEEPER_YOUNG_HOLD_MAX_MS;
1033
+ const since = current && Number.isFinite(job.data.netnsKeeperHoldSinceMs) ? job.data.netnsKeeperHoldSinceMs : at;
1034
+ const startedMs = current && Number.isFinite(job.data.netnsKeeperHoldStartedMs) ? job.data.netnsKeeperHoldStartedMs : null;
1035
+ return { at, since, startedMs };
1036
+ }
1037
+
739
1038
  export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
740
1039
  // One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
741
1040
  // check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
@@ -768,6 +1067,10 @@ export function createWorker({ connection, name, stopContainer, containerName, h
768
1067
  // Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
769
1068
  // and the shared handle could not reach it.
770
1069
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
1070
+ // Issue #287: the name the cancel poll writes into its ack, so the operator's terminal can say
1071
+ // WHICH host took the cancel. `name` is already the registry/client identity; "" for a
1072
+ // deployment that never declared one, and the ack's reader prints it as such.
1073
+ hostName: name ?? "",
771
1074
  // #227. INJECTED, not built here. This was a one-line `docker stop` literal, which meant the abort
772
1075
  // path -- the only thing that can end a runaway job -- was the one backend function unreachable
773
1076
  // from `startWorker`. The wiring now passes the registry's per-job stop, so the container is
@@ -811,6 +1114,8 @@ export function createWorker({ connection, name, stopContainer, containerName, h
811
1114
  recordRun,
812
1115
  });
813
1116
 
1117
+ // Issue #464: only a connection `parseConnection` built, which judges and pins the Valkey it dials.
1118
+ assertJudgedConnection(connection);
814
1119
  worker = new Worker(queueName, processor, {
815
1120
  // maxRetriesPerRequest: null is REQUIRED for BullMQ's blocking connections, or it throws.
816
1121
  connection: { ...connection, maxRetriesPerRequest: null },
@@ -821,6 +1126,9 @@ export function createWorker({ connection, name, stopContainer, containerName, h
821
1126
  ...(name ? { name } : {}),
822
1127
  ...(limiter ? { limiter } : {}),
823
1128
  });
1129
+ // Issue #468: an error of this worker is one line, its message, never BullMQ's console.error of the whole object
1130
+ // (which carried a failed AUTH's password in `command.args` before connection.mjs scrubbed it).
1131
+ onValkeyError(worker, `worker ${queueName}`);
824
1132
  workers.push(worker);
825
1133
  }
826
1134
 
@@ -829,17 +1137,72 @@ export function createWorker({ connection, name, stopContainer, containerName, h
829
1137
  // returned as a pair so every existing caller keeps receiving exactly what it received before.
830
1138
  primary.hostWorker = workers[1] ?? null;
831
1139
 
832
- const shutdown = async () => {
1140
+ const stop = async () => {
833
1141
  // Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
834
1142
  // worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
835
1143
  // registrations would mean two `process.exit(0)` racing, and the second worker's containers would
836
1144
  // outlive the handler that was meant to stop them.
837
1145
  for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
838
1146
  for (const w of workers) await w.close().catch(() => {});
839
- // Close auxiliary resources (e.g. a cron scheduler) after the worker drains. Per-item catch
840
- // so one failing or absent closer never strands the others or blocks exit -- matches the
841
- // swallow posture on cancelAllJobs above.
842
- await Promise.all(extraClosers.map((c) => Promise.resolve(c.close?.()).catch(() => {})));
1147
+ // Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
1148
+ // Per-item catch so one failing or absent closer never strands the others or blocks exit -- matches
1149
+ // the swallow posture on cancelAllJobs above. The try/catch is NOT redundant with the `.catch`:
1150
+ // `Promise.resolve(x)` does not catch a SYNCHRONOUS throw from `x`, and `c.close` on a null entry
1151
+ // throws before `Promise.resolve` is ever reached. Either would escape this callback, reject the whole
1152
+ // shutdown and skip the `process.exit(0)` below. Jobs and containers are already stopped by then, so
1153
+ // what a stranded loop leaks is the rest of the list: `registry.close()` is the DEL that keeps a
1154
+ // stopped host from lingering as a ghost peer for its full TTL, and a ghost peer with a stale
1155
+ // `fpCron` is what makes a later `reconcileGated` refuse a legitimate reconcile. The comment above
1156
+ // promised this isolation before the code delivered it (issue #295). It bounds nothing, though: a
1157
+ // closer that never settles still blocks exit, which no closer here does.
1158
+ //
1159
+ // Read LATE and deliberately: `start.mjs` pushes its live-edit watchers into this array AFTER handing
1160
+ // it over, because they are armed after the boot reconcile. Anything here that snapshots or copies
1161
+ // the array un-registers them in silence.
1162
+ await Promise.all(
1163
+ extraClosers.map((c) => {
1164
+ try {
1165
+ return Promise.resolve(c?.close?.()).catch(() => {});
1166
+ } catch {
1167
+ return Promise.resolve();
1168
+ }
1169
+ }),
1170
+ );
1171
+ // Release the shared ioredis client LAST (issue #300). Eight consumers ride it -- the budget, the
1172
+ // wait state, the leases, the run mirror, the host registry, the stall guard, the scope-claim sweep
1173
+ // -- and until here nothing in the product ever closed it; only the test harness did, reaching into
1174
+ // the captured wiring, which was the tell.
1175
+ //
1176
+ // SEQUENCED AFTER the drain above, never inside it. The raw client has no `.close`, so pushing it
1177
+ // into `extraClosers` is a silent no-op -- and the natural wrapper, `{ close: () => redis.disconnect() }`,
1178
+ // is worse than nothing: drained CONCURRENTLY by the Promise.all, it takes the connection down beside
1179
+ // `registry.close()`, and a recording server then received NO commands at all where this ordering
1180
+ // delivers the registry's DEL and SREM -- the DEL being what keeps a stopped host from lingering as
1181
+ // a ghost peer for its full TTL.
1182
+ //
1183
+ // `disconnect()`, not `quit()`, for a MEASURED reason rather than the plausible one. `quit()` answers
1184
+ // OK in 0ms against a REFUSED port; the hang it can suffer is a server that accepts the TCP
1185
+ // connection and never answers, where the client sits in status "connect" awaiting its ready check --
1186
+ // independent of `maxRetriesPerRequest` and of `enableOfflineQueue`, both measured. `disconnect()`
1187
+ // returns immediately in every case, and everything whose replies matter has already drained above.
1188
+ // Guarded, because the wiring tests hand createWorker a bare `redis: {}`. And wrapped, on the loop's
1189
+ // own rule stated above: a SYNCHRONOUS throw from this line would skip the `process.exit(0)` below
1190
+ // and hang the stop. The real client's disconnect does not throw; a test-injected one is one edit
1191
+ // away from doing so.
1192
+ try {
1193
+ redis?.disconnect?.();
1194
+ } catch {
1195
+ // A release that failed has already stopped mattering; the exit that follows is what a stop owes.
1196
+ }
1197
+ };
1198
+ // The signal path is `stop()` then exit, and the split is issue #299's: a boot that refuses AFTER the
1199
+ // Worker exists must be able to undo what it built WITHOUT exiting, because the refusal's own error --
1200
+ // not a 0 -- has to reach `cli.mjs`'s entryExitCode, and with every handle released above the process
1201
+ // drains to that code on its own. The closure keeps the name `shutdown` because the source pin in
1202
+ // `wiring.test.mjs` reads the registration lines below, deliberately, rather than constructing a
1203
+ // Worker to observe them.
1204
+ const shutdown = async () => {
1205
+ await stop();
843
1206
  process.exit(0);
844
1207
  };
845
1208
  process.once("SIGTERM", shutdown);
@@ -849,6 +1212,13 @@ export function createWorker({ connection, name, stopContainer, containerName, h
849
1212
  // worker still aborts in-flight jobs and docker-stops their containers rather than orphaning them.
850
1213
  if (process.platform === "win32") process.once("SIGBREAK", shutdown);
851
1214
 
1215
+ // Beside `hostWorker` and for the same reason it rides the return value rather than changing it:
1216
+ // every existing caller keeps receiving exactly what it received before, and the one new caller --
1217
+ // `startWorker`'s post-handoff catch (issue #299) -- reaches the teardown through the worker it was
1218
+ // handed. `stop` is the shutdown minus the exit; it is safe to call more than once, because every
1219
+ // step it takes is idempotent by that step's own contract.
1220
+ primary.stop = stop;
1221
+
852
1222
  return primary;
853
1223
  }
854
1224