@edgehero/pi-dispatch 1.10.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +303 -150
- package/README.md +52 -0
- package/deploy/com.pi-dispatch.worker.plist +10 -4
- package/deploy/docker-compose.yml +49 -16
- package/deploy/egress-proxy.conf +32 -2
- package/deploy/nssm-install.cmd +12 -6
- package/deploy/pi-dispatch-egress-out.network +10 -0
- package/deploy/pi-dispatch-egress-proxy.container +50 -0
- package/deploy/pi-dispatch-netns-keeper.container +80 -0
- package/deploy/pi-dispatch-netns-keeper.network +18 -0
- package/deploy/pi-dispatch-valkey.container +51 -0
- package/deploy/pi-dispatch-valkey.network +16 -0
- package/deploy/receiver.service +6 -0
- package/deploy/worker-env-wrapper.cmd +12 -1
- package/deploy/worker-env-wrapper.sh +63 -37
- package/deploy/worker.service +18 -8
- package/package.json +15 -5
- package/src/azure-host.mjs +19 -0
- package/src/azure-identity.mjs +18 -2
- package/src/backend-conformance.mjs +71 -18
- package/src/backend-local.mjs +637 -21
- package/src/backend-podman.mjs +1168 -0
- package/src/backend-registry.mjs +86 -3
- package/src/backends.mjs +489 -37
- package/src/branch.mjs +7 -2
- package/src/cancel-cli.mjs +174 -0
- package/src/cancel-state.mjs +125 -0
- package/src/cli.mjs +188 -90
- package/src/config.mjs +503 -43
- package/src/connection.mjs +374 -8
- package/src/container-spec.mjs +102 -7
- package/src/daemon-facts.mjs +167 -0
- package/src/deployment-venue.mjs +158 -0
- package/src/docker-run.mjs +146 -15
- package/src/doctor.mjs +4756 -394
- package/src/egress-conf-copy.mjs +166 -0
- package/src/egress-proxy-state.mjs +151 -0
- package/src/egress.mjs +456 -25
- package/src/entry.mjs +27 -0
- package/src/env-allowlist.mjs +245 -40
- package/src/env-file.mjs +1869 -33
- package/src/exit-code.mjs +15 -0
- package/src/flow-gate.mjs +5 -3
- package/src/forgejo-host.mjs +19 -0
- package/src/forgejo-identity.mjs +21 -2
- package/src/get-token.mjs +67 -18
- package/src/git-dirty.mjs +9 -1
- package/src/git-hardening.mjs +33 -0
- package/src/github-app-setup.mjs +29 -12
- package/src/github-prompt.mjs +4 -1
- package/src/gitlab-host.mjs +19 -0
- package/src/gitlab-identity.mjs +19 -2
- package/src/host-pi.mjs +19 -3
- package/src/host-registry.mjs +29 -2
- package/src/identity.mjs +29 -4
- package/src/image-preflight.mjs +46 -11
- package/src/image-ref.mjs +21 -0
- package/src/index.mjs +363 -13
- package/src/init.mjs +197 -38
- package/src/job-user.mjs +252 -0
- package/src/json-duplicates.mjs +204 -0
- package/src/live-probes.mjs +1020 -0
- package/src/materialize.mjs +4 -11
- package/src/netns-keeper.mjs +264 -0
- package/src/on-failure.mjs +119 -0
- package/src/outbox.mjs +7 -0
- package/src/packages.mjs +2 -2
- package/src/podman-stack.mjs +1304 -0
- package/src/prepare-github.mjs +6 -6
- package/src/prepare-local.mjs +51 -17
- package/src/prepare.mjs +27 -6
- package/src/pricing.mjs +9 -5
- package/src/processor.mjs +506 -26
- package/src/provider-key.mjs +66 -0
- package/src/provider-steering.mjs +185 -0
- package/src/queue.mjs +35 -8
- package/src/redact.mjs +84 -0
- package/src/reserved-env.mjs +7 -3
- package/src/retention-sweep.mjs +178 -0
- package/src/run-container.mjs +181 -14
- package/src/run-history.mjs +105 -16
- package/src/runtime-observations.mjs +1152 -0
- package/src/runtime-settings.mjs +13 -8
- package/src/sandbox-cli.mjs +100 -95
- package/src/sandbox-store.mjs +612 -45
- package/src/sandbox.mjs +1459 -37
- package/src/schedules.mjs +16 -3
- package/src/secret-profiles.mjs +2 -1
- package/src/secrets.mjs +24 -6
- package/src/service-env.mjs +247 -0
- package/src/service.mjs +618 -28
- package/src/session-store.mjs +678 -53
- package/src/start.mjs +1348 -326
- package/src/subscriptions.mjs +7 -3
- package/src/transient.mjs +240 -0
- package/src/triggers-file.mjs +71 -15
- package/src/triggers.mjs +179 -19
- package/src/up.mjs +1399 -85
- package/src/valkey-auth.mjs +529 -0
- package/src/valkey-endpoint.mjs +367 -0
- package/src/watch-closer.mjs +158 -0
package/src/index.mjs
CHANGED
|
@@ -1,6 +1,12 @@
|
|
|
1
1
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
2
|
+
import { assertJudgedConnection, onValkeyError } from "./connection.mjs";
|
|
2
3
|
import { jobContainerName } from "./backend-local.mjs";
|
|
3
|
-
import {
|
|
4
|
+
import { scrubCredentials } from "./redact.mjs";
|
|
5
|
+
import { BACKEND_NOT_REGISTERED } from "./backend-registry.mjs";
|
|
6
|
+
import { CANCEL_ACK_TTL_MS, cancelAckKey, cancelReqKey } from "./cancel-state.mjs";
|
|
7
|
+
import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING, TERMINAL_COMMENTS, runJob } from "./processor.mjs";
|
|
8
|
+
import { NETNS_KEEPER_YOUNG_HOLD_MAX_MS, netnsKeeperCrashLoopSentence, netnsKeeperLoopAgainSentence } from "./netns-keeper.mjs";
|
|
9
|
+
import { PODMAN_RESTART_HOLD_EXPIRED, PODMAN_RESTART_HOLD_MAX_MS, PODMAN_RESTART_HOLD_RECHECK_MS } from "./runtime-observations.mjs";
|
|
4
10
|
import { targetFor } from "./run-history.mjs";
|
|
5
11
|
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
|
|
6
12
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
@@ -74,7 +80,7 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
|
74
80
|
if (settled) return;
|
|
75
81
|
settled = true;
|
|
76
82
|
log("stop_did_not_take", { job: job.id, graceMs });
|
|
77
|
-
resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null });
|
|
83
|
+
resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null });
|
|
78
84
|
}, graceMs);
|
|
79
85
|
// A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
|
|
80
86
|
// has finished, and this timer only matters while a job is still in flight.
|
|
@@ -106,7 +112,7 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
|
106
112
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
107
113
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
108
114
|
*/
|
|
109
|
-
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
115
|
+
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
110
116
|
return async function processor(job, token, signal) {
|
|
111
117
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
112
118
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -533,14 +539,55 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
533
539
|
let startedAt;
|
|
534
540
|
let name;
|
|
535
541
|
let timer;
|
|
542
|
+
let cancelPoll;
|
|
543
|
+
let cancelPolling = false;
|
|
544
|
+
// The poll tick in flight, KEPT (gate round 2 of PR #479), so the error handler can stop the poll and wait for it
|
|
545
|
+
// before it asks whether the job was cancelled: a flag says a tick runs, only its promise says when it is done.
|
|
546
|
+
let cancelTick = null;
|
|
547
|
+
// THE ONE STOP (gate rounds 2 to 4 of PR #479), used wherever this job's outcome is about to be decided: at the
|
|
548
|
+
// container's observed exit, when runJob returns, and in the error handler. `pollStopped` is set first and read by
|
|
549
|
+
// the tick synchronously right before it aborts and acknowledges, so after a stop no cancel is ever acknowledged;
|
|
550
|
+
// a tick already past that point has aborted the signal and writes its ack before the stop's await returns. So
|
|
551
|
+
// "acknowledged" and "the signal says operator-cancel" are the same fact by the time anything is decided.
|
|
552
|
+
//
|
|
553
|
+
// BOUNDED (gate round 4 of PR #479): the worker's Valkey client queues commands while offline, so a tick's read can
|
|
554
|
+
// hang for a whole outage, and an unbounded await held the job's outcome (its record, comment and slot) with it: a
|
|
555
|
+
// restart in that window lost the record. The bound is safe because the flag is set FIRST: a read still hanging
|
|
556
|
+
// returns without acknowledging, and a tick already past the flag check aborted the signal synchronously, so the
|
|
557
|
+
// decision sees the cancel even while its ack write waits. Said once per job when the bound is hit.
|
|
558
|
+
let pollStopped = false;
|
|
559
|
+
let stopBoundSaid = false;
|
|
560
|
+
const stopCancelPoll = async () => {
|
|
561
|
+
pollStopped = true;
|
|
562
|
+
clearInterval(cancelPoll);
|
|
563
|
+
if (!cancelTick) return;
|
|
564
|
+
let bound;
|
|
565
|
+
const hit = await Promise.race([
|
|
566
|
+
Promise.resolve(cancelTick).then(
|
|
567
|
+
() => false,
|
|
568
|
+
() => false,
|
|
569
|
+
),
|
|
570
|
+
new Promise((resolve) => {
|
|
571
|
+
bound = setTimeout(() => resolve(true), cancelStopBoundMs);
|
|
572
|
+
bound.unref?.();
|
|
573
|
+
}),
|
|
574
|
+
]);
|
|
575
|
+
clearTimeout(bound);
|
|
576
|
+
if (hit && !stopBoundSaid) {
|
|
577
|
+
stopBoundSaid = true;
|
|
578
|
+
deps?.log?.("cancel_poll_stop_bounded", { jobId: job.id, boundMs: cancelStopBoundMs });
|
|
579
|
+
}
|
|
580
|
+
};
|
|
536
581
|
let onAbort;
|
|
537
582
|
let venue;
|
|
538
583
|
try {
|
|
539
584
|
// Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
|
|
540
585
|
// finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
|
|
541
|
-
// scope until a worker restart. Nothing in this block
|
|
542
|
-
// addEventListener on the bullmq-allocated controller are total at processor arity
|
|
543
|
-
//
|
|
586
|
+
// scope until a worker restart. Nothing in this block throws past its own guards (setTimeout,
|
|
587
|
+
// setInterval and addEventListener on the bullmq-allocated controller are total at processor arity
|
|
588
|
+
// 3, and the cancel poll's redis reads happen inside its own guarded ticks). `containerName` is the
|
|
589
|
+
// exception, and it is named: the registry's refusal of an unheld venue is caught at the call, and
|
|
590
|
+
// anything else it throws propagates to the catch below, which releases what was acquired.
|
|
544
591
|
startedAt = new Date().toISOString();
|
|
545
592
|
// The producer of the name both boot reapers sweep by substring. Built from the shared prefix
|
|
546
593
|
// rather than typed here, so a rename cannot land in the producer and not in the sweeps (#227).
|
|
@@ -558,7 +605,25 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
558
605
|
// the trigger's, which lives in `data`. Built once so the name and the stop cannot resolve
|
|
559
606
|
// different backends -- the whole reason the name moved onto the registry in the first place.
|
|
560
607
|
venue = { ...job.data, id: job.id };
|
|
561
|
-
|
|
608
|
+
// The registry REFUSES a venue it does not hold (#277), and that one refusal is caught here.
|
|
609
|
+
// Unguarded, it skipped `runJob` and every `recordRun` below, so a job naming a venue this worker
|
|
610
|
+
// never built -- a producer on a newer build that knows a venue this one does not -- left NO
|
|
611
|
+
// record, no refusal comment, and was retried as if the infrastructure had failed. Such a venue is
|
|
612
|
+
// never blessed here (the registry refuses a blessed name it does not hold, at boot), so `runJob`
|
|
613
|
+
// refuses it pre-spend -- as `backend-unblessed`, or by an earlier refusal that happens to win --
|
|
614
|
+
// and records it; nothing starts, so there is no container for this name to stop. A null name
|
|
615
|
+
// reaches the abort path's stop only if an abort lands in that window, and that stop fails into
|
|
616
|
+
// its own log line.
|
|
617
|
+
//
|
|
618
|
+
// ONLY that refusal, by its code. Any other throw is a REGISTERED venue's own `containerName`
|
|
619
|
+
// failing, a broken adapter: swallowing it would reserve the budget and start a container under a
|
|
620
|
+
// null name that the timeout could not stop and no reaper sweeps, so it propagates as before.
|
|
621
|
+
try {
|
|
622
|
+
name = containerName(venue);
|
|
623
|
+
} catch (err) {
|
|
624
|
+
if (err?.code !== BACKEND_NOT_REGISTERED) throw err;
|
|
625
|
+
name = null;
|
|
626
|
+
}
|
|
562
627
|
timer = setTimeout(() => {
|
|
563
628
|
// BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
|
|
564
629
|
Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
|
|
@@ -580,12 +645,51 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
580
645
|
// host. Losing the kill for one job is bad; losing the process is worse.
|
|
581
646
|
const note = deps.log ?? (() => {});
|
|
582
647
|
try {
|
|
583
|
-
Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: err?.message }));
|
|
648
|
+
Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: scrubCredentials(err?.message) }));
|
|
584
649
|
} catch (err) {
|
|
585
|
-
note("stop_container_failed", { job: job.id, reason: err?.message });
|
|
650
|
+
note("stop_container_failed", { job: job.id, reason: scrubCredentials(err?.message) });
|
|
586
651
|
}
|
|
587
652
|
};
|
|
588
653
|
signal.addEventListener("abort", onAbort, { once: true });
|
|
654
|
+
|
|
655
|
+
// The operator-cancel poll (issue #287). `cancelJob` is reachable only from THIS process, so the
|
|
656
|
+
// CLI and the panel leave a `cancel:req:<jobId>` key instead, and the worker that holds the job
|
|
657
|
+
// answers. Polled beside the kill timer rather than subscribed, for the reasons cancel-state.mjs
|
|
658
|
+
// records; one GET per 2s per active job is invisible against the 30s abort grace. Guarded on
|
|
659
|
+
// `redis.get` because bare test wirings pass a redis with only `incr`/`expire` -- they arm
|
|
660
|
+
// nothing and stay byte-identical. Each tick is re-entrancy-guarded and swallows redis faults:
|
|
661
|
+
// a blip may cost the ack, never the job.
|
|
662
|
+
if (typeof redis?.get === "function") {
|
|
663
|
+
cancelPoll = setInterval(() => {
|
|
664
|
+
if (cancelPolling) return;
|
|
665
|
+
cancelPolling = true;
|
|
666
|
+
cancelTick = (async () => {
|
|
667
|
+
try {
|
|
668
|
+
const req = await redis.get(cancelReqKey(job.id));
|
|
669
|
+
if (req === null || req === undefined) return;
|
|
670
|
+
// Stopped while this read was in flight: the outcome is being decided, so this request is left
|
|
671
|
+
// unanswered (its requester's timeout re-reads the job), never acknowledged after the fact.
|
|
672
|
+
if (pollStopped) return;
|
|
673
|
+
// Bound to this worker (the injection comment in createWorker): `false` means the job
|
|
674
|
+
// left the tracked map in this same instant, in which case the finally below clears
|
|
675
|
+
// this interval anyway and the request's TTL reaps the key.
|
|
676
|
+
const took = await Promise.resolve(cancelJob(job.id, "operator-cancel")).catch(() => false);
|
|
677
|
+
if (took === false) return;
|
|
678
|
+
// Ack first, then consume the request: losing the DEL to a blip costs one redundant
|
|
679
|
+
// re-read, losing the ack costs the operator a false "nobody answered".
|
|
680
|
+
await redis.set(cancelAckKey(job.id), hostName, "PX", CANCEL_ACK_TTL_MS).catch(() => {});
|
|
681
|
+
await redis.del(cancelReqKey(job.id)).catch(() => {});
|
|
682
|
+
clearInterval(cancelPoll);
|
|
683
|
+
} catch {
|
|
684
|
+
// Fail open: the next tick re-asks.
|
|
685
|
+
} finally {
|
|
686
|
+
cancelPolling = false;
|
|
687
|
+
}
|
|
688
|
+
})();
|
|
689
|
+
}, cancelPollMs);
|
|
690
|
+
// Like the abort-grace timer: this must never keep an otherwise-finished worker alive.
|
|
691
|
+
cancelPoll.unref?.();
|
|
692
|
+
}
|
|
589
693
|
} catch (error) {
|
|
590
694
|
// Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
|
|
591
695
|
// scope must never be releasable twice -- a double release frees another holder's slot.
|
|
@@ -600,6 +704,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
600
704
|
void scopeSlot?.release?.();
|
|
601
705
|
scopeSlot = null;
|
|
602
706
|
clearTimeout(timer);
|
|
707
|
+
clearInterval(cancelPoll);
|
|
603
708
|
throw error;
|
|
604
709
|
}
|
|
605
710
|
|
|
@@ -669,7 +774,26 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
669
774
|
// next boot reaper sweeps; what is NOT leaked is the slot, the lease and the reservation.
|
|
670
775
|
// The `stop_did_not_take` line is the loud half, because a host whose daemon ignores a stop
|
|
671
776
|
// is a fact an operator has to learn from somewhere.
|
|
672
|
-
|
|
777
|
+
// Issue #287: WHO aborted rides the result, read off the signal at the one seam both abort
|
|
778
|
+
// paths flow through (a real stop resolves through here, and boundAfterAbort's synthesised
|
|
779
|
+
// `{ code: 137, aborted: true }` does too). `cancelJob(id, reason)` becomes `signal.reason`,
|
|
780
|
+
// so the processor can classify an operator's cancel apart from the kill timer's without
|
|
781
|
+
// this file growing a second channel. Mapped only when `aborted` -- an unaborted result
|
|
782
|
+
// stays byte-identical -- and only a STRING rides: a non-string reason (AbortError objects,
|
|
783
|
+
// future bullmq surprises) collapses to null, which classifies as worker-abort, the
|
|
784
|
+
// conservative direction.
|
|
785
|
+
// Gate round 4 of PR #479: the poll STOPS at the container's observed exit, since after it there is nothing
|
|
786
|
+
// left for a cancel to stop. A cancel acknowledged before that moment raced the exit and lost nothing it was
|
|
787
|
+
// promised: the run is recorded as that cancel (the existing path, "partial work may exist") even when the
|
|
788
|
+
// container happened to exit on its own, because the operator was told the record would say operator-cancel.
|
|
789
|
+
runContainer: (ctx) =>
|
|
790
|
+
boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})).then(async (r) => {
|
|
791
|
+
await stopCancelPoll();
|
|
792
|
+
const raced = r && r.aborted !== true && signal.aborted === true && signal.reason === "operator-cancel";
|
|
793
|
+
if (raced) deps.log?.("cancel_acked_as_container_exited", { jobId: job.id, exitCode: r.code ?? null });
|
|
794
|
+
const run = raced ? { ...r, aborted: true } : r;
|
|
795
|
+
return run?.aborted ? { ...run, abortReason: typeof signal.reason === "string" ? signal.reason : null } : run;
|
|
796
|
+
}),
|
|
673
797
|
// REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
|
|
674
798
|
// to be abortable for the same reason runContainer does: a resolver blocking on an unreachable
|
|
675
799
|
// vault would otherwise hold its slot until its own timeout, and an abort landing mid-resolution
|
|
@@ -711,9 +835,149 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
711
835
|
// replacing wrapper would lose it in the same silence.
|
|
712
836
|
...(deps.checkOnceSpent ? { checkOnceSpent: (j, opts) => deps.checkOnceSpent(j, { ...opts, queueJobId: job.id }) } : {}),
|
|
713
837
|
});
|
|
714
|
-
|
|
715
|
-
|
|
838
|
+
// Gate round 4 of PR #479: stopped again as runJob returns, before anything is recorded (a result with no
|
|
839
|
+
// container never passed the exit stop above). A cancel that lands after is never acknowledged, and its
|
|
840
|
+
// requester's re-read says the job finished. One acknowledged before, on a result that is not the cancel (a
|
|
841
|
+
// pre-spend refusal that finished after the ack, since a container would have seen the abort), is recorded as
|
|
842
|
+
// the cancel the operator was told of, the result it replaced named in the log.
|
|
843
|
+
await stopCancelPoll();
|
|
844
|
+
let outcome = result;
|
|
845
|
+
if (signal.aborted === true && signal.reason === "operator-cancel" && result?.reason !== "operator-cancel") {
|
|
846
|
+
deps.log?.("cancel_acked_before_result", { jobId: job.id, was: `${result?.outcome}/${result?.reason ?? ""}` });
|
|
847
|
+
outcome = { ...result, outcome: "policy", reason: "operator-cancel" };
|
|
848
|
+
}
|
|
849
|
+
recordRun({ job, result: outcome, startedAt, endedAt: new Date().toISOString() });
|
|
850
|
+
return outcome;
|
|
716
851
|
} catch (error) {
|
|
852
|
+
// Issue #448 (gate round 2 of PR #473): a local job held until rootful Podman's service restarts goes back to the
|
|
853
|
+
// delayed set WITHOUT spending an attempt, the pause gate's move, since a service held up past one retry backoff
|
|
854
|
+
// failed a job that heals by itself (measured). Bounded by the hold's own start, stored on the job because a
|
|
855
|
+
// deferral leaves `attemptsMade` alone and a worker restart must not reset it; past the bound it fails for good
|
|
856
|
+
// with its own reason token, whose comment names the restart. Nothing is recorded per deferral: the job never
|
|
857
|
+
// started, and one record a minute for an hour would bury the run history.
|
|
858
|
+
//
|
|
859
|
+
// Gate round 3: the hour counts only while the worker kept checking. A check that comes more than two recheck
|
|
860
|
+
// periods after the last one (the queue was paused, `pi-dispatch pause`, or no worker ran) starts the hold
|
|
861
|
+
// afresh, since the time between was no evidence that the service stayed up; the last check's time is stored
|
|
862
|
+
// beside the start. And the expired hold is RECORDED with its own token, the one its comment and the failure
|
|
863
|
+
// hook carry, never the `container-never-started` of the throw it ends.
|
|
864
|
+
// A CANCELLED JOB IS NEVER RUN AGAIN (gate of PR #479). The operator-cancel poll runs from pickup to the finally
|
|
865
|
+
// below, so a cancel can be acknowledged (the signal aborted with "operator-cancel") after the job was picked up
|
|
866
|
+
// and before this handler hands it back to the queue: a hold's `moveToDelayed`, or an `InfraRetry` BullMQ
|
|
867
|
+
// retries after its backoff. Either ran the job later, with the request key already consumed, so nothing
|
|
868
|
+
// cancelled it again. So every error this handler would hand back (every `InfraRetry`, the two holds' included,
|
|
869
|
+
// since both are one) is asked ONE question first, in ONE place, so the three paths cannot drift apart: was this
|
|
870
|
+
// job cancelled? A request not yet polled is read too, and acknowledged and consumed as the poll would. Either
|
|
871
|
+
// ends the job as the operator's cancel, the policy result the aborted-container path already returns, recorded,
|
|
872
|
+
// one fixed comment, never retried. A shutdown's abort is not a cancel: that job is held or retried as before,
|
|
873
|
+
// so it survives the restart.
|
|
874
|
+
// `readPending: false` asks only whether the poll ALREADY acknowledged a cancel (the signal), and leaves a request it
|
|
875
|
+
// had not read unanswered.
|
|
876
|
+
const operatorCancelled = async ({ readPending = true } = {}) => {
|
|
877
|
+
if (signal?.aborted === true) return signal.reason === "operator-cancel";
|
|
878
|
+
if (!readPending) return false;
|
|
879
|
+
if (typeof redis?.get !== "function") return false;
|
|
880
|
+
let req = null;
|
|
881
|
+
try {
|
|
882
|
+
req = await redis.get(cancelReqKey(job.id));
|
|
883
|
+
} catch {
|
|
884
|
+
return false;
|
|
885
|
+
}
|
|
886
|
+
if (req === null || req === undefined) return false;
|
|
887
|
+
await Promise.resolve(redis.set?.(cancelAckKey(job.id), hostName, "PX", CANCEL_ACK_TTL_MS)).catch(() => {});
|
|
888
|
+
await Promise.resolve(redis.del?.(cancelReqKey(job.id))).catch(() => {});
|
|
889
|
+
return true;
|
|
890
|
+
};
|
|
891
|
+
// A retry that already spent (a container that ran and exited as infra) keeps what it spent on the record, and
|
|
892
|
+
// says the processor's own operator-cancel sentence; one that never started says it was cancelled before it did.
|
|
893
|
+
// The session rides along as the aborted-container path keeps it (`mergeSession`, already applied to the throw).
|
|
894
|
+
// A NON-retryable error (gate round 3) is not known to have started nothing, so it says the processor's own
|
|
895
|
+
// operator-cancel sentence, keeps whatever `budgetReserved` it carried, and its own words go to the log, never lost.
|
|
896
|
+
const endCancelled = async ({ retryable }) => {
|
|
897
|
+
const spent = error?.budgetReserved === true;
|
|
898
|
+
const beforeStart = retryable && !spent;
|
|
899
|
+
const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null) };
|
|
900
|
+
deps?.log?.("job_cancelled_instead_of_retry", { jobId: job.id, spent, retryable, ...(retryable ? {} : { failure: scrubCredentials(String(error?.message ?? error)).slice(0, 300) }) });
|
|
901
|
+
if (deps?.comment) await Promise.resolve(deps.comment(job.data, beforeStart ? CANCELLED_BEFORE_START_COMMENT : TERMINAL_COMMENTS["operator-cancel"])).catch(() => {});
|
|
902
|
+
recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
903
|
+
return result;
|
|
904
|
+
};
|
|
905
|
+
// STOP THE POLL BEFORE ASKING (gate round 2 of PR #479). The poll ran until the finally below, so a request that
|
|
906
|
+
// landed after the question was still acknowledged by it (ack written, request consumed, the CLI saying "cancel
|
|
907
|
+
// accepted") while the job went on to its hold or its retry and ran again. So the poll is stopped first and any
|
|
908
|
+
// tick already in flight awaited, and only then is the question asked: a cancel the poll took is seen on the
|
|
909
|
+
// signal, one it had not read is read here, and one that lands after is never acknowledged, so its requester
|
|
910
|
+
// truthfully says nothing was changed while the job is held or retried, and a second cancel removes it.
|
|
911
|
+
//
|
|
912
|
+
// EVERY ERROR, NOT ONLY A RETRY (gate round 3 of PR #479). On a non-retryable error the still-running poll could
|
|
913
|
+
// acknowledge a cancel, so the CLI said the record would say operator-cancel while it said failed. The poll is
|
|
914
|
+
// stopped the same way before the terminal path, and the question differs by one thing only. A cancel the poll
|
|
915
|
+
// ALREADY acknowledged (the signal) ends the job as operator-cancel: the requester was told that, so the record
|
|
916
|
+
// must agree, and the failure's own words go to the log. A request it had NOT read is left unanswered: the job
|
|
917
|
+
// has failed for its own reason, the record says so, and the requester's re-read then says the job finished
|
|
918
|
+
// before its worker read the cancel and nothing was changed. Consistent either way, never one said and the
|
|
919
|
+
// other recorded. A retryable error reads that request too, since the job would otherwise run again.
|
|
920
|
+
await stopCancelPoll();
|
|
921
|
+
const retryable = error instanceof InfraRetry;
|
|
922
|
+
if (await operatorCancelled({ readPending: retryable })) return await endCancelled({ retryable });
|
|
923
|
+
if (error?.holdUntilRestart === true) {
|
|
924
|
+
const heldAt = now();
|
|
925
|
+
const last = job.data?.podmanRestartHoldLastMs;
|
|
926
|
+
const resumed = !Number.isFinite(last) || heldAt - last > 2 * PODMAN_RESTART_HOLD_RECHECK_MS;
|
|
927
|
+
const since = !resumed && Number.isFinite(job.data?.podmanRestartHoldSinceMs) ? job.data.podmanRestartHoldSinceMs : heldAt;
|
|
928
|
+
if (heldAt - since < PODMAN_RESTART_HOLD_MAX_MS) {
|
|
929
|
+
await job.updateData({ ...job.data, podmanRestartHoldSinceMs: since, podmanRestartHoldLastMs: heldAt });
|
|
930
|
+
deps?.log?.("podman_restart_hold", { jobId: job.id, heldForMs: heldAt - since, delayMs: PODMAN_RESTART_HOLD_RECHECK_MS, reason: String(error.message).slice(0, 300) });
|
|
931
|
+
await job.moveToDelayed(heldAt + PODMAN_RESTART_HOLD_RECHECK_MS, token);
|
|
932
|
+
throw new DelayedError();
|
|
933
|
+
}
|
|
934
|
+
const expired = Object.assign(new UnrecoverableError(`held ${Math.round((heldAt - since) / 60_000)} min for rootful Podman's service to restart, and it did not: ${error.message}`), { reason: PODMAN_RESTART_HOLD_EXPIRED, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
935
|
+
recordRun({ job, error: expired, startedAt, endedAt: new Date().toISOString() });
|
|
936
|
+
throw expired;
|
|
937
|
+
}
|
|
938
|
+
// Issue #476: a job whose egress preflight found the rootless network keeper running on its own bridge but younger
|
|
939
|
+
// than the minimum age is HELD until it is old enough (one `moveToDelayed`, no attempt), the move the hold above
|
|
940
|
+
// makes, since a keeper merely young is not a keeper that does not hold: every joint start of the stack read it
|
|
941
|
+
// at under a second and spent a queued job's attempt (measured on 4.9.3). A crash loop is young at every start,
|
|
942
|
+
// so the hold is BOUNDED, and ends early on the evidence of one: the keeper the job waited on is gone, or another
|
|
943
|
+
// start has taken its place. Either ends in an ordinary retry (an attempt, as any keeper that does not hold
|
|
944
|
+
// costs) with its own reason token naming the loop. The hold's state is stored on the job, as the podman.service
|
|
945
|
+
// hold's is, and read only while the checks come within the bound of each other: a later one starts afresh.
|
|
946
|
+
const keeperHold = keeperHoldState(job, now());
|
|
947
|
+
if (error?.keeperYoung && typeof error.keeperYoung === "object") {
|
|
948
|
+
const young = error.keeperYoung;
|
|
949
|
+
const restarted = keeperHold.startedMs !== null && keeperHold.startedMs !== young.startedMs;
|
|
950
|
+
const heldMs = keeperHold.at - keeperHold.since;
|
|
951
|
+
if (!restarted && heldMs < NETNS_KEEPER_YOUNG_HOLD_MAX_MS) {
|
|
952
|
+
await job.updateData({ ...job.data, netnsKeeperHoldSinceMs: keeperHold.since, netnsKeeperHoldLastMs: keeperHold.at, netnsKeeperHoldStartedMs: young.startedMs });
|
|
953
|
+
deps?.log?.("netns_keeper_young_hold", { jobId: job.id, keeperAgeMs: young.ageMs, heldForMs: heldMs, delayMs: young.waitMs });
|
|
954
|
+
await job.moveToDelayed(keeperHold.at + young.waitMs, token);
|
|
955
|
+
throw new DelayedError();
|
|
956
|
+
}
|
|
957
|
+
const was = keeperHold.startedMs ?? young.startedMs;
|
|
958
|
+
await markKeeperLoop(job, was === young.startedMs ? [was] : [was, young.startedMs]);
|
|
959
|
+
const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was, now: young.startedMs, heldMs, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
960
|
+
recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
961
|
+
throw loop;
|
|
962
|
+
}
|
|
963
|
+
if (error?.reason === NETNS_KEEPER_NOT_HOLDING && keeperHold.startedMs !== null) {
|
|
964
|
+
// The keeper this job was waiting on is no longer running on its bridge: it died young, the loop's other face.
|
|
965
|
+
await markKeeperLoop(job, [keeperHold.startedMs]);
|
|
966
|
+
const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was: keeperHold.startedMs, now: null, problem: error.keeperProblem ?? null, heldMs: keeperHold.at - keeperHold.since, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
967
|
+
recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
968
|
+
throw loop;
|
|
969
|
+
}
|
|
970
|
+
// A LATER ATTEMPT OF A JOB THAT SAW THE LOOP (gate of PR #479). The loop's retry comes after the queue's 60 s
|
|
971
|
+
// backoff, when the hold's 30 s window has closed, and it meets a keeper that restarted out of order against the
|
|
972
|
+
// proxy, or none: measured, it failed as `netns-keeper-not-holding`, and that attempt's record and terminal
|
|
973
|
+
// comment replaced the loop's. The marker (the keeper starts seen) is not reset by the window, so any keeper
|
|
974
|
+
// failure of this job after it is still named the loop.
|
|
975
|
+
const loopSeen = job.data?.netnsKeeperLoopSeen;
|
|
976
|
+
if (error?.reason === NETNS_KEEPER_NOT_HOLDING && Array.isArray(loopSeen) && loopSeen.length > 0) {
|
|
977
|
+
const loop = new InfraRetry(netnsKeeperLoopAgainSentence({ seen: loopSeen, problem: error.keeperProblem ?? null, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
978
|
+
recordRun({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
979
|
+
throw loop;
|
|
980
|
+
}
|
|
717
981
|
recordRun({ job, error, startedAt, endedAt: new Date().toISOString() });
|
|
718
982
|
if (error instanceof InfraRetry) throw error; // retryable: BullMQ retries per attempts
|
|
719
983
|
// A non-retryable, non-infra error (our bug) must not retry forever. UnrecoverableError
|
|
@@ -731,11 +995,46 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
731
995
|
// an async function, and `release` never throws.
|
|
732
996
|
await scopeSlot?.release?.();
|
|
733
997
|
clearTimeout(timer);
|
|
998
|
+
clearInterval(cancelPoll);
|
|
734
999
|
signal.removeEventListener("abort", onAbort);
|
|
735
1000
|
}
|
|
736
1001
|
};
|
|
737
1002
|
}
|
|
738
1003
|
|
|
1004
|
+
/**
|
|
1005
|
+
* How long the stop of a job's cancel poll waits for a tick already in flight (gate round 4 of PR #479). About one Valkey
|
|
1006
|
+
* round trip under load, and far under the 30 s abort grace; past it a read still hanging cannot acknowledge anything.
|
|
1007
|
+
*/
|
|
1008
|
+
export const CANCEL_STOP_BOUND_MS = 1_000;
|
|
1009
|
+
|
|
1010
|
+
/** The comment for a job the operator cancelled before it started, where it would have been held or retried (gate of PR #479). */
|
|
1011
|
+
export const CANCELLED_BEFORE_START_COMMENT = "Stopped: the operator cancelled this run before it started. Nothing was spent. Not retried.";
|
|
1012
|
+
|
|
1013
|
+
/**
|
|
1014
|
+
* Store the keeper starts a crash loop was seen at on the job (gate of PR #479), outside the hold's 30 s window, so a
|
|
1015
|
+
* later attempt that fails on the keeper is still named the loop. Fail open: a lost marker costs only the name.
|
|
1016
|
+
*/
|
|
1017
|
+
async function markKeeperLoop(job, seen) {
|
|
1018
|
+
try {
|
|
1019
|
+
await job.updateData({ ...job.data, netnsKeeperLoopSeen: seen.filter((ms) => Number.isFinite(ms)) });
|
|
1020
|
+
} catch {
|
|
1021
|
+
// The loop is still named on this attempt; only a later attempt's name depends on the marker.
|
|
1022
|
+
}
|
|
1023
|
+
}
|
|
1024
|
+
|
|
1025
|
+
/**
|
|
1026
|
+
* A job's young-keeper hold as stored on it (issue #476): `{ at, since, startedMs }`, `at` being now. The stored start
|
|
1027
|
+
* and the keeper start it waited on count only while the last check was within `NETNS_KEEPER_YOUNG_HOLD_MAX_MS` of
|
|
1028
|
+
* now; an older one (a retry after its backoff, a paused queue) is a fresh hold with no keeper start seen.
|
|
1029
|
+
*/
|
|
1030
|
+
export function keeperHoldState(job, at) {
|
|
1031
|
+
const last = job?.data?.netnsKeeperHoldLastMs;
|
|
1032
|
+
const current = Number.isFinite(last) && at - last >= 0 && at - last <= NETNS_KEEPER_YOUNG_HOLD_MAX_MS;
|
|
1033
|
+
const since = current && Number.isFinite(job.data.netnsKeeperHoldSinceMs) ? job.data.netnsKeeperHoldSinceMs : at;
|
|
1034
|
+
const startedMs = current && Number.isFinite(job.data.netnsKeeperHoldStartedMs) ? job.data.netnsKeeperHoldStartedMs : null;
|
|
1035
|
+
return { at, since, startedMs };
|
|
1036
|
+
}
|
|
1037
|
+
|
|
739
1038
|
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
|
|
740
1039
|
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
741
1040
|
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
@@ -768,6 +1067,10 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
768
1067
|
// Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
|
|
769
1068
|
// and the shared handle could not reach it.
|
|
770
1069
|
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
1070
|
+
// Issue #287: the name the cancel poll writes into its ack, so the operator's terminal can say
|
|
1071
|
+
// WHICH host took the cancel. `name` is already the registry/client identity; "" for a
|
|
1072
|
+
// deployment that never declared one, and the ack's reader prints it as such.
|
|
1073
|
+
hostName: name ?? "",
|
|
771
1074
|
// #227. INJECTED, not built here. This was a one-line `docker stop` literal, which meant the abort
|
|
772
1075
|
// path -- the only thing that can end a runaway job -- was the one backend function unreachable
|
|
773
1076
|
// from `startWorker`. The wiring now passes the registry's per-job stop, so the container is
|
|
@@ -811,6 +1114,8 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
811
1114
|
recordRun,
|
|
812
1115
|
});
|
|
813
1116
|
|
|
1117
|
+
// Issue #464: only a connection `parseConnection` built, which judges and pins the Valkey it dials.
|
|
1118
|
+
assertJudgedConnection(connection);
|
|
814
1119
|
worker = new Worker(queueName, processor, {
|
|
815
1120
|
// maxRetriesPerRequest: null is REQUIRED for BullMQ's blocking connections, or it throws.
|
|
816
1121
|
connection: { ...connection, maxRetriesPerRequest: null },
|
|
@@ -821,6 +1126,9 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
821
1126
|
...(name ? { name } : {}),
|
|
822
1127
|
...(limiter ? { limiter } : {}),
|
|
823
1128
|
});
|
|
1129
|
+
// Issue #468: an error of this worker is one line, its message, never BullMQ's console.error of the whole object
|
|
1130
|
+
// (which carried a failed AUTH's password in `command.args` before connection.mjs scrubbed it).
|
|
1131
|
+
onValkeyError(worker, `worker ${queueName}`);
|
|
824
1132
|
workers.push(worker);
|
|
825
1133
|
}
|
|
826
1134
|
|
|
@@ -829,7 +1137,7 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
829
1137
|
// returned as a pair so every existing caller keeps receiving exactly what it received before.
|
|
830
1138
|
primary.hostWorker = workers[1] ?? null;
|
|
831
1139
|
|
|
832
|
-
const
|
|
1140
|
+
const stop = async () => {
|
|
833
1141
|
// Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
|
|
834
1142
|
// worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
|
|
835
1143
|
// registrations would mean two `process.exit(0)` racing, and the second worker's containers would
|
|
@@ -860,6 +1168,41 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
860
1168
|
}
|
|
861
1169
|
}),
|
|
862
1170
|
);
|
|
1171
|
+
// Release the shared ioredis client LAST (issue #300). Eight consumers ride it -- the budget, the
|
|
1172
|
+
// wait state, the leases, the run mirror, the host registry, the stall guard, the scope-claim sweep
|
|
1173
|
+
// -- and until here nothing in the product ever closed it; only the test harness did, reaching into
|
|
1174
|
+
// the captured wiring, which was the tell.
|
|
1175
|
+
//
|
|
1176
|
+
// SEQUENCED AFTER the drain above, never inside it. The raw client has no `.close`, so pushing it
|
|
1177
|
+
// into `extraClosers` is a silent no-op -- and the natural wrapper, `{ close: () => redis.disconnect() }`,
|
|
1178
|
+
// is worse than nothing: drained CONCURRENTLY by the Promise.all, it takes the connection down beside
|
|
1179
|
+
// `registry.close()`, and a recording server then received NO commands at all where this ordering
|
|
1180
|
+
// delivers the registry's DEL and SREM -- the DEL being what keeps a stopped host from lingering as
|
|
1181
|
+
// a ghost peer for its full TTL.
|
|
1182
|
+
//
|
|
1183
|
+
// `disconnect()`, not `quit()`, for a MEASURED reason rather than the plausible one. `quit()` answers
|
|
1184
|
+
// OK in 0ms against a REFUSED port; the hang it can suffer is a server that accepts the TCP
|
|
1185
|
+
// connection and never answers, where the client sits in status "connect" awaiting its ready check --
|
|
1186
|
+
// independent of `maxRetriesPerRequest` and of `enableOfflineQueue`, both measured. `disconnect()`
|
|
1187
|
+
// returns immediately in every case, and everything whose replies matter has already drained above.
|
|
1188
|
+
// Guarded, because the wiring tests hand createWorker a bare `redis: {}`. And wrapped, on the loop's
|
|
1189
|
+
// own rule stated above: a SYNCHRONOUS throw from this line would skip the `process.exit(0)` below
|
|
1190
|
+
// and hang the stop. The real client's disconnect does not throw; a test-injected one is one edit
|
|
1191
|
+
// away from doing so.
|
|
1192
|
+
try {
|
|
1193
|
+
redis?.disconnect?.();
|
|
1194
|
+
} catch {
|
|
1195
|
+
// A release that failed has already stopped mattering; the exit that follows is what a stop owes.
|
|
1196
|
+
}
|
|
1197
|
+
};
|
|
1198
|
+
// The signal path is `stop()` then exit, and the split is issue #299's: a boot that refuses AFTER the
|
|
1199
|
+
// Worker exists must be able to undo what it built WITHOUT exiting, because the refusal's own error --
|
|
1200
|
+
// not a 0 -- has to reach `cli.mjs`'s entryExitCode, and with every handle released above the process
|
|
1201
|
+
// drains to that code on its own. The closure keeps the name `shutdown` because the source pin in
|
|
1202
|
+
// `wiring.test.mjs` reads the registration lines below, deliberately, rather than constructing a
|
|
1203
|
+
// Worker to observe them.
|
|
1204
|
+
const shutdown = async () => {
|
|
1205
|
+
await stop();
|
|
863
1206
|
process.exit(0);
|
|
864
1207
|
};
|
|
865
1208
|
process.once("SIGTERM", shutdown);
|
|
@@ -869,6 +1212,13 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
869
1212
|
// worker still aborts in-flight jobs and docker-stops their containers rather than orphaning them.
|
|
870
1213
|
if (process.platform === "win32") process.once("SIGBREAK", shutdown);
|
|
871
1214
|
|
|
1215
|
+
// Beside `hostWorker` and for the same reason it rides the return value rather than changing it:
|
|
1216
|
+
// every existing caller keeps receiving exactly what it received before, and the one new caller --
|
|
1217
|
+
// `startWorker`'s post-handoff catch (issue #299) -- reaches the teardown through the worker it was
|
|
1218
|
+
// handed. `stop` is the shutdown minus the exit; it is safe to call more than once, because every
|
|
1219
|
+
// step it takes is idempotent by that step's own contract.
|
|
1220
|
+
primary.stop = stop;
|
|
1221
|
+
|
|
872
1222
|
return primary;
|
|
873
1223
|
}
|
|
874
1224
|
|