@edgehero/pi-dispatch 1.10.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +32 -5
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +387 -17
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1395 -268
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
package/src/branch.mjs CHANGED
@@ -29,7 +29,11 @@
29
29
  export function normalizeNumber(number) {
30
30
  const n = Number(number);
31
31
  if (!Number.isInteger(n) || n <= 0) {
32
- const error = new Error(`invalid target number (must be a positive integer): ${String(number)}`);
32
+ // The TYPE, never the value: `number` is a forge-payload field, and this message's history is
33
+ // exactly the leak class -- since #310 the piDispatchConfig classifier contains it, but a message
34
+ // that would carry attacker text if the classifier ever missed is a property held by luck, not
35
+ // locally (issue #289 made it local; failedReason and the job_failed line are where it would land).
36
+ const error = new Error(`invalid target number (must be a positive integer; got ${typeof number})`);
33
37
  error.piDispatchConfig = true;
34
38
  throw error;
35
39
  }
@@ -47,7 +51,8 @@ export function normalizeNumber(number) {
47
51
  */
48
52
  function positiveReplica(replica) {
49
53
  if (!Number.isInteger(replica) || replica <= 0) {
50
- const error = new Error(`invalid replica index (must be a positive integer): ${String(replica)}`);
54
+ // Type-only for normalizeNumber's reason -- a caller bug's value adds nothing a type does not.
55
+ const error = new Error(`invalid replica index (must be a positive integer; got ${typeof replica})`);
51
56
  error.piDispatchConfig = true;
52
57
  throw error;
53
58
  }
@@ -0,0 +1,174 @@
1
+ /**
2
+ * `pi-dispatch cancel <jobId>` (issue #287): the operator's stop for ONE job, whatever state it is in.
3
+ *
4
+ * VALKEY_URL-only like `pause`, and for the same reason: stopping a job must work even when forge auth is
5
+ * misconfigured. The verb finds the job across every queue this deployment drains (the kill switch's own
6
+ * fleet discovery), then dispatches on what the job IS:
7
+ *
8
+ * held on run.waitFor -> the shared removeHeldJob sequence (hold keys first, then the job); no record
9
+ * (the dispatch_wait_cancel rule).
10
+ * delayed / queued -> job.remove(); no record, same rule.
11
+ * active -> a `cancel:req` key + a brief ack poll (cancel-state.mjs), because the abort can
12
+ * only be raised by the process that holds the job. No ack within the window is a
13
+ * NAMED failure, never a silent one: the job may be on a host that is down, or on
14
+ * a worker predating this verb.
15
+ *
16
+ * Held is checked BEFORE the plain state dispatch: a held job IS delayed, and removing it through the plain
17
+ * path would strand its wait:* keys as a panel row for a job that no longer exists.
18
+ *
19
+ * What the line says about the job's PAST is read off the job, not assumed from its state (issue #477): a delayed job
20
+ * can be one waiting to retry after a failed attempt, which usually wrote a run record and may have left a retained
21
+ * sandbox, and "it never ran" was printed about it. `ranBefore` (cancel-state.mjs, shared with the panel) words it from the
22
+ * job's attempt count.
23
+ */
24
+
25
+ import { cancelReqKey, ranBefore, removeHeldJob, requestCancel } from "./cancel-state.mjs";
26
+ import { jobKey } from "./wait-state.mjs";
27
+
28
+ /** Same budget the kill switch gives the host registry before acting on what the keyspace alone says. */
29
+ const FLEET_READ_TIMEOUT_MS = 2_000;
30
+
31
+ /**
32
+ * Run the cancel. Returns the process exit code. Every collaborator is a seam with the production default,
33
+ * so tests drive states and races without a queue; `write` is the stdout seam cli.mjs already injects.
34
+ */
35
+ export async function runCancel(jobId, url, { write = (chunk) => process.stdout.write(chunk), errWrite = (chunk) => process.stderr.write(chunk), redisFn, queueFn, parseConnectionFn, readLiveHostsFn, discoverHostQueuesFn, fleetQueueNamesFn, unionQueueNamesFn, ackTimeoutMs = 10_000, pollMs = 250, sleep, refusalFn } = {}) {
36
+ const fail = (message) => {
37
+ errWrite(`error: ${message}\n`);
38
+ return 1;
39
+ };
40
+ if (typeof jobId !== "string" || jobId === "") {
41
+ return fail("a job id is required: pi-dispatch cancel <jobId> (ids are in `pi-dispatch status`, the panel, and the run log)");
42
+ }
43
+ // Lazy imports, the pause verb's shape: a mistyped id must not load bullmq before it can be refused.
44
+ const { parseConnection, makeRedisClient } = await import("./connection.mjs");
45
+ const { fleetQueueNames, discoverHostQueues, unionQueueNames, makeQueue } = await import("./queue.mjs");
46
+ const { readLiveHosts } = await import("./host-registry.mjs");
47
+ const parseConn = parseConnectionFn ?? parseConnection;
48
+ const mkRedis = redisFn ?? makeRedisClient;
49
+ const mkQueue = queueFn ?? makeQueue;
50
+ const liveHosts = readLiveHostsFn ?? readLiveHosts;
51
+ const discover = discoverHostQueuesFn ?? discoverHostQueues;
52
+ const fleetNames = fleetQueueNamesFn ?? fleetQueueNames;
53
+ const unionNames = unionQueueNamesFn ?? unionQueueNames;
54
+
55
+ // Issue #464 (gate round 3): judged before anything is sent, so a refused Valkey is said as the refusal it is. Only
56
+ // where this command builds its own clients (a test's fakes stand in for Valkey itself), or a test's `refusalFn`.
57
+ const refusalOf =
58
+ refusalFn ??
59
+ (redisFn
60
+ ? async () => null
61
+ : async (u) => {
62
+ const { judgeValkeyAtStart, defaultValkeyContext } = await import("./connection.mjs");
63
+ try {
64
+ await judgeValkeyAtStart(u, defaultValkeyContext(), { waitMs: 0 });
65
+ return null;
66
+ } catch (error) {
67
+ return error?.valkeyRefused ? error.message : null;
68
+ }
69
+ });
70
+ const refused = await refusalOf(url);
71
+ if (refused) return fail(refused);
72
+ const probe = mkRedis(url);
73
+ probe.on?.("error", () => {});
74
+ const queues = [];
75
+ let requested = false;
76
+ try {
77
+ // The job could sit on the shared queue or on any host's own (issue #57): find it the way pause
78
+ // spans them, and fail OPEN but LOUDLY on a degraded registry read, exactly as the kill switch does.
79
+ const [fleet, existing] = await Promise.all([
80
+ liveHosts(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }).catch((error) => ({ unreachable: error?.message ?? String(error) })),
81
+ discover(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }),
82
+ ]);
83
+ const blind = fleet?.unreachable ?? null;
84
+ const names = unionNames(fleetNames(fleet?.hosts), existing);
85
+
86
+ // Constructed INSIDE the try (the pause loop's leak posture); first queue that knows the id wins.
87
+ // Job ids are delivery GUIDs, `local-<hex>` or `repeat:<id>:<millis>`, so a cross-queue duplicate
88
+ // is not a state this project can produce.
89
+ let job = null;
90
+ let owner = null;
91
+ for (const name of names) {
92
+ const q = mkQueue(parseConn(url, { failFast: true }), { name });
93
+ queues.push(q);
94
+ job = await q.getJob(jobId);
95
+ if (job) {
96
+ owner = q;
97
+ break;
98
+ }
99
+ }
100
+ if (!job) {
101
+ return fail(`no job ${jobId} in ${names.length} queue(s)${blind ? ` [registry unreadable: ${blind} — a host that is not registered may still hold it]` : ""}`);
102
+ }
103
+
104
+ // Held FIRST (see the header). The hash tells held apart from plain-delayed; `since` is the clock
105
+ // that makes it a real hold rather than a counter key.
106
+ const hash = await probe.hgetall(jobKey(jobId)).catch(() => null);
107
+ if (hash?.since) {
108
+ const res = await removeHeldJob({ redis: probe, queue: owner, jobId });
109
+ if (res.ok) {
110
+ // A hold consumes no attempt (`skipAttempt`), so a held job that ran is a retry held again by its conditions.
111
+ write(`cancelled ${jobId}: it was held on its wait condition, and ${ranBefore(res)}\n`);
112
+ return 0;
113
+ }
114
+ return fail(`could not cancel held job ${jobId}: ${res.invalid}`);
115
+ }
116
+
117
+ let state = await job.getState().catch(() => "unknown");
118
+ if (state === "delayed" || state === "waiting" || state === "prioritized" || state === "paused") {
119
+ try {
120
+ await job.remove();
121
+ write(`removed ${jobId} (${state}): ${ranBefore(job)}\n`);
122
+ return 0;
123
+ } catch (error) {
124
+ // The one race worth handling by name: picked up between getState and remove. remove() throws
125
+ // on a locked job, so re-read and fall through to the active path rather than failing an
126
+ // operator whose job is now exactly the case the active path exists for.
127
+ state = await job.getState().catch(() => "unknown");
128
+ // The re-read state and the throw's own words together, labelled apart: the error came from
129
+ // the remove attempt, the state from after it, and gluing them unlabelled read as one
130
+ // mismatched diagnosis (review finding).
131
+ if (state !== "active") return fail(`could not remove ${jobId} — it is now ${state}; the remove failed with: ${error.message}`);
132
+ }
133
+ }
134
+
135
+ if (state === "active") {
136
+ requested = true;
137
+ const res = await requestCancel({ redis: probe, jobId, ackTimeoutMs, pollMs, ...(sleep ? { sleep } : {}) });
138
+ if (res.ack !== undefined) {
139
+ write(`cancel accepted by ${res.ack === "" ? "the worker" : res.ack} — stopping the container (allow ~30s); the run record will say operator-cancel\n`);
140
+ return 0;
141
+ }
142
+ // Gate round 2 of PR #479: the worker stops its cancel poll before it holds or retries a job, so a request that
143
+ // lands in that moment is never acknowledged and the job is back in the queue. Re-read, so the words are true.
144
+ // Gate round 3: by the state, since a job that finished or left the queue cannot be removed by a second cancel.
145
+ const after = await job.getState().catch(() => "unknown");
146
+ const waited = `no worker acknowledged within ${Math.round(ackTimeoutMs / 1000)}s`;
147
+ if (after === "waiting" || after === "delayed" || after === "prioritized" || after === "paused") return fail(`${waited}: the job went back to the queue before its worker read the cancel (it is now ${after}), and nothing was changed. Run \`pi-dispatch cancel ${jobId}\` again to remove it`);
148
+ if (after === "completed" || after === "failed") return fail(`${waited}: the job finished (${after}) before its worker read the cancel, and nothing was changed`);
149
+ if (after !== "active") return fail(`${waited}: the job is no longer in the queue (its state reads ${after}), and nothing was changed`);
150
+ return fail(`no worker acknowledged within ${Math.round(ackTimeoutMs / 1000)}s — the job is active but no reachable worker owns it (host down, or a worker predating cancel); nothing was changed`);
151
+ }
152
+
153
+ return fail(`job ${jobId} is ${state} — nothing to cancel`);
154
+ } catch (error) {
155
+ // Best-effort, and ONLY when a request was actually placed: an abandoned request must not fire
156
+ // after the operator read an error and walked away -- but against a Valkey that is DOWN, a bare
157
+ // del would sit in ioredis's retry queue and hold this process open long past the error message.
158
+ // The bound timer is CLEARED when the del wins the race (discoverHostQueues' own rule): cli.mjs
159
+ // sets exitCode rather than exiting, so a stray pending timer is a second of pure hang.
160
+ if (requested) {
161
+ let bound;
162
+ await Promise.race([
163
+ Promise.resolve(probe.del?.(cancelReqKey(jobId))).catch(() => {}),
164
+ new Promise((r) => (bound = setTimeout(r, 1000))),
165
+ ]);
166
+ clearTimeout(bound);
167
+ }
168
+ return fail(error?.valkeyRefused ? error.message : `could not reach Valkey at ${(await import("./connection.mjs")).urlShown(url)}: ${(await import("./valkey-auth.mjs")).valkeyDownHint(url)}\n ${error.message}`);
169
+ } finally {
170
+ probe.disconnect?.();
171
+ for (const q of queues) await q.close().catch(() => {});
172
+ }
173
+ }
174
+
@@ -0,0 +1,125 @@
1
+ /**
2
+ * The `cancel:` keyspace: how an operator's cancel reaches the worker that owns a running job (issue #287).
3
+ *
4
+ * BullMQ's `cancelJob` aborts an entry in ONE process's tracked-job map -- it is unreachable from another
5
+ * process, returns `false` silently for a job it does not hold, and nothing anywhere maps a jobId to the
6
+ * host draining it. So the CLI and the panel cannot call it; they can only ask. This keyspace is the ask:
7
+ *
8
+ * cancel:req:<jobId> STRING "operator-cancel", TTL 60s -- the request. Written by the CLI or the
9
+ * panel; consumed (DEL) by the worker that holds the job; deleted by the
10
+ * requester itself when it gives up, so a request the operator was told failed
11
+ * cannot fire a minute later. The TTL is the backstop for a requester that died.
12
+ * cancel:ack:<jobId> STRING <workerName> -- the answer, written by the worker AFTER it raised the
13
+ * abort. A name, possibly "", never a path (INT-HOST-REGISTRY-CONTRACT's content
14
+ * rule). TTL 5m: long enough for a requester that polls slowly, gone before the
15
+ * id could plausibly be reused.
16
+ *
17
+ * WHY A DURABLE KEY AND NOT PUB/SUB. A subscriber is a dedicated connection type this repo has zero of, and
18
+ * fire-and-forget loses a cancel across a worker restart or a blip with no error anywhere -- the silent
19
+ * no-op this project refuses. The key survives until someone answers or the TTL says nobody will, it is
20
+ * operator-inspectable (`KEYS cancel:*`, the wait-state doctrine), and the ack the acceptance requires
21
+ * ("a cancel of a job this host does not own says so") needs a readable key regardless.
22
+ *
23
+ * WHY THIS IS NOT THE REDIS STATE OQ-008 REFUSED. That refusal is about durable CONFIG whose deletion
24
+ * silently loses an operator's edit. This is a transient, attended, TTL-bounded one-shot whose loss is
25
+ * REPORTED: the requester is watching the ack poll, and a request that reaches nobody comes back as a
26
+ * named timeout, never as silence.
27
+ */
28
+
29
+ import { HELD_SET, jobKey, leaseKey } from "./wait-state.mjs";
30
+
31
+ /** How long an unanswered request may sit before redis reaps it (the requester deletes it sooner). */
32
+ export const CANCEL_REQ_TTL_MS = 60_000;
33
+
34
+ /** How long the worker's answer stays readable. Generous: an ack outliving its poller costs nothing. */
35
+ export const CANCEL_ACK_TTL_MS = 5 * 60_000;
36
+
37
+ export const cancelReqKey = (jobId) => `cancel:req:${jobId}`;
38
+ export const cancelAckKey = (jobId) => `cancel:ack:${jobId}`;
39
+
40
+ /**
41
+ * Ask whichever worker holds `jobId` to cancel it, and wait briefly for the answer.
42
+ *
43
+ * Resolves `{ ack: host }` when a worker acknowledged (host may be ""), or `{ timeout: true }` when nobody
44
+ * did within `ackTimeoutMs` -- in which case the request key is deleted first, so the operator's "nothing
45
+ * was changed" stays true after they walk away. Redis errors propagate: the callers (CLI, panel deps) each
46
+ * already own a could-not-reach-Valkey message, and swallowing here would turn an unreachable Valkey into
47
+ * a lying "no worker acknowledged".
48
+ *
49
+ * `pollMs`/`sleep` are injectable so tests never wait on a wall clock.
50
+ */
51
+ export async function requestCancel({ redis, jobId, ackTimeoutMs = 10_000, pollMs = 250, sleep = (ms) => new Promise((r) => setTimeout(r, ms)) }) {
52
+ await redis.set(cancelReqKey(jobId), "operator-cancel", "PX", CANCEL_REQ_TTL_MS);
53
+ for (let waited = 0; ; waited += pollMs) {
54
+ const host = await redis.get(cancelAckKey(jobId));
55
+ if (host !== null && host !== undefined) return { ack: host };
56
+ if (waited + pollMs > ackTimeoutMs) break;
57
+ await sleep(pollMs);
58
+ }
59
+ // Best-effort: losing this DEL to a blip leaves the 60s TTL as the backstop, and the worker-side GET
60
+ // racing it by one tick is a named residual (the operator is told to re-check `status`).
61
+ await redis.del(cancelReqKey(jobId)).catch(() => {});
62
+ return { timeout: true };
63
+ }
64
+
65
+ /**
66
+ * What a removed job had done before a cancel removed it, in the CLI's and the panel's words (issue #477), from BullMQ's
67
+ * own counter. `attemptsMade` counts the attempts that FINISHED (moveToFinished/moveToFailed, bullmq 5.80.4). Most of
68
+ * them wrote a run record, but not all: a gate above the processor's `try` (a `moveToDelayed` or a wait-state read that
69
+ * rejects) fails the attempt with nothing recorded, by design (index.mjs). So the sentence says an attempt was made and
70
+ * that whatever it recorded stays, never that a record exists. NOT `attemptsStarted`: a hold on run.waitFor, a
71
+ * pause-gate move and a rootful-Podman deferral each take the job and hand it back to the delayed set with
72
+ * `skipAttempt`, which counts a start and no attempt (moveToDelayed), so that counter is non-zero on jobs that never
73
+ * ran. Cancel itself writes no record in any case.
74
+ */
75
+ export function ranBefore(job) {
76
+ const made = Number.isInteger(job?.attemptsMade) && job.attemptsMade > 0 ? job.attemptsMade : 0;
77
+ if (made === 0) return "it never ran; no record written";
78
+ return made === 1
79
+ ? "it made 1 attempt before and was waiting to retry; whatever that attempt recorded stays, and cancel writes none"
80
+ : `it made ${made} attempts before and was waiting to retry; whatever those attempts recorded stays, and cancel writes none`;
81
+ }
82
+
83
+ /**
84
+ * Remove a job that is held on a `run.waitFor` condition: hold keys first, then the job (issue #230's
85
+ * sequence, issue #287's shared home). This is the inner body of the admin's `cancelHeldJob`, moved here
86
+ * so the CLI's held path and the panel's are ONE sequence rather than two copies that can drift -- the
87
+ * read-model's own rule is that key derivations are imported from the worker, never re-implemented.
88
+ *
89
+ * Returns `{ ok: true, jobId, attemptsMade }` or `{ invalid: <reason> }`; never throws. Writes no run record of its
90
+ * own (INT-RUN-HISTORY-FILE-CONTRACT records terminal states of runs, and a removed hold is none); `attemptsMade` is
91
+ * what the job had done before, for the caller's sentence (`ranBefore`, issue #477): a job held again on a retry has
92
+ * made attempts, and whatever they recorded stays.
93
+ */
94
+ export async function removeHeldJob({ redis, queue, jobId }) {
95
+ try {
96
+ const hash = await redis.hgetall(jobKey(jobId));
97
+ // A hold is a hash with a CLOCK, not merely a non-empty hash: the worker's own counters create this
98
+ // key before anything is held, so "non-empty" would let this tool reach a job that is not waiting.
99
+ if (!hash || !hash.since) return { invalid: `job ${jobId} is not waiting on a condition` };
100
+ const job = await queue.getJob(jobId);
101
+ if (!job) return { invalid: `job ${jobId} is no longer in the queue` };
102
+ // STATE, not existence. `release` is fail-open by design, so a redis blip can leave the hash behind
103
+ // while the job wakes, runs and completes -- and bullmq will happily `remove()` a completed job. That
104
+ // would answer `applied: true` to an operator who approved a dialog reading "It will never run", for
105
+ // a job whose run record is already on disk.
106
+ const state = await job.getState().catch(() => null);
107
+ if (state !== "delayed" && state !== "waiting" && state !== "prioritized") {
108
+ return { invalid: `job ${jobId} is ${state ?? "in an unknown state"}, not waiting -- it has already left the hold` };
109
+ }
110
+ // The hold goes FIRST. If `remove` throws (an active job is locked) or a caller's timeout fires
111
+ // mid-sequence, an orphaned hash would keep a row on the panel for a job that no longer exists; an
112
+ // orphaned JOB is merely a job that still runs, which is the state the operator was already in.
113
+ await redis.del(jobKey(jobId));
114
+ await redis.srem(HELD_SET, jobId).catch(() => {});
115
+ if (hash.dedupId) {
116
+ const holder = await redis.get(leaseKey(hash.dedupId));
117
+ if (holder === jobId) await redis.del(leaseKey(hash.dedupId));
118
+ }
119
+ await job.remove();
120
+ // Issue #477: what the job had done, for the caller's sentence (`ranBefore`); read off the job just removed.
121
+ return { ok: true, jobId, attemptsMade: Number.isInteger(job.attemptsMade) && job.attemptsMade > 0 ? job.attemptsMade : 0 };
122
+ } catch (err) {
123
+ return { invalid: err?.message ?? String(err) };
124
+ }
125
+ }