@edgehero/pi-dispatch 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,315 @@
1
+ /**
2
+ * `run.waitFor`: holding a job until a condition clears (issue #230).
3
+ *
4
+ * Two tiers, split by WHO CAN ANSWER, and the split is not cosmetic: one tier is free and the other
5
+ * spawns a process.
6
+ *
7
+ * { "after": "2026-09-01T09:00:00Z" } tier 1 -- a wall-clock instant, answered in-process from the
8
+ * clock alone. One exact `moveToDelayed`, no polling at all.
9
+ * { "profile": "jira" } tier 2 -- an operator-declared executable answers go / not yet
10
+ * / cannot tell / never. Polled, and every poll is bounded.
11
+ *
12
+ * This module holds the two halves that BOTH the loader and the gate must agree on, so neither side
13
+ * re-derives them: what a legal `after` instant is, and what `PI_WAIT_PROFILES` declares. The trigger-side
14
+ * validator lives in `triggers.mjs` beside its siblings (`validateSecretsProfile`'s split), and reads this
15
+ * file rather than restating it. The spawner that runs a profile will do the same when it lands; NOTHING
16
+ * evaluates a wait yet, and this module is the only part of the feature currently wired to anything.
17
+ *
18
+ * PURE, FS-FREE AND ENV-FREE, for `secret-profiles.mjs`' reason: `config.mjs` imports this module and
19
+ * `admin/build.mjs` inlines that chain into the published console. Whether a declared path EXISTS is asked
20
+ * once, on the worker that is about to spawn it.
21
+ *
22
+ * `configError` is a local copy rather than an import from `config.mjs`, which imports this module. The
23
+ * cycle would resolve (function declarations hoist) but would be a trap for the next reader -- the same
24
+ * duplication, for the same reason, is in `secret-profiles.mjs` and `env-allowlist.mjs`.
25
+ */
26
+
27
+ import { isAbsolute } from "node:path";
28
+
29
+ function configError(message) {
30
+ const error = new Error(message);
31
+ error.piDispatchConfig = true;
32
+ return error;
33
+ }
34
+
35
+ /**
36
+ * The condition keys this version understands. `exclusive` is deliberately ABSENT and is refused by name
37
+ * in `triggers.mjs`: issue #242 shipped an always-on one-job-per-folder mutex at this very gate, so the
38
+ * condition it asked for is structural now, and `scoped-limits.json` covers the wider per-scope case.
39
+ * Accepting-and-ignoring it would be the silent no-op this project refuses.
40
+ */
41
+ export const WAIT_CONDITION_KEYS = ["after", "profile"];
42
+
43
+ /**
44
+ * The ceiling on conditions per trigger. Four, on `SECRETS_MAX`'s reasoning rather than `REPLICAS_MAX`':
45
+ * this multiplies no spend, but every `profile` condition is a subprocess evaluated before the container,
46
+ * so an unbounded array is an unbounded slot occupancy. Four is generous for a conjunction a human wrote.
47
+ */
48
+ export const WAIT_CONDITION_MAX = 4;
49
+
50
+ /**
51
+ * The re-check cadence floor, and the ceiling the elapsed-derived backoff climbs to.
52
+ *
53
+ * CLAMPED UP, NEVER REFUSED, which is the poller's posture and its reason verbatim: "a typo'd `1` must not
54
+ * turn the harness into a hammer". Note what that does and does not cover -- `positiveInt` still refuses
55
+ * `0`, a negative, a fraction and a non-number at boot, so only a positive integer BELOW the floor is
56
+ * quietly raised. The floor is 30s rather than the poller's because there is no third party asking for
57
+ * politeness here; what it protects is this worker's own concurrency slots.
58
+ *
59
+ * The floor is also load-bearing in a second place, and lowering it would break that silently: it is what
60
+ * lets a held job be told apart from a scope deferral, whose re-check is `SCOPE_BUSY_RECHECK_MS` (5s). Any
61
+ * value at or below that would make the two indistinguishable by wake instant.
62
+ *
63
+ * `WAIT_INTERVAL_MAX_MS` is the ceiling the backoff climbs to, and it is a ceiling on the BACKOFF, never on
64
+ * the operator: the rule the backoff must implement is `min(max(base·2^k, base), max(WAIT_INTERVAL_MAX_MS,
65
+ * base))`, so an operator who deliberately configured an hourly cadence to save money keeps it. Writing the
66
+ * cap as a bare `min(..., 900_000)` is the obvious spelling and it silently turns their hour into fifteen
67
+ * minutes -- a 4x cost overrun, in the direction they were trying to avoid, from a knob this file documents
68
+ * as clamped UPWARD.
69
+ */
70
+ export const WAIT_INTERVAL_FLOOR_MS = 30_000;
71
+ export const WAIT_INTERVAL_MAX_MS = 900_000;
72
+
73
+ /**
74
+ * The default ceiling on how far out an `after` instant may sit, and deliberately NOT the maximum hold.
75
+ *
76
+ * An `after` polls nothing: it is one exact `moveToDelayed` to an instant, self-terminating and costing
77
+ * nothing while it waits, so bounding it by the polling budget would refuse "hold this until the
78
+ * maintenance window next month" for a reason about subprocesses it never runs. Thirty days is a bound on
79
+ * how far ahead a REVIEWED FILE may schedule, not a bound on cost.
80
+ *
81
+ * A literal shared by `config.mjs` and the processor's own default, so a bare `makeProcessor` under test
82
+ * and a wired worker agree on what "too far" means.
83
+ */
84
+ export const WAIT_AFTER_MAX_DEFAULT_MS = 30 * 24 * 3600 * 1000;
85
+
86
+ /**
87
+ * How long to wait before the next check, given the configured base and how long this job has been held.
88
+ *
89
+ * Doubles every ten base periods and settles at `WAIT_INTERVAL_MAX_MS` -- **or at `base`, whichever is
90
+ * larger**. That `Math.max` is the whole reason this is a function and not a constant the caller clamps
91
+ * against: the obvious spelling, `Math.min(grown, WAIT_INTERVAL_MAX_MS)`, silently turns a deliberately
92
+ * configured hourly cadence into a fifteen-minute one, which is a 4x cost overrun in exactly the direction
93
+ * the operator was economising. Shipped with the constant so the rule cannot be re-derived wrongly later.
94
+ *
95
+ * DERIVED FROM ELAPSED, never stored: a deferral consumes no attempt (`moveToDelayed` passes
96
+ * `skipAttempt: true`), so `attemptsMade` cannot carry a retry count and nothing else may either. Reading
97
+ * the schedule off the clock makes it survive a worker restart for free.
98
+ */
99
+ export function waitBackoffMs(baseMs, elapsedMs) {
100
+ const base = Number.isFinite(baseMs) && baseMs > 0 ? baseMs : WAIT_INTERVAL_FLOOR_MS;
101
+ const elapsed = Number.isFinite(elapsedMs) && elapsedMs > 0 ? elapsedMs : 0;
102
+ const grown = base * 2 ** Math.floor(elapsed / (10 * base));
103
+ return Math.min(Math.max(grown, base), Math.max(WAIT_INTERVAL_MAX_MS, base));
104
+ }
105
+
106
+ /**
107
+ * An ISO-8601 instant that carries its own zone: `Z`, `z`, or a numeric offset. Seconds and fractional
108
+ * seconds are optional.
109
+ *
110
+ * THE ZONE IS REQUIRED, and that is the whole point of not using bare `Date.parse`. `Date.parse` accepts
111
+ * "2026-09-01T09:00:00" and resolves it against the WORKER'S local zone, so the same reviewed file would
112
+ * hold a job for a different instant on a machine in Amsterdam than on one in UTC, silently, with nothing
113
+ * in the file to explain the difference. `pause-windows.mjs` solves the same problem the other way (an
114
+ * explicit `tz` field, defaulting to UTC) because its times recur; a one-shot instant can simply carry its
115
+ * own offset, so requiring one is cheaper than inventing a second timezone field.
116
+ *
117
+ * The TIME fields are range-bounded in the pattern rather than left to `Date.parse`, for the same reason the
118
+ * calendar is re-checked below: `Date.parse` reads `T24:00:00` as the next day's midnight, so an hour field
119
+ * nobody can write on a clock would silently produce a hold one day later than the file says. `:60` in the
120
+ * minute or second field is refused for the same reason (there are no leap seconds to honour here, and a
121
+ * value that rolls is worse than a value that refuses).
122
+ */
123
+ const AFTER_INSTANT = /^\d{4}-\d{2}-\d{2}[Tt]([01]\d|2[0-3]):[0-5]\d(:[0-5]\d(\.\d+)?)?([Zz]|[+-]\d{2}:\d{2})$/;
124
+
125
+ /**
126
+ * A profile name. The same charset `triggers.mjs` allows in `run.waitFor[].profile`, and a local copy for
127
+ * `secret-profiles.mjs`' stated reason: a name that failed here after passing there would be an
128
+ * operator-visible contradiction between two files meant to agree. `,` and `:` are excluded because they
129
+ * are `PI_WAIT_PROFILES`' own separators, so a name carrying either could not round-trip.
130
+ */
131
+ const PROFILE_NAME = /^[A-Za-z0-9._-]+$/;
132
+
133
+ /**
134
+ * The epoch ms of a legal `after` value, or `null` when the text is not one.
135
+ *
136
+ * ONE DEFINITION, READ BY BOTH SIDES. The loader calls this to refuse a malformed instant at load; the
137
+ * pickup gate calls it to decide how long to hold. A second spelling in either place is how a file that
138
+ * loads clean starts holding for an instant nobody wrote.
139
+ */
140
+ export function parseAfterInstant(text) {
141
+ if (typeof text !== "string" || !AFTER_INSTANT.test(text)) return { error: "shape" };
142
+ const ms = Date.parse(text);
143
+ if (!Number.isFinite(ms)) return { error: "calendar" }; // an out-of-range month: "2026-13-01T00:00Z"
144
+ // An out-of-range DAY does not reach that check, and this is the surprise worth refusing: `Date.parse`
145
+ // rejects month 13 but silently ROLLS "2026-02-31T00:00:00Z" forward to March 3rd. An operator who
146
+ // mistyped a date would get a hold that ends on a day they did not write, with the file still reading
147
+ // as though they had. The calendar fields are checked against the date part alone, so an offset in the
148
+ // text cannot move the day being validated: the operator wrote that date, whatever zone it is in.
149
+ const [year, month, day] = text.slice(0, 10).split("-").map(Number);
150
+ const probe = new Date(Date.UTC(year, month - 1, day));
151
+ // `Date.UTC` maps a two-digit year onto 19xx, so a year below 100 would fail the comparison below and
152
+ // refuse for a reason that has nothing to do with the operator's date. Absurd as a wait, but a refusal
153
+ // the code did not mean to make is still a refusal nobody can act on.
154
+ if (year < 100) probe.setUTCFullYear(year);
155
+ if (probe.getUTCFullYear() !== year || probe.getUTCMonth() !== month - 1 || probe.getUTCDate() !== day) {
156
+ return { error: "calendar" };
157
+ }
158
+ return { ms };
159
+ }
160
+
161
+ /**
162
+ * The epoch ms of a legal `after` value, or `null`. The runtime spelling of `parseAfterInstant`, which the
163
+ * gate wants because at pickup there is nothing to say: the loader already refused anything unparseable, so
164
+ * by then the only question is what instant to hold until.
165
+ */
166
+ export function afterInstantMs(text) {
167
+ const parsed = parseAfterInstant(text);
168
+ return parsed.error ? null : parsed.ms;
169
+ }
170
+
171
+ /**
172
+ * Parse `PI_WAIT_PROFILES`: `name:/abs/path,other:/abs/other`. Returns `{ [name]: path }`, empty when
173
+ * unset. Throws (config-tagged, so the worker refuses to boot) on anything malformed.
174
+ *
175
+ * The twin of `parseSecretProfiles`, deliberately duplicated rather than generalised into a shared helper.
176
+ * The two variables mean different things (one resolves a value, one answers a question), they will drift
177
+ * apart as this one grows bounds the other has no use for, and a shared parser would have to be told which
178
+ * variable name to put in every error message -- which is the only part an operator reads.
179
+ *
180
+ * SET-BUT-GARBLED FAILS LOUD: a silently dropped entry is a profile the operator believes is wired, and
181
+ * every trigger naming it would refuse at delivery time with the operator looking at the line that appears
182
+ * to declare it. Each entry splits on its FIRST colon, so `prod:C:\pi\wait.cmd` parses on Windows.
183
+ */
184
+ export function parseWaitProfiles(raw) {
185
+ if (raw === undefined || raw === null || String(raw).trim() === "") return Object.create(null);
186
+ // PROTOTYPE-FREE, and that is a correctness requirement rather than hygiene: a profile named `toString`
187
+ // or `constructor` passes ID_CHARSET, so on a `{}` table `profiles[name]` answers with an inherited
188
+ // FUNCTION instead of `undefined` and the gate's "is this profile declared?" check silently says yes --
189
+ // for a name no operator declared, with the variable unset entirely. `Object.create(null)` also stops
190
+ // `name in profiles` from reporting a duplicate that was never written.
191
+ const profiles = Object.create(null);
192
+ for (const entry of String(raw).split(",")) {
193
+ const text = entry.trim();
194
+ if (text === "") continue; // a trailing comma is a typo, not a declaration
195
+ const cut = text.indexOf(":");
196
+ if (cut <= 0) {
197
+ throw configError(`PI_WAIT_PROFILES entries must be name:/absolute/path, got ${JSON.stringify(text)}`);
198
+ }
199
+ const name = text.slice(0, cut).trim();
200
+ const path = text.slice(cut + 1).trim();
201
+ if (!PROFILE_NAME.test(name)) {
202
+ throw configError(`PI_WAIT_PROFILES profile name ${JSON.stringify(name)} may use letters, digits, dot, dash and underscore only`);
203
+ }
204
+ if (name in profiles) {
205
+ throw configError(`PI_WAIT_PROFILES declares ${JSON.stringify(name)} twice -- one of the two is not the check you think is running`);
206
+ }
207
+ if (path === "" || !isAbsolutePath(path)) {
208
+ throw configError(`PI_WAIT_PROFILES profile ${JSON.stringify(name)} needs an ABSOLUTE path to its check -- a service manager's working directory is not your shell's, so a relative path is a different file on every host`);
209
+ }
210
+ profiles[name] = path;
211
+ }
212
+ return profiles;
213
+ }
214
+
215
+ /** Absolute on this platform, accepting a Windows drive letter or UNC root as `service.mjs` does. */
216
+ function isAbsolutePath(path) {
217
+ return isAbsolute(path) || /^([A-Za-z]:[\\/]|\\\\)/.test(path);
218
+ }
219
+
220
+ /**
221
+ * True when this job carries any wait condition at all. The ONE guard every wait code path is behind, so
222
+ * an unflagged job takes zero new branches and its record, job data and key set stay byte-identical to a
223
+ * run prepared before this field existed.
224
+ */
225
+ export function waitArmed(job) {
226
+ return Array.isArray(job?.waitFor) && job.waitFor.length > 0;
227
+ }
228
+
229
+ /**
230
+ * The `after` instant this job is holding for, or `null`. At most one `after` per array is legal (the
231
+ * loader refuses a second), so this is a lookup, not a fold.
232
+ */
233
+ export function afterMs(job) {
234
+ if (!waitArmed(job)) return null;
235
+ for (const condition of job.waitFor) {
236
+ const ms = afterInstantMs(condition?.after);
237
+ if (ms !== null) return ms;
238
+ }
239
+ return null;
240
+ }
241
+
242
+ /**
243
+ * The conditions on this job that this worker cannot read, as written.
244
+ *
245
+ * The loader refuses all of these, and the loader is a DIFFERENT PROCESS: `job.data.waitFor` reaches the
246
+ * gate over Redis from the receiver, so a receiver newer than the worker can enqueue a condition shape this
247
+ * build has no branch for. Returning them rather than ignoring them is what lets the gate refuse instead of
248
+ * silently treating an unreadable condition as a satisfied one -- the forward half of the same version skew
249
+ * `makeCheckWaitSkew` closes backward.
250
+ *
251
+ * Deliberately structural rather than a re-run of the loader's validator: this asks only "can I act on
252
+ * this?", so a future worker that learns a new condition answers differently here without the two
253
+ * validators having to agree on every message.
254
+ */
255
+ export function unreadableConditions(job) {
256
+ if (!waitArmed(job)) return [];
257
+ // A SECOND `after` is unreadable even though each one parses: the conjunction has no defined meaning
258
+ // with two instants, `afterMs` answers with whichever comes first, and the loader refuses the shape --
259
+ // so a job carrying one arrived from something newer or something wrong, which is this function's whole
260
+ // subject. Counted here rather than in the shape filter below, which sees one condition at a time.
261
+ let afters = 0;
262
+ for (const c of job.waitFor) if (c && typeof c === "object" && !Array.isArray(c) && Object.keys(c)[0] === "after") afters += 1;
263
+ if (afters > 1) return [...job.waitFor];
264
+ return job.waitFor.filter((condition) => {
265
+ if (condition === null || typeof condition !== "object" || Array.isArray(condition)) return true;
266
+ const keys = Object.keys(condition);
267
+ if (keys.length !== 1 || !WAIT_CONDITION_KEYS.includes(keys[0])) return true;
268
+ if (keys[0] === "after") return afterInstantMs(condition.after) === null;
269
+ return !isDeclarableName(condition.profile);
270
+ });
271
+ }
272
+
273
+ /** A profile name this worker will put in a log line and a public comment. See `waitLabel`. */
274
+ function isDeclarableName(value) {
275
+ return typeof value === "string" && PROFILE_NAME.test(value);
276
+ }
277
+
278
+ /**
279
+ * The profile names this job waits on, in the operator's writing order. Order matters here even though the
280
+ * conditions are a CONJUNCTION and the tiers are evaluated cheapest-first: within tier 2 the checks run
281
+ * sequentially, and naming the first one that says "not yet" is what makes a held row readable.
282
+ */
283
+ export function waitProfileNames(job) {
284
+ if (!waitArmed(job)) return [];
285
+ const names = [];
286
+ for (const condition of job.waitFor) {
287
+ // Charset-checked HERE, not merely upstream. The loader enforces the same set, but that runs in a
288
+ // different process and this value ends up in a PUBLIC forge comment and a log line: a name carrying
289
+ // a backtick breaks the code span it is rendered in, and one carrying a newline is a log injection.
290
+ // The module that makes the no-attacker-bytes claim is the one that has to enforce it -- the same
291
+ // reasoning as this file's prototype-free profile table.
292
+ if (isDeclarableName(condition?.profile)) names.push(condition.profile);
293
+ }
294
+ return names;
295
+ }
296
+
297
+ /**
298
+ * A short, PII-free label for what this job is waiting on, for the held row and the log line.
299
+ *
300
+ * Every byte of it is checked HERE against `PROFILE_NAME` or the instant grammar, rather than trusted to
301
+ * have been checked by the loader in another process. Nothing attacker-chosen can reach it,
302
+ * which is the property that lets it sit in a panel row and a log line at all (`INT-RUN-HISTORY-FILE-CONTRACT`'s
303
+ * PII-free-by-construction rule, applied to a surface that is not the record).
304
+ */
305
+ export function waitLabel(job) {
306
+ if (!waitArmed(job)) return null;
307
+ const parts = [];
308
+ for (const condition of job.waitFor) {
309
+ // Both arms are re-validated for the reason waitProfileNames states: this string is rendered to
310
+ // operators, and "it was validated upstream" is a claim about another process.
311
+ if (afterInstantMs(condition?.after) !== null) parts.push(`after ${condition.after}`);
312
+ else if (isDeclarableName(condition?.profile)) parts.push(condition.profile);
313
+ }
314
+ return parts.length > 0 ? parts.join(" + ") : null;
315
+ }
@@ -0,0 +1,263 @@
1
+ /**
2
+ * The `wait:` keyspace: what the worker remembers about a job it is holding (issue #230).
3
+ *
4
+ * Four things need remembering, and none of them can be derived from the delayed set alone:
5
+ *
6
+ * 1. WHEN THE HOLD STARTED. `job.timestamp` is the ENQUEUE instant, so it counts pause-window time,
7
+ * scope-mutex time and retry backoff as "waited": a job enqueued into a 22:00-08:00 quiet window would
8
+ * burn ten hours of its wait budget before the first check existed, then terminate having actually
9
+ * waited fourteen. The panel would report the wrong number for the same reason.
10
+ * 2. WHICH JOB HOLDS A TARGET. A second delivery for the same repo/issue/flow while the first is held
11
+ * would otherwise hold too, and both would run when the condition cleared -- two paid runs for one
12
+ * intent, which the acceptance criteria refuse. The obvious fix, widening the queue's dedup window,
13
+ * is worse: that key carries no trigger identity (so it would suppress an unflagged sibling on the
14
+ * same target) and it OUTLIVES completion (so it would go on suppressing after this job finished).
15
+ * 3. HOW MANY TIMES A CHECK HAS FAILED TO ANSWER (the enforcement slice writes it; the field is
16
+ * reserved here so both writers agree on the shape).
17
+ * 4. WHAT TO SHOW AN OPERATOR: an id-only target and an operator-authored condition label, chosen by
18
+ * the worker rather than projected out of a delayed job's `.data`, which holds the issue title and
19
+ * body.
20
+ *
21
+ * WHY THIS IS NOT THE REDIS STATE `OQ-008` AND #242 REFUSED. That refusal is about a claim that would
22
+ * survive a crash WRONGLY -- an in-flight count asserting a container the boot reaper had just killed.
23
+ * This describes DELAYED JOBS, which are themselves Redis-persisted, so it is the same source of truth
24
+ * rather than a second one. Every key carries a TTL sized to the hold it describes; a polled holder also
25
+ * refreshes its own lease on each wake, while an `after` holder has no wakes to refresh on, which is
26
+ * exactly why the supersede path VERIFIES the holder is still in the queue before it refuses anyone. A
27
+ * holder it cannot verify admits rather than refuses. So the worst a leak costs is a duplicate run or a
28
+ * stale panel row, never work that is dropped.
29
+ *
30
+ * The keys, all under one prefix so an operator can see the whole feature with one `KEYS wait:*`:
31
+ * wait:held SET of held job ids -- an index the READER prunes, never a source of truth
32
+ * wait:job:<jobId> HASH { since, faults, throttles, target, label, dedupId }, TTL sized to the hold
33
+ * wait:done:<dedupId> STRING jobId -- this target's wait was satisfied, briefly remembered
34
+ * wait:key:<dedupId> STRING jobId -- the supersede lease, TTL sized to the expected hold
35
+ */
36
+
37
+ /** Slack added to every lease so a hold that wakes exactly on time never races its own key's expiry. */
38
+ const LEASE_SLACK_MS = 60 * 60 * 1000;
39
+
40
+ /** The floor on a lease, so a short hold still leaves a trace long enough for a panel refresh to see it. */
41
+ const LEASE_MIN_MS = 5 * 60 * 1000;
42
+
43
+ // An INDEX SET, with the leak handled by the reader rather than by not having one.
44
+ //
45
+ // The first cut of this file had no index, on the reasoning that a SET cannot expire its members, so every
46
+ // hold ending by any route except the clean one would leave a member nothing removes. That reasoning was
47
+ // right about the leak and wrong about the cost of avoiding it: enumerating `wait:job:*` instead means a
48
+ // SCAN of the whole keyspace, and the panel does it every second -- worst on deployments with NOTHING held,
49
+ // because the exit-early condition never fires. Measured at 200k unrelated keys that is most of a refresh
50
+ // interval, for a feature nobody is using.
51
+ //
52
+ // So the index exists and the READER prunes it: a member whose hash is gone is stale by definition, and the
53
+ // reader removes it as it goes. Staleness is bounded by the next read, the hashes remain the source of
54
+ // truth, and the cost of listing held jobs becomes O(held) instead of O(keyspace).
55
+ export const HELD_SET = "wait:held";
56
+ export const jobKey = (jobId) => `wait:job:${jobId}`;
57
+ export const leaseKey = (dedupId) => `wait:key:${dedupId}`;
58
+ export const satisfiedKey = (dedupId) => `wait:done:${dedupId}`;
59
+
60
+ // How long "this target's wait has been satisfied" is remembered. The backoff ceiling, because that is the
61
+ // longest two jobs that cleared together can wake apart.
62
+ const SATISFIED_MS = 15 * 60 * 1000;
63
+
64
+ /**
65
+ * Build the state accessor. `redis` is the same client the budget uses; nothing here is on the paid path,
66
+ * so every method is written to FAIL OPEN: a redis blip must never refuse a job or wedge a hold, it may
67
+ * only cost the panel a row. The one exception is `claim`, whose failure mode is spelled out on it.
68
+ */
69
+ export function makeWaitState({ redis, now = () => Date.now(), counterTtlMs = 25 * 3600 * 1000 }) {
70
+ const leaseMs = (untilMs) => Math.max(LEASE_MIN_MS, (untilMs ?? 0) - now() + LEASE_SLACK_MS);
71
+
72
+ return {
73
+ /**
74
+ * Record (or refresh) a hold. `since` is written ONCE with HSETNX so a re-pick does not restart the
75
+ * clock -- that is the whole point of not using `job.timestamp`, and a plain HSET here would
76
+ * reintroduce the bug from the other side by resetting the wait on every wake.
77
+ */
78
+ async hold(jobId, { dedupId, target, label, untilMs }) {
79
+ try {
80
+ const key = jobKey(jobId);
81
+ const ttl = leaseMs(untilMs);
82
+ await redis.hsetnx(key, "since", String(now()));
83
+ await redis.hset(key, "target", target ?? "", "label", label ?? "", "dedupId", dedupId ?? "");
84
+ await redis.pexpire(key, ttl);
85
+ await redis.sadd(HELD_SET, jobId);
86
+ // Refresh OUR lease only, and re-read to find out whether it IS ours: `hold` is no longer
87
+ // reached solely after a successful claim (the throttle path holds to stamp a clock without
88
+ // ever winning a lease), so ownership is a question rather than an assumption.
89
+ // Extending a lease this job LOST would hand the winner a longer deafening window than its own
90
+ // hold asked for, which is the failure the three-way claim exists to avoid.
91
+ if (dedupId && (await redis.get(leaseKey(dedupId))) === jobId) await redis.pexpire(leaseKey(dedupId), ttl);
92
+ } catch {
93
+ // Fail open: an unrecorded hold is an invisible row, not a wrong decision.
94
+ }
95
+ },
96
+
97
+ /**
98
+ * Take the supersede lease for a target, or say what to do instead. Returns `{ ok: true }` when this
99
+ * job may hold, `{ heldBy }` when another job verifiably still does, or `{ retry: true }` when the
100
+ * holder could not be checked.
101
+ *
102
+ * `SET NX PX` is the coalescing mechanism: atomic, so two deliveries arriving together cannot both win.
103
+ *
104
+ * THE THREE-WAY ANSWER IS THE POINT, and a two-way one is wrong in both directions. Admitting an
105
+ * unverified holder means two jobs hold the same target and BOTH are paid when it clears -- the exact
106
+ * accumulation `REQ-WAIT-FOR` promises will not happen. Refusing one means a holder that is gone
107
+ * deafens the target for the rest of its lease, and a refused forge delivery is gone for good. So a
108
+ * holder we cannot check produces neither: the caller re-defers and asks again, which costs one wake
109
+ * and decides nothing until the answer is known.
110
+ *
111
+ * `isLive` returns true, false, or null/undefined for "cannot tell" (no probe wired, or it threw).
112
+ */
113
+ async claim(jobId, { dedupId, untilMs, isLive }) {
114
+ if (!dedupId) return { ok: true }; // no semantic identity to coalesce on (a job id is unique already)
115
+ try {
116
+ const key = leaseKey(dedupId);
117
+ const won = await redis.set(key, jobId, "PX", leaseMs(untilMs), "NX");
118
+ if (won) return { ok: true };
119
+ const holder = await redis.get(key);
120
+ if (!holder || holder === jobId) return { ok: true }; // expired between the two calls, or it is us
121
+
122
+ let live = null;
123
+ if (typeof isLive === "function") {
124
+ try {
125
+ live = await isLive(holder);
126
+ } catch {
127
+ live = null; // a probe that threw has not told us the holder is alive
128
+ }
129
+ }
130
+ if (live === false) {
131
+ // The holder can no longer wake. Take the lease over rather than leaving a key that
132
+ // refuses every later delivery for this target until it expires.
133
+ await redis.set(key, jobId, "PX", leaseMs(untilMs));
134
+ return { ok: true, tookOverFrom: holder };
135
+ }
136
+ if (live !== true) return { retry: true, holder };
137
+ return { heldBy: holder };
138
+ } catch {
139
+ // A redis fault reached us before any decision was made. Ask again rather than guess: both
140
+ // guesses are wrong in a way an operator cannot see.
141
+ return { retry: true, holder: null };
142
+ }
143
+ },
144
+
145
+ /**
146
+ * Mark a target as SATISFIED by this job, and read who satisfied it.
147
+ *
148
+ * The lease alone cannot close one window. Two deliveries can both be holding when a worker outage
149
+ * outlives the lease TTL -- the delayed jobs survive, being Redis-persisted, while the lease does not
150
+ * -- and if the condition cleared during that outage each one wakes, finds no holder, claims cleanly,
151
+ * and runs. Two paid runs for one intent, which is the accumulation `REQ-WAIT-FOR` promises against.
152
+ *
153
+ * So the job that clears a target says so, and a sibling waking after it refuses. The marker lives for
154
+ * `SATISFIED_MS` because that is the longest two jobs which cleared TOGETHER can wake apart: the
155
+ * backoff ceiling. It is deliberately not longer -- a delivery arriving well after a wait completed is
156
+ * a genuinely new intent, and coalescing it would be the dedup-window mistake this design already
157
+ * refused once.
158
+ */
159
+ async markSatisfied(jobId, { dedupId }) {
160
+ if (!dedupId) return;
161
+ try {
162
+ await redis.set(satisfiedKey(dedupId), jobId, "PX", SATISFIED_MS);
163
+ } catch {
164
+ // Fail open: the cost is the duplicate this marker exists to prevent, not a wrong refusal.
165
+ }
166
+ },
167
+
168
+ async satisfiedBy(dedupId) {
169
+ if (!dedupId) return null;
170
+ try {
171
+ return await redis.get(satisfiedKey(dedupId));
172
+ } catch {
173
+ return null;
174
+ }
175
+ },
176
+
177
+ /** Drop a hold: the job is running, refusing, or expiring. Only clears a lease this job actually owns. */
178
+ async release(jobId, { dedupId } = {}) {
179
+ try {
180
+ await redis.srem(HELD_SET, jobId);
181
+ await redis.del(jobKey(jobId));
182
+ if (dedupId) {
183
+ const holder = await redis.get(leaseKey(dedupId));
184
+ if (holder === jobId) await redis.del(leaseKey(dedupId));
185
+ }
186
+ } catch {
187
+ // Fail open: the TTLs are the backstop, and every field here is advisory.
188
+ }
189
+ },
190
+
191
+ /**
192
+ * The two per-job counters the polled tier keeps: how many checks have run, and how many of them in a
193
+ * ROW could not answer. Both live beside `since` so one hash carries the whole hold, and both come
194
+ * back 0 when nothing was written -- an unrecorded hold must read as a fresh one rather than as a
195
+ * job that has already exhausted its budget.
196
+ */
197
+ async counters(jobId) {
198
+ try {
199
+ const h = await redis.hget(jobKey(jobId), "checks");
200
+ const f = await redis.hget(jobKey(jobId), "faults");
201
+ const t = await redis.hget(jobKey(jobId), "throttles");
202
+ return { checks: Number(h) || 0, faults: Number(f) || 0, throttles: Number(t) || 0 };
203
+ } catch {
204
+ return { checks: 0, faults: 0, throttles: 0 };
205
+ }
206
+ },
207
+
208
+ /**
209
+ * Count a wake that was denied the check lease, or clear the count when one is granted.
210
+ *
211
+ * This is the only observable the capacity bound has. The lease caps how much wall-clock a worker
212
+ * spends checking, which is what it is for -- but a cap that is being hit constantly means demand
213
+ * exceeds it, and the plan's own economics say that arrives quietly: paid jobs starve behind checks
214
+ * that spend nothing, and nothing in the product explains why. A consecutive-denial count is the
215
+ * symptom an operator can actually see, so it is kept and its overflow is logged rather than absorbed.
216
+ */
217
+ async noteThrottle(jobId, { denied }) {
218
+ try {
219
+ if (!denied) return void (await redis.hset(jobKey(jobId), "throttles", "0"));
220
+ const n = Number(await redis.hincrby(jobKey(jobId), "throttles", 1)) || 0;
221
+ await redis.pexpire(jobKey(jobId), counterTtlMs); // see noteCheck: a counter may CREATE this hash
222
+ return n;
223
+ } catch {
224
+ return 0;
225
+ }
226
+ },
227
+
228
+ /**
229
+ * Record what a check answered. `checks` only ever grows; `faults` is a CONSECUTIVE count, so a check
230
+ * that answered resets it -- a script that works after an outage has not accumulated a debt.
231
+ */
232
+ async noteCheck(jobId, { fault }) {
233
+ try {
234
+ await redis.hincrby(jobKey(jobId), "checks", 1);
235
+ if (fault) await redis.hincrby(jobKey(jobId), "faults", 1);
236
+ else await redis.hset(jobKey(jobId), "faults", "0");
237
+ // A counter can CREATE this hash: on a job's first polled wake nothing has held yet, and a path
238
+ // that exits before `hold` (a lease denial, an unverified supersede) would otherwise leave a
239
+ // key with NO expiry at all. That breaks the invariant this whole keyspace rests on -- "every
240
+ // hash carries a TTL", which is why there is no index set to leak instead -- and the panel
241
+ // would render a held row nothing could ever remove.
242
+ await redis.pexpire(jobKey(jobId), counterTtlMs);
243
+ } catch {
244
+ // Fail open: a lost counter costs a longer wait, never a wrong verdict.
245
+ }
246
+ },
247
+
248
+ /** How long this job has been held, in ms, or null when nothing was recorded. */
249
+ async heldForMs(jobId) {
250
+ try {
251
+ const since = await redis.hget(jobKey(jobId), "since");
252
+ // ABSENT IS NOT ZERO, and conflating them is not a rounding error: `Number(null)` is 0, so a
253
+ // job with no recorded hold would read as held since the epoch and every bound measured from
254
+ // here -- the maximum hold above all -- would fire on its first check.
255
+ if (typeof since !== "string" || since.trim() === "") return null;
256
+ const ms = Number(since);
257
+ return Number.isFinite(ms) && ms > 0 ? Math.max(0, now() - ms) : null;
258
+ } catch {
259
+ return null;
260
+ }
261
+ },
262
+ };
263
+ }