@edgehero/pi-dispatch 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.mjs CHANGED
@@ -2,7 +2,10 @@ import { execFile } from "node:child_process";
2
2
  import { promisify } from "node:util";
3
3
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
4
4
  import { InfraRetry, runJob } from "./processor.mjs";
5
+ import { targetFor } from "./run-history.mjs";
5
6
  import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
7
+ import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
8
+ import { makeWaitState } from "./wait-state.mjs";
6
9
 
7
10
  const exec = promisify(execFile);
8
11
 
@@ -16,6 +19,28 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
16
19
  // docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
17
20
  export const SCOPE_BUSY_RECHECK_MS = 5_000;
18
21
 
22
+ // How long a job waits before re-asking whether a target's holder is still alive (issue #230). Reached only
23
+ // when the liveness probe could not answer, which is a redis or queue fault rather than a normal state, so
24
+ // this is a short retry rather than a cadence: the job is deciding nothing and holding nothing while it
25
+ // waits, and the fault it is waiting out is usually seconds long.
26
+ export const SUPERSEDE_RECHECK_MS = 15_000;
27
+
28
+ // The one key the check lease counts under. A single global counter rather than one per profile: what it
29
+ // bounds is this worker's wall-clock spent answering questions, and that is shared whatever is being asked.
30
+ const WAIT_CHECK_KEY = "wait-check";
31
+
32
+ // How many consecutive lease denials one job absorbs before the deployment is told its checking capacity is
33
+ // short. Logged ONCE per run of denials rather than per wake: an alarm that repeats every re-check is the
34
+ // always-on amber this project rejects elsewhere, and the operator only needs telling once per episode.
35
+ const THROTTLE_ALARM = 5;
36
+
37
+ // The floor under a throttled or aborted re-ask. Its own constant rather than a borrow of
38
+ // SUPERSEDE_RECHECK_MS, which documents an unrelated concern. Deliberately NOT 5s: that is
39
+ // SCOPE_BUSY_RECHECK_MS, and INT-WAIT-PROFILES-CONTRACT rests on wait deferrals being distinguishable from
40
+ // scope deferrals by wake instant -- nothing records WHY a job sits in the delayed set, so the instants are
41
+ // the only evidence there is. A test pins the two apart.
42
+ const THROTTLE_FLOOR_MS = 11_000;
43
+
19
44
  /**
20
45
  * Build the BullMQ processor.
21
46
  *
@@ -35,7 +60,7 @@ export const SCOPE_BUSY_RECHECK_MS = 5_000;
35
60
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
36
61
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
37
62
  */
38
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
63
+ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
39
64
  return async function processor(job, token, signal) {
40
65
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
41
66
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -53,9 +78,286 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
53
78
  throw new DelayedError();
54
79
  }
55
80
 
81
+ // The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
82
+ // not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
83
+ // because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
84
+ // does. Strictly above the `try` for the two reasons the gates below it document.
85
+ //
86
+ // The order WITHIN the gate is determinate-refusals-then-holds, which is CONST-BUDGET-BEFORE-TOKENS'
87
+ // shape applied to time rather than to money: a condition this deployment can never answer must be
88
+ // refused now, not after a day of waiting.
89
+ //
90
+ // On throwing above the `try`: an exception here escapes into BullMQ's normal failed-attempt handling,
91
+ // which is WANTED for `moveToDelayed` (the scope gate below gives the argument: a transient rejection
92
+ // must stay a transient failure rather than becoming a permanent one) and unwanted everywhere else. So
93
+ // the state and comment seams fail open by construction, and `recordRun` is relied on not to throw --
94
+ // its writer swallows fs errors by contract, which is the same reliance the settings-overlay refusal
95
+ // below already makes.
96
+ if (waitArmed(job.data)) {
97
+ // The supersede identity: the queue's semantic key PLUS the trigger that produced this job.
98
+ // The semantic key alone is `repo<sep>number:flow`, which two DIFFERENT triggers on one target and
99
+ // flow legitimately share -- a label rule that waits a day and a comment rule that waits a minute
100
+ // would coalesce, and the second would be refused with a message claiming they wait on "the same
101
+ // conditions" when they do not. Adding the raw trigger index makes the key mean one intent.
102
+ const matchedIndex = job.data?.trigger?.matched?.index;
103
+ const dedupId = job.deduplicationId ? `${job.deduplicationId}#${Number.isInteger(matchedIndex) ? matchedIndex : "?"}` : null;
104
+ const refuseWait = async (reason, logEvent, fields, sentence) => {
105
+ // The INJECTED clock, like both gates above: a record whose timestamps ignore the test clock
106
+ // is a record no test of this gate can assert about.
107
+ const at = new Date(now()).toISOString();
108
+ await waitState.release(job.id, { dedupId });
109
+ deps?.log?.(logEvent, { jobId: job.id, ...fields });
110
+ // The comment names the FIELD and the operator's own words for the condition, never a
111
+ // resolver path or a vault topology -- `secret-profile-unknown` sets that rule.
112
+ if (sentence && deps?.comment) await Promise.resolve(deps.comment(job.data, sentence)).catch(() => {});
113
+ const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false };
114
+ recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString() });
115
+ return result;
116
+ };
117
+
118
+ // EVERY condition must be one this worker understands, checked before anything else. The loader
119
+ // refuses an unknown condition, but the loader is a DIFFERENT PROCESS: `job.data.waitFor` arrives
120
+ // over Redis from the receiver, and this whole feature exists because receiver-worker version
121
+ // skew is real. `makeCheckWaitSkew` closes the backward direction (the file has conditions the
122
+ // job arrived without); this closes the forward one (a newer receiver enqueues a condition shape
123
+ // this worker cannot read). Without it the gate would fall through, log `wait_cleared`, and run
124
+ // the job -- asserting in the log that conditions cleared which it never evaluated, which is the
125
+ // same undetectable paid run the backward check exists to stop.
126
+ // A sibling that was held on this same target may already have cleared it. Checked FIRST, because
127
+ // it is free and determinate, and because the window it closes is one no lease can: two jobs
128
+ // holding through an outage that outlives their leases would each wake, find no holder, and run.
129
+ if (dedupId) {
130
+ const satisfiedBy = await waitState.satisfiedBy(dedupId);
131
+ if (satisfiedBy && satisfiedBy !== job.id) {
132
+ return await refuseWait("wait-superseded", "wait_superseded", { satisfiedBy }, "Another delivery for this target already finished waiting on the same conditions. Not run.");
133
+ }
134
+ }
135
+
136
+ const unreadable = unreadableConditions(job.data);
137
+ if (unreadable.length > 0) {
138
+ // Its OWN token, not `wait-skew`. Both are version skew, and the REMEDIES are opposites --
139
+ // upgrade the receiver there, upgrade the worker here -- so one token in a durable record
140
+ // would tell an operator that something is out of step and not which way to move.
141
+ return await refuseWait("wait-unreadable", "refused_wait_unreadable", { conditions: unreadable.length }, `Refused: this job carries ${unreadable.length} wait condition${unreadable.length === 1 ? "" : "s"} this worker cannot read, so it cannot honour them. The worker is older than the service that enqueued this job. Not run.`);
142
+ }
143
+
144
+ // A `profile` condition needs a checker, and with none wired NOTHING can answer it. Refused rather
145
+ // than ignored: a wait the deployment cannot perform must not read as a wait that passed.
146
+ const profiles = waitProfileNames(job.data);
147
+ // Declared-ness is a table lookup, so it belongs with the other free refusals rather than inside
148
+ // the check. Without it here, `[{after: "<tomorrow>"}, {profile: "typo"}]` holds for a day and
149
+ // THEN refuses -- which is the exact sentence the ordering rule above promises will not happen.
150
+ const undeclared = deps?.waitProfileDeclared ? profiles.find((name) => !deps.waitProfileDeclared(name)) : undefined;
151
+ if (undeclared !== undefined) {
152
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: undeclared }, `Waiting on \`${undeclared}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
153
+ }
154
+ if (profiles.length > 0 && !deps?.checkWait) {
155
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: profiles[0] }, `Waiting on \`${profiles[0]}\` is not something this deployment can answer. Not run.`);
156
+ }
157
+
158
+ const holdUntil = afterMs(job.data); // named apart from the pause gate's `until` above, which it would otherwise shadow
159
+ // An instant further out than the ceiling is refused at FIRST pickup rather than held toward:
160
+ // holding for a month to then refuse tells the operator nothing they could not have been told now.
161
+ if (holdUntil !== null && holdUntil - nowMs > afterMaxMs()) {
162
+ return await refuseWait("wait-after-beyond-max", "wait_after_beyond_max", { delayMs: holdUntil - nowMs }, `The \`after\` instant is further out than this deployment allows a job to wait. Not run.`);
163
+ }
164
+
165
+ // The pause gate's boundary guard, for its reason: a tick landing on the instant must run rather
166
+ // than busy-defer to a moment already past.
167
+ if (holdUntil !== null && holdUntil > nowMs + 1000) {
168
+ // `isJobLive` is what stops a vanished holder's lease becoming a tombstone that refuses this
169
+ // target for the rest of the hold. Optional: an unwired probe means the holder cannot be
170
+ // checked, which ADMITS and says so -- one duplicate run beats one dropped delivery, which is
171
+ // `OQ-027`'s call ("one wasted vault read beats one dropped job") on this feature's terms.
172
+ const claim = await waitState.claim(job.id, { dedupId, untilMs: holdUntil, isLive: deps?.isJobLive });
173
+ if (claim.heldBy) {
174
+ // Another delivery for this same target and flow is already holding. Both would clear
175
+ // together and both would be paid, which is the accumulation the acceptance forbids.
176
+ return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
177
+ }
178
+ if (claim.retry) {
179
+ // The holder could not be checked. Holding anyway would put two jobs on one target and pay
180
+ // for both; refusing would drop a delivery over a holder that may be gone. So decide
181
+ // nothing: re-defer briefly and ask again once the probe can answer.
182
+ deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
183
+ await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
184
+ throw new DelayedError();
185
+ }
186
+ if (claim.tookOverFrom) deps?.log?.("wait_lease_taken_over", { jobId: job.id, from: claim.tookOverFrom });
187
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: holdUntil });
188
+ deps?.log?.("wait_deferred", { jobId: job.id, until: new Date(holdUntil).toISOString(), label: waitLabel(job.data) });
189
+ await job.moveToDelayed(holdUntil, token);
190
+ throw new DelayedError();
191
+ }
192
+
193
+ // TIER 2: the polled conditions. Last, because it is the only part of this gate that spawns a
194
+ // process -- the free refusals above it are free, and the free hold above it is free.
195
+ if (profiles.length > 0) {
196
+ const held = (await waitState.heldForMs(job.id)) ?? 0;
197
+ const counted = await waitState.counters(job.id);
198
+
199
+ // One check at a time, process-wide, and never the worker's last free slot. This is the bound
200
+ // that keeps a wait from starving the paid work it is waiting for: slots x timeout is the most
201
+ // wall-clock a worker can spend answering questions instead of running jobs. Computed against
202
+ // the LIVE concurrency rather than the boot value, because the overlay can lower it.
203
+ const slots = Math.min(checkSlotCount(), Math.max(1, concurrencyNow() - 1));
204
+ if (!checkSlots.tryAcquire(WAIT_CHECK_KEY, slots)) {
205
+ // Denials are counted, and a run of them is the ONE symptom the capacity bound has. The
206
+ // lease deliberately caps how much wall-clock this worker spends checking; being at that
207
+ // cap constantly means demand exceeds it, which the issue's own economics say arrives
208
+ // silently -- paid jobs starve behind checks that spend nothing and nothing says why.
209
+ const denials = await waitState.noteThrottle(job.id, { denied: true });
210
+ if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots, hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
211
+ // A starved job still needs a CLOCK and a CEILING, or the lease turns into the very
212
+ // starvation it exists to bound: without this the hold is stamped only on a wake that won
213
+ // the lease, so a job that never wins one has no `since`, never reaches the maximum, and
214
+ // re-wakes forever with no record and no bound. There is no deciding check to run first
215
+ // here -- that is the whole condition -- so the bound applies directly.
216
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
217
+ if (held >= maxWaitMs()) {
218
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait-unchecked", denials, heldForMs: held }, `Gave up waiting: this deployment could not run the check often enough to answer within the maximum wait. Not run.`);
219
+ }
220
+ // Denied. Re-ask at a fraction of the cadence rather than the full backoff (which would
221
+ // turn one lost coin-flip into a fifteen-minute penalty) or a flat few seconds (which at
222
+ // scale is a herd). Jittered, so a fleet of denied jobs does not return together.
223
+ const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
224
+ const delay = wait + Math.floor(wait * 0.1 * random());
225
+ deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots });
226
+ await job.moveToDelayed(nowMs + delay, token);
227
+ throw new DelayedError();
228
+ }
229
+
230
+ // Claimed BEFORE the check, not after: a second delivery for an already-held target is a free
231
+ // determinate refusal, and paying for a subprocess first inverts the free-before-costly rule
232
+ // this gate's own header invokes. Tier 1 already claims in this order.
233
+ const claim = await waitState.claim(job.id, { dedupId, untilMs: nowMs + waitBackoffMs(intervalMs(), held), isLive: deps?.isJobLive });
234
+ if (claim.heldBy) {
235
+ return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
236
+ }
237
+ if (claim.retry) {
238
+ deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
239
+ await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
240
+ throw new DelayedError();
241
+ }
242
+
243
+ await waitState.noteThrottle(job.id, { denied: false }); // granted: the run of denials ends here
244
+ let verdict = null;
245
+ let checked = null;
246
+ try {
247
+ // Sequential, in the operator's writing order: the resolver's reason applies unchanged --
248
+ // naming the first condition that did not clear is what makes a held row readable, and a
249
+ // parallel fan-out would blame whichever lost the race on any given wake.
250
+ for (const profile of profiles) {
251
+ checked = profile;
252
+ verdict = await deps.checkWait(profile, targetFor(job.data?.kind, job.data), { signal });
253
+ if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
254
+ }
255
+ } finally {
256
+ checkSlots.release(WAIT_CHECK_KEY);
257
+ }
258
+
259
+ if (verdict?.unusableTarget) {
260
+ // Determinate and unfixable by waiting: the job's own target is a shape no check can be
261
+ // handed. It belongs with the refusals, not the holds -- holding would spend the fault
262
+ // budget and then blame the operator's script for a value it was never given.
263
+ return await refuseWait("wait-unreadable", "refused_wait_unreadable", { profile: checked }, `Refused: this job's target cannot be handed to a wait check, so \`${checked}\` can never be asked. Not run.`);
264
+ }
265
+ if (verdict?.profileUnknown) {
266
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: verdict.profileUnknown }, `Waiting on \`${verdict.profileUnknown}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
267
+ }
268
+ if (verdict?.verdict === "refuse") {
269
+ // Exit 2: the check says this will NEVER clear. Terminal by the protocol's own words, and
270
+ // distinct from every "not yet" above it.
271
+ return await refuseWait("wait-refused", "wait_refused", { profile: checked, heldForMs: held }, `The check \`${checked}\` reports this will never clear. Not run.`);
272
+ }
273
+
274
+ if (verdict?.aborted) {
275
+ // The worker is stopping or this job was cancelled. Nothing was learned and nothing is
276
+ // owed: re-defer at once rather than at the full backoff, and count neither a check nor a
277
+ // fault, or a rolling deploy would spend a job's whole budget on its own restarts and then
278
+ // blame the operator's script for it.
279
+ deps?.log?.("wait_check_aborted", { jobId: job.id, profile: checked });
280
+ await job.moveToDelayed(nowMs + THROTTLE_FLOOR_MS, token);
281
+ throw new DelayedError();
282
+ }
283
+
284
+ if (verdict?.verdict === "hold") {
285
+ const fault = verdict.fault === true;
286
+ await waitState.noteCheck(job.id, { fault });
287
+ const faults = fault ? counted.faults + 1 : 0;
288
+
289
+ // A check that never answers is a broken script, not a slow condition, and OQ-030 is why
290
+ // this bound exists: most CLIs exit 1 for everything, so without it a typo would hold for
291
+ // the whole maximum wait and then blame the CONDITION rather than the check.
292
+ if (faults >= maxFaults()) {
293
+ return await refuseWait("wait-unanswerable", "wait_unanswerable", { profile: checked, faults }, `The check \`${checked}\` could not answer ${faults} times in a row. Not run.`);
294
+ }
295
+
296
+ // BOTH terminal bounds are tested AFTER the check and never before it, so a condition that
297
+ // cleared on the deciding wake runs instead of being recorded as never having cleared.
298
+ // Without that ordering the backoff's own quantisation makes "cleared at t+1s, declared
299
+ // never-cleared at t+900s" a structural lie in the durable record and in a public comment.
300
+ //
301
+ // The count bound reads `checks + 1` because this wake's check has just run: the job gets
302
+ // exactly `maxChecks` checks, the last of which is the deciding one. Putting it before the
303
+ // check instead -- so the act of testing the bound could not exceed it -- was the obvious
304
+ // spelling, and it silently made this whole guarantee untrue at every shipped default,
305
+ // because the count bound is the one that fires first there.
306
+ if (counted.checks + 1 >= maxChecks()) {
307
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-checks", checks: counted.checks + 1, profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\` after ${counted.checks + 1} checks. Not run.`);
308
+ }
309
+ if (held >= maxWaitMs()) {
310
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait", profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\`. Not run.`);
311
+ }
312
+
313
+ // Clamped to what is LEFT of the budget, never just the cadence. Without this an hourly
314
+ // interval under a fifteen-minute maximum holds for the full hour -- 400% of the bound the
315
+ // operator configured -- because the ceiling is only tested when a wake arrives, and the
316
+ // cadence decides when that is. The two knobs are independent `positiveInt`s and nothing
317
+ // cross-validates them, so the clamp is what makes the smaller one actually bind.
318
+ const base = waitBackoffMs(intervalMs(), held);
319
+ const jittered = base + Math.floor(base * 0.1 * random());
320
+ const remaining = Math.max(0, maxWaitMs() - held);
321
+ const delay = Math.max(1000, Math.min(jittered, remaining));
322
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + delay });
323
+ deps?.log?.("wait_deferred", { jobId: job.id, profile: checked, fault, heldForMs: held, delayMs: delay });
324
+ await job.moveToDelayed(nowMs + delay, token);
325
+ throw new DelayedError();
326
+ }
327
+
328
+ // FAIL CLOSED on anything that is not literally go. Everything above tests for a specific
329
+ // shape and falls through otherwise, and "otherwise" at this gate means STARTING A PAID
330
+ // CONTAINER -- so an `undefined`, a `null`, a `{}`, a mis-cased "GO" or a bare string from a
331
+ // checker would run the job silently, with no record field and no log line to distinguish it
332
+ // from a job whose check said yes. The shipped checker is total, and that is exactly the
333
+ // reasoning `unreadableConditions` above rejects: this is a dependency-injection seam, and a
334
+ // seam's guarantees are the caller's to enforce.
335
+ if (verdict?.verdict !== "go") {
336
+ deps?.log?.("wait_check_unintelligible", { jobId: job.id, profile: checked });
337
+ await waitState.noteCheck(job.id, { fault: true });
338
+ const base = waitBackoffMs(intervalMs(), held);
339
+ await job.moveToDelayed(nowMs + base + Math.floor(base * 0.1 * random()), token);
340
+ throw new DelayedError();
341
+ }
342
+
343
+ // Every profile answered go. Record the last check so the count bound sees it.
344
+ await waitState.noteCheck(job.id, { fault: false });
345
+ }
346
+
347
+ // Only a job that actually HELD has cleared. Without the check this line fires on the first
348
+ // pickup of a job whose instant had already passed, and again on every scope-busy re-check
349
+ // afterwards -- asserting a wait ended that never began.
350
+ const heldForMs = await waitState.heldForMs(job.id);
351
+ // Say so before releasing: a sibling held on this target must find the answer, not an empty lease.
352
+ if (heldForMs !== null) await waitState.markSatisfied(job.id, { dedupId });
353
+ await waitState.release(job.id, { dedupId });
354
+ if (heldForMs !== null) deps?.log?.("wait_cleared", { jobId: job.id, label: waitLabel(job.data), heldForMs });
355
+ }
356
+
56
357
  // Per-scope concurrency and the one-job-per-folder mutex (issue #242,
57
- // INT-SCOPED-LIMITS-FILE-CONTRACT). SECOND, after the pause gate (a paused job must not burn
58
- // re-check wakes) and STRICTLY above the `try` below, like the pause gate and for the same two
358
+ // INT-SCOPED-LIMITS-FILE-CONTRACT). LAST of the three gates, after the pause gate (a paused job must
359
+ // not burn re-check wakes) and after the wait gate (a job holding until tomorrow must not sit on a
360
+ // folder while it does), and STRICTLY above the `try` below, like the pause gate and for the same two
59
361
  // reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
60
362
  // catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
61
363
  // handling exactly as the pause gate's does (inside the try it would become a permanent failure
@@ -123,7 +425,9 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
123
425
  // job completed and does not retry a file that can never parse (CONST-RETRY-INFRA-ONLY). Resolved
124
426
  // before runJob, so no budget slot is reserved and no container starts (CONST-BUDGET-BEFORE-TOKENS).
125
427
  // recordRun leaves the durable settings-overlay-invalid trace for the admin extension.
126
- // No provider/model here, alone among the terminal results: the overlay is the thing that failed to parse, so no honest effective value exists -- buildRecord defaults both null.
428
+ // No provider/model here, as in every result this function returns from ABOVE the try (the wait gate's
429
+ // refusals are the others): each is decided before or during the settings read, so no honest effective
430
+ // value exists yet -- buildRecord defaults both null.
127
431
  const result = { outcome: "policy", reason: "settings-overlay-invalid", exitCode: null, turns: null, tokens: null, budgetReserved: false };
128
432
  recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
129
433
  return result;
@@ -210,7 +514,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
210
514
  };
211
515
  }
212
516
 
213
- export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
517
+ export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
214
518
  let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
215
519
  const processor = makeProcessor({
216
520
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
@@ -228,6 +532,21 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
228
532
  // shape means one per daemon).
229
533
  scopedLimits,
230
534
  inFlight,
535
+ // Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
536
+ // redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
537
+ waitState,
538
+ afterMaxMs,
539
+ // Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
540
+ // because the overlay can lower it through `dispatch_set` and a check must never take the last free
541
+ // slot from a paid job. Late-bound over `worker` exactly as `applyConcurrency` is, and for the same
542
+ // reason: the value it needs does not exist until the Worker is constructed.
543
+ checkSlots,
544
+ checkSlotCount,
545
+ concurrencyNow: concurrencyNow ?? (() => worker?.concurrency ?? concurrency),
546
+ intervalMs,
547
+ maxWaitMs,
548
+ maxChecks,
549
+ maxFaults,
231
550
  deps,
232
551
  recordRun,
233
552
  });
package/src/processor.mjs CHANGED
@@ -54,6 +54,9 @@ export async function runJob(job, deps) {
54
54
  // exactly as before, and the gate below only calls it for a job whose matched rule was a
55
55
  // one-shot, so the default is never a probe running on every delivery.
56
56
  checkOnceSpent = async () => ({ ok: true }),
57
+ // Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
58
+ // unwired seam must not refuse, and the wiring is what turns the check on.
59
+ checkWaitSkew = async () => ({ ok: true }),
57
60
  // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
58
61
  // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
59
62
  egressPreflight = async () => ({ ok: true }),
@@ -155,6 +158,28 @@ export async function runJob(job, deps) {
155
158
  }
156
159
  }
157
160
 
161
+ // The wait-skew check (issue #230), second on the ladder and for the first one's reasons: the same
162
+ // file read, free, determinate, credential-less, and pre-spend. It answers a question no other layer
163
+ // can: does the AUTHORED trigger carry wait conditions this job arrived without? That happens when a
164
+ // service below the version floor dropped the field as an unknown key, and the resulting run is
165
+ // byte-identical to a correct one everywhere it is recorded -- so this refusal is the only thing
166
+ // standing between a stale receiver and a paid job that ran when the operator wrote "wait".
167
+ {
168
+ const skew = await checkWaitSkew(job);
169
+ if (skew.skewed) {
170
+ // Named for the operator, not the payload: how many conditions were authored, never what
171
+ // they say. The fix is a version, so the comment says which one.
172
+ // The message names BOTH causes, because the more likely one is not a version at all. In the
173
+ // compose topology the receiver's single-file `:ro` mount pins a dead inode, so an operator
174
+ // who ADDS `waitFor` to an existing rule gets this refusal on every delivery from a service
175
+ // that is perfectly up to date and merely holding an older copy of the file. Naming only the
176
+ // version would send them looking for an upgrade they do not need.
177
+ await comment(job, `Refused: this trigger declares ${skew.conditions} wait condition${skew.conditions === 1 ? "" : "s"}, but the job reached the worker without them, which means it would have run immediately. Either a service in this deployment is below the version that carries the field, or one is still running against an older copy of the triggers file and needs restarting. Not run.`);
178
+ log("refused_wait_skew", { triggerIndex: job.trigger?.matched?.index ?? null, conditions: skew.conditions });
179
+ return { outcome: "policy", reason: "wait-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
180
+ }
181
+ }
182
+
158
183
  // The job image must exist on THIS host before anything else happens. Free, determinate and
159
184
  // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
160
185
  // image refuses without minting a credential it will not use, cloning a repo it will not read, or
package/src/queue.mjs CHANGED
@@ -134,7 +134,7 @@ export async function enqueueGitLabJob(queue, fields) {
134
134
  * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
135
135
  * has always been.
136
136
  */
137
- export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, replica, replicas }) {
137
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
138
138
  const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
139
139
  // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
140
140
  // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
@@ -179,6 +179,16 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
179
179
  // is copied verbatim into /job/event.json, which an agent reads.
180
180
  ...(secrets !== undefined && { secrets }),
181
181
  ...(secretsProfile !== undefined && { secretsProfile }),
182
+ // Issue #230. The conditions the worker holds this job on, carried so the PICKUP gate can read them:
183
+ // that gate runs above the per-job settings read and never re-parses the triggers file for its terms.
184
+ // At JOB level, and here that placement is a correctness requirement rather than a convention --
185
+ // `trigger` is copied VERBATIM into /job/event.json (prepare-local.mjs), so a `trigger.waitFor` would
186
+ // hand the agent the operator's own gate. Conditional like every field above, so an unflagged job's
187
+ // data keeps exactly the keys it has today. The dedup options below are deliberately NOT widened for
188
+ // a waiting job: that key carries no trigger identity and outlives the job it was set for, so a
189
+ // longer window would suppress an unflagged sibling's deliveries and go on suppressing them after
190
+ // this job finished. Coalescing a held target is the worker's `wait:` keyspace's job instead.
191
+ ...(waitFor !== undefined && { waitFor }),
182
192
  // Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
183
193
  // exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
184
194
  // both are integers, so the run record they land in stays PII-free by construction.
@@ -387,7 +387,12 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
387
387
  * separator from the table, so the notation a forge uses is the notation its records carry -- and a forge
388
388
  * added later inherits a label rather than a null.
389
389
  */
390
- function targetFor(kind, data) {
390
+ /*
391
+ * Exported since issue #230: a held job's panel row needs the same id-only label a run record carries, and
392
+ * the wait gate would otherwise re-derive it. Two spellings of "which issue is this" is how one of them
393
+ * starts carrying a title.
394
+ */
395
+ export function targetFor(kind, data) {
391
396
  if (kind === "local") return `local:${basename(data.folder ?? "")}`;
392
397
  if (isForgeKind(kind)) return `${data.repo}${targetSeparator(kind, data.target?.type)}${data.target?.number}`;
393
398
  return null;
package/src/service.mjs CHANGED
@@ -972,6 +972,15 @@ async function doRestart(ctx, values) {
972
972
  await ctx.sleep(2000);
973
973
  ({ active = 0 } = await queue.getJobCounts("active"));
974
974
  }
975
+ // A HELD job is neither active nor waiting, so the loop above has just reported a drained queue with
976
+ // however many jobs still parked on `run.waitFor` (issue #230). They are safe -- a hold spends
977
+ // nothing, survives a restart and reserves no slot -- but the operator is upgrading, and those jobs
978
+ // will wake against the new version. Said plainly rather than left to be discovered, which is what
979
+ // this command would otherwise be doing: reporting a drained queue it cannot see all of.
980
+ const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
981
+ if (delayed > 0) {
982
+ ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
983
+ }
975
984
  const stopped = await doStop(ctx);
976
985
  if (stopped !== 0) {
977
986
  ctx.out("restart did not happen — the queue STAYS PAUSED; fix the service, then `pi-dispatch resume`.\n");
package/src/start.mjs CHANGED
@@ -22,9 +22,11 @@ import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare
22
22
  import { listRunningSandboxes } from "./sandbox.mjs";
23
23
  import { makeSandboxReaper } from "./sandbox-store.mjs";
24
24
  import { makeSessionStore } from "./session-store.mjs";
25
- import { makeCheckOnceSpent, makeDisarmOnce } from "./triggers-file.mjs";
25
+ import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
26
26
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
27
27
  import { loadScopedLimits } from "./scoped-limits.mjs";
28
+ import { makeWaitChecker } from "./wait-check.mjs";
29
+ import { makeWaitState } from "./wait-state.mjs";
28
30
  import { makeQueue } from "./queue.mjs";
29
31
  import { makeRunContainer } from "./run-container.mjs";
30
32
  import { makeSecretsResolver } from "./secrets.mjs";
@@ -466,6 +468,18 @@ export async function startWorker(
466
468
  // Issue #242: the scoped-limits snapshot the pickup gate and the scoped budget read, once per
467
469
  // pickup, from the live-reloaded ref -- same next-job grain as pauseUntil above.
468
470
  scopedLimits: () => scopedLimits.current,
471
+ // Issue #230. The `after` ceiling is read per pickup from config rather than frozen into the
472
+ // processor, so it is one value with one home; the wait state shares the budget's redis client
473
+ // because it describes the same delayed jobs that client already reasons about.
474
+ afterMaxMs: () => config.waitAfterMaxMs,
475
+ waitState: makeWaitState({ redis }),
476
+ // The polled tier's bounds, read per pickup from config so they are one value with one home. The
477
+ // slot count is a CEILING the gate clamps against the live concurrency, never the final number.
478
+ checkSlotCount: () => config.waitCheckSlots,
479
+ intervalMs: () => config.waitIntervalMs,
480
+ maxWaitMs: () => config.waitMaxMs,
481
+ maxChecks: () => config.waitMaxChecks,
482
+ maxFaults: () => config.waitMaxFaults,
469
483
  deps: {
470
484
  collectChain,
471
485
  // The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
@@ -473,6 +487,35 @@ export async function startWorker(
473
487
  // spending delivery is excused). In the compose topology this check is the once-enforcement
474
488
  // layer, because the receiver's single-file :ro mount pins a dead inode until restart.
475
489
  checkOnceSpent: makeCheckOnceSpent({ triggersPath: onceTriggersFile }),
490
+ // Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
491
+ // asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
492
+ // service below the version floor turns a wait into a paid run nothing can tell from a correct
493
+ // one. In the compose topology the worker's read is the live inode while the receiver's is dead
494
+ // until restart, which is exactly the deployment where the skew happens.
495
+ checkWaitSkew: makeCheckWaitSkew({ triggersPath: onceTriggersFile }),
496
+ // Issue #230. Whether a job the supersede lease names is still in the queue. Without it a holder
497
+ // that vanished by any route except the clean one leaves a key that refuses every later delivery
498
+ // for that target until it expires -- and a refused forge delivery is gone, since no webhook
499
+ // resends it. `getJob` answers from the queue rather than from our own bookkeeping, so the two
500
+ // cannot agree with each other while both being wrong.
501
+ // REQ-WAIT-FOR's polled tier. Built here for the image and egress preflights' reason: one
502
+ // deployment value, one place, so the gate that refuses an undeclared profile and the spawn that
503
+ // runs it cannot disagree about which checks exist. The env-declared table is parsed once at boot
504
+ // (it is env, not overlay -- the gate reads its config above the per-job settings read).
505
+ // The free half of the profile check: whether this deployment declares the name at all. A table
506
+ // lookup, so it belongs with the gate's other free refusals rather than inside the subprocess.
507
+ waitProfileDeclared: (name) => typeof config.waitProfiles[name] === "string",
508
+ checkWait: makeWaitChecker({ profiles: config.waitProfiles, timeoutMs: config.waitCheckTimeoutMs, log }),
509
+ isJobLive: async (id) => {
510
+ const held = await runtimeQueue.getJob(id);
511
+ if (!held) return false;
512
+ // EXISTENCE IS NOT LIVENESS, and the difference decides whether a target stays deafened:
513
+ // `removeOnComplete`/`removeOnFail` keep a finished job's hash for 31 days, so a holder that
514
+ // can never wake again would answer "still waiting" for a month. Only a state it can still be
515
+ // picked up from counts.
516
+ const state = await held.getState();
517
+ return state === "delayed" || state === "waiting" || state === "active" || state === "prioritized" || state === "waiting-children";
518
+ },
476
519
  // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
477
520
  // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
478
521
  // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
@@ -376,6 +376,83 @@ export function makeCheckOnceSpent({ triggersPath, fs = nodeFs }) {
376
376
  };
377
377
  }
378
378
 
379
+ /**
380
+ * Detect the version skew `run.waitFor` opens (issue #230), pre-spend, from the worker's own file read.
381
+ *
382
+ * The hazard is `DES-TRIGGERS-UNIFIED-FILE`'s widening rule arriving somewhere it has never bitten. Unknown
383
+ * keys DROP, which for every previous field was a harmless no-op: an old parser meeting `run.image` gives
384
+ * you the default image and a job that ran. `waitFor` is the first field whose ABSENCE is destructive -- a
385
+ * receiver below the floor enqueues the job without it and the worker runs it UNHELD, producing a record, a
386
+ * panel row and a log line byte-identical to one that correctly waited. Success is the least detectable
387
+ * failure available, and the whole point of a wait is that running now is the destructive option.
388
+ *
389
+ * `docs/secrets.md`'s answer to the same skew is documentation plus a version floor, and it is not enough
390
+ * here for two reasons: a dropped secret surfaces as a 401 and an agent report that reads wrong, and
391
+ * `doctor` cannot see the receiver's installed version from the worker host, so its warning would fire on
392
+ * every deployment using the feature, forever -- the always-on amber the panel's own design rejects.
393
+ *
394
+ * So the worker checks the authored file itself. It already reads it per job for the one-shot gate, and in
395
+ * the compose topology this read is the authoritative one: the receiver's single-file `:ro` mount pins a
396
+ * dead inode until restart, which is precisely the deployment where the skew bites.
397
+ *
398
+ * FAIL-OPEN throughout, `readDisarmState`'s posture and for its reason: an unreadable or changed file means
399
+ * "run", because a broken read must never wedge every job. Only a positive, identity-confirmed mismatch --
400
+ * this entry authored conditions, this job carries none -- refuses.
401
+ */
402
+ export function makeCheckWaitSkew({ triggersPath, fs = nodeFs }) {
403
+ // Cached by mtime, unlike `checkOnceSpent` which re-reads every time. The difference is which jobs each
404
+ // one runs for: the one-shot check is gated on `matched.once === true`, so it is rare by construction and
405
+ // its comment can call one read cheap. This one has to look at EVERY forge job, because the whole point
406
+ // is to catch a job that arrived WITHOUT the field, and there is nothing on such a job to narrow by. So a
407
+ // stat replaces a read-and-parse on the hot path. The residual is millisecond mtime granularity: two
408
+ // writes inside one millisecond would serve a stale parse for one job, which fails OPEN like every other
409
+ // uncertainty here.
410
+ let cache = null; // { mtimeMs, size, triggers }
411
+ const read = () => {
412
+ try {
413
+ const st = fs.statSync?.(triggersPath);
414
+ if (cache && st && cache.mtimeMs === st.mtimeMs && cache.size === st.size) return cache.triggers;
415
+ const raw = JSON.parse(fs.readFileSync(triggersPath, "utf8"));
416
+ const triggers = Array.isArray(raw?.triggers) ? raw.triggers : null;
417
+ if (st) cache = { mtimeMs: st.mtimeMs, size: st.size, triggers };
418
+ return triggers;
419
+ } catch {
420
+ cache = null;
421
+ return null;
422
+ }
423
+ };
424
+
425
+ return async function checkWaitSkew(job) {
426
+ if (typeof triggersPath !== "string" || triggersPath === "") return { ok: true };
427
+ // Cron and CLI jobs carry no matched index, and cannot carry `waitFor` at all.
428
+ const index = job?.trigger?.matched?.index;
429
+ if (!Number.isInteger(index)) return { ok: true };
430
+ // The field arrived. Whether its conditions are SATISFIED is the gate's business, not this check's.
431
+ if (Array.isArray(job?.waitFor) && job.waitFor.length > 0) return { ok: true };
432
+
433
+ const triggers = read();
434
+ if (triggers === null) return { ok: true };
435
+ const entry = triggers[index];
436
+ const authored = entry?.run?.waitFor;
437
+ if (!Array.isArray(authored) || authored.length === 0) return { ok: true };
438
+
439
+ // The identity guard readDisarmState keeps, and WIDER than flow alone. `triggerIndex` is a RAW array
440
+ // position, so an insertion, a reorder, or a stray `triggers.json` at the worker's cwd can all put a
441
+ // different trigger here -- and two rules sharing a flow are indistinguishable on flow alone, which
442
+ // made a false refusal reachable by ordinary editing. Kind and on-type are compared too, on
443
+ // `readDisarmState`'s precedent of checking the item number rather than trusting the index.
444
+ //
445
+ // Every mismatch folds to "run", never to a refusal: the true answer to "is this job missing
446
+ // conditions someone wrote for it?" is then unknown, and unknown must not refuse a paid delivery.
447
+ if (entry?.run?.flow !== job?.flow || entry?.run?.command !== job?.command) return { ok: true };
448
+ if (entry?.run?.kind !== job?.kind) return { ok: true };
449
+ const onType = job?.trigger?.matched?.type;
450
+ if (typeof onType === "string" && entry?.on?.type !== onType) return { ok: true };
451
+
452
+ return { skewed: true, conditions: authored.length };
453
+ };
454
+ }
455
+
379
456
  export function readDisarmState({ triggersPath, index, number, flow, command, fs = nodeFs }) {
380
457
  let raw;
381
458
  try {