@edgehero/pi-dispatch 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.mjs CHANGED
@@ -2,7 +2,10 @@ import { execFile } from "node:child_process";
2
2
  import { promisify } from "node:util";
3
3
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
4
4
  import { InfraRetry, runJob } from "./processor.mjs";
5
+ import { targetFor } from "./run-history.mjs";
5
6
  import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
7
+ import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
8
+ import { makeWaitState } from "./wait-state.mjs";
6
9
 
7
10
  const exec = promisify(execFile);
8
11
 
@@ -16,6 +19,28 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
16
19
  // docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
17
20
  export const SCOPE_BUSY_RECHECK_MS = 5_000;
18
21
 
22
+ // How long a job waits before re-asking whether a target's holder is still alive (issue #230). Reached only
23
+ // when the liveness probe could not answer, which is a redis or queue fault rather than a normal state, so
24
+ // this is a short retry rather than a cadence: the job is deciding nothing and holding nothing while it
25
+ // waits, and the fault it is waiting out is usually seconds long.
26
+ export const SUPERSEDE_RECHECK_MS = 15_000;
27
+
28
+ // The one key the check lease counts under. A single global counter rather than one per profile: what it
29
+ // bounds is this worker's wall-clock spent answering questions, and that is shared whatever is being asked.
30
+ const WAIT_CHECK_KEY = "wait-check";
31
+
32
+ // How many consecutive lease denials one job absorbs before the deployment is told its checking capacity is
33
+ // short. Logged ONCE per run of denials rather than per wake: an alarm that repeats every re-check is the
34
+ // always-on amber this project rejects elsewhere, and the operator only needs telling once per episode.
35
+ const THROTTLE_ALARM = 5;
36
+
37
+ // The floor under a throttled or aborted re-ask. Its own constant rather than a borrow of
38
+ // SUPERSEDE_RECHECK_MS, which documents an unrelated concern. Deliberately NOT 5s: that is
39
+ // SCOPE_BUSY_RECHECK_MS, and INT-WAIT-PROFILES-CONTRACT rests on wait deferrals being distinguishable from
40
+ // scope deferrals by wake instant -- nothing records WHY a job sits in the delayed set, so the instants are
41
+ // the only evidence there is. A test pins the two apart.
42
+ const THROTTLE_FLOOR_MS = 11_000;
43
+
19
44
  /**
20
45
  * Build the BullMQ processor.
21
46
  *
@@ -35,7 +60,7 @@ export const SCOPE_BUSY_RECHECK_MS = 5_000;
35
60
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
36
61
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
37
62
  */
38
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
63
+ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
39
64
  return async function processor(job, token, signal) {
40
65
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
41
66
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -53,9 +78,295 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
53
78
  throw new DelayedError();
54
79
  }
55
80
 
81
+ // The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
82
+ // not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
83
+ // because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
84
+ // does. Strictly above the `try` for the two reasons the gates below it document.
85
+ //
86
+ // The order WITHIN the gate is determinate-refusals-then-holds, which is CONST-BUDGET-BEFORE-TOKENS'
87
+ // shape applied to time rather than to money: a condition this deployment can never answer must be
88
+ // refused now, not after a day of waiting.
89
+ //
90
+ // On throwing above the `try`: an exception here escapes into BullMQ's normal failed-attempt handling,
91
+ // which is WANTED for `moveToDelayed` (the scope gate below gives the argument: a transient rejection
92
+ // must stay a transient failure rather than becoming a permanent one) and unwanted everywhere else. So
93
+ // the state and comment seams fail open by construction, and `recordRun` is relied on not to throw --
94
+ // its writer swallows fs errors by contract, which is the same reliance the settings-overlay refusal
95
+ // below already makes.
96
+ if (waitArmed(job.data)) {
97
+ // The supersede identity: the queue's semantic key PLUS the trigger that produced this job.
98
+ // The semantic key alone is `repo<sep>number:flow`, which two DIFFERENT triggers on one target and
99
+ // flow legitimately share -- a label rule that waits a day and a comment rule that waits a minute
100
+ // would coalesce, and the second would be refused with a message claiming they wait on "the same
101
+ // conditions" when they do not. Adding the raw trigger index makes the key mean one intent.
102
+ const matchedIndex = job.data?.trigger?.matched?.index;
103
+ const dedupId = job.deduplicationId ? `${job.deduplicationId}#${Number.isInteger(matchedIndex) ? matchedIndex : "?"}` : null;
104
+ const refuseWait = async (reason, logEvent, fields, sentence) => {
105
+ // The INJECTED clock, like both gates above: a record whose timestamps ignore the test clock
106
+ // is a record no test of this gate can assert about.
107
+ const at = new Date(now()).toISOString();
108
+ await waitState.release(job.id, { dedupId });
109
+ deps?.log?.(logEvent, { jobId: job.id, ...fields });
110
+ // The comment names the FIELD and the operator's own words for the condition, never a
111
+ // resolver path or a vault topology -- `secret-profile-unknown` sets that rule.
112
+ if (sentence && deps?.comment) await Promise.resolve(deps.comment(job.data, sentence)).catch(() => {});
113
+ const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false };
114
+ recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString() });
115
+ return result;
116
+ };
117
+
118
+ // EVERY condition must be one this worker understands, checked before anything else. The loader
119
+ // refuses an unknown condition, but the loader is a DIFFERENT PROCESS: `job.data.waitFor` arrives
120
+ // over Redis from the receiver, and this whole feature exists because receiver-worker version
121
+ // skew is real. `makeCheckWaitSkew` closes the backward direction (the file has conditions the
122
+ // job arrived without); this closes the forward one (a newer receiver enqueues a condition shape
123
+ // this worker cannot read). Without it the gate would fall through, log `wait_cleared`, and run
124
+ // the job -- asserting in the log that conditions cleared which it never evaluated, which is the
125
+ // same undetectable paid run the backward check exists to stop.
126
+ // A sibling that was held on this same target may already have cleared it. Checked FIRST, because
127
+ // it is free and determinate, and because the window it closes is one no lease can: two jobs
128
+ // holding through an outage that outlives their leases would each wake, find no holder, and run.
129
+ if (dedupId) {
130
+ const satisfiedBy = await waitState.satisfiedBy(dedupId);
131
+ if (satisfiedBy && satisfiedBy !== job.id) {
132
+ return await refuseWait("wait-superseded", "wait_superseded", { satisfiedBy }, "Another delivery for this target already finished waiting on the same conditions. Not run.");
133
+ }
134
+ }
135
+
136
+ const unreadable = unreadableConditions(job.data);
137
+ if (unreadable.length > 0) {
138
+ // Its OWN token, not `wait-skew`. Both are version skew, and the REMEDIES are opposites --
139
+ // upgrade the receiver there, upgrade the worker here -- so one token in a durable record
140
+ // would tell an operator that something is out of step and not which way to move.
141
+ return await refuseWait("wait-unreadable", "refused_wait_unreadable", { conditions: unreadable.length }, `Refused: this job carries ${unreadable.length} wait condition${unreadable.length === 1 ? "" : "s"} this worker cannot read, so it cannot honour them. The worker is older than the service that enqueued this job. Not run.`);
142
+ }
143
+
144
+ // A `profile` condition needs a checker, and with none wired NOTHING can answer it. Refused rather
145
+ // than ignored: a wait the deployment cannot perform must not read as a wait that passed.
146
+ const profiles = waitProfileNames(job.data);
147
+ // Declared-ness is a table lookup, so it belongs with the other free refusals rather than inside
148
+ // the check. Without it here, `[{after: "<tomorrow>"}, {profile: "typo"}]` holds for a day and
149
+ // THEN refuses -- which is the exact sentence the ordering rule above promises will not happen.
150
+ const undeclared = deps?.waitProfileDeclared ? profiles.find((name) => !deps.waitProfileDeclared(name)) : undefined;
151
+ if (undeclared !== undefined) {
152
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: undeclared }, `Waiting on \`${undeclared}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
153
+ }
154
+ if (profiles.length > 0 && !deps?.checkWait) {
155
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: profiles[0] }, `Waiting on \`${profiles[0]}\` is not something this deployment can answer. Not run.`);
156
+ }
157
+
158
+ const holdUntil = afterMs(job.data); // named apart from the pause gate's `until` above, which it would otherwise shadow
159
+ // An instant further out than the ceiling is refused at FIRST pickup rather than held toward:
160
+ // holding for a month to then refuse tells the operator nothing they could not have been told now.
161
+ if (holdUntil !== null && holdUntil - nowMs > afterMaxMs()) {
162
+ return await refuseWait("wait-after-beyond-max", "wait_after_beyond_max", { delayMs: holdUntil - nowMs }, `The \`after\` instant is further out than this deployment allows a job to wait. Not run.`);
163
+ }
164
+
165
+ // The pause gate's boundary guard, for its reason: a tick landing on the instant must run rather
166
+ // than busy-defer to a moment already past.
167
+ if (holdUntil !== null && holdUntil > nowMs + 1000) {
168
+ // `isJobLive` is what stops a vanished holder's lease becoming a tombstone that refuses this
169
+ // target for the rest of the hold. Optional: an unwired probe means the holder cannot be
170
+ // checked, which ADMITS and says so -- one duplicate run beats one dropped delivery, which is
171
+ // `OQ-027`'s call ("one wasted vault read beats one dropped job") on this feature's terms.
172
+ const claim = await waitState.claim(job.id, { dedupId, untilMs: holdUntil, isLive: deps?.isJobLive });
173
+ if (claim.heldBy) {
174
+ // Another delivery for this same target and flow is already holding. Both would clear
175
+ // together and both would be paid, which is the accumulation the acceptance forbids.
176
+ return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
177
+ }
178
+ if (claim.retry) {
179
+ // The holder could not be checked. Holding anyway would put two jobs on one target and pay
180
+ // for both; refusing would drop a delivery over a holder that may be gone. So decide
181
+ // nothing: re-defer briefly and ask again once the probe can answer.
182
+ deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
183
+ await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
184
+ throw new DelayedError();
185
+ }
186
+ if (claim.tookOverFrom) deps?.log?.("wait_lease_taken_over", { jobId: job.id, from: claim.tookOverFrom });
187
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: holdUntil });
188
+ deps?.log?.("wait_deferred", { jobId: job.id, until: new Date(holdUntil).toISOString(), label: waitLabel(job.data) });
189
+ await job.moveToDelayed(holdUntil, token);
190
+ throw new DelayedError();
191
+ }
192
+
193
+ // TIER 2: the polled conditions. Last, because it is the only part of this gate that spawns a
194
+ // process -- the free refusals above it are free, and the free hold above it is free.
195
+ if (profiles.length > 0) {
196
+ const held = (await waitState.heldForMs(job.id)) ?? 0;
197
+ const counted = await waitState.counters(job.id);
198
+
199
+ // One check at a time, process-wide, and never the worker's last free slot. This is the bound
200
+ // that keeps a wait from starving the paid work it is waiting for: slots x timeout is the most
201
+ // wall-clock a worker can spend answering questions instead of running jobs. Computed against
202
+ // the LIVE concurrency rather than the boot value, because the overlay can lower it.
203
+ const slots = Math.min(checkSlotCount(), Math.max(1, concurrencyNow() - 1));
204
+ if (!checkSlots.tryAcquire(WAIT_CHECK_KEY, slots)) {
205
+ // Denials are counted, and a run of them is the ONE symptom the capacity bound has. The
206
+ // lease deliberately caps how much wall-clock this worker spends checking; being at that
207
+ // cap constantly means demand exceeds it, which the issue's own economics say arrives
208
+ // silently -- paid jobs starve behind checks that spend nothing and nothing says why.
209
+ const denials = await waitState.noteThrottle(job.id, { denied: true });
210
+ if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots, hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
211
+ // A starved job still needs a CLOCK and a CEILING, or the lease turns into the very
212
+ // starvation it exists to bound: without this the hold is stamped only on a wake that won
213
+ // the lease, so a job that never wins one has no `since`, never reaches the maximum, and
214
+ // re-wakes forever with no record and no bound. There is no deciding check to run first
215
+ // here -- that is the whole condition -- so the bound applies directly.
216
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
217
+ if (held >= maxWaitMs()) {
218
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait-unchecked", denials, heldForMs: held }, `Gave up waiting: this deployment could not run the check often enough to answer within the maximum wait. Not run.`);
219
+ }
220
+ // Denied. Re-ask at a fraction of the cadence rather than the full backoff (which would
221
+ // turn one lost coin-flip into a fifteen-minute penalty) or a flat few seconds (which at
222
+ // scale is a herd). Jittered, so a fleet of denied jobs does not return together.
223
+ const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
224
+ const delay = wait + Math.floor(wait * 0.1 * random());
225
+ deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots });
226
+ await job.moveToDelayed(nowMs + delay, token);
227
+ throw new DelayedError();
228
+ }
229
+
230
+ // Declared outside the try below because the branches AFTER it read both.
231
+ let verdict = null;
232
+ let checked = null;
233
+ // THE LEASE IS HELD FROM THE `tryAcquire` ABOVE, so every exit from here down must release it.
234
+ // The try opens here and not at the check loop, which is where it used to open: the supersede
235
+ // claim sits between the two, and BOTH of its exits leave -- one returns `wait-superseded`,
236
+ // the other re-defers and throws -- so a claim that refused or could not be verified walked
237
+ // out holding the slot. At the shipped default of one slot that wedged every wait check on
238
+ // the worker until it restarted, and the symptom was silent in the worst way: held jobs kept
239
+ // throttling and eventually recorded `wait-expired` with `max-wait-unchecked`, which blames
240
+ // the deployment's capacity for a slot this gate leaked.
241
+ try {
242
+ // Claimed BEFORE the check, not after: a second delivery for an already-held target is a free
243
+ // determinate refusal, and paying for a subprocess first inverts the free-before-costly rule
244
+ // this gate's own header invokes. Tier 1 already claims in this order.
245
+ const claim = await waitState.claim(job.id, { dedupId, untilMs: nowMs + waitBackoffMs(intervalMs(), held), isLive: deps?.isJobLive });
246
+ if (claim.heldBy) {
247
+ return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
248
+ }
249
+ if (claim.retry) {
250
+ deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
251
+ await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
252
+ throw new DelayedError();
253
+ }
254
+
255
+ await waitState.noteThrottle(job.id, { denied: false }); // granted: the run of denials ends here
256
+ // Sequential, in the operator's writing order: the resolver's reason applies unchanged --
257
+ // naming the first condition that did not clear is what makes a held row readable, and a
258
+ // parallel fan-out would blame whichever lost the race on any given wake.
259
+ for (const profile of profiles) {
260
+ checked = profile;
261
+ verdict = await deps.checkWait(profile, targetFor(job.data?.kind, job.data), { signal });
262
+ if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
263
+ }
264
+ } finally {
265
+ checkSlots.release(WAIT_CHECK_KEY);
266
+ }
267
+
268
+ if (verdict?.unusableTarget) {
269
+ // Determinate and unfixable by waiting: the job's own target is a shape no check can be
270
+ // handed. It belongs with the refusals, not the holds -- holding would spend the fault
271
+ // budget and then blame the operator's script for a value it was never given.
272
+ return await refuseWait("wait-unreadable", "refused_wait_unreadable", { profile: checked }, `Refused: this job's target cannot be handed to a wait check, so \`${checked}\` can never be asked. Not run.`);
273
+ }
274
+ if (verdict?.profileUnknown) {
275
+ return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: verdict.profileUnknown }, `Waiting on \`${verdict.profileUnknown}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
276
+ }
277
+ if (verdict?.verdict === "refuse") {
278
+ // Exit 2: the check says this will NEVER clear. Terminal by the protocol's own words, and
279
+ // distinct from every "not yet" above it.
280
+ return await refuseWait("wait-refused", "wait_refused", { profile: checked, heldForMs: held }, `The check \`${checked}\` reports this will never clear. Not run.`);
281
+ }
282
+
283
+ if (verdict?.aborted) {
284
+ // The worker is stopping or this job was cancelled. Nothing was learned and nothing is
285
+ // owed: re-defer at once rather than at the full backoff, and count neither a check nor a
286
+ // fault, or a rolling deploy would spend a job's whole budget on its own restarts and then
287
+ // blame the operator's script for it.
288
+ deps?.log?.("wait_check_aborted", { jobId: job.id, profile: checked });
289
+ await job.moveToDelayed(nowMs + THROTTLE_FLOOR_MS, token);
290
+ throw new DelayedError();
291
+ }
292
+
293
+ if (verdict?.verdict === "hold") {
294
+ const fault = verdict.fault === true;
295
+ await waitState.noteCheck(job.id, { fault });
296
+ const faults = fault ? counted.faults + 1 : 0;
297
+
298
+ // A check that never answers is a broken script, not a slow condition, and OQ-030 is why
299
+ // this bound exists: most CLIs exit 1 for everything, so without it a typo would hold for
300
+ // the whole maximum wait and then blame the CONDITION rather than the check.
301
+ if (faults >= maxFaults()) {
302
+ return await refuseWait("wait-unanswerable", "wait_unanswerable", { profile: checked, faults }, `The check \`${checked}\` could not answer ${faults} times in a row. Not run.`);
303
+ }
304
+
305
+ // BOTH terminal bounds are tested AFTER the check and never before it, so a condition that
306
+ // cleared on the deciding wake runs instead of being recorded as never having cleared.
307
+ // Without that ordering the backoff's own quantisation makes "cleared at t+1s, declared
308
+ // never-cleared at t+900s" a structural lie in the durable record and in a public comment.
309
+ //
310
+ // The count bound reads `checks + 1` because this wake's check has just run: the job gets
311
+ // exactly `maxChecks` checks, the last of which is the deciding one. Putting it before the
312
+ // check instead -- so the act of testing the bound could not exceed it -- was the obvious
313
+ // spelling, and it silently made this whole guarantee untrue at every shipped default,
314
+ // because the count bound is the one that fires first there.
315
+ if (counted.checks + 1 >= maxChecks()) {
316
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-checks", checks: counted.checks + 1, profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\` after ${counted.checks + 1} checks. Not run.`);
317
+ }
318
+ if (held >= maxWaitMs()) {
319
+ return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait", profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\`. Not run.`);
320
+ }
321
+
322
+ // Clamped to what is LEFT of the budget, never just the cadence. Without this an hourly
323
+ // interval under a fifteen-minute maximum holds for the full hour -- 400% of the bound the
324
+ // operator configured -- because the ceiling is only tested when a wake arrives, and the
325
+ // cadence decides when that is. The two knobs are independent `positiveInt`s and nothing
326
+ // cross-validates them, so the clamp is what makes the smaller one actually bind.
327
+ const base = waitBackoffMs(intervalMs(), held);
328
+ const jittered = base + Math.floor(base * 0.1 * random());
329
+ const remaining = Math.max(0, maxWaitMs() - held);
330
+ const delay = Math.max(1000, Math.min(jittered, remaining));
331
+ await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + delay });
332
+ deps?.log?.("wait_deferred", { jobId: job.id, profile: checked, fault, heldForMs: held, delayMs: delay });
333
+ await job.moveToDelayed(nowMs + delay, token);
334
+ throw new DelayedError();
335
+ }
336
+
337
+ // FAIL CLOSED on anything that is not literally go. Everything above tests for a specific
338
+ // shape and falls through otherwise, and "otherwise" at this gate means STARTING A PAID
339
+ // CONTAINER -- so an `undefined`, a `null`, a `{}`, a mis-cased "GO" or a bare string from a
340
+ // checker would run the job silently, with no record field and no log line to distinguish it
341
+ // from a job whose check said yes. The shipped checker is total, and that is exactly the
342
+ // reasoning `unreadableConditions` above rejects: this is a dependency-injection seam, and a
343
+ // seam's guarantees are the caller's to enforce.
344
+ if (verdict?.verdict !== "go") {
345
+ deps?.log?.("wait_check_unintelligible", { jobId: job.id, profile: checked });
346
+ await waitState.noteCheck(job.id, { fault: true });
347
+ const base = waitBackoffMs(intervalMs(), held);
348
+ await job.moveToDelayed(nowMs + base + Math.floor(base * 0.1 * random()), token);
349
+ throw new DelayedError();
350
+ }
351
+
352
+ // Every profile answered go. Record the last check so the count bound sees it.
353
+ await waitState.noteCheck(job.id, { fault: false });
354
+ }
355
+
356
+ // Only a job that actually HELD has cleared. Without the check this line fires on the first
357
+ // pickup of a job whose instant had already passed, and again on every scope-busy re-check
358
+ // afterwards -- asserting a wait ended that never began.
359
+ const heldForMs = await waitState.heldForMs(job.id);
360
+ // Say so before releasing: a sibling held on this target must find the answer, not an empty lease.
361
+ if (heldForMs !== null) await waitState.markSatisfied(job.id, { dedupId });
362
+ await waitState.release(job.id, { dedupId });
363
+ if (heldForMs !== null) deps?.log?.("wait_cleared", { jobId: job.id, label: waitLabel(job.data), heldForMs });
364
+ }
365
+
56
366
  // Per-scope concurrency and the one-job-per-folder mutex (issue #242,
57
- // INT-SCOPED-LIMITS-FILE-CONTRACT). SECOND, after the pause gate (a paused job must not burn
58
- // re-check wakes) and STRICTLY above the `try` below, like the pause gate and for the same two
367
+ // INT-SCOPED-LIMITS-FILE-CONTRACT). LAST of the three gates, after the pause gate (a paused job must
368
+ // not burn re-check wakes) and after the wait gate (a job holding until tomorrow must not sit on a
369
+ // folder while it does), and STRICTLY above the `try` below, like the pause gate and for the same two
59
370
  // reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
60
371
  // catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
61
372
  // handling exactly as the pause gate's does (inside the try it would become a permanent failure
@@ -123,7 +434,9 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
123
434
  // job completed and does not retry a file that can never parse (CONST-RETRY-INFRA-ONLY). Resolved
124
435
  // before runJob, so no budget slot is reserved and no container starts (CONST-BUDGET-BEFORE-TOKENS).
125
436
  // recordRun leaves the durable settings-overlay-invalid trace for the admin extension.
126
- // No provider/model here, alone among the terminal results: the overlay is the thing that failed to parse, so no honest effective value exists -- buildRecord defaults both null.
437
+ // No provider/model here, as in every result this function returns from ABOVE the try (the wait gate's
438
+ // refusals are the others): each is decided before or during the settings read, so no honest effective
439
+ // value exists yet -- buildRecord defaults both null.
127
440
  const result = { outcome: "policy", reason: "settings-overlay-invalid", exitCode: null, turns: null, tokens: null, budgetReserved: false };
128
441
  recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
129
442
  return result;
@@ -210,7 +523,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
210
523
  };
211
524
  }
212
525
 
213
- export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
526
+ export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
214
527
  let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
215
528
  const processor = makeProcessor({
216
529
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
@@ -228,6 +541,21 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
228
541
  // shape means one per daemon).
229
542
  scopedLimits,
230
543
  inFlight,
544
+ // Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
545
+ // redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
546
+ waitState,
547
+ afterMaxMs,
548
+ // Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
549
+ // because the overlay can lower it through `dispatch_set` and a check must never take the last free
550
+ // slot from a paid job. Late-bound over `worker` exactly as `applyConcurrency` is, and for the same
551
+ // reason: the value it needs does not exist until the Worker is constructed.
552
+ checkSlots,
553
+ checkSlotCount,
554
+ concurrencyNow: concurrencyNow ?? (() => worker?.concurrency ?? concurrency),
555
+ intervalMs,
556
+ maxWaitMs,
557
+ maxChecks,
558
+ maxFaults,
231
559
  deps,
232
560
  recordRun,
233
561
  });
package/src/processor.mjs CHANGED
@@ -54,6 +54,9 @@ export async function runJob(job, deps) {
54
54
  // exactly as before, and the gate below only calls it for a job whose matched rule was a
55
55
  // one-shot, so the default is never a probe running on every delivery.
56
56
  checkOnceSpent = async () => ({ ok: true }),
57
+ // Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
58
+ // unwired seam must not refuse, and the wiring is what turns the check on.
59
+ checkWaitSkew = async () => ({ ok: true }),
57
60
  // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
58
61
  // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
59
62
  egressPreflight = async () => ({ ok: true }),
@@ -155,6 +158,28 @@ export async function runJob(job, deps) {
155
158
  }
156
159
  }
157
160
 
161
+ // The wait-skew check (issue #230), second on the ladder and for the first one's reasons: the same
162
+ // file read, free, determinate, credential-less, and pre-spend. It answers a question no other layer
163
+ // can: does the AUTHORED trigger carry wait conditions this job arrived without? That happens when a
164
+ // service below the version floor dropped the field as an unknown key, and the resulting run is
165
+ // byte-identical to a correct one everywhere it is recorded -- so this refusal is the only thing
166
+ // standing between a stale receiver and a paid job that ran when the operator wrote "wait".
167
+ {
168
+ const skew = await checkWaitSkew(job);
169
+ if (skew.skewed) {
170
+ // Named for the operator, not the payload: how many conditions were authored, never what
171
+ // they say. The fix is a version, so the comment says which one.
172
+ // The message names BOTH causes, because the more likely one is not a version at all. In the
173
+ // compose topology the receiver's single-file `:ro` mount pins a dead inode, so an operator
174
+ // who ADDS `waitFor` to an existing rule gets this refusal on every delivery from a service
175
+ // that is perfectly up to date and merely holding an older copy of the file. Naming only the
176
+ // version would send them looking for an upgrade they do not need.
177
+ await comment(job, `Refused: this trigger declares ${skew.conditions} wait condition${skew.conditions === 1 ? "" : "s"}, but the job reached the worker without them, which means it would have run immediately. Either a service in this deployment is below the version that carries the field, or one is still running against an older copy of the triggers file and needs restarting. Not run.`);
178
+ log("refused_wait_skew", { triggerIndex: job.trigger?.matched?.index ?? null, conditions: skew.conditions });
179
+ return { outcome: "policy", reason: "wait-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
180
+ }
181
+ }
182
+
158
183
  // The job image must exist on THIS host before anything else happens. Free, determinate and
159
184
  // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
160
185
  // image refuses without minting a credential it will not use, cloning a repo it will not read, or
package/src/queue.mjs CHANGED
@@ -134,7 +134,7 @@ export async function enqueueGitLabJob(queue, fields) {
134
134
  * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
135
135
  * has always been.
136
136
  */
137
- export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, replica, replicas }) {
137
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
138
138
  const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
139
139
  // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
140
140
  // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
@@ -179,6 +179,16 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
179
179
  // is copied verbatim into /job/event.json, which an agent reads.
180
180
  ...(secrets !== undefined && { secrets }),
181
181
  ...(secretsProfile !== undefined && { secretsProfile }),
182
+ // Issue #230. The conditions the worker holds this job on, carried so the PICKUP gate can read them:
183
+ // that gate runs above the per-job settings read and never re-parses the triggers file for its terms.
184
+ // At JOB level, and here that placement is a correctness requirement rather than a convention --
185
+ // `trigger` is copied VERBATIM into /job/event.json (prepare-local.mjs), so a `trigger.waitFor` would
186
+ // hand the agent the operator's own gate. Conditional like every field above, so an unflagged job's
187
+ // data keeps exactly the keys it has today. The dedup options below are deliberately NOT widened for
188
+ // a waiting job: that key carries no trigger identity and outlives the job it was set for, so a
189
+ // longer window would suppress an unflagged sibling's deliveries and go on suppressing them after
190
+ // this job finished. Coalescing a held target is the worker's `wait:` keyspace's job instead.
191
+ ...(waitFor !== undefined && { waitFor }),
182
192
  // Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
183
193
  // exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
184
194
  // both are integers, so the run record they land in stays PII-free by construction.
@@ -114,7 +114,34 @@ export function parseExitTurns(text) {
114
114
  * Read-only telemetry, exactly like `parseExitTurns`: NEVER throws and MUST NOT feed exit-code or retry
115
115
  * classification (INT-RUNNER-EXIT-CODE-PROTOCOL). A malformed or non-object `tokens` (or one missing a
116
116
  * numeric `total`) is `null`, never a partial that could poison the daily token counter.
117
+ *
118
+ * The admitted object is REBUILT through `rebuildTokens` rather than returned as it arrived. See that
119
+ * function for why: the pass-through it replaces is what made this comment's "integer token counts and
120
+ * numeric cost only" false one level below `buildRecord`'s literal.
121
+ */
122
+ /**
123
+ * The CLOSED `session.reason` enum, verbatim from `INT-RUN-HISTORY-FILE-CONTRACT`. Three producers write
124
+ * this field (resolve, runner, promote) and the contract has always called the set closed; until this
125
+ * list existed, nothing enforced it and the runner's half was an unchecked string. Kept here rather than
126
+ * beside the store because this module is where the container's copy is admitted, and an enum that lives
127
+ * anywhere but the admission point is a comment, not a check.
117
128
  */
129
+ const SESSION_REASONS = new Set([
130
+ "resumed",
131
+ "absent",
132
+ "expired",
133
+ "conversation-too-old",
134
+ "resume-chain-too-long",
135
+ "context-too-full",
136
+ "too-large",
137
+ "unparseable",
138
+ "not-a-regular-file",
139
+ "pi-version-changed",
140
+ "locked",
141
+ "promote-failed",
142
+ "disabled",
143
+ ]);
144
+
118
145
  /**
119
146
  * The runner's `session` object off the exit line: `{ resumed: <bool>, reason: "<enum>" }` or null when
120
147
  * the container died before emitting one (REQ-RESUMABLE-SESSION).
@@ -125,7 +152,8 @@ export function parseExitTurns(text) {
125
152
  * without both numbers it is indistinguishable from an ordinary cold start. A feature that fails open
126
153
  * must still say that it did.
127
154
  *
128
- * PII-free by construction: a boolean and a fixed enum. No key, no branch name, no path.
155
+ * PII-free by construction: a boolean and a fixed enum. No key, no branch name, no path -- and since the
156
+ * `SESSION_REASONS` check below, that sentence is enforced rather than merely intended.
129
157
  */
130
158
  export function parseExitSession(text) {
131
159
  if (typeof text !== "string") return null;
@@ -137,7 +165,14 @@ export function parseExitSession(text) {
137
165
  if (parsed?.event !== "exit") continue;
138
166
  const sess = parsed?.session;
139
167
  if (sess && typeof sess === "object" && !Array.isArray(sess) && typeof sess.resumed === "boolean") {
140
- return { resumed: sess.resumed, reason: typeof sess.reason === "string" ? sess.reason : null };
168
+ // The reason is checked against the CLOSED enum, not merely against `typeof === "string"`, which
169
+ // is what it used to be. The container owns this value, so an unchecked string put an
170
+ // attacker-shapeable one into a record whose PII-free property rests on holding none -- while
171
+ // the comment above claimed "a boolean and a fixed enum". An unrecognised token reads as `null`
172
+ // (the runner said nothing this contract can represent) rather than being carried through: the
173
+ // enum is documented CLOSED in INT-RUN-HISTORY-FILE-CONTRACT, so a value outside it was already
174
+ // contract-violating and every consumer already handles null.
175
+ return { resumed: sess.resumed, reason: SESSION_REASONS.has(sess.reason) ? sess.reason : null };
141
176
  }
142
177
  return null;
143
178
  }
@@ -176,6 +211,43 @@ export function parseExitContext(text) {
176
211
  return null;
177
212
  }
178
213
 
214
+ /**
215
+ * The keys the runner actually emits, in its own emission order: the metered snapshot
216
+ * (`image/runner/src/usage-meter.mjs` -> `snapshot`) plus the token-budget fallback
217
+ * (`image/runner/run-job.mjs` -> `pickTotals`, which sends the first four and `metered: false`). Order
218
+ * matters because it is what makes a conformant runner's object round-trip byte-identically through the
219
+ * rebuild below, so the record's bytes do not move for anyone running a real image.
220
+ */
221
+ const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced"];
222
+
223
+ /**
224
+ * Rebuild the billed totals from a closed key list rather than passing the container's object through.
225
+ *
226
+ * This function exists because the pass-through was a hole. `parseExitTokens` used to `return t`
227
+ * verbatim whenever `t.total` was a number, so any key the container invented -- a path, a branch name,
228
+ * a string it read out of the workspace -- rode into the durable record, and from there into anything
229
+ * that mirrors it. The record's PII-free-by-construction property held at `buildRecord`'s own level and
230
+ * NOT one level down, while this module's own comment claimed "integer token counts and numeric cost
231
+ * only". `parseExitUsage` already rebuilds for exactly this reason and says so; this is the sibling that
232
+ * did not, and the asymmetry was an oversight rather than a decision.
233
+ *
234
+ * A key the runner omitted stays OMITTED rather than becoming null: the fallback shape legitimately
235
+ * carries only five of the twelve, and a null there would read as "measured zero" for a number nobody
236
+ * measured. `typeof === "number"` rather than `Number.isFinite`, deliberately, so this narrows WHICH
237
+ * KEYS survive and never which objects are admitted -- the admission gate above is unchanged.
238
+ */
239
+ function rebuildTokens(t) {
240
+ const out = {};
241
+ for (const key of TOKEN_KEYS) {
242
+ if (key === "metered") {
243
+ if (typeof t.metered === "boolean") out.metered = t.metered;
244
+ } else if (typeof t[key] === "number") {
245
+ out[key] = t[key];
246
+ }
247
+ }
248
+ return out;
249
+ }
250
+
179
251
  export function parseExitTokens(text) {
180
252
  if (typeof text !== "string") return null;
181
253
  const lines = text.split("\n");
@@ -185,7 +257,7 @@ export function parseExitTokens(text) {
185
257
  const parsed = parseTailLine(line);
186
258
  if (parsed?.event !== "exit") continue;
187
259
  const t = parsed?.tokens;
188
- if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return t;
260
+ if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return rebuildTokens(t);
189
261
  return null;
190
262
  }
191
263
  return null;
@@ -387,7 +459,12 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
387
459
  * separator from the table, so the notation a forge uses is the notation its records carry -- and a forge
388
460
  * added later inherits a label rather than a null.
389
461
  */
390
- function targetFor(kind, data) {
462
+ /*
463
+ * Exported since issue #230: a held job's panel row needs the same id-only label a run record carries, and
464
+ * the wait gate would otherwise re-derive it. Two spellings of "which issue is this" is how one of them
465
+ * starts carrying a title.
466
+ */
467
+ export function targetFor(kind, data) {
391
468
  if (kind === "local") return `local:${basename(data.folder ?? "")}`;
392
469
  if (isForgeKind(kind)) return `${data.repo}${targetSeparator(kind, data.target?.type)}${data.target?.number}`;
393
470
  return null;
package/src/service.mjs CHANGED
@@ -972,6 +972,15 @@ async function doRestart(ctx, values) {
972
972
  await ctx.sleep(2000);
973
973
  ({ active = 0 } = await queue.getJobCounts("active"));
974
974
  }
975
+ // A HELD job is neither active nor waiting, so the loop above has just reported a drained queue with
976
+ // however many jobs still parked on `run.waitFor` (issue #230). They are safe -- a hold spends
977
+ // nothing, survives a restart and reserves no slot -- but the operator is upgrading, and those jobs
978
+ // will wake against the new version. Said plainly rather than left to be discovered, which is what
979
+ // this command would otherwise be doing: reporting a drained queue it cannot see all of.
980
+ const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
981
+ if (delayed > 0) {
982
+ ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
983
+ }
975
984
  const stopped = await doStop(ctx);
976
985
  if (stopped !== 0) {
977
986
  ctx.out("restart did not happen — the queue STAYS PAUSED; fix the service, then `pi-dispatch resume`.\n");