@edgehero/pi-dispatch 1.5.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +19 -0
- package/package.json +3 -1
- package/src/config.mjs +34 -0
- package/src/doctor.mjs +128 -12
- package/src/exit-code.mjs +56 -0
- package/src/index.mjs +333 -5
- package/src/processor.mjs +25 -0
- package/src/queue.mjs +11 -1
- package/src/run-history.mjs +81 -4
- package/src/service.mjs +9 -0
- package/src/start.mjs +44 -1
- package/src/triggers-file.mjs +77 -0
- package/src/triggers.mjs +150 -4
- package/src/wait-check.mjs +172 -0
- package/src/wait-for.mjs +315 -0
- package/src/wait-state.mjs +263 -0
package/src/index.mjs
CHANGED
|
@@ -2,7 +2,10 @@ import { execFile } from "node:child_process";
|
|
|
2
2
|
import { promisify } from "node:util";
|
|
3
3
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
4
4
|
import { InfraRetry, runJob } from "./processor.mjs";
|
|
5
|
+
import { targetFor } from "./run-history.mjs";
|
|
5
6
|
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
|
|
7
|
+
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
8
|
+
import { makeWaitState } from "./wait-state.mjs";
|
|
6
9
|
|
|
7
10
|
const exec = promisify(execFile);
|
|
8
11
|
|
|
@@ -16,6 +19,28 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
|
16
19
|
// docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
|
|
17
20
|
export const SCOPE_BUSY_RECHECK_MS = 5_000;
|
|
18
21
|
|
|
22
|
+
// How long a job waits before re-asking whether a target's holder is still alive (issue #230). Reached only
|
|
23
|
+
// when the liveness probe could not answer, which is a redis or queue fault rather than a normal state, so
|
|
24
|
+
// this is a short retry rather than a cadence: the job is deciding nothing and holding nothing while it
|
|
25
|
+
// waits, and the fault it is waiting out is usually seconds long.
|
|
26
|
+
export const SUPERSEDE_RECHECK_MS = 15_000;
|
|
27
|
+
|
|
28
|
+
// The one key the check lease counts under. A single global counter rather than one per profile: what it
|
|
29
|
+
// bounds is this worker's wall-clock spent answering questions, and that is shared whatever is being asked.
|
|
30
|
+
const WAIT_CHECK_KEY = "wait-check";
|
|
31
|
+
|
|
32
|
+
// How many consecutive lease denials one job absorbs before the deployment is told its checking capacity is
|
|
33
|
+
// short. Logged ONCE per run of denials rather than per wake: an alarm that repeats every re-check is the
|
|
34
|
+
// always-on amber this project rejects elsewhere, and the operator only needs telling once per episode.
|
|
35
|
+
const THROTTLE_ALARM = 5;
|
|
36
|
+
|
|
37
|
+
// The floor under a throttled or aborted re-ask. Its own constant rather than a borrow of
|
|
38
|
+
// SUPERSEDE_RECHECK_MS, which documents an unrelated concern. Deliberately NOT 5s: that is
|
|
39
|
+
// SCOPE_BUSY_RECHECK_MS, and INT-WAIT-PROFILES-CONTRACT rests on wait deferrals being distinguishable from
|
|
40
|
+
// scope deferrals by wake instant -- nothing records WHY a job sits in the delayed set, so the instants are
|
|
41
|
+
// the only evidence there is. A test pins the two apart.
|
|
42
|
+
const THROTTLE_FLOOR_MS = 11_000;
|
|
43
|
+
|
|
19
44
|
/**
|
|
20
45
|
* Build the BullMQ processor.
|
|
21
46
|
*
|
|
@@ -35,7 +60,7 @@ export const SCOPE_BUSY_RECHECK_MS = 5_000;
|
|
|
35
60
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
36
61
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
37
62
|
*/
|
|
38
|
-
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
|
|
63
|
+
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
39
64
|
return async function processor(job, token, signal) {
|
|
40
65
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
41
66
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -53,9 +78,295 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
53
78
|
throw new DelayedError();
|
|
54
79
|
}
|
|
55
80
|
|
|
81
|
+
// The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
|
|
82
|
+
// not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
|
|
83
|
+
// because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
|
|
84
|
+
// does. Strictly above the `try` for the two reasons the gates below it document.
|
|
85
|
+
//
|
|
86
|
+
// The order WITHIN the gate is determinate-refusals-then-holds, which is CONST-BUDGET-BEFORE-TOKENS'
|
|
87
|
+
// shape applied to time rather than to money: a condition this deployment can never answer must be
|
|
88
|
+
// refused now, not after a day of waiting.
|
|
89
|
+
//
|
|
90
|
+
// On throwing above the `try`: an exception here escapes into BullMQ's normal failed-attempt handling,
|
|
91
|
+
// which is WANTED for `moveToDelayed` (the scope gate below gives the argument: a transient rejection
|
|
92
|
+
// must stay a transient failure rather than becoming a permanent one) and unwanted everywhere else. So
|
|
93
|
+
// the state and comment seams fail open by construction, and `recordRun` is relied on not to throw --
|
|
94
|
+
// its writer swallows fs errors by contract, which is the same reliance the settings-overlay refusal
|
|
95
|
+
// below already makes.
|
|
96
|
+
if (waitArmed(job.data)) {
|
|
97
|
+
// The supersede identity: the queue's semantic key PLUS the trigger that produced this job.
|
|
98
|
+
// The semantic key alone is `repo<sep>number:flow`, which two DIFFERENT triggers on one target and
|
|
99
|
+
// flow legitimately share -- a label rule that waits a day and a comment rule that waits a minute
|
|
100
|
+
// would coalesce, and the second would be refused with a message claiming they wait on "the same
|
|
101
|
+
// conditions" when they do not. Adding the raw trigger index makes the key mean one intent.
|
|
102
|
+
const matchedIndex = job.data?.trigger?.matched?.index;
|
|
103
|
+
const dedupId = job.deduplicationId ? `${job.deduplicationId}#${Number.isInteger(matchedIndex) ? matchedIndex : "?"}` : null;
|
|
104
|
+
const refuseWait = async (reason, logEvent, fields, sentence) => {
|
|
105
|
+
// The INJECTED clock, like both gates above: a record whose timestamps ignore the test clock
|
|
106
|
+
// is a record no test of this gate can assert about.
|
|
107
|
+
const at = new Date(now()).toISOString();
|
|
108
|
+
await waitState.release(job.id, { dedupId });
|
|
109
|
+
deps?.log?.(logEvent, { jobId: job.id, ...fields });
|
|
110
|
+
// The comment names the FIELD and the operator's own words for the condition, never a
|
|
111
|
+
// resolver path or a vault topology -- `secret-profile-unknown` sets that rule.
|
|
112
|
+
if (sentence && deps?.comment) await Promise.resolve(deps.comment(job.data, sentence)).catch(() => {});
|
|
113
|
+
const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
114
|
+
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString() });
|
|
115
|
+
return result;
|
|
116
|
+
};
|
|
117
|
+
|
|
118
|
+
// EVERY condition must be one this worker understands, checked before anything else. The loader
|
|
119
|
+
// refuses an unknown condition, but the loader is a DIFFERENT PROCESS: `job.data.waitFor` arrives
|
|
120
|
+
// over Redis from the receiver, and this whole feature exists because receiver-worker version
|
|
121
|
+
// skew is real. `makeCheckWaitSkew` closes the backward direction (the file has conditions the
|
|
122
|
+
// job arrived without); this closes the forward one (a newer receiver enqueues a condition shape
|
|
123
|
+
// this worker cannot read). Without it the gate would fall through, log `wait_cleared`, and run
|
|
124
|
+
// the job -- asserting in the log that conditions cleared which it never evaluated, which is the
|
|
125
|
+
// same undetectable paid run the backward check exists to stop.
|
|
126
|
+
// A sibling that was held on this same target may already have cleared it. Checked FIRST, because
|
|
127
|
+
// it is free and determinate, and because the window it closes is one no lease can: two jobs
|
|
128
|
+
// holding through an outage that outlives their leases would each wake, find no holder, and run.
|
|
129
|
+
if (dedupId) {
|
|
130
|
+
const satisfiedBy = await waitState.satisfiedBy(dedupId);
|
|
131
|
+
if (satisfiedBy && satisfiedBy !== job.id) {
|
|
132
|
+
return await refuseWait("wait-superseded", "wait_superseded", { satisfiedBy }, "Another delivery for this target already finished waiting on the same conditions. Not run.");
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
const unreadable = unreadableConditions(job.data);
|
|
137
|
+
if (unreadable.length > 0) {
|
|
138
|
+
// Its OWN token, not `wait-skew`. Both are version skew, and the REMEDIES are opposites --
|
|
139
|
+
// upgrade the receiver there, upgrade the worker here -- so one token in a durable record
|
|
140
|
+
// would tell an operator that something is out of step and not which way to move.
|
|
141
|
+
return await refuseWait("wait-unreadable", "refused_wait_unreadable", { conditions: unreadable.length }, `Refused: this job carries ${unreadable.length} wait condition${unreadable.length === 1 ? "" : "s"} this worker cannot read, so it cannot honour them. The worker is older than the service that enqueued this job. Not run.`);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// A `profile` condition needs a checker, and with none wired NOTHING can answer it. Refused rather
|
|
145
|
+
// than ignored: a wait the deployment cannot perform must not read as a wait that passed.
|
|
146
|
+
const profiles = waitProfileNames(job.data);
|
|
147
|
+
// Declared-ness is a table lookup, so it belongs with the other free refusals rather than inside
|
|
148
|
+
// the check. Without it here, `[{after: "<tomorrow>"}, {profile: "typo"}]` holds for a day and
|
|
149
|
+
// THEN refuses -- which is the exact sentence the ordering rule above promises will not happen.
|
|
150
|
+
const undeclared = deps?.waitProfileDeclared ? profiles.find((name) => !deps.waitProfileDeclared(name)) : undefined;
|
|
151
|
+
if (undeclared !== undefined) {
|
|
152
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: undeclared }, `Waiting on \`${undeclared}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
|
|
153
|
+
}
|
|
154
|
+
if (profiles.length > 0 && !deps?.checkWait) {
|
|
155
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: profiles[0] }, `Waiting on \`${profiles[0]}\` is not something this deployment can answer. Not run.`);
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
const holdUntil = afterMs(job.data); // named apart from the pause gate's `until` above, which it would otherwise shadow
|
|
159
|
+
// An instant further out than the ceiling is refused at FIRST pickup rather than held toward:
|
|
160
|
+
// holding for a month to then refuse tells the operator nothing they could not have been told now.
|
|
161
|
+
if (holdUntil !== null && holdUntil - nowMs > afterMaxMs()) {
|
|
162
|
+
return await refuseWait("wait-after-beyond-max", "wait_after_beyond_max", { delayMs: holdUntil - nowMs }, `The \`after\` instant is further out than this deployment allows a job to wait. Not run.`);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// The pause gate's boundary guard, for its reason: a tick landing on the instant must run rather
|
|
166
|
+
// than busy-defer to a moment already past.
|
|
167
|
+
if (holdUntil !== null && holdUntil > nowMs + 1000) {
|
|
168
|
+
// `isJobLive` is what stops a vanished holder's lease becoming a tombstone that refuses this
|
|
169
|
+
// target for the rest of the hold. Optional: an unwired probe means the holder cannot be
|
|
170
|
+
// checked, which ADMITS and says so -- one duplicate run beats one dropped delivery, which is
|
|
171
|
+
// `OQ-027`'s call ("one wasted vault read beats one dropped job") on this feature's terms.
|
|
172
|
+
const claim = await waitState.claim(job.id, { dedupId, untilMs: holdUntil, isLive: deps?.isJobLive });
|
|
173
|
+
if (claim.heldBy) {
|
|
174
|
+
// Another delivery for this same target and flow is already holding. Both would clear
|
|
175
|
+
// together and both would be paid, which is the accumulation the acceptance forbids.
|
|
176
|
+
return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
|
|
177
|
+
}
|
|
178
|
+
if (claim.retry) {
|
|
179
|
+
// The holder could not be checked. Holding anyway would put two jobs on one target and pay
|
|
180
|
+
// for both; refusing would drop a delivery over a holder that may be gone. So decide
|
|
181
|
+
// nothing: re-defer briefly and ask again once the probe can answer.
|
|
182
|
+
deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
|
|
183
|
+
await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
|
|
184
|
+
throw new DelayedError();
|
|
185
|
+
}
|
|
186
|
+
if (claim.tookOverFrom) deps?.log?.("wait_lease_taken_over", { jobId: job.id, from: claim.tookOverFrom });
|
|
187
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: holdUntil });
|
|
188
|
+
deps?.log?.("wait_deferred", { jobId: job.id, until: new Date(holdUntil).toISOString(), label: waitLabel(job.data) });
|
|
189
|
+
await job.moveToDelayed(holdUntil, token);
|
|
190
|
+
throw new DelayedError();
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// TIER 2: the polled conditions. Last, because it is the only part of this gate that spawns a
|
|
194
|
+
// process -- the free refusals above it are free, and the free hold above it is free.
|
|
195
|
+
if (profiles.length > 0) {
|
|
196
|
+
const held = (await waitState.heldForMs(job.id)) ?? 0;
|
|
197
|
+
const counted = await waitState.counters(job.id);
|
|
198
|
+
|
|
199
|
+
// One check at a time, process-wide, and never the worker's last free slot. This is the bound
|
|
200
|
+
// that keeps a wait from starving the paid work it is waiting for: slots x timeout is the most
|
|
201
|
+
// wall-clock a worker can spend answering questions instead of running jobs. Computed against
|
|
202
|
+
// the LIVE concurrency rather than the boot value, because the overlay can lower it.
|
|
203
|
+
const slots = Math.min(checkSlotCount(), Math.max(1, concurrencyNow() - 1));
|
|
204
|
+
if (!checkSlots.tryAcquire(WAIT_CHECK_KEY, slots)) {
|
|
205
|
+
// Denials are counted, and a run of them is the ONE symptom the capacity bound has. The
|
|
206
|
+
// lease deliberately caps how much wall-clock this worker spends checking; being at that
|
|
207
|
+
// cap constantly means demand exceeds it, which the issue's own economics say arrives
|
|
208
|
+
// silently -- paid jobs starve behind checks that spend nothing and nothing says why.
|
|
209
|
+
const denials = await waitState.noteThrottle(job.id, { denied: true });
|
|
210
|
+
if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots, hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
|
|
211
|
+
// A starved job still needs a CLOCK and a CEILING, or the lease turns into the very
|
|
212
|
+
// starvation it exists to bound: without this the hold is stamped only on a wake that won
|
|
213
|
+
// the lease, so a job that never wins one has no `since`, never reaches the maximum, and
|
|
214
|
+
// re-wakes forever with no record and no bound. There is no deciding check to run first
|
|
215
|
+
// here -- that is the whole condition -- so the bound applies directly.
|
|
216
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
|
|
217
|
+
if (held >= maxWaitMs()) {
|
|
218
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait-unchecked", denials, heldForMs: held }, `Gave up waiting: this deployment could not run the check often enough to answer within the maximum wait. Not run.`);
|
|
219
|
+
}
|
|
220
|
+
// Denied. Re-ask at a fraction of the cadence rather than the full backoff (which would
|
|
221
|
+
// turn one lost coin-flip into a fifteen-minute penalty) or a flat few seconds (which at
|
|
222
|
+
// scale is a herd). Jittered, so a fleet of denied jobs does not return together.
|
|
223
|
+
const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
|
|
224
|
+
const delay = wait + Math.floor(wait * 0.1 * random());
|
|
225
|
+
deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots });
|
|
226
|
+
await job.moveToDelayed(nowMs + delay, token);
|
|
227
|
+
throw new DelayedError();
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// Declared outside the try below because the branches AFTER it read both.
|
|
231
|
+
let verdict = null;
|
|
232
|
+
let checked = null;
|
|
233
|
+
// THE LEASE IS HELD FROM THE `tryAcquire` ABOVE, so every exit from here down must release it.
|
|
234
|
+
// The try opens here and not at the check loop, which is where it used to open: the supersede
|
|
235
|
+
// claim sits between the two, and BOTH of its exits leave -- one returns `wait-superseded`,
|
|
236
|
+
// the other re-defers and throws -- so a claim that refused or could not be verified walked
|
|
237
|
+
// out holding the slot. At the shipped default of one slot that wedged every wait check on
|
|
238
|
+
// the worker until it restarted, and the symptom was silent in the worst way: held jobs kept
|
|
239
|
+
// throttling and eventually recorded `wait-expired` with `max-wait-unchecked`, which blames
|
|
240
|
+
// the deployment's capacity for a slot this gate leaked.
|
|
241
|
+
try {
|
|
242
|
+
// Claimed BEFORE the check, not after: a second delivery for an already-held target is a free
|
|
243
|
+
// determinate refusal, and paying for a subprocess first inverts the free-before-costly rule
|
|
244
|
+
// this gate's own header invokes. Tier 1 already claims in this order.
|
|
245
|
+
const claim = await waitState.claim(job.id, { dedupId, untilMs: nowMs + waitBackoffMs(intervalMs(), held), isLive: deps?.isJobLive });
|
|
246
|
+
if (claim.heldBy) {
|
|
247
|
+
return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
|
|
248
|
+
}
|
|
249
|
+
if (claim.retry) {
|
|
250
|
+
deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
|
|
251
|
+
await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
|
|
252
|
+
throw new DelayedError();
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
await waitState.noteThrottle(job.id, { denied: false }); // granted: the run of denials ends here
|
|
256
|
+
// Sequential, in the operator's writing order: the resolver's reason applies unchanged --
|
|
257
|
+
// naming the first condition that did not clear is what makes a held row readable, and a
|
|
258
|
+
// parallel fan-out would blame whichever lost the race on any given wake.
|
|
259
|
+
for (const profile of profiles) {
|
|
260
|
+
checked = profile;
|
|
261
|
+
verdict = await deps.checkWait(profile, targetFor(job.data?.kind, job.data), { signal });
|
|
262
|
+
if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
|
|
263
|
+
}
|
|
264
|
+
} finally {
|
|
265
|
+
checkSlots.release(WAIT_CHECK_KEY);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
if (verdict?.unusableTarget) {
|
|
269
|
+
// Determinate and unfixable by waiting: the job's own target is a shape no check can be
|
|
270
|
+
// handed. It belongs with the refusals, not the holds -- holding would spend the fault
|
|
271
|
+
// budget and then blame the operator's script for a value it was never given.
|
|
272
|
+
return await refuseWait("wait-unreadable", "refused_wait_unreadable", { profile: checked }, `Refused: this job's target cannot be handed to a wait check, so \`${checked}\` can never be asked. Not run.`);
|
|
273
|
+
}
|
|
274
|
+
if (verdict?.profileUnknown) {
|
|
275
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: verdict.profileUnknown }, `Waiting on \`${verdict.profileUnknown}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
|
|
276
|
+
}
|
|
277
|
+
if (verdict?.verdict === "refuse") {
|
|
278
|
+
// Exit 2: the check says this will NEVER clear. Terminal by the protocol's own words, and
|
|
279
|
+
// distinct from every "not yet" above it.
|
|
280
|
+
return await refuseWait("wait-refused", "wait_refused", { profile: checked, heldForMs: held }, `The check \`${checked}\` reports this will never clear. Not run.`);
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
if (verdict?.aborted) {
|
|
284
|
+
// The worker is stopping or this job was cancelled. Nothing was learned and nothing is
|
|
285
|
+
// owed: re-defer at once rather than at the full backoff, and count neither a check nor a
|
|
286
|
+
// fault, or a rolling deploy would spend a job's whole budget on its own restarts and then
|
|
287
|
+
// blame the operator's script for it.
|
|
288
|
+
deps?.log?.("wait_check_aborted", { jobId: job.id, profile: checked });
|
|
289
|
+
await job.moveToDelayed(nowMs + THROTTLE_FLOOR_MS, token);
|
|
290
|
+
throw new DelayedError();
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
if (verdict?.verdict === "hold") {
|
|
294
|
+
const fault = verdict.fault === true;
|
|
295
|
+
await waitState.noteCheck(job.id, { fault });
|
|
296
|
+
const faults = fault ? counted.faults + 1 : 0;
|
|
297
|
+
|
|
298
|
+
// A check that never answers is a broken script, not a slow condition, and OQ-030 is why
|
|
299
|
+
// this bound exists: most CLIs exit 1 for everything, so without it a typo would hold for
|
|
300
|
+
// the whole maximum wait and then blame the CONDITION rather than the check.
|
|
301
|
+
if (faults >= maxFaults()) {
|
|
302
|
+
return await refuseWait("wait-unanswerable", "wait_unanswerable", { profile: checked, faults }, `The check \`${checked}\` could not answer ${faults} times in a row. Not run.`);
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// BOTH terminal bounds are tested AFTER the check and never before it, so a condition that
|
|
306
|
+
// cleared on the deciding wake runs instead of being recorded as never having cleared.
|
|
307
|
+
// Without that ordering the backoff's own quantisation makes "cleared at t+1s, declared
|
|
308
|
+
// never-cleared at t+900s" a structural lie in the durable record and in a public comment.
|
|
309
|
+
//
|
|
310
|
+
// The count bound reads `checks + 1` because this wake's check has just run: the job gets
|
|
311
|
+
// exactly `maxChecks` checks, the last of which is the deciding one. Putting it before the
|
|
312
|
+
// check instead -- so the act of testing the bound could not exceed it -- was the obvious
|
|
313
|
+
// spelling, and it silently made this whole guarantee untrue at every shipped default,
|
|
314
|
+
// because the count bound is the one that fires first there.
|
|
315
|
+
if (counted.checks + 1 >= maxChecks()) {
|
|
316
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-checks", checks: counted.checks + 1, profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\` after ${counted.checks + 1} checks. Not run.`);
|
|
317
|
+
}
|
|
318
|
+
if (held >= maxWaitMs()) {
|
|
319
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait", profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\`. Not run.`);
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
// Clamped to what is LEFT of the budget, never just the cadence. Without this an hourly
|
|
323
|
+
// interval under a fifteen-minute maximum holds for the full hour -- 400% of the bound the
|
|
324
|
+
// operator configured -- because the ceiling is only tested when a wake arrives, and the
|
|
325
|
+
// cadence decides when that is. The two knobs are independent `positiveInt`s and nothing
|
|
326
|
+
// cross-validates them, so the clamp is what makes the smaller one actually bind.
|
|
327
|
+
const base = waitBackoffMs(intervalMs(), held);
|
|
328
|
+
const jittered = base + Math.floor(base * 0.1 * random());
|
|
329
|
+
const remaining = Math.max(0, maxWaitMs() - held);
|
|
330
|
+
const delay = Math.max(1000, Math.min(jittered, remaining));
|
|
331
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + delay });
|
|
332
|
+
deps?.log?.("wait_deferred", { jobId: job.id, profile: checked, fault, heldForMs: held, delayMs: delay });
|
|
333
|
+
await job.moveToDelayed(nowMs + delay, token);
|
|
334
|
+
throw new DelayedError();
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
// FAIL CLOSED on anything that is not literally go. Everything above tests for a specific
|
|
338
|
+
// shape and falls through otherwise, and "otherwise" at this gate means STARTING A PAID
|
|
339
|
+
// CONTAINER -- so an `undefined`, a `null`, a `{}`, a mis-cased "GO" or a bare string from a
|
|
340
|
+
// checker would run the job silently, with no record field and no log line to distinguish it
|
|
341
|
+
// from a job whose check said yes. The shipped checker is total, and that is exactly the
|
|
342
|
+
// reasoning `unreadableConditions` above rejects: this is a dependency-injection seam, and a
|
|
343
|
+
// seam's guarantees are the caller's to enforce.
|
|
344
|
+
if (verdict?.verdict !== "go") {
|
|
345
|
+
deps?.log?.("wait_check_unintelligible", { jobId: job.id, profile: checked });
|
|
346
|
+
await waitState.noteCheck(job.id, { fault: true });
|
|
347
|
+
const base = waitBackoffMs(intervalMs(), held);
|
|
348
|
+
await job.moveToDelayed(nowMs + base + Math.floor(base * 0.1 * random()), token);
|
|
349
|
+
throw new DelayedError();
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
// Every profile answered go. Record the last check so the count bound sees it.
|
|
353
|
+
await waitState.noteCheck(job.id, { fault: false });
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// Only a job that actually HELD has cleared. Without the check this line fires on the first
|
|
357
|
+
// pickup of a job whose instant had already passed, and again on every scope-busy re-check
|
|
358
|
+
// afterwards -- asserting a wait ended that never began.
|
|
359
|
+
const heldForMs = await waitState.heldForMs(job.id);
|
|
360
|
+
// Say so before releasing: a sibling held on this target must find the answer, not an empty lease.
|
|
361
|
+
if (heldForMs !== null) await waitState.markSatisfied(job.id, { dedupId });
|
|
362
|
+
await waitState.release(job.id, { dedupId });
|
|
363
|
+
if (heldForMs !== null) deps?.log?.("wait_cleared", { jobId: job.id, label: waitLabel(job.data), heldForMs });
|
|
364
|
+
}
|
|
365
|
+
|
|
56
366
|
// Per-scope concurrency and the one-job-per-folder mutex (issue #242,
|
|
57
|
-
// INT-SCOPED-LIMITS-FILE-CONTRACT).
|
|
58
|
-
// re-check wakes) and
|
|
367
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT). LAST of the three gates, after the pause gate (a paused job must
|
|
368
|
+
// not burn re-check wakes) and after the wait gate (a job holding until tomorrow must not sit on a
|
|
369
|
+
// folder while it does), and STRICTLY above the `try` below, like the pause gate and for the same two
|
|
59
370
|
// reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
|
|
60
371
|
// catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
|
|
61
372
|
// handling exactly as the pause gate's does (inside the try it would become a permanent failure
|
|
@@ -123,7 +434,9 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
123
434
|
// job completed and does not retry a file that can never parse (CONST-RETRY-INFRA-ONLY). Resolved
|
|
124
435
|
// before runJob, so no budget slot is reserved and no container starts (CONST-BUDGET-BEFORE-TOKENS).
|
|
125
436
|
// recordRun leaves the durable settings-overlay-invalid trace for the admin extension.
|
|
126
|
-
// No provider/model here,
|
|
437
|
+
// No provider/model here, as in every result this function returns from ABOVE the try (the wait gate's
|
|
438
|
+
// refusals are the others): each is decided before or during the settings read, so no honest effective
|
|
439
|
+
// value exists yet -- buildRecord defaults both null.
|
|
127
440
|
const result = { outcome: "policy", reason: "settings-overlay-invalid", exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
128
441
|
recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
129
442
|
return result;
|
|
@@ -210,7 +523,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
210
523
|
};
|
|
211
524
|
}
|
|
212
525
|
|
|
213
|
-
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
|
|
526
|
+
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
|
|
214
527
|
let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
|
|
215
528
|
const processor = makeProcessor({
|
|
216
529
|
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
@@ -228,6 +541,21 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
|
|
|
228
541
|
// shape means one per daemon).
|
|
229
542
|
scopedLimits,
|
|
230
543
|
inFlight,
|
|
544
|
+
// Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
|
|
545
|
+
// redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
|
|
546
|
+
waitState,
|
|
547
|
+
afterMaxMs,
|
|
548
|
+
// Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
|
|
549
|
+
// because the overlay can lower it through `dispatch_set` and a check must never take the last free
|
|
550
|
+
// slot from a paid job. Late-bound over `worker` exactly as `applyConcurrency` is, and for the same
|
|
551
|
+
// reason: the value it needs does not exist until the Worker is constructed.
|
|
552
|
+
checkSlots,
|
|
553
|
+
checkSlotCount,
|
|
554
|
+
concurrencyNow: concurrencyNow ?? (() => worker?.concurrency ?? concurrency),
|
|
555
|
+
intervalMs,
|
|
556
|
+
maxWaitMs,
|
|
557
|
+
maxChecks,
|
|
558
|
+
maxFaults,
|
|
231
559
|
deps,
|
|
232
560
|
recordRun,
|
|
233
561
|
});
|
package/src/processor.mjs
CHANGED
|
@@ -54,6 +54,9 @@ export async function runJob(job, deps) {
|
|
|
54
54
|
// exactly as before, and the gate below only calls it for a job whose matched rule was a
|
|
55
55
|
// one-shot, so the default is never a probe running on every delivery.
|
|
56
56
|
checkOnceSpent = async () => ({ ok: true }),
|
|
57
|
+
// Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
|
|
58
|
+
// unwired seam must not refuse, and the wiring is what turns the check on.
|
|
59
|
+
checkWaitSkew = async () => ({ ok: true }),
|
|
57
60
|
// REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
|
|
58
61
|
// deployment with no egress policy does -- which is also what the real factory returns when unarmed.
|
|
59
62
|
egressPreflight = async () => ({ ok: true }),
|
|
@@ -155,6 +158,28 @@ export async function runJob(job, deps) {
|
|
|
155
158
|
}
|
|
156
159
|
}
|
|
157
160
|
|
|
161
|
+
// The wait-skew check (issue #230), second on the ladder and for the first one's reasons: the same
|
|
162
|
+
// file read, free, determinate, credential-less, and pre-spend. It answers a question no other layer
|
|
163
|
+
// can: does the AUTHORED trigger carry wait conditions this job arrived without? That happens when a
|
|
164
|
+
// service below the version floor dropped the field as an unknown key, and the resulting run is
|
|
165
|
+
// byte-identical to a correct one everywhere it is recorded -- so this refusal is the only thing
|
|
166
|
+
// standing between a stale receiver and a paid job that ran when the operator wrote "wait".
|
|
167
|
+
{
|
|
168
|
+
const skew = await checkWaitSkew(job);
|
|
169
|
+
if (skew.skewed) {
|
|
170
|
+
// Named for the operator, not the payload: how many conditions were authored, never what
|
|
171
|
+
// they say. The fix is a version, so the comment says which one.
|
|
172
|
+
// The message names BOTH causes, because the more likely one is not a version at all. In the
|
|
173
|
+
// compose topology the receiver's single-file `:ro` mount pins a dead inode, so an operator
|
|
174
|
+
// who ADDS `waitFor` to an existing rule gets this refusal on every delivery from a service
|
|
175
|
+
// that is perfectly up to date and merely holding an older copy of the file. Naming only the
|
|
176
|
+
// version would send them looking for an upgrade they do not need.
|
|
177
|
+
await comment(job, `Refused: this trigger declares ${skew.conditions} wait condition${skew.conditions === 1 ? "" : "s"}, but the job reached the worker without them, which means it would have run immediately. Either a service in this deployment is below the version that carries the field, or one is still running against an older copy of the triggers file and needs restarting. Not run.`);
|
|
178
|
+
log("refused_wait_skew", { triggerIndex: job.trigger?.matched?.index ?? null, conditions: skew.conditions });
|
|
179
|
+
return { outcome: "policy", reason: "wait-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
158
183
|
// The job image must exist on THIS host before anything else happens. Free, determinate and
|
|
159
184
|
// credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
|
|
160
185
|
// image refuses without minting a credential it will not use, cloning a repo it will not read, or
|
package/src/queue.mjs
CHANGED
|
@@ -134,7 +134,7 @@ export async function enqueueGitLabJob(queue, fields) {
|
|
|
134
134
|
* window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
|
|
135
135
|
* has always been.
|
|
136
136
|
*/
|
|
137
|
-
export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, replica, replicas }) {
|
|
137
|
+
export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
|
|
138
138
|
const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
|
|
139
139
|
// `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
|
|
140
140
|
// come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
|
|
@@ -179,6 +179,16 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
|
|
|
179
179
|
// is copied verbatim into /job/event.json, which an agent reads.
|
|
180
180
|
...(secrets !== undefined && { secrets }),
|
|
181
181
|
...(secretsProfile !== undefined && { secretsProfile }),
|
|
182
|
+
// Issue #230. The conditions the worker holds this job on, carried so the PICKUP gate can read them:
|
|
183
|
+
// that gate runs above the per-job settings read and never re-parses the triggers file for its terms.
|
|
184
|
+
// At JOB level, and here that placement is a correctness requirement rather than a convention --
|
|
185
|
+
// `trigger` is copied VERBATIM into /job/event.json (prepare-local.mjs), so a `trigger.waitFor` would
|
|
186
|
+
// hand the agent the operator's own gate. Conditional like every field above, so an unflagged job's
|
|
187
|
+
// data keeps exactly the keys it has today. The dedup options below are deliberately NOT widened for
|
|
188
|
+
// a waiting job: that key carries no trigger identity and outlives the job it was set for, so a
|
|
189
|
+
// longer window would suppress an unflagged sibling's deliveries and go on suppressing them after
|
|
190
|
+
// this job finished. Coalescing a held target is the worker's `wait:` keyspace's job instead.
|
|
191
|
+
...(waitFor !== undefined && { waitFor }),
|
|
182
192
|
// Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
|
|
183
193
|
// exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
|
|
184
194
|
// both are integers, so the run record they land in stays PII-free by construction.
|
package/src/run-history.mjs
CHANGED
|
@@ -114,7 +114,34 @@ export function parseExitTurns(text) {
|
|
|
114
114
|
* Read-only telemetry, exactly like `parseExitTurns`: NEVER throws and MUST NOT feed exit-code or retry
|
|
115
115
|
* classification (INT-RUNNER-EXIT-CODE-PROTOCOL). A malformed or non-object `tokens` (or one missing a
|
|
116
116
|
* numeric `total`) is `null`, never a partial that could poison the daily token counter.
|
|
117
|
+
*
|
|
118
|
+
* The admitted object is REBUILT through `rebuildTokens` rather than returned as it arrived. See that
|
|
119
|
+
* function for why: the pass-through it replaces is what made this comment's "integer token counts and
|
|
120
|
+
* numeric cost only" false one level below `buildRecord`'s literal.
|
|
121
|
+
*/
|
|
122
|
+
/**
|
|
123
|
+
* The CLOSED `session.reason` enum, verbatim from `INT-RUN-HISTORY-FILE-CONTRACT`. Three producers write
|
|
124
|
+
* this field (resolve, runner, promote) and the contract has always called the set closed; until this
|
|
125
|
+
* list existed, nothing enforced it and the runner's half was an unchecked string. Kept here rather than
|
|
126
|
+
* beside the store because this module is where the container's copy is admitted, and an enum that lives
|
|
127
|
+
* anywhere but the admission point is a comment, not a check.
|
|
117
128
|
*/
|
|
129
|
+
const SESSION_REASONS = new Set([
|
|
130
|
+
"resumed",
|
|
131
|
+
"absent",
|
|
132
|
+
"expired",
|
|
133
|
+
"conversation-too-old",
|
|
134
|
+
"resume-chain-too-long",
|
|
135
|
+
"context-too-full",
|
|
136
|
+
"too-large",
|
|
137
|
+
"unparseable",
|
|
138
|
+
"not-a-regular-file",
|
|
139
|
+
"pi-version-changed",
|
|
140
|
+
"locked",
|
|
141
|
+
"promote-failed",
|
|
142
|
+
"disabled",
|
|
143
|
+
]);
|
|
144
|
+
|
|
118
145
|
/**
|
|
119
146
|
* The runner's `session` object off the exit line: `{ resumed: <bool>, reason: "<enum>" }` or null when
|
|
120
147
|
* the container died before emitting one (REQ-RESUMABLE-SESSION).
|
|
@@ -125,7 +152,8 @@ export function parseExitTurns(text) {
|
|
|
125
152
|
* without both numbers it is indistinguishable from an ordinary cold start. A feature that fails open
|
|
126
153
|
* must still say that it did.
|
|
127
154
|
*
|
|
128
|
-
* PII-free by construction: a boolean and a fixed enum. No key, no branch name, no path
|
|
155
|
+
* PII-free by construction: a boolean and a fixed enum. No key, no branch name, no path -- and since the
|
|
156
|
+
* `SESSION_REASONS` check below, that sentence is enforced rather than merely intended.
|
|
129
157
|
*/
|
|
130
158
|
export function parseExitSession(text) {
|
|
131
159
|
if (typeof text !== "string") return null;
|
|
@@ -137,7 +165,14 @@ export function parseExitSession(text) {
|
|
|
137
165
|
if (parsed?.event !== "exit") continue;
|
|
138
166
|
const sess = parsed?.session;
|
|
139
167
|
if (sess && typeof sess === "object" && !Array.isArray(sess) && typeof sess.resumed === "boolean") {
|
|
140
|
-
|
|
168
|
+
// The reason is checked against the CLOSED enum, not merely against `typeof === "string"`, which
|
|
169
|
+
// is what it used to be. The container owns this value, so an unchecked string put an
|
|
170
|
+
// attacker-shapeable one into a record whose PII-free property rests on holding none -- while
|
|
171
|
+
// the comment above claimed "a boolean and a fixed enum". An unrecognised token reads as `null`
|
|
172
|
+
// (the runner said nothing this contract can represent) rather than being carried through: the
|
|
173
|
+
// enum is documented CLOSED in INT-RUN-HISTORY-FILE-CONTRACT, so a value outside it was already
|
|
174
|
+
// contract-violating and every consumer already handles null.
|
|
175
|
+
return { resumed: sess.resumed, reason: SESSION_REASONS.has(sess.reason) ? sess.reason : null };
|
|
141
176
|
}
|
|
142
177
|
return null;
|
|
143
178
|
}
|
|
@@ -176,6 +211,43 @@ export function parseExitContext(text) {
|
|
|
176
211
|
return null;
|
|
177
212
|
}
|
|
178
213
|
|
|
214
|
+
/**
|
|
215
|
+
* The keys the runner actually emits, in its own emission order: the metered snapshot
|
|
216
|
+
* (`image/runner/src/usage-meter.mjs` -> `snapshot`) plus the token-budget fallback
|
|
217
|
+
* (`image/runner/run-job.mjs` -> `pickTotals`, which sends the first four and `metered: false`). Order
|
|
218
|
+
* matters because it is what makes a conformant runner's object round-trip byte-identically through the
|
|
219
|
+
* rebuild below, so the record's bytes do not move for anyone running a real image.
|
|
220
|
+
*/
|
|
221
|
+
const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced"];
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* Rebuild the billed totals from a closed key list rather than passing the container's object through.
|
|
225
|
+
*
|
|
226
|
+
* This function exists because the pass-through was a hole. `parseExitTokens` used to `return t`
|
|
227
|
+
* verbatim whenever `t.total` was a number, so any key the container invented -- a path, a branch name,
|
|
228
|
+
* a string it read out of the workspace -- rode into the durable record, and from there into anything
|
|
229
|
+
* that mirrors it. The record's PII-free-by-construction property held at `buildRecord`'s own level and
|
|
230
|
+
* NOT one level down, while this module's own comment claimed "integer token counts and numeric cost
|
|
231
|
+
* only". `parseExitUsage` already rebuilds for exactly this reason and says so; this is the sibling that
|
|
232
|
+
* did not, and the asymmetry was an oversight rather than a decision.
|
|
233
|
+
*
|
|
234
|
+
* A key the runner omitted stays OMITTED rather than becoming null: the fallback shape legitimately
|
|
235
|
+
* carries only five of the twelve, and a null there would read as "measured zero" for a number nobody
|
|
236
|
+
* measured. `typeof === "number"` rather than `Number.isFinite`, deliberately, so this narrows WHICH
|
|
237
|
+
* KEYS survive and never which objects are admitted -- the admission gate above is unchanged.
|
|
238
|
+
*/
|
|
239
|
+
function rebuildTokens(t) {
|
|
240
|
+
const out = {};
|
|
241
|
+
for (const key of TOKEN_KEYS) {
|
|
242
|
+
if (key === "metered") {
|
|
243
|
+
if (typeof t.metered === "boolean") out.metered = t.metered;
|
|
244
|
+
} else if (typeof t[key] === "number") {
|
|
245
|
+
out[key] = t[key];
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
return out;
|
|
249
|
+
}
|
|
250
|
+
|
|
179
251
|
export function parseExitTokens(text) {
|
|
180
252
|
if (typeof text !== "string") return null;
|
|
181
253
|
const lines = text.split("\n");
|
|
@@ -185,7 +257,7 @@ export function parseExitTokens(text) {
|
|
|
185
257
|
const parsed = parseTailLine(line);
|
|
186
258
|
if (parsed?.event !== "exit") continue;
|
|
187
259
|
const t = parsed?.tokens;
|
|
188
|
-
if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return t;
|
|
260
|
+
if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return rebuildTokens(t);
|
|
189
261
|
return null;
|
|
190
262
|
}
|
|
191
263
|
return null;
|
|
@@ -387,7 +459,12 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
|
|
|
387
459
|
* separator from the table, so the notation a forge uses is the notation its records carry -- and a forge
|
|
388
460
|
* added later inherits a label rather than a null.
|
|
389
461
|
*/
|
|
390
|
-
|
|
462
|
+
/*
|
|
463
|
+
* Exported since issue #230: a held job's panel row needs the same id-only label a run record carries, and
|
|
464
|
+
* the wait gate would otherwise re-derive it. Two spellings of "which issue is this" is how one of them
|
|
465
|
+
* starts carrying a title.
|
|
466
|
+
*/
|
|
467
|
+
export function targetFor(kind, data) {
|
|
391
468
|
if (kind === "local") return `local:${basename(data.folder ?? "")}`;
|
|
392
469
|
if (isForgeKind(kind)) return `${data.repo}${targetSeparator(kind, data.target?.type)}${data.target?.number}`;
|
|
393
470
|
return null;
|
package/src/service.mjs
CHANGED
|
@@ -972,6 +972,15 @@ async function doRestart(ctx, values) {
|
|
|
972
972
|
await ctx.sleep(2000);
|
|
973
973
|
({ active = 0 } = await queue.getJobCounts("active"));
|
|
974
974
|
}
|
|
975
|
+
// A HELD job is neither active nor waiting, so the loop above has just reported a drained queue with
|
|
976
|
+
// however many jobs still parked on `run.waitFor` (issue #230). They are safe -- a hold spends
|
|
977
|
+
// nothing, survives a restart and reserves no slot -- but the operator is upgrading, and those jobs
|
|
978
|
+
// will wake against the new version. Said plainly rather than left to be discovered, which is what
|
|
979
|
+
// this command would otherwise be doing: reporting a drained queue it cannot see all of.
|
|
980
|
+
const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
|
|
981
|
+
if (delayed > 0) {
|
|
982
|
+
ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
|
|
983
|
+
}
|
|
975
984
|
const stopped = await doStop(ctx);
|
|
976
985
|
if (stopped !== 0) {
|
|
977
986
|
ctx.out("restart did not happen — the queue STAYS PAUSED; fix the service, then `pi-dispatch resume`.\n");
|