@edgehero/pi-dispatch 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +21 -0
- package/package.json +4 -1
- package/src/config.mjs +35 -0
- package/src/doctor.mjs +213 -12
- package/src/exit-code.mjs +56 -0
- package/src/index.mjs +403 -15
- package/src/init.mjs +6 -0
- package/src/processor.mjs +80 -3
- package/src/queue.mjs +11 -1
- package/src/run-history.mjs +6 -1
- package/src/scoped-limits.mjs +277 -0
- package/src/service.mjs +9 -0
- package/src/start.mjs +95 -1
- package/src/triggers-file.mjs +77 -0
- package/src/triggers.mjs +150 -4
- package/src/wait-check.mjs +172 -0
- package/src/wait-for.mjs +315 -0
- package/src/wait-state.mjs +263 -0
package/src/index.mjs
CHANGED
|
@@ -2,11 +2,44 @@ import { execFile } from "node:child_process";
|
|
|
2
2
|
import { promisify } from "node:util";
|
|
3
3
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
4
4
|
import { InfraRetry, runJob } from "./processor.mjs";
|
|
5
|
+
import { targetFor } from "./run-history.mjs";
|
|
6
|
+
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
|
|
7
|
+
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
8
|
+
import { makeWaitState } from "./wait-state.mjs";
|
|
5
9
|
|
|
6
10
|
const exec = promisify(execFile);
|
|
7
11
|
|
|
8
12
|
export const QUEUE = "pi-jobs";
|
|
9
13
|
export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
14
|
+
// The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
|
|
15
|
+
// JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
|
|
16
|
+
// (<=360 wakes across a 30-minute hold, each ~1ms of synchronous predicate briefly occupying a slot)
|
|
17
|
+
// while a same-folder CHAINED job -- enqueued by its parent before the parent's finally releases the
|
|
18
|
+
// folder -- pays exactly one re-check, not fifteen seconds of dead air. No jitter: one worker per
|
|
19
|
+
// docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
|
|
20
|
+
export const SCOPE_BUSY_RECHECK_MS = 5_000;
|
|
21
|
+
|
|
22
|
+
// How long a job waits before re-asking whether a target's holder is still alive (issue #230). Reached only
|
|
23
|
+
// when the liveness probe could not answer, which is a redis or queue fault rather than a normal state, so
|
|
24
|
+
// this is a short retry rather than a cadence: the job is deciding nothing and holding nothing while it
|
|
25
|
+
// waits, and the fault it is waiting out is usually seconds long.
|
|
26
|
+
export const SUPERSEDE_RECHECK_MS = 15_000;
|
|
27
|
+
|
|
28
|
+
// The one key the check lease counts under. A single global counter rather than one per profile: what it
|
|
29
|
+
// bounds is this worker's wall-clock spent answering questions, and that is shared whatever is being asked.
|
|
30
|
+
const WAIT_CHECK_KEY = "wait-check";
|
|
31
|
+
|
|
32
|
+
// How many consecutive lease denials one job absorbs before the deployment is told its checking capacity is
|
|
33
|
+
// short. Logged ONCE per run of denials rather than per wake: an alarm that repeats every re-check is the
|
|
34
|
+
// always-on amber this project rejects elsewhere, and the operator only needs telling once per episode.
|
|
35
|
+
const THROTTLE_ALARM = 5;
|
|
36
|
+
|
|
37
|
+
// The floor under a throttled or aborted re-ask. Its own constant rather than a borrow of
|
|
38
|
+
// SUPERSEDE_RECHECK_MS, which documents an unrelated concern. Deliberately NOT 5s: that is
|
|
39
|
+
// SCOPE_BUSY_RECHECK_MS, and INT-WAIT-PROFILES-CONTRACT rests on wait deferrals being distinguishable from
|
|
40
|
+
// scope deferrals by wake instant -- nothing records WHY a job sits in the delayed set, so the instants are
|
|
41
|
+
// the only evidence there is. A test pins the two apart.
|
|
42
|
+
const THROTTLE_FLOOR_MS = 11_000;
|
|
10
43
|
|
|
11
44
|
/**
|
|
12
45
|
* Build the BullMQ processor.
|
|
@@ -27,7 +60,7 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
|
27
60
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
28
61
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
29
62
|
*/
|
|
30
|
-
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
|
|
63
|
+
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
31
64
|
return async function processor(job, token, signal) {
|
|
32
65
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
33
66
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -45,19 +78,345 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
45
78
|
throw new DelayedError();
|
|
46
79
|
}
|
|
47
80
|
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
81
|
+
// The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
|
|
82
|
+
// not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
|
|
83
|
+
// because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
|
|
84
|
+
// does. Strictly above the `try` for the two reasons the gates below it document.
|
|
85
|
+
//
|
|
86
|
+
// The order WITHIN the gate is determinate-refusals-then-holds, which is CONST-BUDGET-BEFORE-TOKENS'
|
|
87
|
+
// shape applied to time rather than to money: a condition this deployment can never answer must be
|
|
88
|
+
// refused now, not after a day of waiting.
|
|
89
|
+
//
|
|
90
|
+
// On throwing above the `try`: an exception here escapes into BullMQ's normal failed-attempt handling,
|
|
91
|
+
// which is WANTED for `moveToDelayed` (the scope gate below gives the argument: a transient rejection
|
|
92
|
+
// must stay a transient failure rather than becoming a permanent one) and unwanted everywhere else. So
|
|
93
|
+
// the state and comment seams fail open by construction, and `recordRun` is relied on not to throw --
|
|
94
|
+
// its writer swallows fs errors by contract, which is the same reliance the settings-overlay refusal
|
|
95
|
+
// below already makes.
|
|
96
|
+
if (waitArmed(job.data)) {
|
|
97
|
+
// The supersede identity: the queue's semantic key PLUS the trigger that produced this job.
|
|
98
|
+
// The semantic key alone is `repo<sep>number:flow`, which two DIFFERENT triggers on one target and
|
|
99
|
+
// flow legitimately share -- a label rule that waits a day and a comment rule that waits a minute
|
|
100
|
+
// would coalesce, and the second would be refused with a message claiming they wait on "the same
|
|
101
|
+
// conditions" when they do not. Adding the raw trigger index makes the key mean one intent.
|
|
102
|
+
const matchedIndex = job.data?.trigger?.matched?.index;
|
|
103
|
+
const dedupId = job.deduplicationId ? `${job.deduplicationId}#${Number.isInteger(matchedIndex) ? matchedIndex : "?"}` : null;
|
|
104
|
+
const refuseWait = async (reason, logEvent, fields, sentence) => {
|
|
105
|
+
// The INJECTED clock, like both gates above: a record whose timestamps ignore the test clock
|
|
106
|
+
// is a record no test of this gate can assert about.
|
|
107
|
+
const at = new Date(now()).toISOString();
|
|
108
|
+
await waitState.release(job.id, { dedupId });
|
|
109
|
+
deps?.log?.(logEvent, { jobId: job.id, ...fields });
|
|
110
|
+
// The comment names the FIELD and the operator's own words for the condition, never a
|
|
111
|
+
// resolver path or a vault topology -- `secret-profile-unknown` sets that rule.
|
|
112
|
+
if (sentence && deps?.comment) await Promise.resolve(deps.comment(job.data, sentence)).catch(() => {});
|
|
113
|
+
const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
114
|
+
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString() });
|
|
115
|
+
return result;
|
|
116
|
+
};
|
|
117
|
+
|
|
118
|
+
// EVERY condition must be one this worker understands, checked before anything else. The loader
|
|
119
|
+
// refuses an unknown condition, but the loader is a DIFFERENT PROCESS: `job.data.waitFor` arrives
|
|
120
|
+
// over Redis from the receiver, and this whole feature exists because receiver-worker version
|
|
121
|
+
// skew is real. `makeCheckWaitSkew` closes the backward direction (the file has conditions the
|
|
122
|
+
// job arrived without); this closes the forward one (a newer receiver enqueues a condition shape
|
|
123
|
+
// this worker cannot read). Without it the gate would fall through, log `wait_cleared`, and run
|
|
124
|
+
// the job -- asserting in the log that conditions cleared which it never evaluated, which is the
|
|
125
|
+
// same undetectable paid run the backward check exists to stop.
|
|
126
|
+
// A sibling that was held on this same target may already have cleared it. Checked FIRST, because
|
|
127
|
+
// it is free and determinate, and because the window it closes is one no lease can: two jobs
|
|
128
|
+
// holding through an outage that outlives their leases would each wake, find no holder, and run.
|
|
129
|
+
if (dedupId) {
|
|
130
|
+
const satisfiedBy = await waitState.satisfiedBy(dedupId);
|
|
131
|
+
if (satisfiedBy && satisfiedBy !== job.id) {
|
|
132
|
+
return await refuseWait("wait-superseded", "wait_superseded", { satisfiedBy }, "Another delivery for this target already finished waiting on the same conditions. Not run.");
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
const unreadable = unreadableConditions(job.data);
|
|
137
|
+
if (unreadable.length > 0) {
|
|
138
|
+
// Its OWN token, not `wait-skew`. Both are version skew, and the REMEDIES are opposites --
|
|
139
|
+
// upgrade the receiver there, upgrade the worker here -- so one token in a durable record
|
|
140
|
+
// would tell an operator that something is out of step and not which way to move.
|
|
141
|
+
return await refuseWait("wait-unreadable", "refused_wait_unreadable", { conditions: unreadable.length }, `Refused: this job carries ${unreadable.length} wait condition${unreadable.length === 1 ? "" : "s"} this worker cannot read, so it cannot honour them. The worker is older than the service that enqueued this job. Not run.`);
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// A `profile` condition needs a checker, and with none wired NOTHING can answer it. Refused rather
|
|
145
|
+
// than ignored: a wait the deployment cannot perform must not read as a wait that passed.
|
|
146
|
+
const profiles = waitProfileNames(job.data);
|
|
147
|
+
// Declared-ness is a table lookup, so it belongs with the other free refusals rather than inside
|
|
148
|
+
// the check. Without it here, `[{after: "<tomorrow>"}, {profile: "typo"}]` holds for a day and
|
|
149
|
+
// THEN refuses -- which is the exact sentence the ordering rule above promises will not happen.
|
|
150
|
+
const undeclared = deps?.waitProfileDeclared ? profiles.find((name) => !deps.waitProfileDeclared(name)) : undefined;
|
|
151
|
+
if (undeclared !== undefined) {
|
|
152
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: undeclared }, `Waiting on \`${undeclared}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
|
|
153
|
+
}
|
|
154
|
+
if (profiles.length > 0 && !deps?.checkWait) {
|
|
155
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: profiles[0] }, `Waiting on \`${profiles[0]}\` is not something this deployment can answer. Not run.`);
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
const holdUntil = afterMs(job.data); // named apart from the pause gate's `until` above, which it would otherwise shadow
|
|
159
|
+
// An instant further out than the ceiling is refused at FIRST pickup rather than held toward:
|
|
160
|
+
// holding for a month to then refuse tells the operator nothing they could not have been told now.
|
|
161
|
+
if (holdUntil !== null && holdUntil - nowMs > afterMaxMs()) {
|
|
162
|
+
return await refuseWait("wait-after-beyond-max", "wait_after_beyond_max", { delayMs: holdUntil - nowMs }, `The \`after\` instant is further out than this deployment allows a job to wait. Not run.`);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// The pause gate's boundary guard, for its reason: a tick landing on the instant must run rather
|
|
166
|
+
// than busy-defer to a moment already past.
|
|
167
|
+
if (holdUntil !== null && holdUntil > nowMs + 1000) {
|
|
168
|
+
// `isJobLive` is what stops a vanished holder's lease becoming a tombstone that refuses this
|
|
169
|
+
// target for the rest of the hold. Optional: an unwired probe means the holder cannot be
|
|
170
|
+
// checked, which ADMITS and says so -- one duplicate run beats one dropped delivery, which is
|
|
171
|
+
// `OQ-027`'s call ("one wasted vault read beats one dropped job") on this feature's terms.
|
|
172
|
+
const claim = await waitState.claim(job.id, { dedupId, untilMs: holdUntil, isLive: deps?.isJobLive });
|
|
173
|
+
if (claim.heldBy) {
|
|
174
|
+
// Another delivery for this same target and flow is already holding. Both would clear
|
|
175
|
+
// together and both would be paid, which is the accumulation the acceptance forbids.
|
|
176
|
+
return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
|
|
177
|
+
}
|
|
178
|
+
if (claim.retry) {
|
|
179
|
+
// The holder could not be checked. Holding anyway would put two jobs on one target and pay
|
|
180
|
+
// for both; refusing would drop a delivery over a holder that may be gone. So decide
|
|
181
|
+
// nothing: re-defer briefly and ask again once the probe can answer.
|
|
182
|
+
deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
|
|
183
|
+
await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
|
|
184
|
+
throw new DelayedError();
|
|
185
|
+
}
|
|
186
|
+
if (claim.tookOverFrom) deps?.log?.("wait_lease_taken_over", { jobId: job.id, from: claim.tookOverFrom });
|
|
187
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: holdUntil });
|
|
188
|
+
deps?.log?.("wait_deferred", { jobId: job.id, until: new Date(holdUntil).toISOString(), label: waitLabel(job.data) });
|
|
189
|
+
await job.moveToDelayed(holdUntil, token);
|
|
190
|
+
throw new DelayedError();
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// TIER 2: the polled conditions. Last, because it is the only part of this gate that spawns a
|
|
194
|
+
// process -- the free refusals above it are free, and the free hold above it is free.
|
|
195
|
+
if (profiles.length > 0) {
|
|
196
|
+
const held = (await waitState.heldForMs(job.id)) ?? 0;
|
|
197
|
+
const counted = await waitState.counters(job.id);
|
|
198
|
+
|
|
199
|
+
// One check at a time, process-wide, and never the worker's last free slot. This is the bound
|
|
200
|
+
// that keeps a wait from starving the paid work it is waiting for: slots x timeout is the most
|
|
201
|
+
// wall-clock a worker can spend answering questions instead of running jobs. Computed against
|
|
202
|
+
// the LIVE concurrency rather than the boot value, because the overlay can lower it.
|
|
203
|
+
const slots = Math.min(checkSlotCount(), Math.max(1, concurrencyNow() - 1));
|
|
204
|
+
if (!checkSlots.tryAcquire(WAIT_CHECK_KEY, slots)) {
|
|
205
|
+
// Denials are counted, and a run of them is the ONE symptom the capacity bound has. The
|
|
206
|
+
// lease deliberately caps how much wall-clock this worker spends checking; being at that
|
|
207
|
+
// cap constantly means demand exceeds it, which the issue's own economics say arrives
|
|
208
|
+
// silently -- paid jobs starve behind checks that spend nothing and nothing says why.
|
|
209
|
+
const denials = await waitState.noteThrottle(job.id, { denied: true });
|
|
210
|
+
if (denials === THROTTLE_ALARM) deps?.log?.("wait_capacity_exceeded", { jobId: job.id, denials, slots, hint: "raise PI_WAIT_CHECK_SLOTS or PI_CONCURRENCY, lengthen PI_WAIT_INTERVAL_MS, or hold fewer jobs" });
|
|
211
|
+
// A starved job still needs a CLOCK and a CEILING, or the lease turns into the very
|
|
212
|
+
// starvation it exists to bound: without this the hold is stamped only on a wake that won
|
|
213
|
+
// the lease, so a job that never wins one has no `since`, never reaches the maximum, and
|
|
214
|
+
// re-wakes forever with no record and no bound. There is no deciding check to run first
|
|
215
|
+
// here -- that is the whole condition -- so the bound applies directly.
|
|
216
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + maxWaitMs() });
|
|
217
|
+
if (held >= maxWaitMs()) {
|
|
218
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait-unchecked", denials, heldForMs: held }, `Gave up waiting: this deployment could not run the check often enough to answer within the maximum wait. Not run.`);
|
|
219
|
+
}
|
|
220
|
+
// Denied. Re-ask at a fraction of the cadence rather than the full backoff (which would
|
|
221
|
+
// turn one lost coin-flip into a fifteen-minute penalty) or a flat few seconds (which at
|
|
222
|
+
// scale is a herd). Jittered, so a fleet of denied jobs does not return together.
|
|
223
|
+
const wait = Math.max(THROTTLE_FLOOR_MS, Math.floor(waitBackoffMs(intervalMs(), held) / 4));
|
|
224
|
+
const delay = wait + Math.floor(wait * 0.1 * random());
|
|
225
|
+
deps?.log?.("wait_check_throttled", { jobId: job.id, delayMs: delay, slots });
|
|
226
|
+
await job.moveToDelayed(nowMs + delay, token);
|
|
227
|
+
throw new DelayedError();
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// Claimed BEFORE the check, not after: a second delivery for an already-held target is a free
|
|
231
|
+
// determinate refusal, and paying for a subprocess first inverts the free-before-costly rule
|
|
232
|
+
// this gate's own header invokes. Tier 1 already claims in this order.
|
|
233
|
+
const claim = await waitState.claim(job.id, { dedupId, untilMs: nowMs + waitBackoffMs(intervalMs(), held), isLive: deps?.isJobLive });
|
|
234
|
+
if (claim.heldBy) {
|
|
235
|
+
return await refuseWait("wait-superseded", "wait_superseded", { heldBy: claim.heldBy }, "Another delivery for this target is already waiting on the same conditions. Not run.");
|
|
236
|
+
}
|
|
237
|
+
if (claim.retry) {
|
|
238
|
+
deps?.log?.("wait_supersede_unverified", { jobId: job.id, heldBy: claim.holder ?? null, delayMs: SUPERSEDE_RECHECK_MS });
|
|
239
|
+
await job.moveToDelayed(nowMs + SUPERSEDE_RECHECK_MS, token);
|
|
240
|
+
throw new DelayedError();
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
await waitState.noteThrottle(job.id, { denied: false }); // granted: the run of denials ends here
|
|
244
|
+
let verdict = null;
|
|
245
|
+
let checked = null;
|
|
246
|
+
try {
|
|
247
|
+
// Sequential, in the operator's writing order: the resolver's reason applies unchanged --
|
|
248
|
+
// naming the first condition that did not clear is what makes a held row readable, and a
|
|
249
|
+
// parallel fan-out would blame whichever lost the race on any given wake.
|
|
250
|
+
for (const profile of profiles) {
|
|
251
|
+
checked = profile;
|
|
252
|
+
verdict = await deps.checkWait(profile, targetFor(job.data?.kind, job.data), { signal });
|
|
253
|
+
if (verdict?.profileUnknown || verdict?.verdict !== "go") break;
|
|
254
|
+
}
|
|
255
|
+
} finally {
|
|
256
|
+
checkSlots.release(WAIT_CHECK_KEY);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
if (verdict?.unusableTarget) {
|
|
260
|
+
// Determinate and unfixable by waiting: the job's own target is a shape no check can be
|
|
261
|
+
// handed. It belongs with the refusals, not the holds -- holding would spend the fault
|
|
262
|
+
// budget and then blame the operator's script for a value it was never given.
|
|
263
|
+
return await refuseWait("wait-unreadable", "refused_wait_unreadable", { profile: checked }, `Refused: this job's target cannot be handed to a wait check, so \`${checked}\` can never be asked. Not run.`);
|
|
264
|
+
}
|
|
265
|
+
if (verdict?.profileUnknown) {
|
|
266
|
+
return await refuseWait("wait-profile-unknown", "wait_profile_unknown", { profile: verdict.profileUnknown }, `Waiting on \`${verdict.profileUnknown}\` is not something this deployment can answer: no such wait profile is declared here. Not run.`);
|
|
267
|
+
}
|
|
268
|
+
if (verdict?.verdict === "refuse") {
|
|
269
|
+
// Exit 2: the check says this will NEVER clear. Terminal by the protocol's own words, and
|
|
270
|
+
// distinct from every "not yet" above it.
|
|
271
|
+
return await refuseWait("wait-refused", "wait_refused", { profile: checked, heldForMs: held }, `The check \`${checked}\` reports this will never clear. Not run.`);
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
if (verdict?.aborted) {
|
|
275
|
+
// The worker is stopping or this job was cancelled. Nothing was learned and nothing is
|
|
276
|
+
// owed: re-defer at once rather than at the full backoff, and count neither a check nor a
|
|
277
|
+
// fault, or a rolling deploy would spend a job's whole budget on its own restarts and then
|
|
278
|
+
// blame the operator's script for it.
|
|
279
|
+
deps?.log?.("wait_check_aborted", { jobId: job.id, profile: checked });
|
|
280
|
+
await job.moveToDelayed(nowMs + THROTTLE_FLOOR_MS, token);
|
|
281
|
+
throw new DelayedError();
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
if (verdict?.verdict === "hold") {
|
|
285
|
+
const fault = verdict.fault === true;
|
|
286
|
+
await waitState.noteCheck(job.id, { fault });
|
|
287
|
+
const faults = fault ? counted.faults + 1 : 0;
|
|
288
|
+
|
|
289
|
+
// A check that never answers is a broken script, not a slow condition, and OQ-030 is why
|
|
290
|
+
// this bound exists: most CLIs exit 1 for everything, so without it a typo would hold for
|
|
291
|
+
// the whole maximum wait and then blame the CONDITION rather than the check.
|
|
292
|
+
if (faults >= maxFaults()) {
|
|
293
|
+
return await refuseWait("wait-unanswerable", "wait_unanswerable", { profile: checked, faults }, `The check \`${checked}\` could not answer ${faults} times in a row. Not run.`);
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
// BOTH terminal bounds are tested AFTER the check and never before it, so a condition that
|
|
297
|
+
// cleared on the deciding wake runs instead of being recorded as never having cleared.
|
|
298
|
+
// Without that ordering the backoff's own quantisation makes "cleared at t+1s, declared
|
|
299
|
+
// never-cleared at t+900s" a structural lie in the durable record and in a public comment.
|
|
300
|
+
//
|
|
301
|
+
// The count bound reads `checks + 1` because this wake's check has just run: the job gets
|
|
302
|
+
// exactly `maxChecks` checks, the last of which is the deciding one. Putting it before the
|
|
303
|
+
// check instead -- so the act of testing the bound could not exceed it -- was the obvious
|
|
304
|
+
// spelling, and it silently made this whole guarantee untrue at every shipped default,
|
|
305
|
+
// because the count bound is the one that fires first there.
|
|
306
|
+
if (counted.checks + 1 >= maxChecks()) {
|
|
307
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-checks", checks: counted.checks + 1, profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\` after ${counted.checks + 1} checks. Not run.`);
|
|
308
|
+
}
|
|
309
|
+
if (held >= maxWaitMs()) {
|
|
310
|
+
return await refuseWait("wait-expired", "wait_expired", { reason: "max-wait", profile: checked, heldForMs: held }, `Gave up waiting on \`${checked}\`. Not run.`);
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
// Clamped to what is LEFT of the budget, never just the cadence. Without this an hourly
|
|
314
|
+
// interval under a fifteen-minute maximum holds for the full hour -- 400% of the bound the
|
|
315
|
+
// operator configured -- because the ceiling is only tested when a wake arrives, and the
|
|
316
|
+
// cadence decides when that is. The two knobs are independent `positiveInt`s and nothing
|
|
317
|
+
// cross-validates them, so the clamp is what makes the smaller one actually bind.
|
|
318
|
+
const base = waitBackoffMs(intervalMs(), held);
|
|
319
|
+
const jittered = base + Math.floor(base * 0.1 * random());
|
|
320
|
+
const remaining = Math.max(0, maxWaitMs() - held);
|
|
321
|
+
const delay = Math.max(1000, Math.min(jittered, remaining));
|
|
322
|
+
await waitState.hold(job.id, { dedupId, target: targetFor(job.data?.kind, job.data), label: waitLabel(job.data), untilMs: nowMs + delay });
|
|
323
|
+
deps?.log?.("wait_deferred", { jobId: job.id, profile: checked, fault, heldForMs: held, delayMs: delay });
|
|
324
|
+
await job.moveToDelayed(nowMs + delay, token);
|
|
325
|
+
throw new DelayedError();
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
// FAIL CLOSED on anything that is not literally go. Everything above tests for a specific
|
|
329
|
+
// shape and falls through otherwise, and "otherwise" at this gate means STARTING A PAID
|
|
330
|
+
// CONTAINER -- so an `undefined`, a `null`, a `{}`, a mis-cased "GO" or a bare string from a
|
|
331
|
+
// checker would run the job silently, with no record field and no log line to distinguish it
|
|
332
|
+
// from a job whose check said yes. The shipped checker is total, and that is exactly the
|
|
333
|
+
// reasoning `unreadableConditions` above rejects: this is a dependency-injection seam, and a
|
|
334
|
+
// seam's guarantees are the caller's to enforce.
|
|
335
|
+
if (verdict?.verdict !== "go") {
|
|
336
|
+
deps?.log?.("wait_check_unintelligible", { jobId: job.id, profile: checked });
|
|
337
|
+
await waitState.noteCheck(job.id, { fault: true });
|
|
338
|
+
const base = waitBackoffMs(intervalMs(), held);
|
|
339
|
+
await job.moveToDelayed(nowMs + base + Math.floor(base * 0.1 * random()), token);
|
|
340
|
+
throw new DelayedError();
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
// Every profile answered go. Record the last check so the count bound sees it.
|
|
344
|
+
await waitState.noteCheck(job.id, { fault: false });
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
// Only a job that actually HELD has cleared. Without the check this line fires on the first
|
|
348
|
+
// pickup of a job whose instant had already passed, and again on every scope-busy re-check
|
|
349
|
+
// afterwards -- asserting a wait ended that never began.
|
|
350
|
+
const heldForMs = await waitState.heldForMs(job.id);
|
|
351
|
+
// Say so before releasing: a sibling held on this target must find the answer, not an empty lease.
|
|
352
|
+
if (heldForMs !== null) await waitState.markSatisfied(job.id, { dedupId });
|
|
353
|
+
await waitState.release(job.id, { dedupId });
|
|
354
|
+
if (heldForMs !== null) deps?.log?.("wait_cleared", { jobId: job.id, label: waitLabel(job.data), heldForMs });
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
// Per-scope concurrency and the one-job-per-folder mutex (issue #242,
|
|
358
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT). LAST of the three gates, after the pause gate (a paused job must
|
|
359
|
+
// not burn re-check wakes) and after the wait gate (a job holding until tomorrow must not sit on a
|
|
360
|
+
// folder while it does), and STRICTLY above the `try` below, like the pause gate and for the same two
|
|
361
|
+
// reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
|
|
362
|
+
// catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
|
|
363
|
+
// handling exactly as the pause gate's does (inside the try it would become a permanent failure
|
|
364
|
+
// plus a failure record for what was a transient blip). The limits snapshot is read ONCE here and
|
|
365
|
+
// shared with `scopedCaps` below, so the gate and the money ledger cannot disagree mid-job.
|
|
366
|
+
// tryAcquire is a synchronous check-and-increment -- no await between read and take, so Node's
|
|
367
|
+
// single thread makes it atomic at any concurrency -- and the local-folder limit is a structural 1
|
|
368
|
+
// (concurrencyFor) with no file and no off-switch: the scheduler mints a cron trigger's next
|
|
369
|
+
// occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
|
|
370
|
+
// (measured: 301ms of live container overlap through this very processor) unless this gate holds.
|
|
371
|
+
// Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
|
|
372
|
+
const limits = scopedLimits();
|
|
373
|
+
const scope = canonicalScope(job.data);
|
|
374
|
+
let held = false;
|
|
375
|
+
if (scope) {
|
|
376
|
+
if (!inFlight.tryAcquire(scope, concurrencyFor(job.data, limits))) {
|
|
377
|
+
// Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
|
|
378
|
+
// The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
|
|
379
|
+
// host path); the delayed count and the job id are what an operator needs to see it.
|
|
380
|
+
deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
|
|
381
|
+
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
382
|
+
throw new DelayedError();
|
|
383
|
+
}
|
|
384
|
+
held = true;
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
let startedAt;
|
|
388
|
+
let name;
|
|
389
|
+
let timer;
|
|
390
|
+
let onAbort;
|
|
391
|
+
try {
|
|
392
|
+
// Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
|
|
393
|
+
// finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
|
|
394
|
+
// scope until a worker restart. Nothing in this block CAN throw today (setTimeout and
|
|
395
|
+
// addEventListener on the bullmq-allocated controller are total at processor arity 3); the
|
|
396
|
+
// guard is structural, not observational.
|
|
397
|
+
startedAt = new Date().toISOString();
|
|
398
|
+
name = `pi-job-${job.id}`;
|
|
399
|
+
timer = setTimeout(() => {
|
|
400
|
+
// BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
|
|
401
|
+
Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
|
|
402
|
+
}, timeoutMs);
|
|
54
403
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
404
|
+
// Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
|
|
405
|
+
// after the grace period; the runner exits and runContainer returns/throws.
|
|
406
|
+
onAbort = () => {
|
|
407
|
+
Promise.resolve(stopContainer(name)).catch(() => {});
|
|
408
|
+
};
|
|
409
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
410
|
+
} catch (error) {
|
|
411
|
+
// Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
|
|
412
|
+
// scope must never be releasable twice -- a double release frees another holder's slot.
|
|
413
|
+
if (held) {
|
|
414
|
+
inFlight.release(scope);
|
|
415
|
+
held = false;
|
|
416
|
+
}
|
|
417
|
+
clearTimeout(timer);
|
|
418
|
+
throw error;
|
|
419
|
+
}
|
|
61
420
|
|
|
62
421
|
try {
|
|
63
422
|
const settings = await getSettings();
|
|
@@ -66,7 +425,9 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
66
425
|
// job completed and does not retry a file that can never parse (CONST-RETRY-INFRA-ONLY). Resolved
|
|
67
426
|
// before runJob, so no budget slot is reserved and no container starts (CONST-BUDGET-BEFORE-TOKENS).
|
|
68
427
|
// recordRun leaves the durable settings-overlay-invalid trace for the admin extension.
|
|
69
|
-
// No provider/model here,
|
|
428
|
+
// No provider/model here, as in every result this function returns from ABOVE the try (the wait gate's
|
|
429
|
+
// refusals are the others): each is decided before or during the settings read, so no honest effective
|
|
430
|
+
// value exists yet -- buildRecord defaults both null.
|
|
70
431
|
const result = { outcome: "policy", reason: "settings-overlay-invalid", exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
71
432
|
recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
72
433
|
return result;
|
|
@@ -99,6 +460,10 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
99
460
|
// The daily TOKEN cap (issue #25), same overlay > env resolution. Check-AFTER, so it gates the
|
|
100
461
|
// NEXT job on prior recorded spend; null => the daily token counter is disabled.
|
|
101
462
|
tokenCap: settings.dailyTokenCap,
|
|
463
|
+
// This job's scoped budget windows (issue #242), from the SAME limits snapshot the pickup
|
|
464
|
+
// gate above read -- one read per pickup, so gate and ledger agree for this job's whole
|
|
465
|
+
// life. Null when no row carries a money window for this scope.
|
|
466
|
+
scopedCaps: budgetCapsFor(job.data, limits),
|
|
102
467
|
...deps,
|
|
103
468
|
runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
|
|
104
469
|
// REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
|
|
@@ -140,13 +505,16 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
140
505
|
// records it as failed-and-distinct in the queue's failed set without a retry.
|
|
141
506
|
throw new UnrecoverableError(error.message);
|
|
142
507
|
} finally {
|
|
508
|
+
// Release FIRST and never throw (release clamps at zero by construction): a throw here would
|
|
509
|
+
// mask the job's real error, and a missed release wedges the scope until a worker restart.
|
|
510
|
+
if (held) inFlight.release(scope);
|
|
143
511
|
clearTimeout(timer);
|
|
144
512
|
signal.removeEventListener("abort", onAbort);
|
|
145
513
|
}
|
|
146
514
|
};
|
|
147
515
|
}
|
|
148
516
|
|
|
149
|
-
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, extraClosers = [] }) {
|
|
517
|
+
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, waitState, afterMaxMs, checkSlots, checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, extraClosers = [] }) {
|
|
150
518
|
let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
|
|
151
519
|
const processor = makeProcessor({
|
|
152
520
|
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
@@ -159,6 +527,26 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
|
|
|
159
527
|
if (Number.isInteger(n) && worker.concurrency !== n) worker.concurrency = n;
|
|
160
528
|
},
|
|
161
529
|
pauseUntil,
|
|
530
|
+
// Undefined pass-throughs take makeProcessor's own defaults (no limits; a fresh per-processor
|
|
531
|
+
// in-flight map -- one per worker process, which under DES-CONCURRENCY-3's one-worker-per-daemon
|
|
532
|
+
// shape means one per daemon).
|
|
533
|
+
scopedLimits,
|
|
534
|
+
inFlight,
|
|
535
|
+
// Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
|
|
536
|
+
// redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
|
|
537
|
+
waitState,
|
|
538
|
+
afterMaxMs,
|
|
539
|
+
// Issue #230, the polled tier. `concurrencyNow` reads the LIVE slot count rather than the boot value,
|
|
540
|
+
// because the overlay can lower it through `dispatch_set` and a check must never take the last free
|
|
541
|
+
// slot from a paid job. Late-bound over `worker` exactly as `applyConcurrency` is, and for the same
|
|
542
|
+
// reason: the value it needs does not exist until the Worker is constructed.
|
|
543
|
+
checkSlots,
|
|
544
|
+
checkSlotCount,
|
|
545
|
+
concurrencyNow: concurrencyNow ?? (() => worker?.concurrency ?? concurrency),
|
|
546
|
+
intervalMs,
|
|
547
|
+
maxWaitMs,
|
|
548
|
+
maxChecks,
|
|
549
|
+
maxFaults,
|
|
162
550
|
deps,
|
|
163
551
|
recordRun,
|
|
164
552
|
});
|
package/src/init.mjs
CHANGED
|
@@ -19,6 +19,11 @@ const EMPTY_PACKAGES = `${JSON.stringify({ packages: [] }, null, 2)}\n`;
|
|
|
19
19
|
// Operator-declared subscription plans (issue #53), read by the admin extension only — never at job
|
|
20
20
|
// time. Versioned because a newer file must fail loud, and that cannot be retrofitted into a v1 reader.
|
|
21
21
|
const EMPTY_SUBSCRIPTIONS = `${JSON.stringify({ version: 1, subscriptions: [] }, null, 2)}\n`;
|
|
22
|
+
// Scoped limits (issue #242): per repo/folder run caps and concurrency. Empty is inert -- and the
|
|
23
|
+
// one-job-per-folder mutex for local jobs is code, not configuration, so it needs no scaffold line.
|
|
24
|
+
// Versioned for the subscriptions reason, sharpened: this is enforcement config, and a silently
|
|
25
|
+
// down-read newer file would be a silently widened spend limit.
|
|
26
|
+
const EMPTY_SCOPED_LIMITS = `${JSON.stringify({ version: 1, limits: [] }, null, 2)}\n`;
|
|
22
27
|
/**
|
|
23
28
|
* The egress allowlist (REQ-EGRESS-ALLOWLIST): the hosts a job container may reach, one bare hostname per
|
|
24
29
|
* line. Scaffolded with the three a job cannot work without, and NOT empty -- unlike every other scaffold
|
|
@@ -73,6 +78,7 @@ export function runInit(cwd = process.cwd(), deps = {}) {
|
|
|
73
78
|
scaffold(fs, results, join(cwd, "pause-windows.json"), EMPTY_PAUSE_WINDOWS, "empty pause-windows list");
|
|
74
79
|
scaffold(fs, results, join(cwd, "pi-packages.json"), EMPTY_PACKAGES, "empty pi package list (stage with import-pi --with-packages)");
|
|
75
80
|
scaffold(fs, results, join(cwd, "subscriptions.json"), EMPTY_SUBSCRIPTIONS, "empty subscription list (declare plan prices for the admin's cost analytics)");
|
|
81
|
+
scaffold(fs, results, join(cwd, "scoped-limits.json"), EMPTY_SCOPED_LIMITS, "empty scoped-limits list (per repo/folder caps; the folder mutex needs no file)");
|
|
76
82
|
scaffold(fs, results, join(cwd, "egress-allowlist.conf"), DEFAULT_EGRESS_ALLOWLIST, "egress allowlist (provider + forge + registry; the egress policy is on unless PI_EGRESS=0)");
|
|
77
83
|
|
|
78
84
|
for (const [verb, name, note] of results) {
|
package/src/processor.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { lstatSync } from "node:fs";
|
|
2
2
|
import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
|
|
3
3
|
import { configError } from "./config.mjs";
|
|
4
|
+
import { scopeKeyPrefix } from "./scoped-limits.mjs";
|
|
4
5
|
import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
|
|
5
6
|
import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
|
|
6
7
|
|
|
@@ -37,6 +38,12 @@ export async function runJob(job, deps) {
|
|
|
37
38
|
caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
|
|
38
39
|
softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
|
|
39
40
|
tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
|
|
41
|
+
// { scope, caps: { day, week, month } } | null -- this job's scoped budget windows (issue #242,
|
|
42
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT), resolved by the wiring from the same watched-limits snapshot the
|
|
43
|
+
// pickup gate read. Null when the file is unset or the scope's row is concurrency-only; the default
|
|
44
|
+
// keeps an unwired processor byte-identical. The folder MUTEX does not live here -- it is the pickup
|
|
45
|
+
// gate's, pre-everything; this is only the money half.
|
|
46
|
+
scopedCaps = null,
|
|
40
47
|
recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
|
|
41
48
|
// (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
|
|
42
49
|
// this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
|
|
@@ -47,6 +54,9 @@ export async function runJob(job, deps) {
|
|
|
47
54
|
// exactly as before, and the gate below only calls it for a job whose matched rule was a
|
|
48
55
|
// one-shot, so the default is never a probe running on every delivery.
|
|
49
56
|
checkOnceSpent = async () => ({ ok: true }),
|
|
57
|
+
// Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
|
|
58
|
+
// unwired seam must not refuse, and the wiring is what turns the check on.
|
|
59
|
+
checkWaitSkew = async () => ({ ok: true }),
|
|
50
60
|
// REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
|
|
51
61
|
// deployment with no egress policy does -- which is also what the real factory returns when unarmed.
|
|
52
62
|
egressPreflight = async () => ({ ok: true }),
|
|
@@ -124,6 +134,7 @@ export async function runJob(job, deps) {
|
|
|
124
134
|
let token = null;
|
|
125
135
|
let prepared = null;
|
|
126
136
|
let reserved = false;
|
|
137
|
+
let scopedReserved = false;
|
|
127
138
|
|
|
128
139
|
try {
|
|
129
140
|
// The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
|
|
@@ -147,6 +158,28 @@ export async function runJob(job, deps) {
|
|
|
147
158
|
}
|
|
148
159
|
}
|
|
149
160
|
|
|
161
|
+
// The wait-skew check (issue #230), second on the ladder and for the first one's reasons: the same
|
|
162
|
+
// file read, free, determinate, credential-less, and pre-spend. It answers a question no other layer
|
|
163
|
+
// can: does the AUTHORED trigger carry wait conditions this job arrived without? That happens when a
|
|
164
|
+
// service below the version floor dropped the field as an unknown key, and the resulting run is
|
|
165
|
+
// byte-identical to a correct one everywhere it is recorded -- so this refusal is the only thing
|
|
166
|
+
// standing between a stale receiver and a paid job that ran when the operator wrote "wait".
|
|
167
|
+
{
|
|
168
|
+
const skew = await checkWaitSkew(job);
|
|
169
|
+
if (skew.skewed) {
|
|
170
|
+
// Named for the operator, not the payload: how many conditions were authored, never what
|
|
171
|
+
// they say. The fix is a version, so the comment says which one.
|
|
172
|
+
// The message names BOTH causes, because the more likely one is not a version at all. In the
|
|
173
|
+
// compose topology the receiver's single-file `:ro` mount pins a dead inode, so an operator
|
|
174
|
+
// who ADDS `waitFor` to an existing rule gets this refusal on every delivery from a service
|
|
175
|
+
// that is perfectly up to date and merely holding an older copy of the file. Naming only the
|
|
176
|
+
// version would send them looking for an upgrade they do not need.
|
|
177
|
+
await comment(job, `Refused: this trigger declares ${skew.conditions} wait condition${skew.conditions === 1 ? "" : "s"}, but the job reached the worker without them, which means it would have run immediately. Either a service in this deployment is below the version that carries the field, or one is still running against an older copy of the triggers file and needs restarting. Not run.`);
|
|
178
|
+
log("refused_wait_skew", { triggerIndex: job.trigger?.matched?.index ?? null, conditions: skew.conditions });
|
|
179
|
+
return { outcome: "policy", reason: "wait-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
150
183
|
// The job image must exist on THIS host before anything else happens. Free, determinate and
|
|
151
184
|
// credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
|
|
152
185
|
// image refuses without minting a credential it will not use, cloning a repo it will not read, or
|
|
@@ -458,11 +491,51 @@ export async function runJob(job, deps) {
|
|
|
458
491
|
return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
459
492
|
}
|
|
460
493
|
|
|
461
|
-
//
|
|
494
|
+
// Per-scope budget windows (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): the NARROWER ledger
|
|
495
|
+
// reserves FIRST, so a noisy scope's refusals never consume a global slot -- the global INCR below
|
|
496
|
+
// runs only for jobs the scope admitted. Same atomic INCR, same refused-still-counts invariant,
|
|
497
|
+
// through budget.mjs's keyPrefix seam (dayKey/weekKey/monthKey under budget:s:<hash16>). softHoldPct
|
|
498
|
+
// is deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob;
|
|
499
|
+
// scoped windows are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
|
|
500
|
+
if (scopedCaps) {
|
|
501
|
+
// A redis fault BETWEEN this reserve and the global one below strands the scoped INCR with no
|
|
502
|
+
// run and no refund -- the pre-existing mid-reserve posture, shared with the global ledger's
|
|
503
|
+
// own partial-INCR seam; the compensating release below covers REFUSALS, not faults.
|
|
504
|
+
const scoped = await reserveBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
505
|
+
scopedReserved = true;
|
|
506
|
+
if (!scoped.allowed) {
|
|
507
|
+
const w = scoped.blockedWindow;
|
|
508
|
+
const win = scoped.windows[w];
|
|
509
|
+
// A local job's scope is a full host path and its "comment" is not dropped -- the wiring's
|
|
510
|
+
// local adapter LOGS the text (start.mjs forgeFor fallthrough) -- so the path must never
|
|
511
|
+
// enter the message; "this folder" is enough beside the jobId the adapter logs. A forge
|
|
512
|
+
// scope IS the repo the comment posts on, safe to name.
|
|
513
|
+
const scopeLabel = job.kind === "local" ? "this folder" : scopedCaps.scope;
|
|
514
|
+
await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
|
|
515
|
+
// The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would
|
|
516
|
+
// put a full host path in the worker log against no-pii-in-logs (the record keeps only
|
|
517
|
+
// basename(folder) for the same reason). The admin recomputes the key from the configured
|
|
518
|
+
// scope to join it back.
|
|
519
|
+
log("over_scope_budget", { scopeKey: scopeKeyPrefix(scopedCaps.scope), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
|
|
520
|
+
// budgetReserved false: the GLOBAL slot was never touched (scoped reserves first). The scoped
|
|
521
|
+
// counter did INCR and keeps it -- its own refused-reservation-still-counts, per ledger.
|
|
522
|
+
return { outcome: "policy", reason: "scope-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
// GLOBAL budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
|
|
462
527
|
// every active window (day + optional week/month) and the soft-hold band in one atomic pass.
|
|
463
528
|
const budget = await reserveBudget(redis, { caps, softHoldPct, now });
|
|
464
529
|
reserved = true;
|
|
465
530
|
if (!budget.allowed) {
|
|
531
|
+
// The scoped reserve above committed before this global refusal -- give that slot back. Without
|
|
532
|
+
// this, an exhausted global window drains every arriving scope's own day/week/month counters
|
|
533
|
+
// with zero runs to show for it (a storm against a spent global daily cap would empty a repo's
|
|
534
|
+
// week by noon). The scoped ledger's refused-still-counts covers the SCOPE's own refusal above,
|
|
535
|
+
// never a refusal it did not issue.
|
|
536
|
+
if (scopedReserved && scopedCaps) {
|
|
537
|
+
await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
538
|
+
}
|
|
466
539
|
const w = budget.blockedWindow;
|
|
467
540
|
const win = budget.windows[w];
|
|
468
541
|
if (budget.reason === "soft-hold") {
|
|
@@ -561,8 +634,12 @@ export async function runJob(job, deps) {
|
|
|
561
634
|
// budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
|
|
562
635
|
// true for a real container that ran and spent (exit-1 infra / unknown exit).
|
|
563
636
|
if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
|
|
564
|
-
if (
|
|
565
|
-
|
|
637
|
+
if (e instanceof InfraRetry && e.reason === "container-never-started") {
|
|
638
|
+
// Both-or-neither (issue #242): a never-started container can only follow BOTH reserves (the
|
|
639
|
+
// scoped one precedes the global one, and the container follows both), so they refund
|
|
640
|
+
// together -- and a scoped refusal returned above without ever touching the global ledger.
|
|
641
|
+
if (reserved) await releaseBudget(redis, { caps, now });
|
|
642
|
+
if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
566
643
|
}
|
|
567
644
|
throw e;
|
|
568
645
|
} finally {
|