@edgehero/pi-dispatch 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/queue.mjs CHANGED
@@ -134,7 +134,7 @@ export async function enqueueGitLabJob(queue, fields) {
134
134
  * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
135
135
  * has always been.
136
136
  */
137
- export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, replica, replicas }) {
137
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
138
138
  const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
139
139
  // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
140
140
  // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
@@ -179,6 +179,16 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
179
179
  // is copied verbatim into /job/event.json, which an agent reads.
180
180
  ...(secrets !== undefined && { secrets }),
181
181
  ...(secretsProfile !== undefined && { secretsProfile }),
182
+ // Issue #230. The conditions the worker holds this job on, carried so the PICKUP gate can read them:
183
+ // that gate runs above the per-job settings read and never re-parses the triggers file for its terms.
184
+ // At JOB level, and here that placement is a correctness requirement rather than a convention --
185
+ // `trigger` is copied VERBATIM into /job/event.json (prepare-local.mjs), so a `trigger.waitFor` would
186
+ // hand the agent the operator's own gate. Conditional like every field above, so an unflagged job's
187
+ // data keeps exactly the keys it has today. The dedup options below are deliberately NOT widened for
188
+ // a waiting job: that key carries no trigger identity and outlives the job it was set for, so a
189
+ // longer window would suppress an unflagged sibling's deliveries and go on suppressing them after
190
+ // this job finished. Coalescing a held target is the worker's `wait:` keyspace's job instead.
191
+ ...(waitFor !== undefined && { waitFor }),
182
192
  // Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
183
193
  // exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
184
194
  // both are integers, so the run record they land in stays PII-free by construction.
@@ -387,7 +387,12 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
387
387
  * separator from the table, so the notation a forge uses is the notation its records carry -- and a forge
388
388
  * added later inherits a label rather than a null.
389
389
  */
390
- function targetFor(kind, data) {
390
+ /*
391
+ * Exported since issue #230: a held job's panel row needs the same id-only label a run record carries, and
392
+ * the wait gate would otherwise re-derive it. Two spellings of "which issue is this" is how one of them
393
+ * starts carrying a title.
394
+ */
395
+ export function targetFor(kind, data) {
391
396
  if (kind === "local") return `local:${basename(data.folder ?? "")}`;
392
397
  if (isForgeKind(kind)) return `${data.repo}${targetSeparator(kind, data.target?.type)}${data.target?.number}`;
393
398
  return null;
@@ -0,0 +1,277 @@
1
+ /**
2
+ * Scoped limits (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): per-scope run caps and per-scope
3
+ * concurrency, where a scope is what `scopeOf` already answers -- the folder for a local job, the repo
4
+ * for a forge one. One `scoped-limits.json` of `{ scope, day?, week?, month?, concurrent? }` entries:
5
+ * the day/week/month caps refuse a job pre-spend (reason `scope-cap`, a policy refusal), `concurrent`
6
+ * defers the excess through the delayed set (never a refusal -- a busy scope is transient state).
7
+ *
8
+ * This module is pure and fs-injectable (mirrors pause-windows.mjs in every respect): `parseScopedLimits`
9
+ * validates the file TEXT fail-loud, `loadScopedLimits` layers the one fs read on top, and the small
10
+ * helpers below are what the processor gate and the budget wiring consume. It also owns the in-process
11
+ * in-flight counter (`makeInFlight`) so the counter is unit-testable without a bullmq import, the same
12
+ * reason job-id.mjs is queue-free. The wiring that consumes all of this (the gate, the budget calls,
13
+ * the admin surfaces) lands in this issue's later slices; the module ships first so the contract has
14
+ * one implementation to bind to -- sentences below describing enforcement describe THOSE slices.
15
+ *
16
+ * The file is a SIBLING of pause-windows.json, not part of the settings overlay, deliberately: the
17
+ * deferral gate runs before the per-job overlay read, so gate-read config must come from a watched
18
+ * mutable ref, and the overlay's KNOWN_KEYS are flat scalars whose only map-shaped precedent
19
+ * (secretProfiles) is deliberately model-unreachable -- the opposite of what these limits need.
20
+ *
21
+ * `version` is REQUIRED and fail-loud-on-newer (subscriptions.mjs's rule, adopted here because this is a
22
+ * MONEY file): unknown fields are silently dropped per the operator-file policy, so a v2 cap field an old
23
+ * worker drops would be a silently WIDENED spend limit. Pause-windows shipping without a version is a
24
+ * sunk decision, not a precedent to extend to enforcement config.
25
+ *
26
+ * Custom: scoped limits validated inline per triggers.mjs/pause-windows.mjs precedent; zod not in deps
27
+ */
28
+
29
+ import { createHash } from "node:crypto";
30
+ import { existsSync as fsExistsSync, readFileSync as fsReadFileSync } from "node:fs";
31
+ import { isAbsolute, resolve } from "node:path";
32
+ import { configError } from "./config.mjs";
33
+ import { scopeOf } from "./pause-windows.mjs";
34
+
35
+ /** The schema version this build reads and writes. A file declaring a higher one is refused loudly. */
36
+ export const SCOPED_LIMITS_VERSION = 1;
37
+
38
+ /** The four limit fields a row may carry, in display order. */
39
+ const LIMIT_FIELDS = ["day", "week", "month", "concurrent"];
40
+
41
+ function isNonEmptyString(value) {
42
+ return typeof value === "string" && value.trim() !== "";
43
+ }
44
+
45
+ /**
46
+ * The canonical scope string for a job: the RESOLVED folder path for a local job, the repo for a forge
47
+ * one. `scopeOf` alone is not enough for enforcement: nothing on the trigger path normalizes
48
+ * `run.folder`, so `/srv/site`, `/srv/site/`, `/srv//site`, `/srv/x/../site` and a padded spelling are
49
+ * five distinct strings naming ONE directory -- an exact-string mutex keyed on the raw value would run
50
+ * them concurrently in one working tree, which is the exact race the mutex exists to close.
51
+ * `path.resolve` (not `normalize`, which keeps trailing slashes and whitespace) collapses them all; a
52
+ * relative folder resolves against the worker's cwd, the same base `prepareWorkspace`'s existence check
53
+ * uses; Unicode is NFC-normalized on both the job and the row side (see below). Two residuals,
54
+ * deliberate: symlinks are NOT resolved (realpath is an fs call on the hot path and can throw), and
55
+ * neither is filesystem case-insensitivity (on a default macOS/APFS volume `/Srv/Site` and `/srv/site`
56
+ * are one directory and two scopes) -- the pause matcher lives with both.
57
+ *
58
+ * The pause matcher itself keeps the RAW `scopeOf` value: resolving there would silently change which
59
+ * jobs an operator's existing trailing-slash window matches. The two features share the folder-vs-repo
60
+ * split (`scopeOf`, defined once) but not the normalization, and this comment is where that difference
61
+ * is recorded.
62
+ *
63
+ * A useful side effect: a resolved local scope is always an absolute path, and a repo string never is,
64
+ * so a folder named `a/b` and a repo named `a/b` can no longer collide in the counters or the mutex.
65
+ */
66
+ export function canonicalScope(job) {
67
+ const scope = scopeOf(job);
68
+ if (!isNonEmptyString(scope)) return null;
69
+ // NFC on both kinds: macOS's filesystem hands paths back NFD while an admin dialog types NFC, so
70
+ // "wéb" can arrive as two byte sequences naming one thing -- without this, an NFD-spelled forge
71
+ // scope silently escapes an NFC-spelled cap (the local side would at least keep the structural
72
+ // mutex). ASCII is fixed under NFC, so no existing key changes.
73
+ return job?.kind === "local" ? resolve(scope.trim().normalize("NFC")) : scope.normalize("NFC");
74
+ }
75
+
76
+ /**
77
+ * Parse, validate, and normalize the scoped-limits file TEXT. Returns the normalized `limits` array
78
+ * (every row rebuilt as an explicit `{ scope, day, week, month, concurrent }` literal, `null` for absent
79
+ * fields, unknown fields dropped -- the operator-file policy). Throws `configError` (fail-loud) on any
80
+ * malformed entry. `path` is for error messages only -- this function touches no filesystem.
81
+ */
82
+ export function parseScopedLimits(text, path) {
83
+ let parsed;
84
+ try {
85
+ parsed = JSON.parse(text);
86
+ } catch (error) {
87
+ throw configError(`scoped-limits file is not valid JSON: ${path} (${error.message})`);
88
+ }
89
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
90
+ throw configError(`scoped-limits file must be an object with "version" and "limits": ${path}`);
91
+ }
92
+ const version = parsed.version;
93
+ if (!Number.isInteger(version) || version < 1) {
94
+ throw configError(`scoped-limits file must have "version": 1 (an integer >= 1): ${path}`);
95
+ }
96
+ if (version > SCOPED_LIMITS_VERSION) {
97
+ throw configError(`scoped-limits file written by a newer pi-dispatch (version ${version}; this build understands ${SCOPED_LIMITS_VERSION}): ${path}`);
98
+ }
99
+ if (!Array.isArray(parsed.limits)) {
100
+ throw configError(`scoped-limits file must have a "limits" array: ${path}`);
101
+ }
102
+ const rows = parsed.limits.map((row, index) => normalizeLimit(row, index, path));
103
+ const seen = new Map();
104
+ rows.forEach((row, index) => {
105
+ if (seen.has(row.scope)) {
106
+ // Two rows for one scope is a precedence question with no right answer; the admin's
107
+ // edit-in-place never produces one, so a duplicate is always a hand-edit mistake.
108
+ throw configError(`scoped limit at index ${index}: duplicate scope ${JSON.stringify(row.scope)} (first at index ${seen.get(row.scope)}): ${path}`);
109
+ }
110
+ seen.set(row.scope, index);
111
+ });
112
+ return rows;
113
+ }
114
+
115
+ function normalizeLimit(row, index, path) {
116
+ const at = `scoped limit at index ${index}`;
117
+ if (row === null || typeof row !== "object" || Array.isArray(row)) {
118
+ throw configError(`${at}: must be an object: ${path}`);
119
+ }
120
+ if (!isNonEmptyString(row.scope)) throw configError(`${at}: scope must be a non-empty string: ${path}`);
121
+ const trimmed = row.scope.trim().normalize("NFC"); // the same NFC canonicalScope applies job-side
122
+ if (trimmed === "*") {
123
+ // "*" as ONE shared counter is redundant with the global caps, so the only useful reading is a
124
+ // per-scope default -- the OPPOSITE of what "*" means one file over (pause-windows: one rule
125
+ // matching all scopes). Refused rather than shipped divergent; a later version may adopt the
126
+ // per-scope-default reading, with an exact row beating "*" (recorded in the contract).
127
+ throw configError(`${at}: "*" is not supported -- add one row per scope (a per-scope default may adopt "*" later): ${path}`);
128
+ }
129
+ if (trimmed.includes("*")) {
130
+ // No globs, enforced rather than described: an exact matcher makes "acme/*" a row that governs
131
+ // nothing, and a silently inert money limit is the failure class this repo refuses outright.
132
+ throw configError(`${at}: scopes match exactly; a scope containing "*" is refused (no globs): ${path}`);
133
+ }
134
+ const norm = {
135
+ // An absolute path is stored resolved so a `/srv/site/` row governs `/srv/site` jobs -- the same
136
+ // collapse canonicalScope applies on the job side. isAbsolute is PLATFORM-NATIVE on purpose, so a
137
+ // foreign-platform row (a windows drive path on a POSIX worker) stays verbatim and is inert here;
138
+ // the doctor's unreferenced-scope advisory names it. Resolving it instead would "work" only by
139
+ // both sides mangling into the same cwd-prefixed string -- a match by accident, not by contract.
140
+ scope: isAbsolute(trimmed) ? resolve(trimmed) : trimmed,
141
+ day: null,
142
+ week: null,
143
+ month: null,
144
+ concurrent: null,
145
+ };
146
+ let any = false;
147
+ for (const field of LIMIT_FIELDS) {
148
+ const value = row[field];
149
+ // Absent-or-null (subscriptions.mjs's rule): null is the normalizer's OWN output for an unset
150
+ // field, so the parser must accept it back or it cannot re-parse what it produced -- the admin's
151
+ // read-modify-write goes through this parser on both edges.
152
+ if (value === undefined || value === null) continue;
153
+ // 0 is refused, not "never run": budget.mjs's caps treat every configured window as >= 1, and
154
+ // "never run this scope" already has two honest spellings (delete the trigger; a pause window).
155
+ // isSafeInteger, not isInteger: 1e21 passes isInteger and reads as a limit while being
156
+ // indistinguishable from unlimited -- a bound that cannot count is not a bound.
157
+ if (!Number.isSafeInteger(value) || value < 1) {
158
+ throw configError(`${at}: ${field} must be an integer >= 1: ${path}`);
159
+ }
160
+ norm[field] = value;
161
+ any = true;
162
+ }
163
+ if (!any) {
164
+ throw configError(`${at}: at least one of day, week, month, concurrent is required (a row that limits nothing is a row an operator sets and then trusts): ${path}`);
165
+ }
166
+ return norm;
167
+ }
168
+
169
+ /**
170
+ * Load and validate the scoped-limits file named by `config.scopedLimitsFile`. Returns `[]` when the file
171
+ * is unset (no scoped caps or concurrency -- a valid deployment; the folder mutex holds regardless, it is
172
+ * code, not configuration). `readFileSync`/`existsSync` are injectable for tests.
173
+ */
174
+ export function loadScopedLimits(config, { readFileSync = fsReadFileSync, existsSync = fsExistsSync } = {}) {
175
+ const path = config.scopedLimitsFile;
176
+ if (path === null || path === undefined) return [];
177
+ if (!existsSync(path)) throw configError(`scoped-limits file does not exist: ${path}`);
178
+ return parseScopedLimits(readFileSync(path, "utf8"), path);
179
+ }
180
+
181
+ /**
182
+ * The exact-match row for a canonical scope, or null. Exact string equality only -- the pause matcher's
183
+ * semantics minus its "*" (refused above). With duplicates refused there is no precedence ladder.
184
+ */
185
+ export function limitFor(limits, scope) {
186
+ if (!Array.isArray(limits) || !isNonEmptyString(scope)) return null;
187
+ return limits.find((l) => l.scope === scope) ?? null;
188
+ }
189
+
190
+ /**
191
+ * The scoped budget windows this job reserves against, or null when nothing applies (no row for the
192
+ * scope, or the row is concurrency-only). The returned `scope` is CANONICAL so the redis counters are
193
+ * spelling-stable. Shaped like the global `caps` object so `reserveBudget` consumes it unchanged.
194
+ */
195
+ export function budgetCapsFor(job, limits) {
196
+ const scope = canonicalScope(job);
197
+ const row = limitFor(limits, scope);
198
+ if (!row || (row.day === null && row.week === null && row.month === null)) return null;
199
+ return { scope, caps: { day: row.day, week: row.week, month: row.month } };
200
+ }
201
+
202
+ /**
203
+ * The effective in-flight ceiling for this job's scope: `min(configured concurrent, structural)`, where
204
+ * structural is 1 for a local job -- the folder mutex -- and unbounded otherwise. The mutex is
205
+ * UNCONDITIONAL, in code, with no file configured and no off-switch: two agents in one bind-mounted
206
+ * working tree is the race `run.replicas` is already refused on local jobs for, and a cron trigger
207
+ * reaches it with no operator mistake at all (the scheduler mints the next occurrence at pickup and
208
+ * promotes on time alone, so a slow run overlaps its own successor). A configured `concurrent` above 1
209
+ * on a folder scope silently clamps to 1 rather than refusing at parse: scope strings are not reliably
210
+ * typeable as folder-vs-repo (`"a/b"` is a legal relative folder and a legal repo), so a parse-time
211
+ * classifier would misfire; min() cannot.
212
+ *
213
+ * No scope (a malformed payload) means no gate: Infinity, admit -- the job will fail its own validation
214
+ * downstream, and holding a mutex slot under key `null` helps nobody.
215
+ */
216
+ export function concurrencyFor(job, limits) {
217
+ const scope = canonicalScope(job);
218
+ if (scope === null) return Infinity;
219
+ const structural = job?.kind === "local" ? 1 : Infinity;
220
+ const configured = limitFor(limits, scope)?.concurrent ?? Infinity;
221
+ return Math.min(structural, configured);
222
+ }
223
+
224
+ /**
225
+ * The redis key prefix for a scope's budget windows: `budget:s:<16 hex>`. Handed to
226
+ * `reserveBudget`/`releaseBudget` as `keyPrefix`, so `dayKey`/`weekKey`/`monthKey` compose
227
+ * `budget:s:<h>:YYYY-MM-DD` / `:w:...` / `:m:...` with zero new key-shape logic. A hash (the localJobId
228
+ * idiom: sha256, first 16 hex) rather than an escape: a scope legally contains `:` and `/` (folder
229
+ * paths, gitlab group/subgroup/project), which would collide with budget.mjs's own `w:`/`m:`/`t:`
230
+ * sub-namespaces, and a bijective escape grammar is a new thing to get wrong with unbounded key lengths.
231
+ * The one consumer that must map keys BACK to scopes is the admin's counter display, and it knows the
232
+ * configured scopes -- it recomputes keys through this same export, so unreadability in redis-cli is the
233
+ * accepted cost.
234
+ */
235
+ export function scopeKeyPrefix(scope) {
236
+ const h = createHash("sha256").update(String(scope)).digest("hex").slice(0, 16);
237
+ return `budget:s:${h}`;
238
+ }
239
+
240
+ /**
241
+ * The per-process in-flight counter behind per-scope concurrency and the folder mutex. Process memory is
242
+ * the CORRECT store, not a compromise: one worker per docker daemon is the shape DES-CONCURRENCY-3
243
+ * assumes everywhere and `service install` enforces for installed units (`pi-dispatch start` holds no
244
+ * lock, and two hand-run workers are already unsupported -- the second one's boot reaper kills the
245
+ * first's live containers); the reaper removes every surviving `pi-job-*` container before the worker
246
+ * starts draining, so a fresh, empty map is never wrong about a live container except when the reap
247
+ * itself was skipped (`reaper_skipped`: docker missing/down at boot -- a state where no NEW container
248
+ * can start either); and a Redis-held counter would survive a crash WRONGLY -- a claim for a container
249
+ * the reaper just killed, demanding TTL/heartbeat machinery, a second source of truth about "what is
250
+ * running" (the OQ-008 failure mode).
251
+ *
252
+ * `tryAcquire` is a synchronous check-and-increment: no await between the read and the take, so under
253
+ * Node's single thread no interleaving exists at any concurrency. `release` never throws -- it runs in
254
+ * the processor's finally, where a throw would mask the job's real error -- and clamps at zero.
255
+ */
256
+ export function makeInFlight() {
257
+ const counts = new Map();
258
+ return {
259
+ /** True and counted when under `limit`; false WITHOUT counting when at or over it. */
260
+ tryAcquire(scope, limit) {
261
+ const current = counts.get(scope) ?? 0;
262
+ if (current >= limit) return false;
263
+ counts.set(scope, current + 1);
264
+ return true;
265
+ },
266
+ /** Decrement, deleting at zero; a release without a matching acquire is a no-op, never a throw. */
267
+ release(scope) {
268
+ const current = counts.get(scope) ?? 0;
269
+ if (current <= 1) counts.delete(scope);
270
+ else counts.set(scope, current - 1);
271
+ },
272
+ /** The current in-flight count for a scope (tests and future observability). */
273
+ count(scope) {
274
+ return counts.get(scope) ?? 0;
275
+ },
276
+ };
277
+ }
package/src/service.mjs CHANGED
@@ -972,6 +972,15 @@ async function doRestart(ctx, values) {
972
972
  await ctx.sleep(2000);
973
973
  ({ active = 0 } = await queue.getJobCounts("active"));
974
974
  }
975
+ // A HELD job is neither active nor waiting, so the loop above has just reported a drained queue with
976
+ // however many jobs still parked on `run.waitFor` (issue #230). They are safe -- a hold spends
977
+ // nothing, survives a restart and reserves no slot -- but the operator is upgrading, and those jobs
978
+ // will wake against the new version. Said plainly rather than left to be discovered, which is what
979
+ // this command would otherwise be doing: reporting a drained queue it cannot see all of.
980
+ const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
981
+ if (delayed > 0) {
982
+ ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
983
+ }
975
984
  const stopped = await doStop(ctx);
976
985
  if (stopped !== 0) {
977
986
  ctx.out("restart did not happen — the queue STAYS PAUSED; fix the service, then `pi-dispatch resume`.\n");
package/src/start.mjs CHANGED
@@ -22,8 +22,11 @@ import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare
22
22
  import { listRunningSandboxes } from "./sandbox.mjs";
23
23
  import { makeSandboxReaper } from "./sandbox-store.mjs";
24
24
  import { makeSessionStore } from "./session-store.mjs";
25
- import { makeCheckOnceSpent, makeDisarmOnce } from "./triggers-file.mjs";
25
+ import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
26
26
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
27
+ import { loadScopedLimits } from "./scoped-limits.mjs";
28
+ import { makeWaitChecker } from "./wait-check.mjs";
29
+ import { makeWaitState } from "./wait-state.mjs";
27
30
  import { makeQueue } from "./queue.mjs";
28
31
  import { makeRunContainer } from "./run-container.mjs";
29
32
  import { makeSecretsResolver } from "./secrets.mjs";
@@ -103,6 +106,42 @@ function watchPauseWindowsFile(config, ref, log) {
103
106
  }
104
107
  }
105
108
 
109
+ /**
110
+ * The scoped-limits reload, EXPORTED apart from its watcher so keep-last-good is unit-testable without
111
+ * fs.watch (its two watcher siblings above bind theirs inline; this one is money config, so the
112
+ * last-good property carries its own test). A bad edit keeps `ref.current` untouched and logs
113
+ * `scoped_limits_reload_invalid` -- the pause-windows posture, INT-SCOPED-LIMITS-FILE-CONTRACT.
114
+ */
115
+ export function reloadScopedLimits(config, ref, log) {
116
+ try {
117
+ ref.current = loadScopedLimits(config);
118
+ log("scoped_limits_reloaded", { count: ref.current.length });
119
+ } catch (err) {
120
+ log("scoped_limits_reload_invalid", { reason: err?.message });
121
+ }
122
+ }
123
+
124
+ /**
125
+ * Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
126
+ * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort + unref'd.
127
+ */
128
+ function watchScopedLimitsFile(config, ref, log) {
129
+ const path = config.scopedLimitsFile;
130
+ const dir = dirname(path) || ".";
131
+ const file = basename(path);
132
+ let timer = null;
133
+ try {
134
+ watch(dir, (_event, changed) => {
135
+ if (changed && changed !== file) return;
136
+ clearTimeout(timer);
137
+ timer = setTimeout(() => reloadScopedLimits(config, ref, log), 150);
138
+ }).unref?.();
139
+ log("scoped_limits_watching", { path });
140
+ } catch (err) {
141
+ log("scoped_limits_watch_unavailable", { reason: err?.message });
142
+ }
143
+ }
144
+
106
145
  export function makeReaper({ log }) {
107
146
  return async function reap() {
108
147
  try {
@@ -188,6 +227,10 @@ export async function startWorker(
188
227
  // pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
189
228
  const pauseWindows = { current: loadPauseWindows(config) };
190
229
 
230
+ // Issue #242: same posture for the scoped-limits file -- fail-loud with the operator present, mutable
231
+ // ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
232
+ const scopedLimits = { current: loadScopedLimits(config) };
233
+
191
234
  // The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
192
235
  // Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
193
236
  // forge is unconfigured or unreachable), `host` is the three methods github-host.mjs returns. The map
@@ -422,6 +465,21 @@ export async function startWorker(
422
465
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
423
466
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
424
467
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
468
+ // Issue #242: the scoped-limits snapshot the pickup gate and the scoped budget read, once per
469
+ // pickup, from the live-reloaded ref -- same next-job grain as pauseUntil above.
470
+ scopedLimits: () => scopedLimits.current,
471
+ // Issue #230. The `after` ceiling is read per pickup from config rather than frozen into the
472
+ // processor, so it is one value with one home; the wait state shares the budget's redis client
473
+ // because it describes the same delayed jobs that client already reasons about.
474
+ afterMaxMs: () => config.waitAfterMaxMs,
475
+ waitState: makeWaitState({ redis }),
476
+ // The polled tier's bounds, read per pickup from config so they are one value with one home. The
477
+ // slot count is a CEILING the gate clamps against the live concurrency, never the final number.
478
+ checkSlotCount: () => config.waitCheckSlots,
479
+ intervalMs: () => config.waitIntervalMs,
480
+ maxWaitMs: () => config.waitMaxMs,
481
+ maxChecks: () => config.waitMaxChecks,
482
+ maxFaults: () => config.waitMaxFaults,
425
483
  deps: {
426
484
  collectChain,
427
485
  // The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
@@ -429,6 +487,35 @@ export async function startWorker(
429
487
  // spending delivery is excused). In the compose topology this check is the once-enforcement
430
488
  // layer, because the receiver's single-file :ro mount pins a dead inode until restart.
431
489
  checkOnceSpent: makeCheckOnceSpent({ triggersPath: onceTriggersFile }),
490
+ // Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
491
+ // asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
492
+ // service below the version floor turns a wait into a paid run nothing can tell from a correct
493
+ // one. In the compose topology the worker's read is the live inode while the receiver's is dead
494
+ // until restart, which is exactly the deployment where the skew happens.
495
+ checkWaitSkew: makeCheckWaitSkew({ triggersPath: onceTriggersFile }),
496
+ // Issue #230. Whether a job the supersede lease names is still in the queue. Without it a holder
497
+ // that vanished by any route except the clean one leaves a key that refuses every later delivery
498
+ // for that target until it expires -- and a refused forge delivery is gone, since no webhook
499
+ // resends it. `getJob` answers from the queue rather than from our own bookkeeping, so the two
500
+ // cannot agree with each other while both being wrong.
501
+ // REQ-WAIT-FOR's polled tier. Built here for the image and egress preflights' reason: one
502
+ // deployment value, one place, so the gate that refuses an undeclared profile and the spawn that
503
+ // runs it cannot disagree about which checks exist. The env-declared table is parsed once at boot
504
+ // (it is env, not overlay -- the gate reads its config above the per-job settings read).
505
+ // The free half of the profile check: whether this deployment declares the name at all. A table
506
+ // lookup, so it belongs with the gate's other free refusals rather than inside the subprocess.
507
+ waitProfileDeclared: (name) => typeof config.waitProfiles[name] === "string",
508
+ checkWait: makeWaitChecker({ profiles: config.waitProfiles, timeoutMs: config.waitCheckTimeoutMs, log }),
509
+ isJobLive: async (id) => {
510
+ const held = await runtimeQueue.getJob(id);
511
+ if (!held) return false;
512
+ // EXISTENCE IS NOT LIVENESS, and the difference decides whether a target stays deafened:
513
+ // `removeOnComplete`/`removeOnFail` keep a finished job's hash for 31 days, so a holder that
514
+ // can never wake again would answer "still waiting" for a month. Only a state it can still be
515
+ // picked up from counts.
516
+ const state = await held.getState();
517
+ return state === "delayed" || state === "waiting" || state === "active" || state === "prioritized" || state === "waiting-children";
518
+ },
432
519
  // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
433
520
  // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
434
521
  // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
@@ -595,6 +682,11 @@ export async function startWorker(
595
682
  watchPauseWindowsFile(config, pauseWindows, log);
596
683
  }
597
684
 
685
+ // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
686
+ if (config.scopedLimitsFile) {
687
+ watchScopedLimitsFile(config, scopedLimits, log);
688
+ }
689
+
598
690
  log("worker_started", {
599
691
  queue: "pi-jobs",
600
692
  concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
@@ -602,6 +694,8 @@ export async function startWorker(
602
694
  weeklyCap: config.weeklyCap, // null when the weekly window is disabled
603
695
  monthlyCap: config.monthlyCap, // null when the monthly window is disabled
604
696
  softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
697
+ scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
698
+ scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
605
699
  image: config.jobImage,
606
700
  valkey: config.valkeyUrl,
607
701
  logsDir: config.logsDir,
@@ -376,6 +376,83 @@ export function makeCheckOnceSpent({ triggersPath, fs = nodeFs }) {
376
376
  };
377
377
  }
378
378
 
379
+ /**
380
+ * Detect the version skew `run.waitFor` opens (issue #230), pre-spend, from the worker's own file read.
381
+ *
382
+ * The hazard is `DES-TRIGGERS-UNIFIED-FILE`'s widening rule arriving somewhere it has never bitten. Unknown
383
+ * keys DROP, which for every previous field was a harmless no-op: an old parser meeting `run.image` gives
384
+ * you the default image and a job that ran. `waitFor` is the first field whose ABSENCE is destructive -- a
385
+ * receiver below the floor enqueues the job without it and the worker runs it UNHELD, producing a record, a
386
+ * panel row and a log line byte-identical to one that correctly waited. Success is the least detectable
387
+ * failure available, and the whole point of a wait is that running now is the destructive option.
388
+ *
389
+ * `docs/secrets.md`'s answer to the same skew is documentation plus a version floor, and it is not enough
390
+ * here for two reasons: a dropped secret surfaces as a 401 and an agent report that reads wrong, and
391
+ * `doctor` cannot see the receiver's installed version from the worker host, so its warning would fire on
392
+ * every deployment using the feature, forever -- the always-on amber the panel's own design rejects.
393
+ *
394
+ * So the worker checks the authored file itself. It already reads it per job for the one-shot gate, and in
395
+ * the compose topology this read is the authoritative one: the receiver's single-file `:ro` mount pins a
396
+ * dead inode until restart, which is precisely the deployment where the skew bites.
397
+ *
398
+ * FAIL-OPEN throughout, `readDisarmState`'s posture and for its reason: an unreadable or changed file means
399
+ * "run", because a broken read must never wedge every job. Only a positive, identity-confirmed mismatch --
400
+ * this entry authored conditions, this job carries none -- refuses.
401
+ */
402
+ export function makeCheckWaitSkew({ triggersPath, fs = nodeFs }) {
403
+ // Cached by mtime, unlike `checkOnceSpent` which re-reads every time. The difference is which jobs each
404
+ // one runs for: the one-shot check is gated on `matched.once === true`, so it is rare by construction and
405
+ // its comment can call one read cheap. This one has to look at EVERY forge job, because the whole point
406
+ // is to catch a job that arrived WITHOUT the field, and there is nothing on such a job to narrow by. So a
407
+ // stat replaces a read-and-parse on the hot path. The residual is millisecond mtime granularity: two
408
+ // writes inside one millisecond would serve a stale parse for one job, which fails OPEN like every other
409
+ // uncertainty here.
410
+ let cache = null; // { mtimeMs, size, triggers }
411
+ const read = () => {
412
+ try {
413
+ const st = fs.statSync?.(triggersPath);
414
+ if (cache && st && cache.mtimeMs === st.mtimeMs && cache.size === st.size) return cache.triggers;
415
+ const raw = JSON.parse(fs.readFileSync(triggersPath, "utf8"));
416
+ const triggers = Array.isArray(raw?.triggers) ? raw.triggers : null;
417
+ if (st) cache = { mtimeMs: st.mtimeMs, size: st.size, triggers };
418
+ return triggers;
419
+ } catch {
420
+ cache = null;
421
+ return null;
422
+ }
423
+ };
424
+
425
+ return async function checkWaitSkew(job) {
426
+ if (typeof triggersPath !== "string" || triggersPath === "") return { ok: true };
427
+ // Cron and CLI jobs carry no matched index, and cannot carry `waitFor` at all.
428
+ const index = job?.trigger?.matched?.index;
429
+ if (!Number.isInteger(index)) return { ok: true };
430
+ // The field arrived. Whether its conditions are SATISFIED is the gate's business, not this check's.
431
+ if (Array.isArray(job?.waitFor) && job.waitFor.length > 0) return { ok: true };
432
+
433
+ const triggers = read();
434
+ if (triggers === null) return { ok: true };
435
+ const entry = triggers[index];
436
+ const authored = entry?.run?.waitFor;
437
+ if (!Array.isArray(authored) || authored.length === 0) return { ok: true };
438
+
439
+ // The identity guard readDisarmState keeps, and WIDER than flow alone. `triggerIndex` is a RAW array
440
+ // position, so an insertion, a reorder, or a stray `triggers.json` at the worker's cwd can all put a
441
+ // different trigger here -- and two rules sharing a flow are indistinguishable on flow alone, which
442
+ // made a false refusal reachable by ordinary editing. Kind and on-type are compared too, on
443
+ // `readDisarmState`'s precedent of checking the item number rather than trusting the index.
444
+ //
445
+ // Every mismatch folds to "run", never to a refusal: the true answer to "is this job missing
446
+ // conditions someone wrote for it?" is then unknown, and unknown must not refuse a paid delivery.
447
+ if (entry?.run?.flow !== job?.flow || entry?.run?.command !== job?.command) return { ok: true };
448
+ if (entry?.run?.kind !== job?.kind) return { ok: true };
449
+ const onType = job?.trigger?.matched?.type;
450
+ if (typeof onType === "string" && entry?.on?.type !== onType) return { ok: true };
451
+
452
+ return { skewed: true, conditions: authored.length };
453
+ };
454
+ }
455
+
379
456
  export function readDisarmState({ triggersPath, index, number, flow, command, fs = nodeFs }) {
380
457
  let raw;
381
458
  try {