@edgehero/pi-dispatch 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -74,6 +74,8 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
74
74
  # Unset = cron disabled for the worker; the receiver falls back to ./triggers.json in the folder it starts from (what `pi-dispatch init` scaffolds)
75
75
  # and refuses to start when neither exists (it holds the label/comment/pull_request trigger config)
76
76
  # PI_PAUSE_WINDOWS_FILE= # ABSOLUTE path to pause-windows.json — "quiet hours" per folder/repo (pause runs between certain times/days/dates, auto-resume). Unset = feature off. See docs/pause-windows.md
77
+ # PI_SCOPED_LIMITS_FILE= # ABSOLUTE path to scoped-limits.json — per repo/folder job-count caps (day/week/month, refused pre-spend as scope-cap) and max concurrent jobs per scope (excess deferred, never dropped).
78
+ # Unset = no scoped caps or concurrency; the one-job-per-folder mutex for local jobs is always on and needs no file. See docs/scoped-limits.md
77
79
  # PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
78
80
  # Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
79
81
  # PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.4.0",
3
+ "version": "1.5.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -55,6 +55,7 @@
55
55
  "./triggers-file": "./src/triggers-file.mjs",
56
56
  "./packages": "./src/packages.mjs",
57
57
  "./pause-windows": "./src/pause-windows.mjs",
58
+ "./scoped-limits": "./src/scoped-limits.mjs",
58
59
  "./identity": "./src/identity.mjs",
59
60
  "./gitlab-identity": "./src/gitlab-identity.mjs",
60
61
  "./forgejo-identity": "./src/forgejo-identity.mjs",
package/src/config.mjs CHANGED
@@ -244,6 +244,7 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
244
244
  sandboxIdleMinutes: nonNegativeInt(env, "PI_SANDBOX_IDLE_MINUTES", 30), // bash's own TMOUT inside a sandbox; 0 = no idle logout
245
245
  triggersFile: env.PI_TRIGGERS_FILE ?? null, // DES-CRON-VIA-BULLMQ-SCHEDULER: unified triggers file; null = cron disabled for the worker (it selects on.type:"cron")
246
246
  pauseWindowsFile: env.PI_PAUSE_WINDOWS_FILE ?? null, // REQ-SCOPED-PAUSE-WINDOWS: per-folder/repo timed pause; null = no scoped pauses
247
+ scopedLimitsFile: env.PI_SCOPED_LIMITS_FILE ?? null, // issue #242: per-scope run caps + concurrency (INT-SCOPED-LIMITS-FILE-CONTRACT); null = none. The one-job-per-folder mutex for local jobs is code, not configuration, and holds regardless
247
248
  schedulerStallMax: positiveInt(env, "PI_SCHEDULER_STALL_MAX", 2), // CONST-RETRY-INFRA-ONLY: per-scheduler stall backstop; positiveInt rejects <1 so a 0 threshold fails closed
248
249
  logsDir: env.PI_LOGS_DIR || defaultLogsDir(), // || (not ??) so an empty string falls back to the default
249
250
  settingsFile: env.PI_SETTINGS_FILE || defaultSettingsFile(), // || (not ??) so an empty string falls back; INT-CONFIG-OVERLAY-CONTRACT
package/src/doctor.mjs CHANGED
@@ -50,6 +50,7 @@ import { dirname, join, delimiter } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
52
  import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
53
+ import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
53
54
  import { isForgeKind } from "./forges.mjs";
54
55
  import { findLiteralSecret, ADMIN_RE } from "./import-pi.mjs";
55
56
  import { agentDirFrom, readHostPi } from "./host-pi.mjs";
@@ -264,7 +265,8 @@ export async function collectChecks(env, seams) {
264
265
  // image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
265
266
  // `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
266
267
  // run.packages: true, which arms nothing any more but is still an operator statement of intent.
267
- const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
268
+ const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, folders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
269
+ const scopedLimitFacts = readScopedLimitFacts(env, fileExists);
268
270
  // FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
269
271
  // zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
270
272
  // file unconditionally and refuses to start without it, which is the consequence worth naming.
@@ -1308,6 +1310,58 @@ export async function collectChecks(env, seams) {
1308
1310
  }
1309
1311
  }
1310
1312
 
1313
+ // The same trap, scoped-limits edition (issue #242): init scaffolds ./scoped-limits.json, the admin
1314
+ // defaults to it, and the worker reads only PI_SCOPED_LIMITS_FILE. The label's mutex parenthetical is
1315
+ // load-bearing -- the check must not imply local folders run ungated when the file is off.
1316
+ {
1317
+ const scopedLimitsFile = env.PI_SCOPED_LIMITS_FILE;
1318
+ const scaffolded = join(cwd, "scoped-limits.json");
1319
+ if ((typeof scopedLimitsFile !== "string" || scopedLimitsFile.trim() === "") && fileExists(scaffolded)) {
1320
+ checks.push({
1321
+ ok: false,
1322
+ warn: true,
1323
+ label: `${scaffolded} exists but PI_SCOPED_LIMITS_FILE is unset -- the worker ignores it, so scoped caps and concurrency are OFF (the built-in one-job-per-folder mutex stays on)`,
1324
+ fix: `set PI_SCOPED_LIMITS_FILE=${scaffolded} in .env and restart the worker -- unset means the worker enforces no scoped limits at all, while the admin panel defaults to this same file and reports each limit it writes as applied live; delete the file if this deployment has no scoped limits`,
1325
+ });
1326
+ }
1327
+ }
1328
+
1329
+ // Issue #242: a CONFIGURED scoped-limits file is boot-load fail-loud, so a file that does not load
1330
+ // refuses the next worker start -- doctor says it before the restart does. Never-tier: doctor never
1331
+ // rewrites limits content (DES-CLI-SURFACE).
1332
+ if (scopedLimitFacts.path !== null && scopedLimitFacts.parseError !== null) {
1333
+ checks.push({
1334
+ ok: false,
1335
+ label: `scoped-limits file does not load -- the worker will refuse to start: ${scopedLimitFacts.parseError}`,
1336
+ fix: `fix ${scopedLimitFacts.path} by hand, or through the dispatch_limit_* tools / the panel's m key once it parses again -- doctor never rewrites limits content`,
1337
+ });
1338
+ }
1339
+
1340
+ // The dead-scope advisory (issue #242), honest about what doctor can actually judge. A forge repo
1341
+ // always contains "/" and never begins "/", "./" or "../" or carries a backslash, so a scope in any
1342
+ // of THOSE shapes can only ever be a folder -- and a folder row that matches no trigger's canonical
1343
+ // run.folder guards nothing. Rows that COULD be a repo (an "a/b" shape) stay silent, not caveated:
1344
+ // webhook jobs carry their repo in the delivery, which triggers.json cannot enumerate, so a line on
1345
+ // every legitimate repo cap would be standing noise that teaches skimming (`repositories` is empty
1346
+ // for every valid file today -- run.repository is azure-only, its own fact says so). Guarded on the
1347
+ // TRIGGERS facts being readable too: a zeroed `folders` from an absent or unparseable triggers file
1348
+ // has no honest claim to make (readTriggerFacts' own rule). ok:true -- the replica advisory's tier,
1349
+ // and like it, everything the operator needs lives in the LABEL: an ok:true check never prints its
1350
+ // fix line.
1351
+ if (scopedLimitFacts.parseError === null && scopedLimitFacts.limits.length > 0 && parseError === null && triggersFilePath !== null) {
1352
+ const folderSet = new Set(folders);
1353
+ const folderOnly = (s) => s.startsWith("/") || s.startsWith("./") || s.startsWith("../") || s.includes("\\") || !s.includes("/") || /^[A-Za-z]:/.test(s);
1354
+ const dead = scopedLimitFacts.limits.map((l) => l.scope).filter((s) => folderOnly(s) && !folderSet.has(s));
1355
+ if (dead.length > 0) {
1356
+ checks.push({
1357
+ ok: true,
1358
+ warn: true,
1359
+ label: `${dead.length} scoped limit(s) name a folder no trigger runs in (${dead.join(", ")}) -- the cap guards nothing; scopes match exactly (no globs, folders by resolved ABSOLUTE path), so check the spelling against triggers.json run.folder or delete the entry`,
1360
+ fix: `edit ${scopedLimitFacts.path} by hand or via dispatch_limit_edit/_delete -- repo-shaped scopes are never flagged here, because a webhook job's repo comes from the delivery, which triggers.json cannot enumerate`,
1361
+ });
1362
+ }
1363
+ }
1364
+
1311
1365
  // REQ-RESURRECTABLE-SANDBOX. A warning, never a failure: retention is a convenience, and the only thing
1312
1366
  // worth surfacing is that finished runs' directories -- a repository clone plus the run's prompt.md and
1313
1367
  // event.json, so issue text -- are sitting on disk, and how many. An operator who never opens a sandbox
@@ -1496,8 +1550,30 @@ function parseSecretProfilesSafe(raw) {
1496
1550
  }
1497
1551
  }
1498
1552
 
1553
+ /**
1554
+ * The scoped-limits facts (issue #242): the parsed rows when PI_SCOPED_LIMITS_FILE is set, or the
1555
+ * boot-blocking reason when it will not load. Unset is `none` -- the worker enforces no scoped limits
1556
+ * and doctor has nothing to say (the mutex is code and needs no check). A configured-but-missing file
1557
+ * IS a parseError here: loadScopedLimits refuses boot on it, so doctor must too. Raw fs errors
1558
+ * (EACCES, EISDIR) are reported the same way, deliberately unlike readTriggerFacts' tagged-only
1559
+ * filter: the worker's own boot load is an unguarded readFileSync, so those throws refuse startup
1560
+ * exactly as a parse failure does, and the check's claim is "will the worker start", not "is the
1561
+ * content valid".
1562
+ */
1563
+ function readScopedLimitFacts(env, fileExists) {
1564
+ const none = { limits: [], parseError: null, path: null };
1565
+ const path = env.PI_SCOPED_LIMITS_FILE;
1566
+ if (typeof path !== "string" || path.trim() === "") return none;
1567
+ if (!fileExists(path)) return { limits: [], parseError: `scoped-limits file does not exist: ${path}`, path };
1568
+ try {
1569
+ return { limits: parseScopedLimits(readFileSync(path, "utf8"), path), parseError: null, path };
1570
+ } catch (e) {
1571
+ return { limits: [], parseError: e?.message ?? String(e), path };
1572
+ }
1573
+ }
1574
+
1499
1575
  function readTriggerFacts(env, fileExists, cwd) {
1500
- const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1576
+ const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], folders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1501
1577
  try {
1502
1578
  // Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
1503
1579
  // (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
@@ -1539,6 +1615,10 @@ function readTriggerFacts(env, fileExists, cwd) {
1539
1615
  // read-write with no clone, so a credential an agent writes into .env lands in the operator's real
1540
1616
  // repository rather than a temp dir that gets swept. Deduped for skillsDirs' reason.
1541
1617
  localSecretFolders: [...new Set(triggers.filter((t) => t.run.secrets !== undefined && t.run.kind === "local" && typeof t.run.folder === "string").map((t) => t.run.folder))].sort(),
1618
+ // Issue #242: every local run.folder, CANONICALIZED the way the scoped-limits matcher
1619
+ // canonicalizes a job's folder (one derivation -- canonicalScope, never re-spelled here), so
1620
+ // the unreferenced-scope advisory compares like with like across spelling variants.
1621
+ folders: [...new Set(triggers.filter((t) => t.run.kind === "local" && typeof t.run.folder === "string").map((t) => canonicalScope({ kind: "local", folder: t.run.folder })))].sort(),
1542
1622
  optingOut: triggers.filter((t) => t.run.packages === false).length,
1543
1623
  images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
1544
1624
  // REQ-PER-TRIGGER-SKILLS. The distinct host directories the file names, deduped like `images`,
@@ -1569,6 +1649,11 @@ function readTriggerFacts(env, fileExists, cwd) {
1569
1649
  packages: t.run.packages !== false,
1570
1650
  }))
1571
1651
  .filter((f) => typeof f.flow === "string"),
1652
+ // Explicit on the success path too (issue #242): the dead-scope advisory distinguishes
1653
+ // "facts read clean" (path set, no error) from the zeroed `none` -- an implicit undefined
1654
+ // here made that test silently false for every deployment.
1655
+ parseError: null,
1656
+ path,
1572
1657
  };
1573
1658
  } catch (e) {
1574
1659
  // REPORTED, not swallowed. This catch used to justify itself with "a malformed triggers file already
package/src/index.mjs CHANGED
@@ -2,11 +2,19 @@ import { execFile } from "node:child_process";
2
2
  import { promisify } from "node:util";
3
3
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
4
4
  import { InfraRetry, runJob } from "./processor.mjs";
5
+ import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
5
6
 
6
7
  const exec = promisify(execFile);
7
8
 
8
9
  export const QUEUE = "pi-jobs";
9
10
  export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
11
+ // The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
12
+ // JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
13
+ // (<=360 wakes across a 30-minute hold, each ~1ms of synchronous predicate briefly occupying a slot)
14
+ // while a same-folder CHAINED job -- enqueued by its parent before the parent's finally releases the
15
+ // folder -- pays exactly one re-check, not fifteen seconds of dead air. No jitter: one worker per
16
+ // docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
17
+ export const SCOPE_BUSY_RECHECK_MS = 5_000;
10
18
 
11
19
  /**
12
20
  * Build the BullMQ processor.
@@ -27,7 +35,7 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
27
35
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
28
36
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
29
37
  */
30
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
38
+ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
31
39
  return async function processor(job, token, signal) {
32
40
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
33
41
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -45,19 +53,68 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
45
53
  throw new DelayedError();
46
54
  }
47
55
 
48
- const startedAt = new Date().toISOString();
49
- const name = `pi-job-${job.id}`;
50
- const timer = setTimeout(() => {
51
- // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
52
- Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
53
- }, timeoutMs);
56
+ // Per-scope concurrency and the one-job-per-folder mutex (issue #242,
57
+ // INT-SCOPED-LIMITS-FILE-CONTRACT). SECOND, after the pause gate (a paused job must not burn
58
+ // re-check wakes) and STRICTLY above the `try` below, like the pause gate and for the same two
59
+ // reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
60
+ // catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
61
+ // handling exactly as the pause gate's does (inside the try it would become a permanent failure
62
+ // plus a failure record for what was a transient blip). The limits snapshot is read ONCE here and
63
+ // shared with `scopedCaps` below, so the gate and the money ledger cannot disagree mid-job.
64
+ // tryAcquire is a synchronous check-and-increment -- no await between read and take, so Node's
65
+ // single thread makes it atomic at any concurrency -- and the local-folder limit is a structural 1
66
+ // (concurrencyFor) with no file and no off-switch: the scheduler mints a cron trigger's next
67
+ // occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
68
+ // (measured: 301ms of live container overlap through this very processor) unless this gate holds.
69
+ // Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
70
+ const limits = scopedLimits();
71
+ const scope = canonicalScope(job.data);
72
+ let held = false;
73
+ if (scope) {
74
+ if (!inFlight.tryAcquire(scope, concurrencyFor(job.data, limits))) {
75
+ // Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
76
+ // The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
77
+ // host path); the delayed count and the job id are what an operator needs to see it.
78
+ deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
79
+ await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
80
+ throw new DelayedError();
81
+ }
82
+ held = true;
83
+ }
54
84
 
55
- // Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
56
- // after the grace period; the runner exits and runContainer returns/throws.
57
- const onAbort = () => {
58
- Promise.resolve(stopContainer(name)).catch(() => {});
59
- };
60
- signal.addEventListener("abort", onAbort, { once: true });
85
+ let startedAt;
86
+ let name;
87
+ let timer;
88
+ let onAbort;
89
+ try {
90
+ // Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
91
+ // finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
92
+ // scope until a worker restart. Nothing in this block CAN throw today (setTimeout and
93
+ // addEventListener on the bullmq-allocated controller are total at processor arity 3); the
94
+ // guard is structural, not observational.
95
+ startedAt = new Date().toISOString();
96
+ name = `pi-job-${job.id}`;
97
+ timer = setTimeout(() => {
98
+ // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
99
+ Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
100
+ }, timeoutMs);
101
+
102
+ // Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
103
+ // after the grace period; the runner exits and runContainer returns/throws.
104
+ onAbort = () => {
105
+ Promise.resolve(stopContainer(name)).catch(() => {});
106
+ };
107
+ signal.addEventListener("abort", onAbort, { once: true });
108
+ } catch (error) {
109
+ // Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
110
+ // scope must never be releasable twice -- a double release frees another holder's slot.
111
+ if (held) {
112
+ inFlight.release(scope);
113
+ held = false;
114
+ }
115
+ clearTimeout(timer);
116
+ throw error;
117
+ }
61
118
 
62
119
  try {
63
120
  const settings = await getSettings();
@@ -99,6 +156,10 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
99
156
  // The daily TOKEN cap (issue #25), same overlay > env resolution. Check-AFTER, so it gates the
100
157
  // NEXT job on prior recorded spend; null => the daily token counter is disabled.
101
158
  tokenCap: settings.dailyTokenCap,
159
+ // This job's scoped budget windows (issue #242), from the SAME limits snapshot the pickup
160
+ // gate above read -- one read per pickup, so gate and ledger agree for this job's whole
161
+ // life. Null when no row carries a money window for this scope.
162
+ scopedCaps: budgetCapsFor(job.data, limits),
102
163
  ...deps,
103
164
  runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
104
165
  // REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
@@ -140,13 +201,16 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
140
201
  // records it as failed-and-distinct in the queue's failed set without a retry.
141
202
  throw new UnrecoverableError(error.message);
142
203
  } finally {
204
+ // Release FIRST and never throw (release clamps at zero by construction): a throw here would
205
+ // mask the job's real error, and a missed release wedges the scope until a worker restart.
206
+ if (held) inFlight.release(scope);
143
207
  clearTimeout(timer);
144
208
  signal.removeEventListener("abort", onAbort);
145
209
  }
146
210
  };
147
211
  }
148
212
 
149
- export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, extraClosers = [] }) {
213
+ export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
150
214
  let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
151
215
  const processor = makeProcessor({
152
216
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
@@ -159,6 +223,11 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
159
223
  if (Number.isInteger(n) && worker.concurrency !== n) worker.concurrency = n;
160
224
  },
161
225
  pauseUntil,
226
+ // Undefined pass-throughs take makeProcessor's own defaults (no limits; a fresh per-processor
227
+ // in-flight map -- one per worker process, which under DES-CONCURRENCY-3's one-worker-per-daemon
228
+ // shape means one per daemon).
229
+ scopedLimits,
230
+ inFlight,
162
231
  deps,
163
232
  recordRun,
164
233
  });
package/src/init.mjs CHANGED
@@ -19,6 +19,11 @@ const EMPTY_PACKAGES = `${JSON.stringify({ packages: [] }, null, 2)}\n`;
19
19
  // Operator-declared subscription plans (issue #53), read by the admin extension only — never at job
20
20
  // time. Versioned because a newer file must fail loud, and that cannot be retrofitted into a v1 reader.
21
21
  const EMPTY_SUBSCRIPTIONS = `${JSON.stringify({ version: 1, subscriptions: [] }, null, 2)}\n`;
22
+ // Scoped limits (issue #242): per repo/folder run caps and concurrency. Empty is inert -- and the
23
+ // one-job-per-folder mutex for local jobs is code, not configuration, so it needs no scaffold line.
24
+ // Versioned for the subscriptions reason, sharpened: this is enforcement config, and a silently
25
+ // down-read newer file would be a silently widened spend limit.
26
+ const EMPTY_SCOPED_LIMITS = `${JSON.stringify({ version: 1, limits: [] }, null, 2)}\n`;
22
27
  /**
23
28
  * The egress allowlist (REQ-EGRESS-ALLOWLIST): the hosts a job container may reach, one bare hostname per
24
29
  * line. Scaffolded with the three a job cannot work without, and NOT empty -- unlike every other scaffold
@@ -73,6 +78,7 @@ export function runInit(cwd = process.cwd(), deps = {}) {
73
78
  scaffold(fs, results, join(cwd, "pause-windows.json"), EMPTY_PAUSE_WINDOWS, "empty pause-windows list");
74
79
  scaffold(fs, results, join(cwd, "pi-packages.json"), EMPTY_PACKAGES, "empty pi package list (stage with import-pi --with-packages)");
75
80
  scaffold(fs, results, join(cwd, "subscriptions.json"), EMPTY_SUBSCRIPTIONS, "empty subscription list (declare plan prices for the admin's cost analytics)");
81
+ scaffold(fs, results, join(cwd, "scoped-limits.json"), EMPTY_SCOPED_LIMITS, "empty scoped-limits list (per repo/folder caps; the folder mutex needs no file)");
76
82
  scaffold(fs, results, join(cwd, "egress-allowlist.conf"), DEFAULT_EGRESS_ALLOWLIST, "egress allowlist (provider + forge + registry; the egress policy is on unless PI_EGRESS=0)");
77
83
 
78
84
  for (const [verb, name, note] of results) {
package/src/processor.mjs CHANGED
@@ -1,6 +1,7 @@
1
1
  import { lstatSync } from "node:fs";
2
2
  import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
3
3
  import { configError } from "./config.mjs";
4
+ import { scopeKeyPrefix } from "./scoped-limits.mjs";
4
5
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
5
6
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
6
7
 
@@ -37,6 +38,12 @@ export async function runJob(job, deps) {
37
38
  caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
38
39
  softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
39
40
  tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
41
+ // { scope, caps: { day, week, month } } | null -- this job's scoped budget windows (issue #242,
42
+ // INT-SCOPED-LIMITS-FILE-CONTRACT), resolved by the wiring from the same watched-limits snapshot the
43
+ // pickup gate read. Null when the file is unset or the scope's row is concurrency-only; the default
44
+ // keeps an unwired processor byte-identical. The folder MUTEX does not live here -- it is the pickup
45
+ // gate's, pre-everything; this is only the money half.
46
+ scopedCaps = null,
40
47
  recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
41
48
  // (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
42
49
  // this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
@@ -124,6 +131,7 @@ export async function runJob(job, deps) {
124
131
  let token = null;
125
132
  let prepared = null;
126
133
  let reserved = false;
134
+ let scopedReserved = false;
127
135
 
128
136
  try {
129
137
  // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
@@ -458,11 +466,51 @@ export async function runJob(job, deps) {
458
466
  return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
459
467
  }
460
468
 
461
- // Budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
469
+ // Per-scope budget windows (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): the NARROWER ledger
470
+ // reserves FIRST, so a noisy scope's refusals never consume a global slot -- the global INCR below
471
+ // runs only for jobs the scope admitted. Same atomic INCR, same refused-still-counts invariant,
472
+ // through budget.mjs's keyPrefix seam (dayKey/weekKey/monthKey under budget:s:<hash16>). softHoldPct
473
+ // is deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob;
474
+ // scoped windows are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
475
+ if (scopedCaps) {
476
+ // A redis fault BETWEEN this reserve and the global one below strands the scoped INCR with no
477
+ // run and no refund -- the pre-existing mid-reserve posture, shared with the global ledger's
478
+ // own partial-INCR seam; the compensating release below covers REFUSALS, not faults.
479
+ const scoped = await reserveBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
480
+ scopedReserved = true;
481
+ if (!scoped.allowed) {
482
+ const w = scoped.blockedWindow;
483
+ const win = scoped.windows[w];
484
+ // A local job's scope is a full host path and its "comment" is not dropped -- the wiring's
485
+ // local adapter LOGS the text (start.mjs forgeFor fallthrough) -- so the path must never
486
+ // enter the message; "this folder" is enough beside the jobId the adapter logs. A forge
487
+ // scope IS the repo the comment posts on, safe to name.
488
+ const scopeLabel = job.kind === "local" ? "this folder" : scopedCaps.scope;
489
+ await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
490
+ // The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would
491
+ // put a full host path in the worker log against no-pii-in-logs (the record keeps only
492
+ // basename(folder) for the same reason). The admin recomputes the key from the configured
493
+ // scope to join it back.
494
+ log("over_scope_budget", { scopeKey: scopeKeyPrefix(scopedCaps.scope), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
495
+ // budgetReserved false: the GLOBAL slot was never touched (scoped reserves first). The scoped
496
+ // counter did INCR and keeps it -- its own refused-reservation-still-counts, per ledger.
497
+ return { outcome: "policy", reason: "scope-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
498
+ }
499
+ }
500
+
501
+ // GLOBAL budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
462
502
  // every active window (day + optional week/month) and the soft-hold band in one atomic pass.
463
503
  const budget = await reserveBudget(redis, { caps, softHoldPct, now });
464
504
  reserved = true;
465
505
  if (!budget.allowed) {
506
+ // The scoped reserve above committed before this global refusal -- give that slot back. Without
507
+ // this, an exhausted global window drains every arriving scope's own day/week/month counters
508
+ // with zero runs to show for it (a storm against a spent global daily cap would empty a repo's
509
+ // week by noon). The scoped ledger's refused-still-counts covers the SCOPE's own refusal above,
510
+ // never a refusal it did not issue.
511
+ if (scopedReserved && scopedCaps) {
512
+ await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
513
+ }
466
514
  const w = budget.blockedWindow;
467
515
  const win = budget.windows[w];
468
516
  if (budget.reason === "soft-hold") {
@@ -561,8 +609,12 @@ export async function runJob(job, deps) {
561
609
  // budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
562
610
  // true for a real container that ran and spent (exit-1 infra / unknown exit).
563
611
  if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
564
- if (reserved && e instanceof InfraRetry && e.reason === "container-never-started") {
565
- await releaseBudget(redis, { caps, now });
612
+ if (e instanceof InfraRetry && e.reason === "container-never-started") {
613
+ // Both-or-neither (issue #242): a never-started container can only follow BOTH reserves (the
614
+ // scoped one precedes the global one, and the container follows both), so they refund
615
+ // together -- and a scoped refusal returned above without ever touching the global ledger.
616
+ if (reserved) await releaseBudget(redis, { caps, now });
617
+ if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
566
618
  }
567
619
  throw e;
568
620
  } finally {
@@ -0,0 +1,277 @@
1
+ /**
2
+ * Scoped limits (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): per-scope run caps and per-scope
3
+ * concurrency, where a scope is what `scopeOf` already answers -- the folder for a local job, the repo
4
+ * for a forge one. One `scoped-limits.json` of `{ scope, day?, week?, month?, concurrent? }` entries:
5
+ * the day/week/month caps refuse a job pre-spend (reason `scope-cap`, a policy refusal), `concurrent`
6
+ * defers the excess through the delayed set (never a refusal -- a busy scope is transient state).
7
+ *
8
+ * This module is pure and fs-injectable (mirrors pause-windows.mjs in every respect): `parseScopedLimits`
9
+ * validates the file TEXT fail-loud, `loadScopedLimits` layers the one fs read on top, and the small
10
+ * helpers below are what the processor gate and the budget wiring consume. It also owns the in-process
11
+ * in-flight counter (`makeInFlight`) so the counter is unit-testable without a bullmq import, the same
12
+ * reason job-id.mjs is queue-free. The wiring that consumes all of this (the gate, the budget calls,
13
+ * the admin surfaces) lands in this issue's later slices; the module ships first so the contract has
14
+ * one implementation to bind to -- sentences below describing enforcement describe THOSE slices.
15
+ *
16
+ * The file is a SIBLING of pause-windows.json, not part of the settings overlay, deliberately: the
17
+ * deferral gate runs before the per-job overlay read, so gate-read config must come from a watched
18
+ * mutable ref, and the overlay's KNOWN_KEYS are flat scalars whose only map-shaped precedent
19
+ * (secretProfiles) is deliberately model-unreachable -- the opposite of what these limits need.
20
+ *
21
+ * `version` is REQUIRED and fail-loud-on-newer (subscriptions.mjs's rule, adopted here because this is a
22
+ * MONEY file): unknown fields are silently dropped per the operator-file policy, so a v2 cap field an old
23
+ * worker drops would be a silently WIDENED spend limit. Pause-windows shipping without a version is a
24
+ * sunk decision, not a precedent to extend to enforcement config.
25
+ *
26
+ * Custom: scoped limits validated inline per triggers.mjs/pause-windows.mjs precedent; zod not in deps
27
+ */
28
+
29
+ import { createHash } from "node:crypto";
30
+ import { existsSync as fsExistsSync, readFileSync as fsReadFileSync } from "node:fs";
31
+ import { isAbsolute, resolve } from "node:path";
32
+ import { configError } from "./config.mjs";
33
+ import { scopeOf } from "./pause-windows.mjs";
34
+
35
+ /** The schema version this build reads and writes. A file declaring a higher one is refused loudly. */
36
+ export const SCOPED_LIMITS_VERSION = 1;
37
+
38
+ /** The four limit fields a row may carry, in display order. */
39
+ const LIMIT_FIELDS = ["day", "week", "month", "concurrent"];
40
+
41
+ function isNonEmptyString(value) {
42
+ return typeof value === "string" && value.trim() !== "";
43
+ }
44
+
45
+ /**
46
+ * The canonical scope string for a job: the RESOLVED folder path for a local job, the repo for a forge
47
+ * one. `scopeOf` alone is not enough for enforcement: nothing on the trigger path normalizes
48
+ * `run.folder`, so `/srv/site`, `/srv/site/`, `/srv//site`, `/srv/x/../site` and a padded spelling are
49
+ * five distinct strings naming ONE directory -- an exact-string mutex keyed on the raw value would run
50
+ * them concurrently in one working tree, which is the exact race the mutex exists to close.
51
+ * `path.resolve` (not `normalize`, which keeps trailing slashes and whitespace) collapses them all; a
52
+ * relative folder resolves against the worker's cwd, the same base `prepareWorkspace`'s existence check
53
+ * uses; Unicode is NFC-normalized on both the job and the row side (see below). Two residuals,
54
+ * deliberate: symlinks are NOT resolved (realpath is an fs call on the hot path and can throw), and
55
+ * neither is filesystem case-insensitivity (on a default macOS/APFS volume `/Srv/Site` and `/srv/site`
56
+ * are one directory and two scopes) -- the pause matcher lives with both.
57
+ *
58
+ * The pause matcher itself keeps the RAW `scopeOf` value: resolving there would silently change which
59
+ * jobs an operator's existing trailing-slash window matches. The two features share the folder-vs-repo
60
+ * split (`scopeOf`, defined once) but not the normalization, and this comment is where that difference
61
+ * is recorded.
62
+ *
63
+ * A useful side effect: a resolved local scope is always an absolute path, and a repo string never is,
64
+ * so a folder named `a/b` and a repo named `a/b` can no longer collide in the counters or the mutex.
65
+ */
66
+ export function canonicalScope(job) {
67
+ const scope = scopeOf(job);
68
+ if (!isNonEmptyString(scope)) return null;
69
+ // NFC on both kinds: macOS's filesystem hands paths back NFD while an admin dialog types NFC, so
70
+ // "wéb" can arrive as two byte sequences naming one thing -- without this, an NFD-spelled forge
71
+ // scope silently escapes an NFC-spelled cap (the local side would at least keep the structural
72
+ // mutex). ASCII is fixed under NFC, so no existing key changes.
73
+ return job?.kind === "local" ? resolve(scope.trim().normalize("NFC")) : scope.normalize("NFC");
74
+ }
75
+
76
+ /**
77
+ * Parse, validate, and normalize the scoped-limits file TEXT. Returns the normalized `limits` array
78
+ * (every row rebuilt as an explicit `{ scope, day, week, month, concurrent }` literal, `null` for absent
79
+ * fields, unknown fields dropped -- the operator-file policy). Throws `configError` (fail-loud) on any
80
+ * malformed entry. `path` is for error messages only -- this function touches no filesystem.
81
+ */
82
+ export function parseScopedLimits(text, path) {
83
+ let parsed;
84
+ try {
85
+ parsed = JSON.parse(text);
86
+ } catch (error) {
87
+ throw configError(`scoped-limits file is not valid JSON: ${path} (${error.message})`);
88
+ }
89
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
90
+ throw configError(`scoped-limits file must be an object with "version" and "limits": ${path}`);
91
+ }
92
+ const version = parsed.version;
93
+ if (!Number.isInteger(version) || version < 1) {
94
+ throw configError(`scoped-limits file must have "version": 1 (an integer >= 1): ${path}`);
95
+ }
96
+ if (version > SCOPED_LIMITS_VERSION) {
97
+ throw configError(`scoped-limits file written by a newer pi-dispatch (version ${version}; this build understands ${SCOPED_LIMITS_VERSION}): ${path}`);
98
+ }
99
+ if (!Array.isArray(parsed.limits)) {
100
+ throw configError(`scoped-limits file must have a "limits" array: ${path}`);
101
+ }
102
+ const rows = parsed.limits.map((row, index) => normalizeLimit(row, index, path));
103
+ const seen = new Map();
104
+ rows.forEach((row, index) => {
105
+ if (seen.has(row.scope)) {
106
+ // Two rows for one scope is a precedence question with no right answer; the admin's
107
+ // edit-in-place never produces one, so a duplicate is always a hand-edit mistake.
108
+ throw configError(`scoped limit at index ${index}: duplicate scope ${JSON.stringify(row.scope)} (first at index ${seen.get(row.scope)}): ${path}`);
109
+ }
110
+ seen.set(row.scope, index);
111
+ });
112
+ return rows;
113
+ }
114
+
115
+ function normalizeLimit(row, index, path) {
116
+ const at = `scoped limit at index ${index}`;
117
+ if (row === null || typeof row !== "object" || Array.isArray(row)) {
118
+ throw configError(`${at}: must be an object: ${path}`);
119
+ }
120
+ if (!isNonEmptyString(row.scope)) throw configError(`${at}: scope must be a non-empty string: ${path}`);
121
+ const trimmed = row.scope.trim().normalize("NFC"); // the same NFC canonicalScope applies job-side
122
+ if (trimmed === "*") {
123
+ // "*" as ONE shared counter is redundant with the global caps, so the only useful reading is a
124
+ // per-scope default -- the OPPOSITE of what "*" means one file over (pause-windows: one rule
125
+ // matching all scopes). Refused rather than shipped divergent; a later version may adopt the
126
+ // per-scope-default reading, with an exact row beating "*" (recorded in the contract).
127
+ throw configError(`${at}: "*" is not supported -- add one row per scope (a per-scope default may adopt "*" later): ${path}`);
128
+ }
129
+ if (trimmed.includes("*")) {
130
+ // No globs, enforced rather than described: an exact matcher makes "acme/*" a row that governs
131
+ // nothing, and a silently inert money limit is the failure class this repo refuses outright.
132
+ throw configError(`${at}: scopes match exactly; a scope containing "*" is refused (no globs): ${path}`);
133
+ }
134
+ const norm = {
135
+ // An absolute path is stored resolved so a `/srv/site/` row governs `/srv/site` jobs -- the same
136
+ // collapse canonicalScope applies on the job side. isAbsolute is PLATFORM-NATIVE on purpose, so a
137
+ // foreign-platform row (a windows drive path on a POSIX worker) stays verbatim and is inert here;
138
+ // the doctor's unreferenced-scope advisory names it. Resolving it instead would "work" only by
139
+ // both sides mangling into the same cwd-prefixed string -- a match by accident, not by contract.
140
+ scope: isAbsolute(trimmed) ? resolve(trimmed) : trimmed,
141
+ day: null,
142
+ week: null,
143
+ month: null,
144
+ concurrent: null,
145
+ };
146
+ let any = false;
147
+ for (const field of LIMIT_FIELDS) {
148
+ const value = row[field];
149
+ // Absent-or-null (subscriptions.mjs's rule): null is the normalizer's OWN output for an unset
150
+ // field, so the parser must accept it back or it cannot re-parse what it produced -- the admin's
151
+ // read-modify-write goes through this parser on both edges.
152
+ if (value === undefined || value === null) continue;
153
+ // 0 is refused, not "never run": budget.mjs's caps treat every configured window as >= 1, and
154
+ // "never run this scope" already has two honest spellings (delete the trigger; a pause window).
155
+ // isSafeInteger, not isInteger: 1e21 passes isInteger and reads as a limit while being
156
+ // indistinguishable from unlimited -- a bound that cannot count is not a bound.
157
+ if (!Number.isSafeInteger(value) || value < 1) {
158
+ throw configError(`${at}: ${field} must be an integer >= 1: ${path}`);
159
+ }
160
+ norm[field] = value;
161
+ any = true;
162
+ }
163
+ if (!any) {
164
+ throw configError(`${at}: at least one of day, week, month, concurrent is required (a row that limits nothing is a row an operator sets and then trusts): ${path}`);
165
+ }
166
+ return norm;
167
+ }
168
+
169
+ /**
170
+ * Load and validate the scoped-limits file named by `config.scopedLimitsFile`. Returns `[]` when the file
171
+ * is unset (no scoped caps or concurrency -- a valid deployment; the folder mutex holds regardless, it is
172
+ * code, not configuration). `readFileSync`/`existsSync` are injectable for tests.
173
+ */
174
+ export function loadScopedLimits(config, { readFileSync = fsReadFileSync, existsSync = fsExistsSync } = {}) {
175
+ const path = config.scopedLimitsFile;
176
+ if (path === null || path === undefined) return [];
177
+ if (!existsSync(path)) throw configError(`scoped-limits file does not exist: ${path}`);
178
+ return parseScopedLimits(readFileSync(path, "utf8"), path);
179
+ }
180
+
181
+ /**
182
+ * The exact-match row for a canonical scope, or null. Exact string equality only -- the pause matcher's
183
+ * semantics minus its "*" (refused above). With duplicates refused there is no precedence ladder.
184
+ */
185
+ export function limitFor(limits, scope) {
186
+ if (!Array.isArray(limits) || !isNonEmptyString(scope)) return null;
187
+ return limits.find((l) => l.scope === scope) ?? null;
188
+ }
189
+
190
+ /**
191
+ * The scoped budget windows this job reserves against, or null when nothing applies (no row for the
192
+ * scope, or the row is concurrency-only). The returned `scope` is CANONICAL so the redis counters are
193
+ * spelling-stable. Shaped like the global `caps` object so `reserveBudget` consumes it unchanged.
194
+ */
195
+ export function budgetCapsFor(job, limits) {
196
+ const scope = canonicalScope(job);
197
+ const row = limitFor(limits, scope);
198
+ if (!row || (row.day === null && row.week === null && row.month === null)) return null;
199
+ return { scope, caps: { day: row.day, week: row.week, month: row.month } };
200
+ }
201
+
202
+ /**
203
+ * The effective in-flight ceiling for this job's scope: `min(configured concurrent, structural)`, where
204
+ * structural is 1 for a local job -- the folder mutex -- and unbounded otherwise. The mutex is
205
+ * UNCONDITIONAL, in code, with no file configured and no off-switch: two agents in one bind-mounted
206
+ * working tree is the race `run.replicas` is already refused on local jobs for, and a cron trigger
207
+ * reaches it with no operator mistake at all (the scheduler mints the next occurrence at pickup and
208
+ * promotes on time alone, so a slow run overlaps its own successor). A configured `concurrent` above 1
209
+ * on a folder scope silently clamps to 1 rather than refusing at parse: scope strings are not reliably
210
+ * typeable as folder-vs-repo (`"a/b"` is a legal relative folder and a legal repo), so a parse-time
211
+ * classifier would misfire; min() cannot.
212
+ *
213
+ * No scope (a malformed payload) means no gate: Infinity, admit -- the job will fail its own validation
214
+ * downstream, and holding a mutex slot under key `null` helps nobody.
215
+ */
216
+ export function concurrencyFor(job, limits) {
217
+ const scope = canonicalScope(job);
218
+ if (scope === null) return Infinity;
219
+ const structural = job?.kind === "local" ? 1 : Infinity;
220
+ const configured = limitFor(limits, scope)?.concurrent ?? Infinity;
221
+ return Math.min(structural, configured);
222
+ }
223
+
224
+ /**
225
+ * The redis key prefix for a scope's budget windows: `budget:s:<16 hex>`. Handed to
226
+ * `reserveBudget`/`releaseBudget` as `keyPrefix`, so `dayKey`/`weekKey`/`monthKey` compose
227
+ * `budget:s:<h>:YYYY-MM-DD` / `:w:...` / `:m:...` with zero new key-shape logic. A hash (the localJobId
228
+ * idiom: sha256, first 16 hex) rather than an escape: a scope legally contains `:` and `/` (folder
229
+ * paths, gitlab group/subgroup/project), which would collide with budget.mjs's own `w:`/`m:`/`t:`
230
+ * sub-namespaces, and a bijective escape grammar is a new thing to get wrong with unbounded key lengths.
231
+ * The one consumer that must map keys BACK to scopes is the admin's counter display, and it knows the
232
+ * configured scopes -- it recomputes keys through this same export, so unreadability in redis-cli is the
233
+ * accepted cost.
234
+ */
235
+ export function scopeKeyPrefix(scope) {
236
+ const h = createHash("sha256").update(String(scope)).digest("hex").slice(0, 16);
237
+ return `budget:s:${h}`;
238
+ }
239
+
240
+ /**
241
+ * The per-process in-flight counter behind per-scope concurrency and the folder mutex. Process memory is
242
+ * the CORRECT store, not a compromise: one worker per docker daemon is the shape DES-CONCURRENCY-3
243
+ * assumes everywhere and `service install` enforces for installed units (`pi-dispatch start` holds no
244
+ * lock, and two hand-run workers are already unsupported -- the second one's boot reaper kills the
245
+ * first's live containers); the reaper removes every surviving `pi-job-*` container before the worker
246
+ * starts draining, so a fresh, empty map is never wrong about a live container except when the reap
247
+ * itself was skipped (`reaper_skipped`: docker missing/down at boot -- a state where no NEW container
248
+ * can start either); and a Redis-held counter would survive a crash WRONGLY -- a claim for a container
249
+ * the reaper just killed, demanding TTL/heartbeat machinery, a second source of truth about "what is
250
+ * running" (the OQ-008 failure mode).
251
+ *
252
+ * `tryAcquire` is a synchronous check-and-increment: no await between the read and the take, so under
253
+ * Node's single thread no interleaving exists at any concurrency. `release` never throws -- it runs in
254
+ * the processor's finally, where a throw would mask the job's real error -- and clamps at zero.
255
+ */
256
+ export function makeInFlight() {
257
+ const counts = new Map();
258
+ return {
259
+ /** True and counted when under `limit`; false WITHOUT counting when at or over it. */
260
+ tryAcquire(scope, limit) {
261
+ const current = counts.get(scope) ?? 0;
262
+ if (current >= limit) return false;
263
+ counts.set(scope, current + 1);
264
+ return true;
265
+ },
266
+ /** Decrement, deleting at zero; a release without a matching acquire is a no-op, never a throw. */
267
+ release(scope) {
268
+ const current = counts.get(scope) ?? 0;
269
+ if (current <= 1) counts.delete(scope);
270
+ else counts.set(scope, current - 1);
271
+ },
272
+ /** The current in-flight count for a scope (tests and future observability). */
273
+ count(scope) {
274
+ return counts.get(scope) ?? 0;
275
+ },
276
+ };
277
+ }
package/src/start.mjs CHANGED
@@ -24,6 +24,7 @@ import { makeSandboxReaper } from "./sandbox-store.mjs";
24
24
  import { makeSessionStore } from "./session-store.mjs";
25
25
  import { makeCheckOnceSpent, makeDisarmOnce } from "./triggers-file.mjs";
26
26
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
27
+ import { loadScopedLimits } from "./scoped-limits.mjs";
27
28
  import { makeQueue } from "./queue.mjs";
28
29
  import { makeRunContainer } from "./run-container.mjs";
29
30
  import { makeSecretsResolver } from "./secrets.mjs";
@@ -103,6 +104,42 @@ function watchPauseWindowsFile(config, ref, log) {
103
104
  }
104
105
  }
105
106
 
107
+ /**
108
+ * The scoped-limits reload, EXPORTED apart from its watcher so keep-last-good is unit-testable without
109
+ * fs.watch (its two watcher siblings above bind theirs inline; this one is money config, so the
110
+ * last-good property carries its own test). A bad edit keeps `ref.current` untouched and logs
111
+ * `scoped_limits_reload_invalid` -- the pause-windows posture, INT-SCOPED-LIMITS-FILE-CONTRACT.
112
+ */
113
+ export function reloadScopedLimits(config, ref, log) {
114
+ try {
115
+ ref.current = loadScopedLimits(config);
116
+ log("scoped_limits_reloaded", { count: ref.current.length });
117
+ } catch (err) {
118
+ log("scoped_limits_reload_invalid", { reason: err?.message });
119
+ }
120
+ }
121
+
122
+ /**
123
+ * Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
124
+ * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort + unref'd.
125
+ */
126
+ function watchScopedLimitsFile(config, ref, log) {
127
+ const path = config.scopedLimitsFile;
128
+ const dir = dirname(path) || ".";
129
+ const file = basename(path);
130
+ let timer = null;
131
+ try {
132
+ watch(dir, (_event, changed) => {
133
+ if (changed && changed !== file) return;
134
+ clearTimeout(timer);
135
+ timer = setTimeout(() => reloadScopedLimits(config, ref, log), 150);
136
+ }).unref?.();
137
+ log("scoped_limits_watching", { path });
138
+ } catch (err) {
139
+ log("scoped_limits_watch_unavailable", { reason: err?.message });
140
+ }
141
+ }
142
+
106
143
  export function makeReaper({ log }) {
107
144
  return async function reap() {
108
145
  try {
@@ -188,6 +225,10 @@ export async function startWorker(
188
225
  // pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
189
226
  const pauseWindows = { current: loadPauseWindows(config) };
190
227
 
228
+ // Issue #242: same posture for the scoped-limits file -- fail-loud with the operator present, mutable
229
+ // ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
230
+ const scopedLimits = { current: loadScopedLimits(config) };
231
+
191
232
  // The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
192
233
  // Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
193
234
  // forge is unconfigured or unreachable), `host` is the three methods github-host.mjs returns. The map
@@ -422,6 +463,9 @@ export async function startWorker(
422
463
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
423
464
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
424
465
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
466
+ // Issue #242: the scoped-limits snapshot the pickup gate and the scoped budget read, once per
467
+ // pickup, from the live-reloaded ref -- same next-job grain as pauseUntil above.
468
+ scopedLimits: () => scopedLimits.current,
425
469
  deps: {
426
470
  collectChain,
427
471
  // The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
@@ -595,6 +639,11 @@ export async function startWorker(
595
639
  watchPauseWindowsFile(config, pauseWindows, log);
596
640
  }
597
641
 
642
+ // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
643
+ if (config.scopedLimitsFile) {
644
+ watchScopedLimitsFile(config, scopedLimits, log);
645
+ }
646
+
598
647
  log("worker_started", {
599
648
  queue: "pi-jobs",
600
649
  concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
@@ -602,6 +651,8 @@ export async function startWorker(
602
651
  weeklyCap: config.weeklyCap, // null when the weekly window is disabled
603
652
  monthlyCap: config.monthlyCap, // null when the monthly window is disabled
604
653
  softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
654
+ scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
655
+ scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
605
656
  image: config.jobImage,
606
657
  valkey: config.valkeyUrl,
607
658
  logsDir: config.logsDir,