@edgehero/pi-dispatch 1.3.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -74,6 +74,8 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
74
74
  # Unset = cron disabled for the worker; the receiver falls back to ./triggers.json in the folder it starts from (what `pi-dispatch init` scaffolds)
75
75
  # and refuses to start when neither exists (it holds the label/comment/pull_request trigger config)
76
76
  # PI_PAUSE_WINDOWS_FILE= # ABSOLUTE path to pause-windows.json — "quiet hours" per folder/repo (pause runs between certain times/days/dates, auto-resume). Unset = feature off. See docs/pause-windows.md
77
+ # PI_SCOPED_LIMITS_FILE= # ABSOLUTE path to scoped-limits.json — per repo/folder job-count caps (day/week/month, refused pre-spend as scope-cap) and max concurrent jobs per scope (excess deferred, never dropped).
78
+ # Unset = no scoped caps or concurrency; the one-job-per-folder mutex for local jobs is always on and needs no file. See docs/scoped-limits.md
77
79
  # PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
78
80
  # Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
79
81
  # PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
@@ -49,7 +49,11 @@ services:
49
49
  PI_TRIGGERS_FILE: /config/triggers.json
50
50
  # The repo-root triggers.json (`pi-dispatch init` scaffolds it -- run that first: mounting a path
51
51
  # that does not exist makes Docker create it as a DIRECTORY and the boot fails confusingly).
52
- # Read-only: the receiver live-reloads this file on change; it never writes it.
52
+ # Read-only: the receiver live-reloads this file on change; it never writes it. One consequence of
53
+ # a single-FILE bind mount (issue #231): every write to this file is an atomic tmp+rename that
54
+ # SWAPS THE INODE, and the mount stays pinned to the old one -- so the worker's one-shot disarm is
55
+ # invisible in here until the container restarts, and the worker's own pre-spend check is what
56
+ # keeps a spent one-shot from running again in the meantime. A restart picks up the current file.
53
57
  volumes:
54
58
  - ../triggers.json:/config/triggers.json:ro
55
59
  # Loopback only, like Valkey's port above: the operator's reverse proxy or tunnel (TLS, public
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.3.0",
3
+ "version": "1.5.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -52,8 +52,10 @@
52
52
  "./job-id": "./src/job-id.mjs",
53
53
  "./forges": "./src/forges.mjs",
54
54
  "./triggers": "./src/triggers.mjs",
55
+ "./triggers-file": "./src/triggers-file.mjs",
55
56
  "./packages": "./src/packages.mjs",
56
57
  "./pause-windows": "./src/pause-windows.mjs",
58
+ "./scoped-limits": "./src/scoped-limits.mjs",
57
59
  "./identity": "./src/identity.mjs",
58
60
  "./gitlab-identity": "./src/gitlab-identity.mjs",
59
61
  "./forgejo-identity": "./src/forgejo-identity.mjs",
package/src/config.mjs CHANGED
@@ -244,6 +244,7 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
244
244
  sandboxIdleMinutes: nonNegativeInt(env, "PI_SANDBOX_IDLE_MINUTES", 30), // bash's own TMOUT inside a sandbox; 0 = no idle logout
245
245
  triggersFile: env.PI_TRIGGERS_FILE ?? null, // DES-CRON-VIA-BULLMQ-SCHEDULER: unified triggers file; null = cron disabled for the worker (it selects on.type:"cron")
246
246
  pauseWindowsFile: env.PI_PAUSE_WINDOWS_FILE ?? null, // REQ-SCOPED-PAUSE-WINDOWS: per-folder/repo timed pause; null = no scoped pauses
247
+ scopedLimitsFile: env.PI_SCOPED_LIMITS_FILE ?? null, // issue #242: per-scope run caps + concurrency (INT-SCOPED-LIMITS-FILE-CONTRACT); null = none. The one-job-per-folder mutex for local jobs is code, not configuration, and holds regardless
247
248
  schedulerStallMax: positiveInt(env, "PI_SCHEDULER_STALL_MAX", 2), // CONST-RETRY-INFRA-ONLY: per-scheduler stall backstop; positiveInt rejects <1 so a 0 threshold fails closed
248
249
  logsDir: env.PI_LOGS_DIR || defaultLogsDir(), // || (not ??) so an empty string falls back to the default
249
250
  settingsFile: env.PI_SETTINGS_FILE || defaultSettingsFile(), // || (not ??) so an empty string falls back; INT-CONFIG-OVERLAY-CONTRACT
package/src/doctor.mjs CHANGED
@@ -50,6 +50,7 @@ import { dirname, join, delimiter } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
52
  import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
53
+ import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
53
54
  import { isForgeKind } from "./forges.mjs";
54
55
  import { findLiteralSecret, ADMIN_RE } from "./import-pi.mjs";
55
56
  import { agentDirFrom, readHostPi } from "./host-pi.mjs";
@@ -264,7 +265,8 @@ export async function collectChecks(env, seams) {
264
265
  // image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
265
266
  // `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
266
267
  // run.packages: true, which arms nothing any more but is still an operator statement of intent.
267
- const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, secretProfiles, localSecretFolders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
268
+ const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, folders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
269
+ const scopedLimitFacts = readScopedLimitFacts(env, fileExists);
268
270
  // FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
269
271
  // zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
270
272
  // file unconditionally and refuses to start without it, which is the consequence worth naming.
@@ -1242,6 +1244,31 @@ export async function collectChecks(env, seams) {
1242
1244
  });
1243
1245
  }
1244
1246
 
1247
+ // One-shot close triggers (issue #231, DES-ONE-SHOT-DISARM-IN-THE-FILE). Advisory only -- doctor
1248
+ // never touches triggers -- and counted from the RAW file (readTriggerFacts says why). Two lines
1249
+ // with different lives: the armed line names the count and, when PI_TRIGGERS_FILE is unset, warns
1250
+ // that the disarm resolves ./triggers.json against the WORKER SERVICE's working directory -- a
1251
+ // service unit whose WorkingDirectory differs from the receiver's would disarm a file nobody
1252
+ // matches against, the split-file hazard no mechanism can detect. The spent line states the
1253
+ // deliberate degradation: a spent entry counts toward NO parsed fact above (forges, flows,
1254
+ // webhook-secret), mirroring what the receiver serves at its next boot.
1255
+ if (onceArmed > 0) {
1256
+ checks.push({
1257
+ ok: true,
1258
+ warn: env.PI_TRIGGERS_FILE === undefined,
1259
+ label: `${onceArmed} one-shot trigger(s) armed (on.once) -- the worker disarms the entry in ${env.PI_TRIGGERS_FILE === undefined ? "./triggers.json resolved against the worker service's working directory; set PI_TRIGGERS_FILE so worker and receiver name the same file from anywhere" : "PI_TRIGGERS_FILE"} after the run record exists`,
1260
+ fix: "set PI_TRIGGERS_FILE to an absolute path in both services' environments",
1261
+ });
1262
+ }
1263
+ if (onceSpent > 0) {
1264
+ checks.push({
1265
+ ok: true,
1266
+ warn: false,
1267
+ label: `${onceSpent} one-shot trigger(s) already spent (on.disarmed) -- spent entries match nothing and count toward no credential or flow check; delete on.disarmed to re-arm, or delete the entry once its history no longer matters`,
1268
+ fix: "",
1269
+ });
1270
+ }
1271
+
1245
1272
  // REQ-SCOPED-PAUSE-WINDOWS, the panel-writes-what-the-worker-ignores trap (issue #99). Three defaults
1246
1273
  // that are individually defensible and together silent:
1247
1274
  //
@@ -1283,6 +1310,58 @@ export async function collectChecks(env, seams) {
1283
1310
  }
1284
1311
  }
1285
1312
 
1313
+ // The same trap, scoped-limits edition (issue #242): init scaffolds ./scoped-limits.json, the admin
1314
+ // defaults to it, and the worker reads only PI_SCOPED_LIMITS_FILE. The label's mutex parenthetical is
1315
+ // load-bearing -- the check must not imply local folders run ungated when the file is off.
1316
+ {
1317
+ const scopedLimitsFile = env.PI_SCOPED_LIMITS_FILE;
1318
+ const scaffolded = join(cwd, "scoped-limits.json");
1319
+ if ((typeof scopedLimitsFile !== "string" || scopedLimitsFile.trim() === "") && fileExists(scaffolded)) {
1320
+ checks.push({
1321
+ ok: false,
1322
+ warn: true,
1323
+ label: `${scaffolded} exists but PI_SCOPED_LIMITS_FILE is unset -- the worker ignores it, so scoped caps and concurrency are OFF (the built-in one-job-per-folder mutex stays on)`,
1324
+ fix: `set PI_SCOPED_LIMITS_FILE=${scaffolded} in .env and restart the worker -- unset means the worker enforces no scoped limits at all, while the admin panel defaults to this same file and reports each limit it writes as applied live; delete the file if this deployment has no scoped limits`,
1325
+ });
1326
+ }
1327
+ }
1328
+
1329
+ // Issue #242: a CONFIGURED scoped-limits file is boot-load fail-loud, so a file that does not load
1330
+ // refuses the next worker start -- doctor says it before the restart does. Never-tier: doctor never
1331
+ // rewrites limits content (DES-CLI-SURFACE).
1332
+ if (scopedLimitFacts.path !== null && scopedLimitFacts.parseError !== null) {
1333
+ checks.push({
1334
+ ok: false,
1335
+ label: `scoped-limits file does not load -- the worker will refuse to start: ${scopedLimitFacts.parseError}`,
1336
+ fix: `fix ${scopedLimitFacts.path} by hand, or through the dispatch_limit_* tools / the panel's m key once it parses again -- doctor never rewrites limits content`,
1337
+ });
1338
+ }
1339
+
1340
+ // The dead-scope advisory (issue #242), honest about what doctor can actually judge. A forge repo
1341
+ // always contains "/" and never begins "/", "./" or "../" or carries a backslash, so a scope in any
1342
+ // of THOSE shapes can only ever be a folder -- and a folder row that matches no trigger's canonical
1343
+ // run.folder guards nothing. Rows that COULD be a repo (an "a/b" shape) stay silent, not caveated:
1344
+ // webhook jobs carry their repo in the delivery, which triggers.json cannot enumerate, so a line on
1345
+ // every legitimate repo cap would be standing noise that teaches skimming (`repositories` is empty
1346
+ // for every valid file today -- run.repository is azure-only, its own fact says so). Guarded on the
1347
+ // TRIGGERS facts being readable too: a zeroed `folders` from an absent or unparseable triggers file
1348
+ // has no honest claim to make (readTriggerFacts' own rule). ok:true -- the replica advisory's tier,
1349
+ // and like it, everything the operator needs lives in the LABEL: an ok:true check never prints its
1350
+ // fix line.
1351
+ if (scopedLimitFacts.parseError === null && scopedLimitFacts.limits.length > 0 && parseError === null && triggersFilePath !== null) {
1352
+ const folderSet = new Set(folders);
1353
+ const folderOnly = (s) => s.startsWith("/") || s.startsWith("./") || s.startsWith("../") || s.includes("\\") || !s.includes("/") || /^[A-Za-z]:/.test(s);
1354
+ const dead = scopedLimitFacts.limits.map((l) => l.scope).filter((s) => folderOnly(s) && !folderSet.has(s));
1355
+ if (dead.length > 0) {
1356
+ checks.push({
1357
+ ok: true,
1358
+ warn: true,
1359
+ label: `${dead.length} scoped limit(s) name a folder no trigger runs in (${dead.join(", ")}) -- the cap guards nothing; scopes match exactly (no globs, folders by resolved ABSOLUTE path), so check the spelling against triggers.json run.folder or delete the entry`,
1360
+ fix: `edit ${scopedLimitFacts.path} by hand or via dispatch_limit_edit/_delete -- repo-shaped scopes are never flagged here, because a webhook job's repo comes from the delivery, which triggers.json cannot enumerate`,
1361
+ });
1362
+ }
1363
+ }
1364
+
1286
1365
  // REQ-RESURRECTABLE-SANDBOX. A warning, never a failure: retention is a convenience, and the only thing
1287
1366
  // worth surfacing is that finished runs' directories -- a repository clone plus the run's prompt.md and
1288
1367
  // event.json, so issue text -- are sitting on disk, and how many. An operator who never opens a sandbox
@@ -1471,16 +1550,48 @@ function parseSecretProfilesSafe(raw) {
1471
1550
  }
1472
1551
  }
1473
1552
 
1553
+ /**
1554
+ * The scoped-limits facts (issue #242): the parsed rows when PI_SCOPED_LIMITS_FILE is set, or the
1555
+ * boot-blocking reason when it will not load. Unset is `none` -- the worker enforces no scoped limits
1556
+ * and doctor has nothing to say (the mutex is code and needs no check). A configured-but-missing file
1557
+ * IS a parseError here: loadScopedLimits refuses boot on it, so doctor must too. Raw fs errors
1558
+ * (EACCES, EISDIR) are reported the same way, deliberately unlike readTriggerFacts' tagged-only
1559
+ * filter: the worker's own boot load is an unguarded readFileSync, so those throws refuse startup
1560
+ * exactly as a parse failure does, and the check's claim is "will the worker start", not "is the
1561
+ * content valid".
1562
+ */
1563
+ function readScopedLimitFacts(env, fileExists) {
1564
+ const none = { limits: [], parseError: null, path: null };
1565
+ const path = env.PI_SCOPED_LIMITS_FILE;
1566
+ if (typeof path !== "string" || path.trim() === "") return none;
1567
+ if (!fileExists(path)) return { limits: [], parseError: `scoped-limits file does not exist: ${path}`, path };
1568
+ try {
1569
+ return { limits: parseScopedLimits(readFileSync(path, "utf8"), path), parseError: null, path };
1570
+ } catch (e) {
1571
+ return { limits: [], parseError: e?.message ?? String(e), path };
1572
+ }
1573
+ }
1574
+
1474
1575
  function readTriggerFacts(env, fileExists, cwd) {
1475
- const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, secretProfiles: [], localSecretFolders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1576
+ const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], folders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1476
1577
  try {
1477
1578
  // Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
1478
1579
  // (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
1479
1580
  // the receiver will not boot. An absent file still means "no triggers at all", exactly as before.
1480
1581
  const path = env.PI_TRIGGERS_FILE ?? join(cwd, "triggers.json");
1481
1582
  if (!fileExists(path)) return none;
1482
- const triggers = parseTriggers(readFileSync(path, "utf8"), path);
1583
+ const text = readFileSync(path, "utf8");
1584
+ const triggers = parseTriggers(text, path);
1585
+ // The one-shot facts are counted from the RAW entries, not the parsed records, because the
1586
+ // validator collapses a disarmed entry to a sentinel that carries neither `once` nor
1587
+ // `disarmed` -- exactly so nothing can match it -- which also erases it from every parsed
1588
+ // count above. Doctor is the surface that must still SEE the spent entry: "why did nothing
1589
+ // fire" is answered by a spent row, and only the raw file still holds it. Safe unguarded:
1590
+ // parseTriggers just accepted this same text, so JSON.parse cannot throw here.
1591
+ const rawEntries = JSON.parse(text)?.triggers ?? [];
1483
1592
  return {
1593
+ onceArmed: rawEntries.filter((t) => t?.on?.once === true && t.on.disarmed === undefined).length,
1594
+ onceSpent: rawEntries.filter((t) => t?.on?.disarmed !== undefined).length,
1484
1595
  requiring: triggers.filter((t) => t.run.packages === true).length,
1485
1596
  resuming: triggers.filter((t) => t.run.resume === true).length,
1486
1597
  // REQ-PER-TRIGGER-INSTRUCTION. Counted beside `resuming` for the same reason: it is a per-trigger
@@ -1504,6 +1615,10 @@ function readTriggerFacts(env, fileExists, cwd) {
1504
1615
  // read-write with no clone, so a credential an agent writes into .env lands in the operator's real
1505
1616
  // repository rather than a temp dir that gets swept. Deduped for skillsDirs' reason.
1506
1617
  localSecretFolders: [...new Set(triggers.filter((t) => t.run.secrets !== undefined && t.run.kind === "local" && typeof t.run.folder === "string").map((t) => t.run.folder))].sort(),
1618
+ // Issue #242: every local run.folder, CANONICALIZED the way the scoped-limits matcher
1619
+ // canonicalizes a job's folder (one derivation -- canonicalScope, never re-spelled here), so
1620
+ // the unreferenced-scope advisory compares like with like across spelling variants.
1621
+ folders: [...new Set(triggers.filter((t) => t.run.kind === "local" && typeof t.run.folder === "string").map((t) => canonicalScope({ kind: "local", folder: t.run.folder })))].sort(),
1507
1622
  optingOut: triggers.filter((t) => t.run.packages === false).length,
1508
1623
  images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
1509
1624
  // REQ-PER-TRIGGER-SKILLS. The distinct host directories the file names, deduped like `images`,
@@ -1534,6 +1649,11 @@ function readTriggerFacts(env, fileExists, cwd) {
1534
1649
  packages: t.run.packages !== false,
1535
1650
  }))
1536
1651
  .filter((f) => typeof f.flow === "string"),
1652
+ // Explicit on the success path too (issue #242): the dead-scope advisory distinguishes
1653
+ // "facts read clean" (path set, no error) from the zeroed `none` -- an implicit undefined
1654
+ // here made that test silently false for every deployment.
1655
+ parseError: null,
1656
+ path,
1537
1657
  };
1538
1658
  } catch (e) {
1539
1659
  // REPORTED, not swallowed. This catch used to justify itself with "a malformed triggers file already
package/src/get-token.mjs CHANGED
@@ -104,10 +104,15 @@ export async function makeGitHubAuth(cfg, deps = {}) {
104
104
  );
105
105
  }
106
106
  const repositoryNames = [repoNameOf(repo)]; // scope to the ONE repo; owner stripped
107
+ // An optional PERMISSIONS narrowing (issue #231): the receiver's closer-permission lookup asks
108
+ // for `{ metadata: "read" }`, so the token it holds for that one question cannot write anything
109
+ // even if leaked. Job mints never pass this and keep the installation's full grant -- narrowing
110
+ // is the caller's statement of intent, not a default this mint could guess.
111
+ const permissions = job?.permissions;
107
112
  let minted;
108
113
  try {
109
114
  const appAuth = createAppAuth(auth);
110
- minted = await appAuth({ type: "installation", repositoryNames });
115
+ minted = await appAuth({ type: "installation", repositoryNames, ...(permissions && { permissions }) });
111
116
  } catch (error) {
112
117
  throw classifyAppMintError(error);
113
118
  }
package/src/index.mjs CHANGED
@@ -2,11 +2,19 @@ import { execFile } from "node:child_process";
2
2
  import { promisify } from "node:util";
3
3
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
4
4
  import { InfraRetry, runJob } from "./processor.mjs";
5
+ import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
5
6
 
6
7
  const exec = promisify(execFile);
7
8
 
8
9
  export const QUEUE = "pi-jobs";
9
10
  export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
11
+ // The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
12
+ // JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
13
+ // (<=360 wakes across a 30-minute hold, each ~1ms of synchronous predicate briefly occupying a slot)
14
+ // while a same-folder CHAINED job -- enqueued by its parent before the parent's finally releases the
15
+ // folder -- pays exactly one re-check, not fifteen seconds of dead air. No jitter: one worker per
16
+ // docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
17
+ export const SCOPE_BUSY_RECHECK_MS = 5_000;
10
18
 
11
19
  /**
12
20
  * Build the BullMQ processor.
@@ -27,7 +35,7 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
27
35
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
28
36
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
29
37
  */
30
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
38
+ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
31
39
  return async function processor(job, token, signal) {
32
40
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
33
41
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -45,19 +53,68 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
45
53
  throw new DelayedError();
46
54
  }
47
55
 
48
- const startedAt = new Date().toISOString();
49
- const name = `pi-job-${job.id}`;
50
- const timer = setTimeout(() => {
51
- // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
52
- Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
53
- }, timeoutMs);
56
+ // Per-scope concurrency and the one-job-per-folder mutex (issue #242,
57
+ // INT-SCOPED-LIMITS-FILE-CONTRACT). SECOND, after the pause gate (a paused job must not burn
58
+ // re-check wakes) and STRICTLY above the `try` below, like the pause gate and for the same two
59
+ // reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
60
+ // catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
61
+ // handling exactly as the pause gate's does (inside the try it would become a permanent failure
62
+ // plus a failure record for what was a transient blip). The limits snapshot is read ONCE here and
63
+ // shared with `scopedCaps` below, so the gate and the money ledger cannot disagree mid-job.
64
+ // tryAcquire is a synchronous check-and-increment -- no await between read and take, so Node's
65
+ // single thread makes it atomic at any concurrency -- and the local-folder limit is a structural 1
66
+ // (concurrencyFor) with no file and no off-switch: the scheduler mints a cron trigger's next
67
+ // occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
68
+ // (measured: 301ms of live container overlap through this very processor) unless this gate holds.
69
+ // Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
70
+ const limits = scopedLimits();
71
+ const scope = canonicalScope(job.data);
72
+ let held = false;
73
+ if (scope) {
74
+ if (!inFlight.tryAcquire(scope, concurrencyFor(job.data, limits))) {
75
+ // Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
76
+ // The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
77
+ // host path); the delayed count and the job id are what an operator needs to see it.
78
+ deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
79
+ await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
80
+ throw new DelayedError();
81
+ }
82
+ held = true;
83
+ }
54
84
 
55
- // Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
56
- // after the grace period; the runner exits and runContainer returns/throws.
57
- const onAbort = () => {
58
- Promise.resolve(stopContainer(name)).catch(() => {});
59
- };
60
- signal.addEventListener("abort", onAbort, { once: true });
85
+ let startedAt;
86
+ let name;
87
+ let timer;
88
+ let onAbort;
89
+ try {
90
+ // Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
91
+ // finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
92
+ // scope until a worker restart. Nothing in this block CAN throw today (setTimeout and
93
+ // addEventListener on the bullmq-allocated controller are total at processor arity 3); the
94
+ // guard is structural, not observational.
95
+ startedAt = new Date().toISOString();
96
+ name = `pi-job-${job.id}`;
97
+ timer = setTimeout(() => {
98
+ // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
99
+ Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
100
+ }, timeoutMs);
101
+
102
+ // Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
103
+ // after the grace period; the runner exits and runContainer returns/throws.
104
+ onAbort = () => {
105
+ Promise.resolve(stopContainer(name)).catch(() => {});
106
+ };
107
+ signal.addEventListener("abort", onAbort, { once: true });
108
+ } catch (error) {
109
+ // Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
110
+ // scope must never be releasable twice -- a double release frees another holder's slot.
111
+ if (held) {
112
+ inFlight.release(scope);
113
+ held = false;
114
+ }
115
+ clearTimeout(timer);
116
+ throw error;
117
+ }
61
118
 
62
119
  try {
63
120
  const settings = await getSettings();
@@ -99,6 +156,10 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
99
156
  // The daily TOKEN cap (issue #25), same overlay > env resolution. Check-AFTER, so it gates the
100
157
  // NEXT job on prior recorded spend; null => the daily token counter is disabled.
101
158
  tokenCap: settings.dailyTokenCap,
159
+ // This job's scoped budget windows (issue #242), from the SAME limits snapshot the pickup
160
+ // gate above read -- one read per pickup, so gate and ledger agree for this job's whole
161
+ // life. Null when no row carries a money window for this scope.
162
+ scopedCaps: budgetCapsFor(job.data, limits),
102
163
  ...deps,
103
164
  runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
104
165
  // REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
@@ -125,6 +186,11 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
125
186
  // `queueJobId`, mirroring the collectChain injection above. Omitted when unwired so a bare
126
187
  // processor keeps runJob's plain (job, token) call.
127
188
  ...(deps.prepareWorkspace ? { prepareWorkspace: (j, t) => deps.prepareWorkspace(j, t, { queueJobId: job.id }) } : {}),
189
+ // The one-shot pre-spend check (issue #231) needs the REAL BullMQ job's `.id` to excuse this
190
+ // delivery's own earlier attempt -- runJob's effectiveJob has no `.id`, prepareWorkspace's
191
+ // own injection above states why, and this one mirrors it. Omitted when unwired so a bare
192
+ // processor keeps runJob's admit-everything default.
193
+ ...(deps.checkOnceSpent ? { checkOnceSpent: (j) => deps.checkOnceSpent(j, { queueJobId: job.id }) } : {}),
128
194
  });
129
195
  recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
130
196
  return result;
@@ -135,13 +201,16 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
135
201
  // records it as failed-and-distinct in the queue's failed set without a retry.
136
202
  throw new UnrecoverableError(error.message);
137
203
  } finally {
204
+ // Release FIRST and never throw (release clamps at zero by construction): a throw here would
205
+ // mask the job's real error, and a missed release wedges the scope until a worker restart.
206
+ if (held) inFlight.release(scope);
138
207
  clearTimeout(timer);
139
208
  signal.removeEventListener("abort", onAbort);
140
209
  }
141
210
  };
142
211
  }
143
212
 
144
- export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, extraClosers = [] }) {
213
+ export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
145
214
  let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
146
215
  const processor = makeProcessor({
147
216
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
@@ -154,6 +223,11 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
154
223
  if (Number.isInteger(n) && worker.concurrency !== n) worker.concurrency = n;
155
224
  },
156
225
  pauseUntil,
226
+ // Undefined pass-throughs take makeProcessor's own defaults (no limits; a fresh per-processor
227
+ // in-flight map -- one per worker process, which under DES-CONCURRENCY-3's one-worker-per-daemon
228
+ // shape means one per daemon).
229
+ scopedLimits,
230
+ inFlight,
157
231
  deps,
158
232
  recordRun,
159
233
  });
package/src/init.mjs CHANGED
@@ -19,6 +19,11 @@ const EMPTY_PACKAGES = `${JSON.stringify({ packages: [] }, null, 2)}\n`;
19
19
  // Operator-declared subscription plans (issue #53), read by the admin extension only — never at job
20
20
  // time. Versioned because a newer file must fail loud, and that cannot be retrofitted into a v1 reader.
21
21
  const EMPTY_SUBSCRIPTIONS = `${JSON.stringify({ version: 1, subscriptions: [] }, null, 2)}\n`;
22
+ // Scoped limits (issue #242): per repo/folder run caps and concurrency. Empty is inert -- and the
23
+ // one-job-per-folder mutex for local jobs is code, not configuration, so it needs no scaffold line.
24
+ // Versioned for the subscriptions reason, sharpened: this is enforcement config, and a silently
25
+ // down-read newer file would be a silently widened spend limit.
26
+ const EMPTY_SCOPED_LIMITS = `${JSON.stringify({ version: 1, limits: [] }, null, 2)}\n`;
22
27
  /**
23
28
  * The egress allowlist (REQ-EGRESS-ALLOWLIST): the hosts a job container may reach, one bare hostname per
24
29
  * line. Scaffolded with the three a job cannot work without, and NOT empty -- unlike every other scaffold
@@ -73,6 +78,7 @@ export function runInit(cwd = process.cwd(), deps = {}) {
73
78
  scaffold(fs, results, join(cwd, "pause-windows.json"), EMPTY_PAUSE_WINDOWS, "empty pause-windows list");
74
79
  scaffold(fs, results, join(cwd, "pi-packages.json"), EMPTY_PACKAGES, "empty pi package list (stage with import-pi --with-packages)");
75
80
  scaffold(fs, results, join(cwd, "subscriptions.json"), EMPTY_SUBSCRIPTIONS, "empty subscription list (declare plan prices for the admin's cost analytics)");
81
+ scaffold(fs, results, join(cwd, "scoped-limits.json"), EMPTY_SCOPED_LIMITS, "empty scoped-limits list (per repo/folder caps; the folder mutex needs no file)");
76
82
  scaffold(fs, results, join(cwd, "egress-allowlist.conf"), DEFAULT_EGRESS_ALLOWLIST, "egress allowlist (provider + forge + registry; the egress policy is on unless PI_EGRESS=0)");
77
83
 
78
84
  for (const [verb, name, note] of results) {
package/src/processor.mjs CHANGED
@@ -1,6 +1,7 @@
1
1
  import { lstatSync } from "node:fs";
2
2
  import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
3
3
  import { configError } from "./config.mjs";
4
+ import { scopeKeyPrefix } from "./scoped-limits.mjs";
4
5
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
5
6
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
6
7
 
@@ -37,11 +38,22 @@ export async function runJob(job, deps) {
37
38
  caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
38
39
  softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
39
40
  tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
41
+ // { scope, caps: { day, week, month } } | null -- this job's scoped budget windows (issue #242,
42
+ // INT-SCOPED-LIMITS-FILE-CONTRACT), resolved by the wiring from the same watched-limits snapshot the
43
+ // pickup gate read. Null when the file is unset or the scope's row is concurrency-only; the default
44
+ // keeps an unwired processor byte-identical. The folder MUTEX does not live here -- it is the pickup
45
+ // gate's, pre-everything; this is only the money half.
46
+ scopedCaps = null,
40
47
  recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
41
48
  // (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
42
49
  // this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
43
50
  // omits it behaves exactly as before -- the container's own failure stays the backstop.
44
51
  imagePreflight = async () => ({ ok: true }),
52
+ // (job) => { ok } | { refused, at, jobId }. The one-shot pre-spend check (issue #231,
53
+ // DES-ONE-SHOT-DISARM-IN-THE-FILE). Default admits everything -- an unwired processor behaves
54
+ // exactly as before, and the gate below only calls it for a job whose matched rule was a
55
+ // one-shot, so the default is never a probe running on every delivery.
56
+ checkOnceSpent = async () => ({ ok: true }),
45
57
  // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
46
58
  // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
47
59
  egressPreflight = async () => ({ ok: true }),
@@ -119,8 +131,30 @@ export async function runJob(job, deps) {
119
131
  let token = null;
120
132
  let prepared = null;
121
133
  let reserved = false;
134
+ let scopedReserved = false;
122
135
 
123
136
  try {
137
+ // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
138
+ // the docker inspect below, free, determinate, credential-less. Only a FOREIGN positive
139
+ // disarmed mark refuses -- the check excuses this queue job's own id, so a retry of the
140
+ // delivery that spent the trigger still runs (attempts:2 stays attempts:2) -- and anything
141
+ // unreadable or changed means "run": fail-open, because the disarm writer owns the loud
142
+ // refusals, and a broken read must never wedge every once job. In the compose topology the
143
+ // receiver reads a dead inode until restart, so this check is the once-enforcement layer
144
+ // there, not optional hardening.
145
+ if (job.trigger?.matched?.once === true) {
146
+ const spent = await checkOnceSpent(job);
147
+ if (spent.refused) {
148
+ // Commented like every sibling policy refusal: explainability is this refusal's whole
149
+ // purpose, and only a DISTINCT re-close reaches it past the GUID dedup, so the noise
150
+ // bound is the operator's own reopen-close rate. `at`/`jobId` are harness-written
151
+ // provenance, never payload text.
152
+ await comment(job, `Refused: this one-shot trigger was already spent${spent.at ? ` at ${spent.at}` : ""}${spent.jobId ? ` by job ${spent.jobId}` : ""}. The close that armed it has already produced a run; delete on.disarmed from the trigger entry to re-arm it. Not run.`);
153
+ log("refused_once_already_spent", { triggerIndex: job.trigger?.matched?.index ?? null });
154
+ return { outcome: "policy", reason: "once-already-spent", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
155
+ }
156
+ }
157
+
124
158
  // The job image must exist on THIS host before anything else happens. Free, determinate and
125
159
  // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
126
160
  // image refuses without minting a credential it will not use, cloning a repo it will not read, or
@@ -432,11 +466,51 @@ export async function runJob(job, deps) {
432
466
  return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
433
467
  }
434
468
 
435
- // Budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
469
+ // Per-scope budget windows (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): the NARROWER ledger
470
+ // reserves FIRST, so a noisy scope's refusals never consume a global slot -- the global INCR below
471
+ // runs only for jobs the scope admitted. Same atomic INCR, same refused-still-counts invariant,
472
+ // through budget.mjs's keyPrefix seam (dayKey/weekKey/monthKey under budget:s:<hash16>). softHoldPct
473
+ // is deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob;
474
+ // scoped windows are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
475
+ if (scopedCaps) {
476
+ // A redis fault BETWEEN this reserve and the global one below strands the scoped INCR with no
477
+ // run and no refund -- the pre-existing mid-reserve posture, shared with the global ledger's
478
+ // own partial-INCR seam; the compensating release below covers REFUSALS, not faults.
479
+ const scoped = await reserveBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
480
+ scopedReserved = true;
481
+ if (!scoped.allowed) {
482
+ const w = scoped.blockedWindow;
483
+ const win = scoped.windows[w];
484
+ // A local job's scope is a full host path and its "comment" is not dropped -- the wiring's
485
+ // local adapter LOGS the text (start.mjs forgeFor fallthrough) -- so the path must never
486
+ // enter the message; "this folder" is enough beside the jobId the adapter logs. A forge
487
+ // scope IS the repo the comment posts on, safe to name.
488
+ const scopeLabel = job.kind === "local" ? "this folder" : scopedCaps.scope;
489
+ await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
490
+ // The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would
491
+ // put a full host path in the worker log against no-pii-in-logs (the record keeps only
492
+ // basename(folder) for the same reason). The admin recomputes the key from the configured
493
+ // scope to join it back.
494
+ log("over_scope_budget", { scopeKey: scopeKeyPrefix(scopedCaps.scope), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
495
+ // budgetReserved false: the GLOBAL slot was never touched (scoped reserves first). The scoped
496
+ // counter did INCR and keeps it -- its own refused-reservation-still-counts, per ledger.
497
+ return { outcome: "policy", reason: "scope-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
498
+ }
499
+ }
500
+
501
+ // GLOBAL budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
436
502
  // every active window (day + optional week/month) and the soft-hold band in one atomic pass.
437
503
  const budget = await reserveBudget(redis, { caps, softHoldPct, now });
438
504
  reserved = true;
439
505
  if (!budget.allowed) {
506
+ // The scoped reserve above committed before this global refusal -- give that slot back. Without
507
+ // this, an exhausted global window drains every arriving scope's own day/week/month counters
508
+ // with zero runs to show for it (a storm against a spent global daily cap would empty a repo's
509
+ // week by noon). The scoped ledger's refused-still-counts covers the SCOPE's own refusal above,
510
+ // never a refusal it did not issue.
511
+ if (scopedReserved && scopedCaps) {
512
+ await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
513
+ }
440
514
  const w = budget.blockedWindow;
441
515
  const win = budget.windows[w];
442
516
  if (budget.reason === "soft-hold") {
@@ -535,8 +609,12 @@ export async function runJob(job, deps) {
535
609
  // budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
536
610
  // true for a real container that ran and spent (exit-1 infra / unknown exit).
537
611
  if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
538
- if (reserved && e instanceof InfraRetry && e.reason === "container-never-started") {
539
- await releaseBudget(redis, { caps, now });
612
+ if (e instanceof InfraRetry && e.reason === "container-never-started") {
613
+ // Both-or-neither (issue #242): a never-started container can only follow BOTH reserves (the
614
+ // scoped one precedes the global one, and the container follows both), so they refund
615
+ // together -- and a scoped refusal returned above without ever touching the global ledger.
616
+ if (reserved) await releaseBudget(redis, { caps, now });
617
+ if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
540
618
  }
541
619
  throw e;
542
620
  } finally {
package/src/queue.mjs CHANGED
@@ -1,6 +1,11 @@
1
1
  import { Queue } from "bullmq";
2
2
  import { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId } from "./job-id.mjs";
3
3
  import { targetSeparator } from "./forges.mjs";
4
+ import { PR_CLOSE_ACTIONS } from "./triggers.mjs";
5
+
6
+ // The close words in every forge's spelling, derived from the one table (never re-typed here): a
7
+ // matched PR action in this set marks a close job for the semantic-key discriminant below.
8
+ const PR_CLOSE_WORDS = new Set(Object.values(PR_CLOSE_ACTIONS));
4
9
 
5
10
  export const QUEUE = "pi-jobs";
6
11
  export { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId };
@@ -180,6 +185,21 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
180
185
  ...(replica !== undefined && { replica }),
181
186
  ...(replicas !== undefined && { replicas }),
182
187
  };
188
+ // A close-triggered job (issue #231) leads the semantic key's flow slot with `closed:`. Without it,
189
+ // a label/comment/PR job on the same target and flow inside the 10-minute window silently swallows
190
+ // the close job -- and because a swallowed close job writes no run record, the once trigger it was
191
+ // meant to spend never disarms: a permanently dead one-shot with nothing in the panel to say why.
192
+ // The discriminant is DERIVED from the matched rule (`issue` type, or a PR close action word) rather
193
+ // than carried as a job field: an execution detail of dedup is not a fact about the delivery, and
194
+ // `data`/`event.json` stay byte-identical. `:` is outside the skill-name charset -- enforced at load
195
+ // since #231 -- so no real flow can spell either prefixed form, and `closed:cmd:<name>` composes for
196
+ // close-dispatched commands (outermost discriminant first, then the entry-point prefix).
197
+ const matched = trigger?.matched;
198
+ // `type === "issue"` reads as "close" only while the issue vocabulary is close-only (it is; the
199
+ // tables say "one word each so far"). If that type ever grows a non-close action, this test must
200
+ // narrow to the matched action word, like the PR half already does.
201
+ const isCloseJob = matched?.type === "issue" || (matched?.type === "pull_request" && PR_CLOSE_WORDS.has(matched?.action));
202
+ const flowSlot = `${isCloseJob ? "closed:" : ""}${command !== undefined ? `cmd:${command}` : flow}`;
183
203
  await queue.add(kind, data, {
184
204
  jobId,
185
205
  // A command job (issue #189) fills the semantic key's flow slot with `cmd:<command>`: a command
@@ -188,7 +208,7 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
188
208
  // `cmd:` prefix keeps a command named X from coalescing against a flow named X -- `:` is outside
189
209
  // the skill-name charset, so no real flow can spell the prefixed form -- and a flow job's key
190
210
  // stays byte-identical to before the feature.
191
- deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${command !== undefined ? `cmd:${command}` : flow}${replica !== undefined ? `:r${replica}` : ""}`, ttl: SEMANTIC_WINDOW_MS }, // ttl in ms
211
+ deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${flowSlot}${replica !== undefined ? `:r${replica}` : ""}`, ttl: SEMANTIC_WINDOW_MS }, // ttl in ms
192
212
  attempts: 2,
193
213
  backoff: { type: "exponential", delay: 60_000 },
194
214
  removeOnComplete: { age: 31 * 24 * 3600 }, // age in seconds -- do not cross units with the ms ttl above