@edgehero/pi-dispatch 1.3.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +2 -0
- package/deploy/docker-compose.yml +5 -1
- package/package.json +3 -1
- package/src/config.mjs +1 -0
- package/src/doctor.mjs +123 -3
- package/src/get-token.mjs +6 -1
- package/src/index.mjs +88 -14
- package/src/init.mjs +6 -0
- package/src/processor.mjs +81 -3
- package/src/queue.mjs +21 -1
- package/src/run-history.mjs +50 -31
- package/src/scoped-limits.mjs +277 -0
- package/src/start.mjs +76 -2
- package/src/triggers-file.mjs +403 -0
- package/src/triggers.mjs +312 -15
package/.env.example
CHANGED
|
@@ -74,6 +74,8 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
|
|
|
74
74
|
# Unset = cron disabled for the worker; the receiver falls back to ./triggers.json in the folder it starts from (what `pi-dispatch init` scaffolds)
|
|
75
75
|
# and refuses to start when neither exists (it holds the label/comment/pull_request trigger config)
|
|
76
76
|
# PI_PAUSE_WINDOWS_FILE= # ABSOLUTE path to pause-windows.json — "quiet hours" per folder/repo (pause runs between certain times/days/dates, auto-resume). Unset = feature off. See docs/pause-windows.md
|
|
77
|
+
# PI_SCOPED_LIMITS_FILE= # ABSOLUTE path to scoped-limits.json — per repo/folder job-count caps (day/week/month, refused pre-spend as scope-cap) and max concurrent jobs per scope (excess deferred, never dropped).
|
|
78
|
+
# Unset = no scoped caps or concurrency; the one-job-per-folder mutex for local jobs is always on and needs no file. See docs/scoped-limits.md
|
|
77
79
|
# PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
|
|
78
80
|
# Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
|
|
79
81
|
# PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
|
|
@@ -49,7 +49,11 @@ services:
|
|
|
49
49
|
PI_TRIGGERS_FILE: /config/triggers.json
|
|
50
50
|
# The repo-root triggers.json (`pi-dispatch init` scaffolds it -- run that first: mounting a path
|
|
51
51
|
# that does not exist makes Docker create it as a DIRECTORY and the boot fails confusingly).
|
|
52
|
-
# Read-only: the receiver live-reloads this file on change; it never writes it.
|
|
52
|
+
# Read-only: the receiver live-reloads this file on change; it never writes it. One consequence of
|
|
53
|
+
# a single-FILE bind mount (issue #231): every write to this file is an atomic tmp+rename that
|
|
54
|
+
# SWAPS THE INODE, and the mount stays pinned to the old one -- so the worker's one-shot disarm is
|
|
55
|
+
# invisible in here until the container restarts, and the worker's own pre-spend check is what
|
|
56
|
+
# keeps a spent one-shot from running again in the meantime. A restart picks up the current file.
|
|
53
57
|
volumes:
|
|
54
58
|
- ../triggers.json:/config/triggers.json:ro
|
|
55
59
|
# Loopback only, like Valkey's port above: the operator's reverse proxy or tunnel (TLS, public
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.5.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
|
@@ -52,8 +52,10 @@
|
|
|
52
52
|
"./job-id": "./src/job-id.mjs",
|
|
53
53
|
"./forges": "./src/forges.mjs",
|
|
54
54
|
"./triggers": "./src/triggers.mjs",
|
|
55
|
+
"./triggers-file": "./src/triggers-file.mjs",
|
|
55
56
|
"./packages": "./src/packages.mjs",
|
|
56
57
|
"./pause-windows": "./src/pause-windows.mjs",
|
|
58
|
+
"./scoped-limits": "./src/scoped-limits.mjs",
|
|
57
59
|
"./identity": "./src/identity.mjs",
|
|
58
60
|
"./gitlab-identity": "./src/gitlab-identity.mjs",
|
|
59
61
|
"./forgejo-identity": "./src/forgejo-identity.mjs",
|
package/src/config.mjs
CHANGED
|
@@ -244,6 +244,7 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
|
|
|
244
244
|
sandboxIdleMinutes: nonNegativeInt(env, "PI_SANDBOX_IDLE_MINUTES", 30), // bash's own TMOUT inside a sandbox; 0 = no idle logout
|
|
245
245
|
triggersFile: env.PI_TRIGGERS_FILE ?? null, // DES-CRON-VIA-BULLMQ-SCHEDULER: unified triggers file; null = cron disabled for the worker (it selects on.type:"cron")
|
|
246
246
|
pauseWindowsFile: env.PI_PAUSE_WINDOWS_FILE ?? null, // REQ-SCOPED-PAUSE-WINDOWS: per-folder/repo timed pause; null = no scoped pauses
|
|
247
|
+
scopedLimitsFile: env.PI_SCOPED_LIMITS_FILE ?? null, // issue #242: per-scope run caps + concurrency (INT-SCOPED-LIMITS-FILE-CONTRACT); null = none. The one-job-per-folder mutex for local jobs is code, not configuration, and holds regardless
|
|
247
248
|
schedulerStallMax: positiveInt(env, "PI_SCHEDULER_STALL_MAX", 2), // CONST-RETRY-INFRA-ONLY: per-scheduler stall backstop; positiveInt rejects <1 so a 0 threshold fails closed
|
|
248
249
|
logsDir: env.PI_LOGS_DIR || defaultLogsDir(), // || (not ??) so an empty string falls back to the default
|
|
249
250
|
settingsFile: env.PI_SETTINGS_FILE || defaultSettingsFile(), // || (not ??) so an empty string falls back; INT-CONFIG-OVERLAY-CONTRACT
|
package/src/doctor.mjs
CHANGED
|
@@ -50,6 +50,7 @@ import { dirname, join, delimiter } from "node:path";
|
|
|
50
50
|
import { fileURLToPath } from "node:url";
|
|
51
51
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
52
52
|
import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
|
|
53
|
+
import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
|
|
53
54
|
import { isForgeKind } from "./forges.mjs";
|
|
54
55
|
import { findLiteralSecret, ADMIN_RE } from "./import-pi.mjs";
|
|
55
56
|
import { agentDirFrom, readHostPi } from "./host-pi.mjs";
|
|
@@ -264,7 +265,8 @@ export async function collectChecks(env, seams) {
|
|
|
264
265
|
// image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
|
|
265
266
|
// `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
|
|
266
267
|
// run.packages: true, which arms nothing any more but is still an operator statement of intent.
|
|
267
|
-
const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, secretProfiles, localSecretFolders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
|
|
268
|
+
const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, folders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
|
|
269
|
+
const scopedLimitFacts = readScopedLimitFacts(env, fileExists);
|
|
268
270
|
// FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
|
|
269
271
|
// zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
|
|
270
272
|
// file unconditionally and refuses to start without it, which is the consequence worth naming.
|
|
@@ -1242,6 +1244,31 @@ export async function collectChecks(env, seams) {
|
|
|
1242
1244
|
});
|
|
1243
1245
|
}
|
|
1244
1246
|
|
|
1247
|
+
// One-shot close triggers (issue #231, DES-ONE-SHOT-DISARM-IN-THE-FILE). Advisory only -- doctor
|
|
1248
|
+
// never touches triggers -- and counted from the RAW file (readTriggerFacts says why). Two lines
|
|
1249
|
+
// with different lives: the armed line names the count and, when PI_TRIGGERS_FILE is unset, warns
|
|
1250
|
+
// that the disarm resolves ./triggers.json against the WORKER SERVICE's working directory -- a
|
|
1251
|
+
// service unit whose WorkingDirectory differs from the receiver's would disarm a file nobody
|
|
1252
|
+
// matches against, the split-file hazard no mechanism can detect. The spent line states the
|
|
1253
|
+
// deliberate degradation: a spent entry counts toward NO parsed fact above (forges, flows,
|
|
1254
|
+
// webhook-secret), mirroring what the receiver serves at its next boot.
|
|
1255
|
+
if (onceArmed > 0) {
|
|
1256
|
+
checks.push({
|
|
1257
|
+
ok: true,
|
|
1258
|
+
warn: env.PI_TRIGGERS_FILE === undefined,
|
|
1259
|
+
label: `${onceArmed} one-shot trigger(s) armed (on.once) -- the worker disarms the entry in ${env.PI_TRIGGERS_FILE === undefined ? "./triggers.json resolved against the worker service's working directory; set PI_TRIGGERS_FILE so worker and receiver name the same file from anywhere" : "PI_TRIGGERS_FILE"} after the run record exists`,
|
|
1260
|
+
fix: "set PI_TRIGGERS_FILE to an absolute path in both services' environments",
|
|
1261
|
+
});
|
|
1262
|
+
}
|
|
1263
|
+
if (onceSpent > 0) {
|
|
1264
|
+
checks.push({
|
|
1265
|
+
ok: true,
|
|
1266
|
+
warn: false,
|
|
1267
|
+
label: `${onceSpent} one-shot trigger(s) already spent (on.disarmed) -- spent entries match nothing and count toward no credential or flow check; delete on.disarmed to re-arm, or delete the entry once its history no longer matters`,
|
|
1268
|
+
fix: "",
|
|
1269
|
+
});
|
|
1270
|
+
}
|
|
1271
|
+
|
|
1245
1272
|
// REQ-SCOPED-PAUSE-WINDOWS, the panel-writes-what-the-worker-ignores trap (issue #99). Three defaults
|
|
1246
1273
|
// that are individually defensible and together silent:
|
|
1247
1274
|
//
|
|
@@ -1283,6 +1310,58 @@ export async function collectChecks(env, seams) {
|
|
|
1283
1310
|
}
|
|
1284
1311
|
}
|
|
1285
1312
|
|
|
1313
|
+
// The same trap, scoped-limits edition (issue #242): init scaffolds ./scoped-limits.json, the admin
|
|
1314
|
+
// defaults to it, and the worker reads only PI_SCOPED_LIMITS_FILE. The label's mutex parenthetical is
|
|
1315
|
+
// load-bearing -- the check must not imply local folders run ungated when the file is off.
|
|
1316
|
+
{
|
|
1317
|
+
const scopedLimitsFile = env.PI_SCOPED_LIMITS_FILE;
|
|
1318
|
+
const scaffolded = join(cwd, "scoped-limits.json");
|
|
1319
|
+
if ((typeof scopedLimitsFile !== "string" || scopedLimitsFile.trim() === "") && fileExists(scaffolded)) {
|
|
1320
|
+
checks.push({
|
|
1321
|
+
ok: false,
|
|
1322
|
+
warn: true,
|
|
1323
|
+
label: `${scaffolded} exists but PI_SCOPED_LIMITS_FILE is unset -- the worker ignores it, so scoped caps and concurrency are OFF (the built-in one-job-per-folder mutex stays on)`,
|
|
1324
|
+
fix: `set PI_SCOPED_LIMITS_FILE=${scaffolded} in .env and restart the worker -- unset means the worker enforces no scoped limits at all, while the admin panel defaults to this same file and reports each limit it writes as applied live; delete the file if this deployment has no scoped limits`,
|
|
1325
|
+
});
|
|
1326
|
+
}
|
|
1327
|
+
}
|
|
1328
|
+
|
|
1329
|
+
// Issue #242: a CONFIGURED scoped-limits file is boot-load fail-loud, so a file that does not load
|
|
1330
|
+
// refuses the next worker start -- doctor says it before the restart does. Never-tier: doctor never
|
|
1331
|
+
// rewrites limits content (DES-CLI-SURFACE).
|
|
1332
|
+
if (scopedLimitFacts.path !== null && scopedLimitFacts.parseError !== null) {
|
|
1333
|
+
checks.push({
|
|
1334
|
+
ok: false,
|
|
1335
|
+
label: `scoped-limits file does not load -- the worker will refuse to start: ${scopedLimitFacts.parseError}`,
|
|
1336
|
+
fix: `fix ${scopedLimitFacts.path} by hand, or through the dispatch_limit_* tools / the panel's m key once it parses again -- doctor never rewrites limits content`,
|
|
1337
|
+
});
|
|
1338
|
+
}
|
|
1339
|
+
|
|
1340
|
+
// The dead-scope advisory (issue #242), honest about what doctor can actually judge. A forge repo
|
|
1341
|
+
// always contains "/" and never begins "/", "./" or "../" or carries a backslash, so a scope in any
|
|
1342
|
+
// of THOSE shapes can only ever be a folder -- and a folder row that matches no trigger's canonical
|
|
1343
|
+
// run.folder guards nothing. Rows that COULD be a repo (an "a/b" shape) stay silent, not caveated:
|
|
1344
|
+
// webhook jobs carry their repo in the delivery, which triggers.json cannot enumerate, so a line on
|
|
1345
|
+
// every legitimate repo cap would be standing noise that teaches skimming (`repositories` is empty
|
|
1346
|
+
// for every valid file today -- run.repository is azure-only, its own fact says so). Guarded on the
|
|
1347
|
+
// TRIGGERS facts being readable too: a zeroed `folders` from an absent or unparseable triggers file
|
|
1348
|
+
// has no honest claim to make (readTriggerFacts' own rule). ok:true -- the replica advisory's tier,
|
|
1349
|
+
// and like it, everything the operator needs lives in the LABEL: an ok:true check never prints its
|
|
1350
|
+
// fix line.
|
|
1351
|
+
if (scopedLimitFacts.parseError === null && scopedLimitFacts.limits.length > 0 && parseError === null && triggersFilePath !== null) {
|
|
1352
|
+
const folderSet = new Set(folders);
|
|
1353
|
+
const folderOnly = (s) => s.startsWith("/") || s.startsWith("./") || s.startsWith("../") || s.includes("\\") || !s.includes("/") || /^[A-Za-z]:/.test(s);
|
|
1354
|
+
const dead = scopedLimitFacts.limits.map((l) => l.scope).filter((s) => folderOnly(s) && !folderSet.has(s));
|
|
1355
|
+
if (dead.length > 0) {
|
|
1356
|
+
checks.push({
|
|
1357
|
+
ok: true,
|
|
1358
|
+
warn: true,
|
|
1359
|
+
label: `${dead.length} scoped limit(s) name a folder no trigger runs in (${dead.join(", ")}) -- the cap guards nothing; scopes match exactly (no globs, folders by resolved ABSOLUTE path), so check the spelling against triggers.json run.folder or delete the entry`,
|
|
1360
|
+
fix: `edit ${scopedLimitFacts.path} by hand or via dispatch_limit_edit/_delete -- repo-shaped scopes are never flagged here, because a webhook job's repo comes from the delivery, which triggers.json cannot enumerate`,
|
|
1361
|
+
});
|
|
1362
|
+
}
|
|
1363
|
+
}
|
|
1364
|
+
|
|
1286
1365
|
// REQ-RESURRECTABLE-SANDBOX. A warning, never a failure: retention is a convenience, and the only thing
|
|
1287
1366
|
// worth surfacing is that finished runs' directories -- a repository clone plus the run's prompt.md and
|
|
1288
1367
|
// event.json, so issue text -- are sitting on disk, and how many. An operator who never opens a sandbox
|
|
@@ -1471,16 +1550,48 @@ function parseSecretProfilesSafe(raw) {
|
|
|
1471
1550
|
}
|
|
1472
1551
|
}
|
|
1473
1552
|
|
|
1553
|
+
/**
|
|
1554
|
+
* The scoped-limits facts (issue #242): the parsed rows when PI_SCOPED_LIMITS_FILE is set, or the
|
|
1555
|
+
* boot-blocking reason when it will not load. Unset is `none` -- the worker enforces no scoped limits
|
|
1556
|
+
* and doctor has nothing to say (the mutex is code and needs no check). A configured-but-missing file
|
|
1557
|
+
* IS a parseError here: loadScopedLimits refuses boot on it, so doctor must too. Raw fs errors
|
|
1558
|
+
* (EACCES, EISDIR) are reported the same way, deliberately unlike readTriggerFacts' tagged-only
|
|
1559
|
+
* filter: the worker's own boot load is an unguarded readFileSync, so those throws refuse startup
|
|
1560
|
+
* exactly as a parse failure does, and the check's claim is "will the worker start", not "is the
|
|
1561
|
+
* content valid".
|
|
1562
|
+
*/
|
|
1563
|
+
function readScopedLimitFacts(env, fileExists) {
|
|
1564
|
+
const none = { limits: [], parseError: null, path: null };
|
|
1565
|
+
const path = env.PI_SCOPED_LIMITS_FILE;
|
|
1566
|
+
if (typeof path !== "string" || path.trim() === "") return none;
|
|
1567
|
+
if (!fileExists(path)) return { limits: [], parseError: `scoped-limits file does not exist: ${path}`, path };
|
|
1568
|
+
try {
|
|
1569
|
+
return { limits: parseScopedLimits(readFileSync(path, "utf8"), path), parseError: null, path };
|
|
1570
|
+
} catch (e) {
|
|
1571
|
+
return { limits: [], parseError: e?.message ?? String(e), path };
|
|
1572
|
+
}
|
|
1573
|
+
}
|
|
1574
|
+
|
|
1474
1575
|
function readTriggerFacts(env, fileExists, cwd) {
|
|
1475
|
-
const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, secretProfiles: [], localSecretFolders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
|
|
1576
|
+
const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], folders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
|
|
1476
1577
|
try {
|
|
1477
1578
|
// Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
|
|
1478
1579
|
// (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
|
|
1479
1580
|
// the receiver will not boot. An absent file still means "no triggers at all", exactly as before.
|
|
1480
1581
|
const path = env.PI_TRIGGERS_FILE ?? join(cwd, "triggers.json");
|
|
1481
1582
|
if (!fileExists(path)) return none;
|
|
1482
|
-
const
|
|
1583
|
+
const text = readFileSync(path, "utf8");
|
|
1584
|
+
const triggers = parseTriggers(text, path);
|
|
1585
|
+
// The one-shot facts are counted from the RAW entries, not the parsed records, because the
|
|
1586
|
+
// validator collapses a disarmed entry to a sentinel that carries neither `once` nor
|
|
1587
|
+
// `disarmed` -- exactly so nothing can match it -- which also erases it from every parsed
|
|
1588
|
+
// count above. Doctor is the surface that must still SEE the spent entry: "why did nothing
|
|
1589
|
+
// fire" is answered by a spent row, and only the raw file still holds it. Safe unguarded:
|
|
1590
|
+
// parseTriggers just accepted this same text, so JSON.parse cannot throw here.
|
|
1591
|
+
const rawEntries = JSON.parse(text)?.triggers ?? [];
|
|
1483
1592
|
return {
|
|
1593
|
+
onceArmed: rawEntries.filter((t) => t?.on?.once === true && t.on.disarmed === undefined).length,
|
|
1594
|
+
onceSpent: rawEntries.filter((t) => t?.on?.disarmed !== undefined).length,
|
|
1484
1595
|
requiring: triggers.filter((t) => t.run.packages === true).length,
|
|
1485
1596
|
resuming: triggers.filter((t) => t.run.resume === true).length,
|
|
1486
1597
|
// REQ-PER-TRIGGER-INSTRUCTION. Counted beside `resuming` for the same reason: it is a per-trigger
|
|
@@ -1504,6 +1615,10 @@ function readTriggerFacts(env, fileExists, cwd) {
|
|
|
1504
1615
|
// read-write with no clone, so a credential an agent writes into .env lands in the operator's real
|
|
1505
1616
|
// repository rather than a temp dir that gets swept. Deduped for skillsDirs' reason.
|
|
1506
1617
|
localSecretFolders: [...new Set(triggers.filter((t) => t.run.secrets !== undefined && t.run.kind === "local" && typeof t.run.folder === "string").map((t) => t.run.folder))].sort(),
|
|
1618
|
+
// Issue #242: every local run.folder, CANONICALIZED the way the scoped-limits matcher
|
|
1619
|
+
// canonicalizes a job's folder (one derivation -- canonicalScope, never re-spelled here), so
|
|
1620
|
+
// the unreferenced-scope advisory compares like with like across spelling variants.
|
|
1621
|
+
folders: [...new Set(triggers.filter((t) => t.run.kind === "local" && typeof t.run.folder === "string").map((t) => canonicalScope({ kind: "local", folder: t.run.folder })))].sort(),
|
|
1507
1622
|
optingOut: triggers.filter((t) => t.run.packages === false).length,
|
|
1508
1623
|
images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
|
|
1509
1624
|
// REQ-PER-TRIGGER-SKILLS. The distinct host directories the file names, deduped like `images`,
|
|
@@ -1534,6 +1649,11 @@ function readTriggerFacts(env, fileExists, cwd) {
|
|
|
1534
1649
|
packages: t.run.packages !== false,
|
|
1535
1650
|
}))
|
|
1536
1651
|
.filter((f) => typeof f.flow === "string"),
|
|
1652
|
+
// Explicit on the success path too (issue #242): the dead-scope advisory distinguishes
|
|
1653
|
+
// "facts read clean" (path set, no error) from the zeroed `none` -- an implicit undefined
|
|
1654
|
+
// here made that test silently false for every deployment.
|
|
1655
|
+
parseError: null,
|
|
1656
|
+
path,
|
|
1537
1657
|
};
|
|
1538
1658
|
} catch (e) {
|
|
1539
1659
|
// REPORTED, not swallowed. This catch used to justify itself with "a malformed triggers file already
|
package/src/get-token.mjs
CHANGED
|
@@ -104,10 +104,15 @@ export async function makeGitHubAuth(cfg, deps = {}) {
|
|
|
104
104
|
);
|
|
105
105
|
}
|
|
106
106
|
const repositoryNames = [repoNameOf(repo)]; // scope to the ONE repo; owner stripped
|
|
107
|
+
// An optional PERMISSIONS narrowing (issue #231): the receiver's closer-permission lookup asks
|
|
108
|
+
// for `{ metadata: "read" }`, so the token it holds for that one question cannot write anything
|
|
109
|
+
// even if leaked. Job mints never pass this and keep the installation's full grant -- narrowing
|
|
110
|
+
// is the caller's statement of intent, not a default this mint could guess.
|
|
111
|
+
const permissions = job?.permissions;
|
|
107
112
|
let minted;
|
|
108
113
|
try {
|
|
109
114
|
const appAuth = createAppAuth(auth);
|
|
110
|
-
minted = await appAuth({ type: "installation", repositoryNames });
|
|
115
|
+
minted = await appAuth({ type: "installation", repositoryNames, ...(permissions && { permissions }) });
|
|
111
116
|
} catch (error) {
|
|
112
117
|
throw classifyAppMintError(error);
|
|
113
118
|
}
|
package/src/index.mjs
CHANGED
|
@@ -2,11 +2,19 @@ import { execFile } from "node:child_process";
|
|
|
2
2
|
import { promisify } from "node:util";
|
|
3
3
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
4
4
|
import { InfraRetry, runJob } from "./processor.mjs";
|
|
5
|
+
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight } from "./scoped-limits.mjs";
|
|
5
6
|
|
|
6
7
|
const exec = promisify(execFile);
|
|
7
8
|
|
|
8
9
|
export const QUEUE = "pi-jobs";
|
|
9
10
|
export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
11
|
+
// The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
|
|
12
|
+
// JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
|
|
13
|
+
// (<=360 wakes across a 30-minute hold, each ~1ms of synchronous predicate briefly occupying a slot)
|
|
14
|
+
// while a same-folder CHAINED job -- enqueued by its parent before the parent's finally releases the
|
|
15
|
+
// folder -- pays exactly one re-check, not fifteen seconds of dead air. No jitter: one worker per
|
|
16
|
+
// docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
|
|
17
|
+
export const SCOPE_BUSY_RECHECK_MS = 5_000;
|
|
10
18
|
|
|
11
19
|
/**
|
|
12
20
|
* Build the BullMQ processor.
|
|
@@ -27,7 +35,7 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
|
27
35
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
28
36
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
29
37
|
*/
|
|
30
|
-
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
|
|
38
|
+
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now() }) {
|
|
31
39
|
return async function processor(job, token, signal) {
|
|
32
40
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
33
41
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -45,19 +53,68 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
45
53
|
throw new DelayedError();
|
|
46
54
|
}
|
|
47
55
|
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
56
|
+
// Per-scope concurrency and the one-job-per-folder mutex (issue #242,
|
|
57
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT). SECOND, after the pause gate (a paused job must not burn
|
|
58
|
+
// re-check wakes) and STRICTLY above the `try` below, like the pause gate and for the same two
|
|
59
|
+
// reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
|
|
60
|
+
// catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
|
|
61
|
+
// handling exactly as the pause gate's does (inside the try it would become a permanent failure
|
|
62
|
+
// plus a failure record for what was a transient blip). The limits snapshot is read ONCE here and
|
|
63
|
+
// shared with `scopedCaps` below, so the gate and the money ledger cannot disagree mid-job.
|
|
64
|
+
// tryAcquire is a synchronous check-and-increment -- no await between read and take, so Node's
|
|
65
|
+
// single thread makes it atomic at any concurrency -- and the local-folder limit is a structural 1
|
|
66
|
+
// (concurrencyFor) with no file and no off-switch: the scheduler mints a cron trigger's next
|
|
67
|
+
// occurrence at pickup and promotes it on time alone, so a slow run overlaps its own successor
|
|
68
|
+
// (measured: 301ms of live container overlap through this very processor) unless this gate holds.
|
|
69
|
+
// Infinity-limited scopes still acquire, so release stays uniform for every scoped job.
|
|
70
|
+
const limits = scopedLimits();
|
|
71
|
+
const scope = canonicalScope(job.data);
|
|
72
|
+
let held = false;
|
|
73
|
+
if (scope) {
|
|
74
|
+
if (!inFlight.tryAcquire(scope, concurrencyFor(job.data, limits))) {
|
|
75
|
+
// Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
|
|
76
|
+
// The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
|
|
77
|
+
// host path); the delayed count and the job id are what an operator needs to see it.
|
|
78
|
+
deps?.log?.("scope_busy_deferred", { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS });
|
|
79
|
+
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
80
|
+
throw new DelayedError();
|
|
81
|
+
}
|
|
82
|
+
held = true;
|
|
83
|
+
}
|
|
54
84
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
85
|
+
let startedAt;
|
|
86
|
+
let name;
|
|
87
|
+
let timer;
|
|
88
|
+
let onAbort;
|
|
89
|
+
try {
|
|
90
|
+
// Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
|
|
91
|
+
// finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
|
|
92
|
+
// scope until a worker restart. Nothing in this block CAN throw today (setTimeout and
|
|
93
|
+
// addEventListener on the bullmq-allocated controller are total at processor arity 3); the
|
|
94
|
+
// guard is structural, not observational.
|
|
95
|
+
startedAt = new Date().toISOString();
|
|
96
|
+
name = `pi-job-${job.id}`;
|
|
97
|
+
timer = setTimeout(() => {
|
|
98
|
+
// BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
|
|
99
|
+
Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
|
|
100
|
+
}, timeoutMs);
|
|
101
|
+
|
|
102
|
+
// Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
|
|
103
|
+
// after the grace period; the runner exits and runContainer returns/throws.
|
|
104
|
+
onAbort = () => {
|
|
105
|
+
Promise.resolve(stopContainer(name)).catch(() => {});
|
|
106
|
+
};
|
|
107
|
+
signal.addEventListener("abort", onAbort, { once: true });
|
|
108
|
+
} catch (error) {
|
|
109
|
+
// Release and CLEAR the flag: this throw never reaches the main finally below, but a shared
|
|
110
|
+
// scope must never be releasable twice -- a double release frees another holder's slot.
|
|
111
|
+
if (held) {
|
|
112
|
+
inFlight.release(scope);
|
|
113
|
+
held = false;
|
|
114
|
+
}
|
|
115
|
+
clearTimeout(timer);
|
|
116
|
+
throw error;
|
|
117
|
+
}
|
|
61
118
|
|
|
62
119
|
try {
|
|
63
120
|
const settings = await getSettings();
|
|
@@ -99,6 +156,10 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
99
156
|
// The daily TOKEN cap (issue #25), same overlay > env resolution. Check-AFTER, so it gates the
|
|
100
157
|
// NEXT job on prior recorded spend; null => the daily token counter is disabled.
|
|
101
158
|
tokenCap: settings.dailyTokenCap,
|
|
159
|
+
// This job's scoped budget windows (issue #242), from the SAME limits snapshot the pickup
|
|
160
|
+
// gate above read -- one read per pickup, so gate and ledger agree for this job's whole
|
|
161
|
+
// life. Null when no row carries a money window for this scope.
|
|
162
|
+
scopedCaps: budgetCapsFor(job.data, limits),
|
|
102
163
|
...deps,
|
|
103
164
|
runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
|
|
104
165
|
// REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
|
|
@@ -125,6 +186,11 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
125
186
|
// `queueJobId`, mirroring the collectChain injection above. Omitted when unwired so a bare
|
|
126
187
|
// processor keeps runJob's plain (job, token) call.
|
|
127
188
|
...(deps.prepareWorkspace ? { prepareWorkspace: (j, t) => deps.prepareWorkspace(j, t, { queueJobId: job.id }) } : {}),
|
|
189
|
+
// The one-shot pre-spend check (issue #231) needs the REAL BullMQ job's `.id` to excuse this
|
|
190
|
+
// delivery's own earlier attempt -- runJob's effectiveJob has no `.id`, prepareWorkspace's
|
|
191
|
+
// own injection above states why, and this one mirrors it. Omitted when unwired so a bare
|
|
192
|
+
// processor keeps runJob's admit-everything default.
|
|
193
|
+
...(deps.checkOnceSpent ? { checkOnceSpent: (j) => deps.checkOnceSpent(j, { queueJobId: job.id }) } : {}),
|
|
128
194
|
});
|
|
129
195
|
recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
130
196
|
return result;
|
|
@@ -135,13 +201,16 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
135
201
|
// records it as failed-and-distinct in the queue's failed set without a retry.
|
|
136
202
|
throw new UnrecoverableError(error.message);
|
|
137
203
|
} finally {
|
|
204
|
+
// Release FIRST and never throw (release clamps at zero by construction): a throw here would
|
|
205
|
+
// mask the job's real error, and a missed release wedges the scope until a worker restart.
|
|
206
|
+
if (held) inFlight.release(scope);
|
|
138
207
|
clearTimeout(timer);
|
|
139
208
|
signal.removeEventListener("abort", onAbort);
|
|
140
209
|
}
|
|
141
210
|
};
|
|
142
211
|
}
|
|
143
212
|
|
|
144
|
-
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, extraClosers = [] }) {
|
|
213
|
+
export function createWorker({ connection, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight, extraClosers = [] }) {
|
|
145
214
|
let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
|
|
146
215
|
const processor = makeProcessor({
|
|
147
216
|
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
@@ -154,6 +223,11 @@ export function createWorker({ connection, concurrency, getSettings, redis, deps
|
|
|
154
223
|
if (Number.isInteger(n) && worker.concurrency !== n) worker.concurrency = n;
|
|
155
224
|
},
|
|
156
225
|
pauseUntil,
|
|
226
|
+
// Undefined pass-throughs take makeProcessor's own defaults (no limits; a fresh per-processor
|
|
227
|
+
// in-flight map -- one per worker process, which under DES-CONCURRENCY-3's one-worker-per-daemon
|
|
228
|
+
// shape means one per daemon).
|
|
229
|
+
scopedLimits,
|
|
230
|
+
inFlight,
|
|
157
231
|
deps,
|
|
158
232
|
recordRun,
|
|
159
233
|
});
|
package/src/init.mjs
CHANGED
|
@@ -19,6 +19,11 @@ const EMPTY_PACKAGES = `${JSON.stringify({ packages: [] }, null, 2)}\n`;
|
|
|
19
19
|
// Operator-declared subscription plans (issue #53), read by the admin extension only — never at job
|
|
20
20
|
// time. Versioned because a newer file must fail loud, and that cannot be retrofitted into a v1 reader.
|
|
21
21
|
const EMPTY_SUBSCRIPTIONS = `${JSON.stringify({ version: 1, subscriptions: [] }, null, 2)}\n`;
|
|
22
|
+
// Scoped limits (issue #242): per repo/folder run caps and concurrency. Empty is inert -- and the
|
|
23
|
+
// one-job-per-folder mutex for local jobs is code, not configuration, so it needs no scaffold line.
|
|
24
|
+
// Versioned for the subscriptions reason, sharpened: this is enforcement config, and a silently
|
|
25
|
+
// down-read newer file would be a silently widened spend limit.
|
|
26
|
+
const EMPTY_SCOPED_LIMITS = `${JSON.stringify({ version: 1, limits: [] }, null, 2)}\n`;
|
|
22
27
|
/**
|
|
23
28
|
* The egress allowlist (REQ-EGRESS-ALLOWLIST): the hosts a job container may reach, one bare hostname per
|
|
24
29
|
* line. Scaffolded with the three a job cannot work without, and NOT empty -- unlike every other scaffold
|
|
@@ -73,6 +78,7 @@ export function runInit(cwd = process.cwd(), deps = {}) {
|
|
|
73
78
|
scaffold(fs, results, join(cwd, "pause-windows.json"), EMPTY_PAUSE_WINDOWS, "empty pause-windows list");
|
|
74
79
|
scaffold(fs, results, join(cwd, "pi-packages.json"), EMPTY_PACKAGES, "empty pi package list (stage with import-pi --with-packages)");
|
|
75
80
|
scaffold(fs, results, join(cwd, "subscriptions.json"), EMPTY_SUBSCRIPTIONS, "empty subscription list (declare plan prices for the admin's cost analytics)");
|
|
81
|
+
scaffold(fs, results, join(cwd, "scoped-limits.json"), EMPTY_SCOPED_LIMITS, "empty scoped-limits list (per repo/folder caps; the folder mutex needs no file)");
|
|
76
82
|
scaffold(fs, results, join(cwd, "egress-allowlist.conf"), DEFAULT_EGRESS_ALLOWLIST, "egress allowlist (provider + forge + registry; the egress policy is on unless PI_EGRESS=0)");
|
|
77
83
|
|
|
78
84
|
for (const [verb, name, note] of results) {
|
package/src/processor.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { lstatSync } from "node:fs";
|
|
2
2
|
import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
|
|
3
3
|
import { configError } from "./config.mjs";
|
|
4
|
+
import { scopeKeyPrefix } from "./scoped-limits.mjs";
|
|
4
5
|
import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
|
|
5
6
|
import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
|
|
6
7
|
|
|
@@ -37,11 +38,22 @@ export async function runJob(job, deps) {
|
|
|
37
38
|
caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
|
|
38
39
|
softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
|
|
39
40
|
tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
|
|
41
|
+
// { scope, caps: { day, week, month } } | null -- this job's scoped budget windows (issue #242,
|
|
42
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT), resolved by the wiring from the same watched-limits snapshot the
|
|
43
|
+
// pickup gate read. Null when the file is unset or the scope's row is concurrency-only; the default
|
|
44
|
+
// keeps an unwired processor byte-identical. The folder MUTEX does not live here -- it is the pickup
|
|
45
|
+
// gate's, pre-everything; this is only the money half.
|
|
46
|
+
scopedCaps = null,
|
|
40
47
|
recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
|
|
41
48
|
// (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
|
|
42
49
|
// this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
|
|
43
50
|
// omits it behaves exactly as before -- the container's own failure stays the backstop.
|
|
44
51
|
imagePreflight = async () => ({ ok: true }),
|
|
52
|
+
// (job) => { ok } | { refused, at, jobId }. The one-shot pre-spend check (issue #231,
|
|
53
|
+
// DES-ONE-SHOT-DISARM-IN-THE-FILE). Default admits everything -- an unwired processor behaves
|
|
54
|
+
// exactly as before, and the gate below only calls it for a job whose matched rule was a
|
|
55
|
+
// one-shot, so the default is never a probe running on every delivery.
|
|
56
|
+
checkOnceSpent = async () => ({ ok: true }),
|
|
45
57
|
// REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
|
|
46
58
|
// deployment with no egress policy does -- which is also what the real factory returns when unarmed.
|
|
47
59
|
egressPreflight = async () => ({ ok: true }),
|
|
@@ -119,8 +131,30 @@ export async function runJob(job, deps) {
|
|
|
119
131
|
let token = null;
|
|
120
132
|
let prepared = null;
|
|
121
133
|
let reserved = false;
|
|
134
|
+
let scopedReserved = false;
|
|
122
135
|
|
|
123
136
|
try {
|
|
137
|
+
// The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
|
|
138
|
+
// the docker inspect below, free, determinate, credential-less. Only a FOREIGN positive
|
|
139
|
+
// disarmed mark refuses -- the check excuses this queue job's own id, so a retry of the
|
|
140
|
+
// delivery that spent the trigger still runs (attempts:2 stays attempts:2) -- and anything
|
|
141
|
+
// unreadable or changed means "run": fail-open, because the disarm writer owns the loud
|
|
142
|
+
// refusals, and a broken read must never wedge every once job. In the compose topology the
|
|
143
|
+
// receiver reads a dead inode until restart, so this check is the once-enforcement layer
|
|
144
|
+
// there, not optional hardening.
|
|
145
|
+
if (job.trigger?.matched?.once === true) {
|
|
146
|
+
const spent = await checkOnceSpent(job);
|
|
147
|
+
if (spent.refused) {
|
|
148
|
+
// Commented like every sibling policy refusal: explainability is this refusal's whole
|
|
149
|
+
// purpose, and only a DISTINCT re-close reaches it past the GUID dedup, so the noise
|
|
150
|
+
// bound is the operator's own reopen-close rate. `at`/`jobId` are harness-written
|
|
151
|
+
// provenance, never payload text.
|
|
152
|
+
await comment(job, `Refused: this one-shot trigger was already spent${spent.at ? ` at ${spent.at}` : ""}${spent.jobId ? ` by job ${spent.jobId}` : ""}. The close that armed it has already produced a run; delete on.disarmed from the trigger entry to re-arm it. Not run.`);
|
|
153
|
+
log("refused_once_already_spent", { triggerIndex: job.trigger?.matched?.index ?? null });
|
|
154
|
+
return { outcome: "policy", reason: "once-already-spent", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
|
|
124
158
|
// The job image must exist on THIS host before anything else happens. Free, determinate and
|
|
125
159
|
// credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
|
|
126
160
|
// image refuses without minting a credential it will not use, cloning a repo it will not read, or
|
|
@@ -432,11 +466,51 @@ export async function runJob(job, deps) {
|
|
|
432
466
|
return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
433
467
|
}
|
|
434
468
|
|
|
435
|
-
//
|
|
469
|
+
// Per-scope budget windows (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): the NARROWER ledger
|
|
470
|
+
// reserves FIRST, so a noisy scope's refusals never consume a global slot -- the global INCR below
|
|
471
|
+
// runs only for jobs the scope admitted. Same atomic INCR, same refused-still-counts invariant,
|
|
472
|
+
// through budget.mjs's keyPrefix seam (dayKey/weekKey/monthKey under budget:s:<hash16>). softHoldPct
|
|
473
|
+
// is deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob;
|
|
474
|
+
// scoped windows are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
|
|
475
|
+
if (scopedCaps) {
|
|
476
|
+
// A redis fault BETWEEN this reserve and the global one below strands the scoped INCR with no
|
|
477
|
+
// run and no refund -- the pre-existing mid-reserve posture, shared with the global ledger's
|
|
478
|
+
// own partial-INCR seam; the compensating release below covers REFUSALS, not faults.
|
|
479
|
+
const scoped = await reserveBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
480
|
+
scopedReserved = true;
|
|
481
|
+
if (!scoped.allowed) {
|
|
482
|
+
const w = scoped.blockedWindow;
|
|
483
|
+
const win = scoped.windows[w];
|
|
484
|
+
// A local job's scope is a full host path and its "comment" is not dropped -- the wiring's
|
|
485
|
+
// local adapter LOGS the text (start.mjs forgeFor fallthrough) -- so the path must never
|
|
486
|
+
// enter the message; "this folder" is enough beside the jobId the adapter logs. A forge
|
|
487
|
+
// scope IS the repo the comment posts on, safe to name.
|
|
488
|
+
const scopeLabel = job.kind === "local" ? "this folder" : scopedCaps.scope;
|
|
489
|
+
await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
|
|
490
|
+
// The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would
|
|
491
|
+
// put a full host path in the worker log against no-pii-in-logs (the record keeps only
|
|
492
|
+
// basename(folder) for the same reason). The admin recomputes the key from the configured
|
|
493
|
+
// scope to join it back.
|
|
494
|
+
log("over_scope_budget", { scopeKey: scopeKeyPrefix(scopedCaps.scope), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
|
|
495
|
+
// budgetReserved false: the GLOBAL slot was never touched (scoped reserves first). The scoped
|
|
496
|
+
// counter did INCR and keeps it -- its own refused-reservation-still-counts, per ledger.
|
|
497
|
+
return { outcome: "policy", reason: "scope-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
498
|
+
}
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
// GLOBAL budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
|
|
436
502
|
// every active window (day + optional week/month) and the soft-hold band in one atomic pass.
|
|
437
503
|
const budget = await reserveBudget(redis, { caps, softHoldPct, now });
|
|
438
504
|
reserved = true;
|
|
439
505
|
if (!budget.allowed) {
|
|
506
|
+
// The scoped reserve above committed before this global refusal -- give that slot back. Without
|
|
507
|
+
// this, an exhausted global window drains every arriving scope's own day/week/month counters
|
|
508
|
+
// with zero runs to show for it (a storm against a spent global daily cap would empty a repo's
|
|
509
|
+
// week by noon). The scoped ledger's refused-still-counts covers the SCOPE's own refusal above,
|
|
510
|
+
// never a refusal it did not issue.
|
|
511
|
+
if (scopedReserved && scopedCaps) {
|
|
512
|
+
await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
513
|
+
}
|
|
440
514
|
const w = budget.blockedWindow;
|
|
441
515
|
const win = budget.windows[w];
|
|
442
516
|
if (budget.reason === "soft-hold") {
|
|
@@ -535,8 +609,12 @@ export async function runJob(job, deps) {
|
|
|
535
609
|
// budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
|
|
536
610
|
// true for a real container that ran and spent (exit-1 infra / unknown exit).
|
|
537
611
|
if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
|
|
538
|
-
if (
|
|
539
|
-
|
|
612
|
+
if (e instanceof InfraRetry && e.reason === "container-never-started") {
|
|
613
|
+
// Both-or-neither (issue #242): a never-started container can only follow BOTH reserves (the
|
|
614
|
+
// scoped one precedes the global one, and the container follows both), so they refund
|
|
615
|
+
// together -- and a scoped refusal returned above without ever touching the global ledger.
|
|
616
|
+
if (reserved) await releaseBudget(redis, { caps, now });
|
|
617
|
+
if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
540
618
|
}
|
|
541
619
|
throw e;
|
|
542
620
|
} finally {
|
package/src/queue.mjs
CHANGED
|
@@ -1,6 +1,11 @@
|
|
|
1
1
|
import { Queue } from "bullmq";
|
|
2
2
|
import { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId } from "./job-id.mjs";
|
|
3
3
|
import { targetSeparator } from "./forges.mjs";
|
|
4
|
+
import { PR_CLOSE_ACTIONS } from "./triggers.mjs";
|
|
5
|
+
|
|
6
|
+
// The close words in every forge's spelling, derived from the one table (never re-typed here): a
|
|
7
|
+
// matched PR action in this set marks a close job for the semantic-key discriminant below.
|
|
8
|
+
const PR_CLOSE_WORDS = new Set(Object.values(PR_CLOSE_ACTIONS));
|
|
4
9
|
|
|
5
10
|
export const QUEUE = "pi-jobs";
|
|
6
11
|
export { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId };
|
|
@@ -180,6 +185,21 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
|
|
|
180
185
|
...(replica !== undefined && { replica }),
|
|
181
186
|
...(replicas !== undefined && { replicas }),
|
|
182
187
|
};
|
|
188
|
+
// A close-triggered job (issue #231) leads the semantic key's flow slot with `closed:`. Without it,
|
|
189
|
+
// a label/comment/PR job on the same target and flow inside the 10-minute window silently swallows
|
|
190
|
+
// the close job -- and because a swallowed close job writes no run record, the once trigger it was
|
|
191
|
+
// meant to spend never disarms: a permanently dead one-shot with nothing in the panel to say why.
|
|
192
|
+
// The discriminant is DERIVED from the matched rule (`issue` type, or a PR close action word) rather
|
|
193
|
+
// than carried as a job field: an execution detail of dedup is not a fact about the delivery, and
|
|
194
|
+
// `data`/`event.json` stay byte-identical. `:` is outside the skill-name charset -- enforced at load
|
|
195
|
+
// since #231 -- so no real flow can spell either prefixed form, and `closed:cmd:<name>` composes for
|
|
196
|
+
// close-dispatched commands (outermost discriminant first, then the entry-point prefix).
|
|
197
|
+
const matched = trigger?.matched;
|
|
198
|
+
// `type === "issue"` reads as "close" only while the issue vocabulary is close-only (it is; the
|
|
199
|
+
// tables say "one word each so far"). If that type ever grows a non-close action, this test must
|
|
200
|
+
// narrow to the matched action word, like the PR half already does.
|
|
201
|
+
const isCloseJob = matched?.type === "issue" || (matched?.type === "pull_request" && PR_CLOSE_WORDS.has(matched?.action));
|
|
202
|
+
const flowSlot = `${isCloseJob ? "closed:" : ""}${command !== undefined ? `cmd:${command}` : flow}`;
|
|
183
203
|
await queue.add(kind, data, {
|
|
184
204
|
jobId,
|
|
185
205
|
// A command job (issue #189) fills the semantic key's flow slot with `cmd:<command>`: a command
|
|
@@ -188,7 +208,7 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
|
|
|
188
208
|
// `cmd:` prefix keeps a command named X from coalescing against a flow named X -- `:` is outside
|
|
189
209
|
// the skill-name charset, so no real flow can spell the prefixed form -- and a flow job's key
|
|
190
210
|
// stays byte-identical to before the feature.
|
|
191
|
-
deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${
|
|
211
|
+
deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${flowSlot}${replica !== undefined ? `:r${replica}` : ""}`, ttl: SEMANTIC_WINDOW_MS }, // ttl in ms
|
|
192
212
|
attempts: 2,
|
|
193
213
|
backoff: { type: "exponential", delay: 60_000 },
|
|
194
214
|
removeOnComplete: { age: 31 * 24 * 3600 }, // age in seconds -- do not cross units with the ms ttl above
|