@edgehero/pi-dispatch 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -102,6 +102,25 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
102
102
  # Default empty = fail-closed: `/dispatch secrets add` can declare nothing, and only PI_SECRET_PROFILES above is honoured.
103
103
  # PI_SECRET_RESOLVE_TIMEOUT_MS= # default 10000, per reference. Sits before a paid container and is multiplied by the reference count, so tighter than doctor's 30s.
104
104
 
105
+ # --- Holding a job until something else happens (docs/wait-for.md, issue #230) ---
106
+ # A trigger may carry "waitFor": [{ "after": "2026-09-01T09:00:00Z" }, { "profile": "jira" }]. The job is
107
+ # enqueued as usual and then HELD: it reserves no budget slot, arms no kill timer, consumes no retry attempt,
108
+ # survives a restart, and runs exactly once when every condition clears. An `after` is answered from the
109
+ # clock and costs nothing. A `profile` names one of YOUR scripts, which the worker runs on the HOST with the
110
+ # job's id-only target as its first argument.
111
+ # PI_WAIT_PROFILES= # name:/absolute/path pairs, comma separated (each entry splits on its FIRST colon, so a Windows C:\ path parses). Unset = feature off.
112
+ # Map the codes yourself, because a bare pipeline cannot: `s=$(jira issue view "$1" --plain) || exit 1` then
113
+ # `case "$s" in *"Status: Done"*) exit 0;; *) exit 3;; esac`. `grep -q` alone exits 1 for "no match", which is a counted FAULT, not "not yet".
114
+ # Exit 0 to go, 3 for not yet, 2 if it will NEVER clear (terminal), 1 if you could not tell (held, and counted).
115
+ # PI_WAIT_CHECK_TIMEOUT_MS= # default 10000, per check. Sits before a paid container and holds a concurrency slot while it runs.
116
+ # PI_WAIT_INTERVAL_MS= # default 60000, floored at 30000: a positive value BELOW the floor is raised to it, while 0, a negative, a fraction or junk still refuses at boot.
117
+ # The base cadence; it backs off toward 15 minutes, or toward YOUR value if you set a longer one.
118
+ # PI_WAIT_CHECK_SLOTS= # default 1. How many checks may run at once. Will be held below PI_CONCURRENCY at the gate, so a check can never take the last slot from a paid job.
119
+ # PI_WAIT_MAX_MS= # default 86400000 (24h). A profile hold terminates here with a named reason: a dependency, unlike a pause window, does not end on its own.
120
+ # PI_WAIT_AFTER_MAX_MS= # default 2592000000 (30d). The separate, larger ceiling on an `after` instant, which polls nothing while it waits.
121
+ # PI_WAIT_MAX_CHECKS= # default 96, per job. Nothing in the spend caps sees a check, so this is the bound that does.
122
+ # PI_WAIT_MAX_FAULTS= # default 5 consecutive "could not tell" answers. What makes a broken check loud in minutes instead of silent for a day.
123
+
105
124
  PI_SCHEDULER_STALL_MAX=2 # tear down a scheduler after N consecutive stalls (money backstop)
106
125
 
107
126
  # --- Egress policy: what a job container may reach on the network (docs/egress.md) ---
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.5.0",
3
+ "version": "1.6.1",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -56,6 +56,8 @@
56
56
  "./packages": "./src/packages.mjs",
57
57
  "./pause-windows": "./src/pause-windows.mjs",
58
58
  "./scoped-limits": "./src/scoped-limits.mjs",
59
+ "./wait-for": "./src/wait-for.mjs",
60
+ "./wait-state": "./src/wait-state.mjs",
59
61
  "./identity": "./src/identity.mjs",
60
62
  "./gitlab-identity": "./src/gitlab-identity.mjs",
61
63
  "./forgejo-identity": "./src/forgejo-identity.mjs",
package/src/config.mjs CHANGED
@@ -10,6 +10,7 @@ import { delimiter } from "node:path";
10
10
  import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
11
11
  import { MINTED_TOKEN_VARS } from "./forges.mjs";
12
12
  import { parseSecretProfiles } from "./secret-profiles.mjs";
13
+ import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, parseWaitProfiles } from "./wait-for.mjs";
13
14
 
14
15
  export function configError(message) {
15
16
  const error = new Error(message);
@@ -296,6 +297,39 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
296
297
  // Per-reference ceiling. Tighter than doctor's 30s on purpose: this runs before a paid container, is
297
298
  // multiplied by the reference count, and holds a PI_CONCURRENCY slot while it waits.
298
299
  secretResolveTimeoutMs: positiveInt(env, "PI_SECRET_RESOLVE_TIMEOUT_MS", 10000),
300
+ // Issue #230, `run.waitFor`. The operator's declared wait checks, same `name:absolute-path` grammar as
301
+ // the resolvers above and deliberately a SEPARATE variable: the two answer different questions (one
302
+ // fetches a value, one says whether to go), they will grow different bounds, and one list would make a
303
+ // resolver reachable as a gate and a gate reachable as a resolver. Unset = the feature is off and any
304
+ // trigger naming a profile refuses pre-spend. There is no `PI_WAIT_RESOLVER_ROOTS` twin because there
305
+ // is no overlay half to bound: the wait gate runs ABOVE the per-job settings read, so a wait profile
306
+ // can only ever be declared here, beside the forge tokens.
307
+ waitProfiles: parseWaitProfiles(env.PI_WAIT_PROFILES),
308
+ // Per-check ceiling, `secretResolveTimeoutMs`' twin and for its reason: this runs before a paid
309
+ // container and holds a PI_CONCURRENCY slot while it waits.
310
+ waitCheckTimeoutMs: positiveInt(env, "PI_WAIT_CHECK_TIMEOUT_MS", 10000),
311
+ // The base re-check cadence, clamped UP to the floor rather than refused (wait-for.mjs states why, and
312
+ // what the clamp does not cover). The backoff derives from elapsed time, so this is a base, not a period.
313
+ waitIntervalMs: Math.max(WAIT_INTERVAL_FLOOR_MS, positiveInt(env, "PI_WAIT_INTERVAL_MS", 60_000)),
314
+ // How long a PROFILE hold may last before it terminates with a named reason. A dependency, unlike a
315
+ // pause window, is not self-terminating by construction, so this is the bound that makes it one.
316
+ waitMaxMs: positiveInt(env, "PI_WAIT_MAX_MS", 24 * 3600 * 1000),
317
+ // The separate, far larger ceiling on an `after` instant. Deliberately NOT waitMaxMs: an `after` is a
318
+ // scheduled instant, not a poll -- one exact moveToDelayed, self-terminating, costing nothing while it
319
+ // waits -- so bounding it by the polling budget would refuse "hold this until the maintenance window
320
+ // next month" for a reason that is about subprocesses it never runs.
321
+ waitAfterMaxMs: positiveInt(env, "PI_WAIT_AFTER_MAX_MS", WAIT_AFTER_MAX_DEFAULT_MS),
322
+ // How many wait checks may run AT ONCE in this worker process. One by default, and the ceiling it
323
+ // really pins is duty cycle: slots x timeout is the most wall-clock a worker can spend answering
324
+ // questions instead of running paid jobs. Clamped below PI_CONCURRENCY at the gate so a check can
325
+ // never take the last free slot.
326
+ waitCheckSlots: positiveInt(env, "PI_WAIT_CHECK_SLOTS", 1),
327
+ // Two bounds on ONE job's checks, both logged on overflow. The count bound is SECRETS_MAX's argument
328
+ // applied over time rather than over a map, and it matters because nothing in the money system sees a
329
+ // check at all: CONST-BUDGET-BEFORE-TOKENS counts container starts. The fault bound is what makes a
330
+ // broken check loud in minutes instead of silent for a day (OQ-027: most CLIs exit 1 for everything).
331
+ waitMaxChecks: positiveInt(env, "PI_WAIT_MAX_CHECKS", 96),
332
+ waitMaxFaults: positiveInt(env, "PI_WAIT_MAX_FAULTS", 5),
299
333
  github: { ...loadGitHubAuth(env, fileExists), allowGhResume: env.PI_SESSIONS_ALLOW_GH_SOURCE === "1" },
300
334
  gitlab: loadGitLabAuth(env),
301
335
  forgejo: loadForgejoAuth(env),
package/src/doctor.mjs CHANGED
@@ -44,13 +44,14 @@
44
44
  * about severity: a --fix run still exits by the same failed/ok logic, warns stay warns, and the fix pass
45
45
  * happens at most once (check, fix, re-check -- never a loop).
46
46
  */
47
- import { chmodSync, closeSync, existsSync, lstatSync, mkdirSync, mkdtempSync, openSync, readdirSync, readFileSync, readSync, rmSync, statSync } from "node:fs";
47
+ import { chmodSync, closeSync, existsSync, lstatSync, mkdirSync, mkdtempSync, openSync, readdirSync, readFileSync, readSync, realpathSync, rmSync, statSync } from "node:fs";
48
48
  import { homedir, tmpdir } from "node:os";
49
49
  import { dirname, join, delimiter } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
52
  import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
53
53
  import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
54
+ import { WAIT_AFTER_MAX_DEFAULT_MS, afterInstantMs, parseWaitProfiles } from "./wait-for.mjs";
54
55
  import { isForgeKind } from "./forges.mjs";
55
56
  import { findLiteralSecret, ADMIN_RE } from "./import-pi.mjs";
56
57
  import { agentDirFrom, readHostPi } from "./host-pi.mjs";
@@ -265,7 +266,7 @@ export async function collectChecks(env, seams) {
265
266
  // image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
266
267
  // `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
267
268
  // run.packages: true, which arms nothing any more but is still an operator statement of intent.
268
- const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, folders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
269
+ const { requiring, waiting, waitProfiles, waitAfters, optingOut, resuming, replicating, instructing, commands, secreting, onceArmed, onceSpent, secretProfiles, localSecretFolders, folders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
269
270
  const scopedLimitFacts = readScopedLimitFacts(env, fileExists);
270
271
  // FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
271
272
  // zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
@@ -1099,14 +1100,84 @@ export async function collectChecks(env, seams) {
1099
1100
  });
1100
1101
  }
1101
1102
 
1103
+ // Issue #230, `run.waitFor`. The PARSE is asked unconditionally, and the rest only when something waits.
1104
+ // That split is not the usual "only report what this deployment uses": `loadConfig` calls
1105
+ // `parseWaitProfiles` on every boot whether or not a trigger holds anything, so a garbled variable is a
1106
+ // worker that will not START, and gating that behind `waiting > 0` would have hidden it from precisely
1107
+ // the operator this check exists for. The ordinary sequence produces that state: declare the variable,
1108
+ // restart, then write the trigger -- doctor is run in the middle, and would have said nothing at all.
1109
+ const { profiles: declaredWaits, error: waitParseFailure } = parseWaitProfilesSafe(env.PI_WAIT_PROFILES);
1110
+ if (waitParseFailure) {
1111
+ checks.push({ ok: false, label: "PI_WAIT_PROFILES does not parse", fix: `${waitParseFailure} -- the worker refuses to BOOT until this is fixed, rather than dropping the entry and leaving you a check you believe is wired` });
1112
+ }
1113
+ // Everything below is about triggers, so it is asked only when a trigger holds something: a deployment
1114
+ // that waits on nothing must not carry a line about a feature it does not use, which is the always-on
1115
+ // advisory this file avoids everywhere else.
1116
+ if (waiting > 0 && !waitParseFailure) {
1117
+ // A HARD FAIL, `secretProfiles`' twin and for its reason: these jobs refuse pre-spend until the
1118
+ // profile is declared, deliberately, rather than starting without ever asking the question.
1119
+ const missing = waitProfiles.filter((name) => !(name in declaredWaits));
1120
+ checks.push({
1121
+ ok: missing.length === 0,
1122
+ label: missing.length === 0 ? `${waiting} trigger(s) hold their jobs, and every wait profile they name is declared` : `${waiting} trigger(s) hold their jobs, but ${missing.length} named wait profile(s) are not declared: ${missing.join(", ")}`,
1123
+ fix: `declare them in PI_WAIT_PROFILES as name:/absolute/path pairs (a check is one line, and its exit code is the answer: 0 go, 3 not yet, 2 never, 1 could not tell) -- these jobs refuse pre-spend until you do`,
1124
+ });
1125
+ // Each declared check is stat'd, exactly as a resolver is: absent, a directory, or not executable is a
1126
+ // check that can never answer. NAMED profiles fail; declared-but-unnamed ones only warn, because no
1127
+ // job looks them up -- a retired entry left in `.env` is untidy, not a deployment that refuses
1128
+ // deliveries, and failing the whole command on it is the same over-reporting the `waiting > 0` gate
1129
+ // exists to prevent. The probe is `wait-check.mjs`'s own, symlinks and all, so doctor and the gate
1130
+ // cannot disagree about what will run.
1131
+ const named = new Set(waitProfiles);
1132
+ for (const name of Object.keys(declaredWaits).sort()) {
1133
+ const path = declaredWaits[name];
1134
+ const st = statPath(path);
1135
+ const used = named.has(name);
1136
+ checks.push({
1137
+ ok: st.ok || !used,
1138
+ ...(st.ok ? { label: `Wait profile ${name} -> ${path}${used ? "" : " (declared, named by no trigger)"}` } : {}),
1139
+ ...(st.ok
1140
+ ? {}
1141
+ : used
1142
+ ? { label: `Wait profile ${name} -> ${path} ${st.why}`, fix: "every job naming this profile refuses pre-spend as wait-profile-unknown until the path resolves to an executable file" }
1143
+ : { warn: true, label: `Wait profile ${name} -> ${path} ${st.why}, and no trigger names it`, fix: "no job looks this up, so nothing refuses today -- fix the path or drop the entry before a trigger starts naming it" }),
1144
+ });
1145
+ }
1146
+ // An `after` further out than the ceiling refuses EVERY delivery at first pickup, and doctor holds
1147
+ // both halves of that arithmetic, so it is the same class of finding as an undeclared profile: a
1148
+ // trigger that cannot deliver, knowable before anything is enqueued. Measured from now, exactly as
1149
+ // the gate measures it.
1150
+ const afterMax = Number(env.PI_WAIT_AFTER_MAX_MS ?? "") > 0 ? Number(env.PI_WAIT_AFTER_MAX_MS) : WAIT_AFTER_MAX_DEFAULT_MS;
1151
+ const beyond = waitAfters.filter((iso) => {
1152
+ const ms = afterInstantMs(iso);
1153
+ return ms !== null && ms - Date.now() > afterMax;
1154
+ });
1155
+ if (beyond.length > 0) {
1156
+ checks.push({
1157
+ ok: false,
1158
+ label: `${beyond.length} wait condition(s) name an instant beyond PI_WAIT_AFTER_MAX_MS: ${beyond.join(", ")}`,
1159
+ fix: "every delivery refuses pre-spend as wait-after-beyond-max at first pickup -- bring the instant inside the ceiling or raise PI_WAIT_AFTER_MAX_MS",
1160
+ });
1161
+ }
1162
+ // The version-floor disclosure. Stated ONCE, as a fact rather than a warning, because it is not a
1163
+ // defect: it is the one thing about this feature an operator cannot check from here. `doctor` runs
1164
+ // on the worker host and cannot see the receiver's installed version, so an unconditional warning
1165
+ // would be the always-on amber the panel's own design rejects -- and the worker's own skew check
1166
+ // already refuses a job that arrives without conditions it should have had.
1167
+ checks.push({
1168
+ ok: true,
1169
+ label: `run.waitFor needs worker >= 1.6.0, receiver >= 1.4.0 and admin >= 1.6.0 (a service below the floor drops the field silently; the worker refuses such a job as wait-skew rather than running it unheld)`,
1170
+ });
1171
+ }
1172
+
1102
1173
  // REQ-TRIGGER-SECRETS. Only reported when a trigger actually binds one, on the run.resume block's
1103
1174
  // reasoning below: a deployment that uses no secrets should not be told about a variable it has no
1104
1175
  // reason to set.
1105
1176
  if (secreting > 0) {
1106
- const declared = parseSecretProfilesSafe(env.PI_SECRET_PROFILES);
1177
+ const { profiles: declared, error: parseFailure } = parseSecretProfilesSafe(env.PI_SECRET_PROFILES);
1107
1178
  const names = Object.keys(declared).sort();
1108
- if (declared.error) {
1109
- checks.push({ ok: false, label: "PI_SECRET_PROFILES does not parse", fix: `${declared.error} -- the worker refuses to boot until this is fixed, rather than dropping the entry and leaving you a profile you believe is wired` });
1179
+ if (parseFailure) {
1180
+ checks.push({ ok: false, label: "PI_SECRET_PROFILES does not parse", fix: `${parseFailure} -- the worker refuses to boot until this is fixed, rather than dropping the entry and leaving you a profile you believe is wired` });
1110
1181
  } else {
1111
1182
  // A HARD FAIL, not a warning, and worded like the run.resume/PI_SESSIONS_DIR check below for the
1112
1183
  // same reason: these jobs refuse pre-spend until it is set, deliberately, rather than running
@@ -1537,16 +1608,46 @@ async function repoFlowAtHead(spawn, folder, flow) {
1537
1608
  }
1538
1609
 
1539
1610
  /**
1540
- * `parseSecretProfiles`, but doctor never throws. A malformed PI_SECRET_PROFILES is a finding to REPORT,
1541
- * not a reason for the diagnostic tool to die: the operator running doctor is very likely running it
1542
- * BECAUSE the worker refused to boot on that exact line, and a stack trace instead of a check is the least
1543
- * useful possible answer. Returns the table, or `{ error }` carrying the parser's own message.
1611
+ * `parseWaitProfiles`, but doctor never throws: a malformed variable is a finding to REPORT, not a reason
1612
+ * for the diagnostic tool to die, since the operator running doctor is very likely running it BECAUSE the
1613
+ * worker refused to boot on that exact line.
1614
+ *
1615
+ * Returns an ENVELOPE, `{ profiles, error }`, rather than the table with an `error` key beside the profiles.
1616
+ * The flat shape reads better and is wrong: `error` is a legal profile name, so `PI_WAIT_PROFILES=error:/x.sh`
1617
+ * declares a profile whose PATH then reads as a parse failure -- doctor reports the variable as unparseable,
1618
+ * quoting the path as the message, and skips every check below it on a deployment that is perfectly fine.
1619
+ */
1620
+ function parseWaitProfilesSafe(raw) {
1621
+ try {
1622
+ return { profiles: parseWaitProfiles(raw), error: null };
1623
+ } catch (err) {
1624
+ return { profiles: Object.create(null), error: err?.message ?? String(err) };
1625
+ }
1626
+ }
1627
+
1628
+ /** Is this path something the worker could actually execute? The resolver's probe, reused verbatim. */
1629
+ function statPath(path) {
1630
+ try {
1631
+ const real = realpathSync(path);
1632
+ const st = statSync(real);
1633
+ if (!st.isFile()) return { ok: false, why: "(not a regular file)" };
1634
+ if ((st.mode & 0o111) === 0) return { ok: false, why: "(not executable)" };
1635
+ return { ok: true };
1636
+ } catch (err) {
1637
+ return { ok: false, why: `(${err?.code ?? "unreadable"})` };
1638
+ }
1639
+ }
1640
+
1641
+ /**
1642
+ * `parseSecretProfiles`, but doctor never throws, for `parseWaitProfilesSafe`'s reason and returning the
1643
+ * same envelope. The `error`-is-a-legal-profile-name defect was found in the wait copy and fixed in both:
1644
+ * the flat shape let one declared profile's PATH read as a parse failure and hide every check below it.
1544
1645
  */
1545
1646
  function parseSecretProfilesSafe(raw) {
1546
1647
  try {
1547
- return parseSecretProfiles(raw);
1648
+ return { profiles: parseSecretProfiles(raw), error: null };
1548
1649
  } catch (err) {
1549
- return { error: err?.message ?? "unparseable" };
1650
+ return { profiles: Object.create(null), error: err?.message ?? "unparseable" };
1550
1651
  }
1551
1652
  }
1552
1653
 
@@ -1573,7 +1674,7 @@ function readScopedLimitFacts(env, fileExists) {
1573
1674
  }
1574
1675
 
1575
1676
  function readTriggerFacts(env, fileExists, cwd) {
1576
- const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], folders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1677
+ const none = { requiring: 0, waiting: 0, waitProfiles: [], waitAfters: [], optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, onceArmed: 0, onceSpent: 0, secretProfiles: [], localSecretFolders: [], folders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1577
1678
  try {
1578
1679
  // Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
1579
1680
  // (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
@@ -1619,6 +1720,21 @@ function readTriggerFacts(env, fileExists, cwd) {
1619
1720
  // canonicalizes a job's folder (one derivation -- canonicalScope, never re-spelled here), so
1620
1721
  // the unreferenced-scope advisory compares like with like across spelling variants.
1621
1722
  folders: [...new Set(triggers.filter((t) => t.run.kind === "local" && typeof t.run.folder === "string").map((t) => canonicalScope({ kind: "local", folder: t.run.folder })))].sort(),
1723
+ // Issue #230. How many triggers hold their jobs, and the distinct profile NAMES they select --
1724
+ // deduped like `secretProfiles` and for its reason: each name costs a lookup, and two triggers
1725
+ // waiting on one profile are one question.
1726
+ waiting: triggers.filter((t) => Array.isArray(t.run.waitFor) && t.run.waitFor.length > 0).length,
1727
+ // The `after` instants as WRITTEN, deduped. Not parsed here: `readTriggerFacts` is a fact reader and
1728
+ // the ceiling it is measured against is env, which belongs at the check. Two triggers naming one
1729
+ // instant are one finding, and the raw string is what the operator has to go and edit.
1730
+ waitAfters: [...new Set(triggers.flatMap((t) => (Array.isArray(t.run.waitFor) ? t.run.waitFor : [])).map((c) => c?.after).filter((v) => typeof v === "string"))].sort(),
1731
+ waitProfiles: [
1732
+ ...new Set(
1733
+ triggers
1734
+ .filter((t) => Array.isArray(t.run.waitFor))
1735
+ .flatMap((t) => t.run.waitFor.map((c) => c?.profile).filter((n) => typeof n === "string")),
1736
+ ),
1737
+ ].sort(),
1622
1738
  optingOut: triggers.filter((t) => t.run.packages === false).length,
1623
1739
  images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
1624
1740
  // REQ-PER-TRIGGER-SKILLS. The distinct host directories the file names, deduped like `images`,
package/src/exit-code.mjs CHANGED
@@ -9,6 +9,21 @@ export const EXIT_COMPLETED = 0; // agent ran, INCLUDING "I cannot fix this" --
9
9
  export const EXIT_INFRA = 1; // container died, network, provider 5xx/429 -- the only retryable class
10
10
  export const EXIT_POLICY = 2; // budget/turn cap/config -- a determinate refusal, never retried
11
11
 
12
+ /**
13
+ * "Not yet, ask again later" (issue #230). Emitted by the WAIT PARTICIPANT ONLY.
14
+ *
15
+ * The protocol's third queue behaviour, and the only code in it that no container and no secret resolver
16
+ * may emit: a container exiting 3 is still an unrecognised code and still infra-retries through
17
+ * `processor.mjs`'s own switch, a resolver exiting 3 is still `unreachable`, and both are pinned. This is
18
+ * therefore a per-participant widening rather than a new meaning for a code anyone else already speaks.
19
+ *
20
+ * Minting a code rather than riding `reason` on an exit log line -- which is what
21
+ * `INT-RUNNER-EXIT-CODE-PROTOCOL` asks of the CONTAINER's new vocabulary -- is possible here for the
22
+ * reason that rule does not reach: a wait participant has no exit log line to ride. It produces no run
23
+ * record and no log line of its own while a job is held, so `reason` is not a channel it has.
24
+ */
25
+ export const EXIT_HOLD = 3;
26
+
12
27
  /**
13
28
  * Decide whether the processor should RETURN (BullMQ records success, no retry) or THROW (BullMQ
14
29
  * retries per `attempts`). Returns `{ retry }`; the caller returns on false and throws on true.
@@ -30,3 +45,44 @@ export function decideRetry(exitCode) {
30
45
  return { retry: true, outcome: `unknown-exit-${exitCode}` };
31
46
  }
32
47
  }
48
+
49
+ /**
50
+ * Classify a WAIT PROFILE's exit code (issue #230). Returns `{ verdict, fault }`, where `verdict` is
51
+ * `"go"` (run the job), `"hold"` (defer and ask again) or `"refuse"` (terminal, never retried).
52
+ *
53
+ * The four codes and why each lands where it does:
54
+ *
55
+ * 0 the condition has cleared -> go
56
+ * 3 not yet -> hold, NOT a fault: this is the normal answer
57
+ * 1 I could not tell -> hold, AND a fault
58
+ * 2 this will never clear -> refuse, terminal
59
+ * * anything else -> hold, AND a fault
60
+ *
61
+ * `1` HOLDS RATHER THAN REFUSING because a check that cannot answer has not answered "no": treating an
62
+ * unreachable Jira as "this will never clear" would drop a paid delivery over a transient outage, which is
63
+ * `CONST-RETRY-INFRA-ONLY` in the expensive direction and the same call `secrets.mjs` makes for a resolver
64
+ * that cannot reach its manager.
65
+ *
66
+ * `fault` is what keeps `1` and `3` from being the same code wearing two hats, and it exists because of
67
+ * `OQ-027`: most CLIs exit 1 for everything, so without it a permanently broken check -- a typo'd `curl`
68
+ * exits 6, a false `jq -e` exits 1 -- would hold for the entire maximum wait, spawn a process every
69
+ * interval, and finally terminate with a reason that blames the CONDITION rather than the script. Counting
70
+ * consecutive faults lets that terminate loudly in minutes, naming the check. A `3` resets the count: a
71
+ * check that answered is a check that works.
72
+ *
73
+ * The unrecognised arm folds into `1` deliberately, which is this protocol's own rule for an unrecognised
74
+ * code, and it is why the naive one-liner an operator writes first (`grep -q ... ` , exit 1 when the
75
+ * pattern is absent) behaves correctly by accident: it holds, and the fault counter bounds it.
76
+ */
77
+ export function decideWait(exitCode) {
78
+ switch (exitCode) {
79
+ case EXIT_COMPLETED:
80
+ return { verdict: "go", fault: false };
81
+ case EXIT_HOLD:
82
+ return { verdict: "hold", fault: false };
83
+ case EXIT_POLICY:
84
+ return { verdict: "refuse", fault: false };
85
+ default:
86
+ return { verdict: "hold", fault: true }; // EXIT_INFRA and every unrecognised code
87
+ }
88
+ }