@edgehero/pi-dispatch 1.8.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Per-scheduler stall accounting -- the money backstop for cron (constitution.md:203-216).
2
+ * Per-scheduler stall accounting -- the money backstop for cron (CONST-RETRY-INFRA-ONLY).
3
3
  *
4
4
  * BullMQ's `maxStalledCount` does not cover scheduler jobs: `moveStalledJobsToWait` derives
5
5
  * `isRepeatableJob` from the job's `rjk` field and skips the stall-fail for a live scheduler, so a
@@ -10,16 +10,41 @@
10
10
  * Injected `redis` (ioredis-compatible), `removeJobScheduler`, and `log` keep the logic testable with
11
11
  * no queue, no bullmq import, and no real Valkey.
12
12
  *
13
- * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- constitution.md:203-216 carve-out ("BullMQ will never do this for us")
13
+ * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- CONST-RETRY-INFRA-ONLY carve-out ("BullMQ will never do this for us")
14
14
  */
15
15
 
16
- // The Redis hash of per-scheduler stall counts (field = schedulerId, value = count). Exported so the admin
17
- // panel can read it (HGETALL) for the cron drill-in without re-deriving the key string.
16
+ // The prefix every stall counter lives under, so `KEYS pi-dispatch:sched-stalls*` still shows an operator
17
+ // the whole feature -- the affordance `wait:`, `slot:` and `budget:` all assume. Exported alongside the
18
+ // builder so the admin panel and the integration teardown compose keys through one definition and cannot
19
+ // drift from the writer.
18
20
  export const STALL_KEY = "pi-dispatch:sched-stalls";
19
21
 
20
- // A rolling window: the EXPIRE is re-set on every stall, so a scheduler that stops stalling for a full
21
- // day drops back to zero. This prevents unrelated transient stalls weeks apart from accumulating into a
22
- // false teardown -- only sustained stalling inside one window trips the threshold.
22
+ /**
23
+ * One scheduler's counter. The id is VALIDATED upstream rather than hashed here: it is operator-declared in
24
+ * `triggers.json`, `triggers.mjs` already refuses a `:` in it precisely to protect this parse, and the value
25
+ * of a readable keyspace is that `GET pi-dispatch:sched-stalls:nightly` answers the question directly.
26
+ */
27
+ export const stallKey = (schedulerId) => `${STALL_KEY}:${schedulerId}`;
28
+
29
+ // ONE KEY PER SCHEDULER, so the window is per scheduler.
30
+ //
31
+ // This was one HASH with a field per scheduler and a single `EXPIRE` on the whole key, which meant any
32
+ // scheduler's stall pushed the TTL forward for EVERY scheduler's count. The window never reset on a
33
+ // deployment where anything stalled regularly, so the guard silently degraded from "sustained stalling
34
+ // inside one window" to "cumulative stalling ever": three stalls ninety days apart tore a scheduler down
35
+ // if a neighbour was stalling twice a day, and did not if the deployment was quiet. Same scheduler, same
36
+ // stalls, opposite outcome, decided by an unrelated trigger (issue #267).
37
+ //
38
+ // Per-field TTLs would have fixed it in place and are not available: `HEXPIRE` does not exist on the pinned
39
+ // `valkey/valkey:8` (verified, `ERR unknown command`, recorded under DES-HOST-REGISTRY). A key per entity is
40
+ // the only shape that gets per-entity expiry.
41
+ //
42
+ // The EXPIRE still ROLLS on every stall, deliberately, and that is not `budget.mjs`'s set-once rule being
43
+ // broken. A budget window is a CALENDAR window and must not be pushed forward by traffic or a busy day
44
+ // never resets. This is a STREAK detector -- "is this scheduler wedged right now" -- and quiet for a day
45
+ // genuinely should forget. `poll:<repo>:close-gate:<deliveryId>` is the in-repo precedent, a bounded
46
+ // consecutive-failure counter given its own key and TTL for exactly this reason: it must decay with the
47
+ // thing it measures rather than with a larger family.
23
48
  const STALL_WINDOW_SECONDS = 24 * 60 * 60;
24
49
 
25
50
  /**
@@ -45,8 +70,9 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
45
70
  return;
46
71
  }
47
72
 
48
- const count = Number(await redis.hincrby(STALL_KEY, schedulerId, 1));
49
- await redis.expire(STALL_KEY, STALL_WINDOW_SECONDS);
73
+ const key = stallKey(schedulerId);
74
+ const count = Number(await redis.incr(key));
75
+ await redis.expire(key, STALL_WINDOW_SECONDS);
50
76
 
51
77
  if (count > threshold) {
52
78
  try {
@@ -56,7 +82,7 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
56
82
  // state, not an error -- swallow it so hdel and the teardown alert still run.
57
83
  log("scheduler_teardown_remove_failed", { schedulerId, error: error?.message });
58
84
  }
59
- await redis.hdel(STALL_KEY, schedulerId);
85
+ await redis.del(key);
60
86
  // The loud log is the "alert" half of the constitution's "removeJobScheduler -- or alert".
61
87
  log("scheduler_torn_down", { schedulerId, stalls: count });
62
88
  }
package/src/schedules.mjs CHANGED
@@ -137,7 +137,7 @@ function normalizeCronSchedule({ on, run }, path, existsSync, fleet) {
137
137
  // key. A command trigger carries no flow/task at all (the validator enforces the XOR), so those two
138
138
  // keys hold undefined here and drop at JSON serialization -- the command schedule's data is exactly
139
139
  // kind/folder/command plus the shared fields.
140
- const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
140
+ const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.backend !== undefined && { backend: run.backend }), ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
141
141
  // Retention only; the deterministic repeat:<id>:<millis> jobId supplies dedup, so no jobId here, and
142
142
  // scheduler jobs are not retried (DES-CRON-VIA-BULLMQ-SCHEDULER) so no attempts/backoff.
143
143
  const opts = { removeOnComplete: { age: 24 * 3600 }, removeOnFail: { age: 7 * 24 * 3600 } };
package/src/start.mjs CHANGED
@@ -1,7 +1,5 @@
1
- import { execFile } from "node:child_process";
2
1
  import { readFileSync, watch } from "node:fs";
3
2
  import { dirname, basename, join } from "node:path";
4
- import { promisify } from "node:util";
5
3
  import { configError, loadConfig } from "./config.mjs";
6
4
  import { makeRedisClient, parseConnection } from "./connection.mjs";
7
5
  import { reconcileGated, reloadSchedules } from "./cron.mjs";
@@ -32,6 +30,10 @@ import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
32
30
  import { makeWaitChecker } from "./wait-check.mjs";
33
31
  import { makeWaitState } from "./wait-state.mjs";
34
32
  import { hostQueueName, makeQueue } from "./queue.mjs";
33
+ import { makeLocalBackend, makeReaper, makeStopContainer } from "./backend-local.mjs";
34
+ import { makeBackendRegistry, reapAll } from "./backend-registry.mjs";
35
+ import { DEFAULT_BACKEND } from "./backends.mjs";
36
+
35
37
  import { makeRunContainer } from "./run-container.mjs";
36
38
  import { makeSecretsResolver } from "./secrets.mjs";
37
39
  import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
@@ -40,7 +42,6 @@ import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
40
42
  import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
41
43
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
42
44
 
43
- const exec = promisify(execFile);
44
45
 
45
46
  /** How long boot will wait for `docker image inspect` before shipping without a digest. */
46
47
  const BOOT_IMAGE_TIMEOUT_MS = 5_000;
@@ -172,49 +173,6 @@ function watchScopedLimitsFile(config, ref, log) {
172
173
  }
173
174
  }
174
175
 
175
- export function makeReaper({ log }) {
176
- return async function reap() {
177
- try {
178
- const { stdout } = await exec("docker", ["ps", "--filter", "name=pi-job-", "--format", "{{.Names}}"]);
179
- const names = stdout
180
- .split("\n")
181
- .map((n) => n.trim())
182
- .filter(Boolean);
183
- for (const name of names) {
184
- await exec("docker", ["rm", "-f", name]);
185
- log("reaped_container", { name });
186
- }
187
- // REQ-EGRESS-ALLOWLIST: the per-job networks those containers were on. Swept AFTER the containers,
188
- // because a network with a member still attached cannot be removed -- and swept by the SAME
189
- // `pi-job-` filter, so the namespace rule that keeps an operator's live sandbox safe from the
190
- // container reaper keeps their sandbox NETWORK safe too, with no second rule to remember.
191
- //
192
- // A crashed worker is the case this exists for: `runContainer`'s own finally removes the network
193
- // on every ordinary path, so anything still here outlived a process that did not get to run it.
194
- // A network still in use by something else fails to remove and is skipped, which is correct: this
195
- // is a best-effort sweep and never a reason not to boot.
196
- const { stdout: nets } = await exec("docker", ["network", "ls", "--filter", "name=pi-job-", "--format", "{{.Name}}"]);
197
- for (const net of nets.split("\n").map((n) => n.trim()).filter(Boolean)) {
198
- try {
199
- await exec("docker", ["network", "rm", net]);
200
- log("reaped_network", { network: net });
201
- } catch {} // still in use, or already gone -- either way not this boot's problem
202
- }
203
- // Whether the enumeration HAPPENED, which the scope-claim sweep depends on: it may only delete a
204
- // claim naming this host once this host has actually established that it holds no containers.
205
- return { reaped: true };
206
- } catch (err) {
207
- // The `docker ps` is inside this try, so this path CANNOT establish that this host holds no
208
- // containers -- whether it failed before listing anything or after reaping some and then losing
209
- // the daemon. Either way the claim "I hold nothing" is unproven, and sweeping on it would free
210
- // slots for containers that may STILL BE RUNNING, letting another host start more alongside
211
- // them: a money overrun rather than a tidy-up. Conservative in the only safe direction.
212
- log("reaper_skipped", { reason: err?.message });
213
- return { reaped: false };
214
- }
215
- };
216
- }
217
-
218
176
  /**
219
177
  * The runnable worker. Reads config, connects to Valkey, wires every REAL dependency the processor
220
178
  * needs, and starts draining the queue. `createWorker` already installs the timeout, the
@@ -233,10 +191,23 @@ export function makeReaper({ log }) {
233
191
  export async function startWorker(
234
192
  env = process.env,
235
193
  {
194
+ // WHERE THE BOOT LOG BYTES GO. Defaults to the real stdout, so production is byte-identical; a test
195
+ // passes a collector instead of reassigning `process.stdout.write`.
196
+ //
197
+ // That distinction is not stylistic. `node --test` runs each file in a CHILD PROCESS that serialises
198
+ // its own results over `process.stdout`, so a test that replaces the global and holds the replacement
199
+ // across an `await` swallows the runner's result frames for whatever completes in that window. Three
200
+ // tests in `start-wiring.test.mjs` were reported as not existing at all -- no name, no count, exit 0 --
201
+ // because this function's log line went through the same channel the runner needed (issue #266).
202
+ write = (chunk) => process.stdout.write(chunk),
236
203
  makeAuth = makeGitHubAuth,
237
204
  makeHost = makeGitHubHost,
238
205
  createWorkerFn = createWorker,
239
206
  makeReaper: makeReaperFn = makeReaper,
207
+ makeBackendRegistry: makeBackendRegistryFn = makeBackendRegistry,
208
+ // Additional backend bundles, in registration order after `local`. The deployment still decides which
209
+ // are BLESSED (PI_BACKENDS) and which is default; this only says which exist.
210
+ extraBackends = [],
240
211
  makeLogSink: makeLogSinkFn = makeLogSink,
241
212
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
242
213
  makeRunMirror: makeRunMirrorFn = makeRunMirror,
@@ -263,7 +234,7 @@ export async function startWorker(
263
234
  // other module takes `log` injected, and two tests pin the KEY SET of the fields object handed to an
264
235
  // injected log (`run_record_failed`, `wait_check`); a `host` added at any call site would break them,
265
236
  // while one added inside this closure cannot reach them.
266
- const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
237
+ const log = (event, fields = {}) => write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
267
238
 
268
239
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
269
240
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
@@ -344,9 +315,24 @@ export async function startWorker(
344
315
  // Clear strays left by a previous crash before the worker starts draining. Best-effort: the reaper
345
316
  // swallows its own docker errors; this guard keeps any reaper failure from blocking boot.
346
317
  // Whether the container reaper actually ENUMERATED, which the scope-claim sweep below depends on.
318
+ // #227. ONE map, read twice: the boot sweep below combines every entry, and each backend's bundle takes
319
+ // its own reaper from it by name. Held here rather than derived from the bundles because the bundles
320
+ // cannot exist yet -- they need the log sink, the package resolver and the image preflight, all built
321
+ // further down -- and reaching for one here is a temporal dead zone the boot try/catch would swallow.
322
+ let backendReaps = {};
323
+
347
324
  let reaped = false;
348
325
  try {
349
- reaped = (await makeReaperFn({ log })())?.reaped === true;
326
+ // CONSTRUCTED INSIDE THE GUARD, not above it. The comment on this try says it "keeps any reaper
327
+ // failure from blocking boot", and a factory that throws is a reaper failure -- an earlier draft
328
+ // hoisted the construction out and quietly made that sentence false.
329
+ // DERIVED from the bundles, not a hand-kept parallel map. An earlier draft had a literal here and the
330
+ // registry cross-checking it, which made adding a venue three edits that nothing forced to agree --
331
+ // and a forgotten one is INVISIBLE, because `reapAll` is conservative over the reapers it is handed
332
+ // rather than over the venues that exist. A bundle already carries its own `reap`, so taking it from
333
+ // there is one place. `local`'s is built here because its bundle cannot exist yet.
334
+ backendReaps = { [DEFAULT_BACKEND]: makeReaperFn({ log }), ...Object.fromEntries(extraBackends.map((b) => [b?.name, b?.reap])) };
335
+ reaped = (await reapAll(Object.values(backendReaps), { log }))?.reaped === true;
350
336
  } catch (err) {
351
337
  log("reaper_skipped", { reason: err?.message });
352
338
  }
@@ -620,8 +606,85 @@ export async function startWorker(
620
606
  });
621
607
 
622
608
 
609
+ // #227: WHERE this job's container runs. The three functions that decide whether a container may start and
610
+ // then start it -- two pre-spend gates and the launcher -- bundled into one value with a completeness
611
+ // check, so the set has a name instead of being three unrelated `deps` keys. Byte-identical to passing them individually -- the same three functions reach the same three keys,
612
+ // built from the same config, behind the same injectable factories -- and this is the seam a second
613
+ // backend is selected at once there is one to select.
614
+ //
615
+ // Assigned key by key below rather than spread, because the bundle also carries `name` and `declares`,
616
+ // and `deps` is the processor's namespace: a spread would put a backend's name into it under a key the
617
+ // processor is free to mean something else by.
618
+ const localBackend = makeLocalBackend({
619
+ // #227. The two the earlier slices deferred, now real seams. `reap` keeps its tri-state: the boot
620
+ // sweep below only sweeps this host's scope claims once the reaper has PROVEN this host holds no
621
+ // job containers, and an unproven answer must never free a slot.
622
+ stopContainer: makeStopContainer(),
623
+ reap: backendReaps[DEFAULT_BACKEND],
624
+ // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
625
+ // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
626
+ // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
627
+ // Nothing is memoised (see its construction above): `docker image inspect` costs ~tens of ms against a
628
+ // container run of minutes, and a cache would be wrong in both directions -- an operator who builds the
629
+ // image mid-day would stay refused, one who removes it would stay admitted. Contrast the staged-package
630
+ // manifest, correctly read once at boot because it is deploy-time state under a :ro mount.
631
+ imagePreflight,
632
+ // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
633
+ // value, one place, so the gate that checks the proxy and the runner that attaches to its network
634
+ // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
635
+ // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
636
+ // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
637
+ egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
638
+ runContainer: makeRunContainerFn({
639
+ image: config.jobImage,
640
+ hostEnv: env,
641
+ egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
642
+ egressProxy: config.egressProxy,
643
+ openJobLog,
644
+ globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
645
+ allowGlobalExtensions: config.allowGlobalExtensions,
646
+ // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
647
+ // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
648
+ // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
649
+ packagePaths: getPackagePaths,
650
+ forwardEnv: config.forwardEnv,
651
+ authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
652
+ // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
653
+ // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
654
+ // adding one does not widen this signature again.
655
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
656
+ }),
657
+ });
658
+
659
+ // #227. WHICH backend runs which job, and the one place that decides. One bundle today, so every
660
+ // resolution returns it -- but the mechanism is real, so `run.backend` stops being a validated label and
661
+ // the abort path can reach a venue it did not build. `config.backends[0]` is the deployment's default,
662
+ // and the registry refuses a default it does not hold rather than discovering it at the first pickup.
663
+ const backends = makeBackendRegistryFn({
664
+ // #227. A SEAM, not a literal. `docs/backends.md` tells an adapter author to register their bundle
665
+ // here, and until this was injectable that instruction described code nobody could run: the array
666
+ // was hard-coded, so a venue could pass the conformance suite, get a table entry and be blessed in
667
+ // PI_BACKENDS, and then be refused at boot as blessed-but-unbuilt with nowhere to put it. It is also
668
+ // what lets a wiring test prove `startWorker` actually CONNECTS the registry to the processor --
669
+ // six mutations reverting that connection survived the whole suite, which is the same shape as the
670
+ // bug that shipped: invisible while there is one venue.
671
+ bundles: [localBackend, ...extraBackends],
672
+ defaultName: config.defaultBackend,
673
+ // Cross-checked at boot rather than discovered at the first pickup: a name PI_BACKENDS blesses but
674
+ // nothing builds passes both the loader and the pre-spend gate, and a venue with no boot reaper is
675
+ // swept by nothing while still reporting the host as proven clean.
676
+ blessed: config.backends,
677
+ reaps: backendReaps,
678
+ });
679
+
623
680
  const worker = createWorkerFn({
624
681
  connection: parseConnection(config.valkeyUrl),
682
+ // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
683
+ // not enough to find the runtime holding it once there is more than one venue.
684
+ stopContainer: backends.stopContainer,
685
+ // The NAME the abort stops, built by the venue that will build the container rather than by the local
686
+ // adapter reached for directly -- the name and the stop have to come from the same venue.
687
+ containerName: backends.containerName,
625
688
  hostQueue,
626
689
  // Names the BullMQ Worker, which makes `getWorkers()` rows tell hosts apart -- bullmq appends
627
690
  // `:w:<name>` to the client name and `moveToActive` stamps `processedBy` onto each active job's
@@ -709,20 +772,8 @@ export async function startWorker(
709
772
  const state = await held.getState();
710
773
  return state === "delayed" || state === "waiting" || state === "active" || state === "prioritized" || state === "waiting-children";
711
774
  },
712
- // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
713
- // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
714
- // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
715
- // Nothing is memoised: `docker image inspect` costs ~tens of ms against a container run of minutes, and a
716
- // cache would be wrong in both directions -- an operator who builds the image mid-day would stay refused,
717
- // one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
718
- // boot because it is deploy-time state under a :ro mount; the host's image set is not.
719
- imagePreflight,
720
- // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
721
- // value, one place, so the gate that checks the proxy and the runner that attaches to its network
722
- // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
723
- // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
724
- // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
725
- egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
775
+ imagePreflight: backends.imagePreflight,
776
+ egressPreflight: backends.egressPreflight,
726
777
  // Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
727
778
  // retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
728
779
  promoteSession: sessionStore.promoteSession,
@@ -739,25 +790,21 @@ export async function startWorker(
739
790
  // the panel should not have to restart the worker to use it. A deployment that declares nothing
740
791
  // spawns nothing at all: the gate only calls this when a trigger is armed.
741
792
  resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, log }),
742
- runContainer: makeRunContainerFn({
743
- image: config.jobImage,
744
- hostEnv: env,
745
- egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
746
- egressProxy: config.egressProxy,
747
- openJobLog,
748
- globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
749
- allowGlobalExtensions: config.allowGlobalExtensions,
750
- // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
751
- // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
752
- // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
753
- packagePaths: getPackagePaths,
754
- forwardEnv: config.forwardEnv,
755
- authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
756
- // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
757
- // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
758
- // adding one does not widen this signature again.
759
- forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
760
- }),
793
+ // #227. What PI_BACKENDS blessed, so a trigger naming an unblessed venue refuses pre-spend. The
794
+ // panel's picker is bounded by the same list, and this is the half that binds: the overlay is
795
+ // not the reviewed artifact (DES-PER-TRIGGER-SECRET-PROFILE).
796
+ blessedBackends: config.backends,
797
+ runContainer: backends.runContainer,
798
+ // #227. Resolved through the registry like every other per-job backend fact, so the integers the
799
+ // processor treats as "never started" are the ones the venue that ran the job actually uses.
800
+ // NOT `?? []`. An empty list means "this venue never reports a never-started exit", which sends a
801
+ // 125 to the unknown-exit branch -- no `reason`, so the budget slot is KEPT and BullMQ retries,
802
+ // burning a second one per never-started job. That is the bug the explicit case group was added
803
+ // to fix. A bundle that omits the field is a wiring defect, so it fails loudly here rather than
804
+ // silently in the money direction; `backends.mjs` requires it of every bundle.
805
+ // Off the REGISTRY's surface, like every other per-job backend fact. Rebuilding it at the call site
806
+ // is how a fact ends up dispatched on one path and hardcoded on another.
807
+ neverStartedExits: backends.neverStartedExits,
761
808
  prepareWorkspace: makePrepareWorkspace({
762
809
  jobsDir: config.jobsDir,
763
810
  forgeFor,
@@ -842,7 +889,7 @@ export async function startWorker(
842
889
  // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
843
890
  // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
844
891
  // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
845
- const guard = makeStallGuard({
892
+ const onStalled = makeStallGuard({
846
893
  redis,
847
894
  threshold: config.schedulerStallMax,
848
895
  // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
@@ -850,7 +897,10 @@ export async function startWorker(
850
897
  removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
851
898
  log,
852
899
  });
853
- for (const w of allWorkers) w.on("stalled", (jobId) => void guard.onStalled(jobId));
900
+ // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
901
+ // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
902
+ // so every stall threw a TypeError and the money backstop never counted one (issue #267).
903
+ for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
854
904
 
855
905
  // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
856
906
  // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
package/src/triggers.mjs CHANGED
@@ -23,6 +23,7 @@
23
23
  * Custom: triggers validated inline per config.mjs/schedules.mjs precedent; zod not in deps
24
24
  */
25
25
 
26
+ import { BACKEND_NAMES, backendFor } from "./backends.mjs";
26
27
  import { EGRESS_ENV_VARS, WORKER_ONLY_SECRET_VARS, configError } from "./config.mjs";
27
28
  // SKILL_NAME_RE is the single-sourced skill charset (flow-gate exports it for exactly this reason:
28
29
  // materialize.mjs and the admin already import it, and a keep-in-sync copy would drift where a
@@ -350,6 +351,7 @@ function normalizeCron(on, run, index, path, state) {
350
351
  validateDisarmed(on, `cron trigger "${id}"`, path, { onType: "cron" });
351
352
  const secrets = validateSecrets(run, `cron trigger "${id}"`, path);
352
353
  const secretsProfile = validateSecretsProfile(run, `cron trigger "${id}"`, path);
354
+ const backend = validateBackend(on, run, `cron trigger "${id}"`, path, { localWorkspace: true });
353
355
  // Called for its refusal only: the returned `run` below deliberately grows no `waitFor` key, exactly as
354
356
  // it grows no `replicas` one, because a cron entry can never carry either (issue #230).
355
357
  validateWaitFor(on, run, `cron trigger "${id}"`, path, { onType: "cron" });
@@ -361,7 +363,7 @@ function normalizeCron(on, run, index, path, state) {
361
363
  // freeze today's default into every stored repeatable.
362
364
  return {
363
365
  on: { type: "cron", id, pattern },
364
- run: { kind: "local", folder: run.folder, flow: run.flow, task: run.task, provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages, image, resume, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }) },
366
+ run: { kind: "local", folder: run.folder, flow: run.flow, task: run.task, provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages, image, resume, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(backend !== undefined && { backend }) },
365
367
  };
366
368
  }
367
369
 
@@ -992,6 +994,115 @@ function validateSecretsProfile(run, at, path) {
992
994
  return profile;
993
995
  }
994
996
 
997
+ /**
998
+ * Would running a folder-bound job in this venue be impossible? Exported ONLY so it can be driven.
999
+ *
1000
+ * The refusal it answers cannot be reached through `parseTriggers` today: the table holds one entry and it
1001
+ * is local, so no trigger can name a remote venue and the branch below is unreachable. A mutation pass duly
1002
+ * deleted that branch and the whole suite stayed green. A rule nothing can exercise is a rule that will be
1003
+ * wrong on the day it finally matters, which is the day a second backend lands -- so the decision lives
1004
+ * here, as one line over an explicit entry, and a test drives it against a synthetic remote one.
1005
+ *
1006
+ * `!== false` and not `=== true`: `remote` is a plain field on the table entry rather than a member of the
1007
+ * closed PROPERTIES list, so an entry that simply omits it would otherwise read as local and be admitted
1008
+ * onto a folder-bound trigger. That is the fail-open direction `backends.mjs`'s own rule forbids ("a backend
1009
+ * that omits one is not admitted for it"), and it would defeat with an absent key a refusal this file calls
1010
+ * physics. A test pins that every entry declares it, so both halves have to fail together.
1011
+ */
1012
+ export function refusesLocalWorkspace(entry, localWorkspace) {
1013
+ return Boolean(localWorkspace) && entry?.remote !== false;
1014
+ }
1015
+
1016
+ /**
1017
+ * `run.backend` -- WHICH venue this trigger's container is built in (issue #227).
1018
+ *
1019
+ * A NAME, never a configuration. `validateSecretsProfile` above is the template end to end, and the reason
1020
+ * is the same: a trigger SELECTS among what the deployment already blessed, and never configures a posture.
1021
+ * `run.network` was rejected outright for that reason, and this field must not become a way back to it --
1022
+ * naming `remote-vendor` cannot turn egress off, because what egress a venue can provide is the backend
1023
+ * table's answer and the deployment's floor bounds which venues are reachable at all.
1024
+ *
1025
+ * WHAT IS CHECKED HERE AND WHAT IS NOT. This file answers what the FILE knows: that the name is a name, and
1026
+ * that the venue could ever run this kind of job. Whether the DEPLOYMENT blessed it is `PI_BACKENDS`, which
1027
+ * lives in the environment, and a loader that read it would refuse a reviewed file on a per-host setting --
1028
+ * so the processor refuses that pre-spend instead. Same split `secretsProfile` already draws between the
1029
+ * charset check here and `secret-profile-unknown` there.
1030
+ *
1031
+ * A NEAR-MISS SPELLING IS REFUSED, which puts this in `waitFor`'s class rather than `run.imgae`'s. A
1032
+ * misspelled image gives you the default image and a job that ran; a misspelled `backend` gives you the
1033
+ * DEFAULT VENUE and a job that ran -- byte-identical in the record, the panel and the log to one that
1034
+ * correctly named a venue, while the operator reads the file as though it chose. That is the destructive
1035
+ * absence `validateWaitFor` refuses near-misses for, one field over.
1036
+ */
1037
+ function validateBackend(on, run, at, path, { localWorkspace }) {
1038
+ // The near-miss sweep, exactly `validateWaitFor`'s: on `run` the exact spelling is the field, on `on`
1039
+ // every spelling is wrong including the correct one, because a venue is a property of the run.
1040
+ for (const [label, source, exactIsLegal] of [
1041
+ ["run", run, true],
1042
+ ["on", on, false],
1043
+ ]) {
1044
+ for (const key of Object.keys(source ?? {})) {
1045
+ if (exactIsLegal && key === "backend") continue;
1046
+ // The PLURAL is included, and it is the likeliest miss of all: the deployment-side variable is
1047
+ // `PI_BACKENDS`, so `run.backends` is what an operator writes from memory. Nothing else in this
1048
+ // file needs that, which is why the comparison is against a small set rather than `waitFor`'s
1049
+ // single normalized string.
1050
+ // A HOMOGLYPH is caught too. `replace(/[^a-z0-9]/gi, "")` DELETES a non-ASCII character rather
1051
+ // than failing on it, so a Cyrillic "a" in `b<U+0430>ckend` normalizes to `bckend` and an
1052
+ // Armenian "n" in `backe<U+0578>d` to `backed` -- both miss an equality test and are dropped in
1053
+ // exactly the silence this sweep exists to prevent. Such a key arrives by paste from a rendered
1054
+ // doc or a chat client, the same route homoglyphs take into package names.
1055
+ //
1056
+ // So a key carrying non-ASCII is compared as a SUBSEQUENCE: whatever ASCII letters survive must
1057
+ // still be obtainable from `backend`/`backends` in order. That catches any number of substituted
1058
+ // glyphs without guessing which ones, and a genuinely unrelated non-ASCII key (`fl<U+00F6>w`)
1059
+ // keeps falling through to this file's documented tolerance of unknown `run` keys.
1060
+ const normalized = key.replace(/[^a-z0-9]/gi, "").toLowerCase();
1061
+ const targets = ["backend", "backends"];
1062
+ const isSubsequence = (needle, hay) => {
1063
+ let i = 0;
1064
+ for (const ch of hay) if (i < needle.length && needle[i] === ch) i++;
1065
+ return i === needle.length;
1066
+ };
1067
+ // The floor is 4 rather than 5 so a key with three substituted glyphs is still caught. Only a
1068
+ // non-ASCII key reaches this branch, so the false-positive surface is a non-ASCII key whose
1069
+ // surviving letters happen to be an in-order subset of "backend", which no field here is.
1070
+ const suspicious = /^[\x20-\x7E]*$/.test(key)
1071
+ ? targets.includes(normalized)
1072
+ : normalized.length >= 4 && targets.some((t) => isSubsequence(normalized, t));
1073
+ if (!suspicious) continue;
1074
+ throw configError(`${at}: ${label}.${key} is not a field -- did you mean run.backend? A venue the loader drops runs the job somewhere else while the file reads as though it chose, so a near miss is refused rather than dropped: ${path}`);
1075
+ }
1076
+ }
1077
+
1078
+ const backend = run.backend;
1079
+ if (backend === undefined) return undefined;
1080
+
1081
+ if (!isNonEmptyString(backend)) {
1082
+ throw configError(`${at}: run.backend must be a non-empty string naming one of the backends this deployment can run jobs in (got ${JSON.stringify(backend)}): ${path}`);
1083
+ }
1084
+ if (!ID_CHARSET.test(backend)) {
1085
+ throw configError(`${at}: run.backend ${JSON.stringify(backend)} may use letters, digits, dot, dash and underscore only -- PI_BACKENDS is a comma-separated list, so a name carrying that separator could not be blessed: ${path}`);
1086
+ }
1087
+ const entry = backendFor(backend);
1088
+ if (!entry) {
1089
+ throw configError(`${at}: run.backend ${JSON.stringify(backend)} is not a backend this build knows (known: ${BACKEND_NAMES.join(", ")}) -- whether your deployment BLESSES it is PI_BACKENDS, checked before the job spends: ${path}`);
1090
+ }
1091
+
1092
+ // A LOCAL job cannot run in a remote venue, and this is physics rather than policy. `DES-WORKER-ON-HOST`
1093
+ // finding (2): "the operator's own folder must be bind-mounted as /workspace, edited in place. There is
1094
+ // no volume to hide behind." A cron trigger normalizes to a local run for the same reason. Refused at
1095
+ // LOAD and permanently, because no amount of deployment configuration can make a folder on this machine
1096
+ // appear inside someone else's daemon -- so this is not a bound an operator could widen, it is one the
1097
+ // world imposes, and discovering it per job would burn a pickup for an answer the file already had.
1098
+ // The decision itself is `refusesLocalWorkspace` above, so it can be driven; see its comment.
1099
+ if (refusesLocalWorkspace(entry, localWorkspace)) {
1100
+ throw configError(`${at}: run.backend ${JSON.stringify(backend)} is remote, and this trigger runs against a folder on THIS machine that has to be bind-mounted and edited in place -- there is no volume to hide behind (DES-WORKER-ON-HOST): ${path}`);
1101
+ }
1102
+
1103
+ return backend;
1104
+ }
1105
+
995
1106
  /**
996
1107
  * `run.waitFor` -- the conditions that must all clear before this trigger's job starts (issue #230).
997
1108
  *
@@ -1147,10 +1258,11 @@ function normalizeLabel(on, run, index, path) {
1147
1258
  const replicas = validateReplicas(run, at, path);
1148
1259
  const secrets = validateSecrets(run, at, path);
1149
1260
  const secretsProfile = validateSecretsProfile(run, at, path);
1261
+ const backend = validateBackend(on, run, at, path, { localWorkspace: false });
1150
1262
  const waitFor = validateWaitFor(on, run, at, path, { onType: on.type });
1151
1263
  return {
1152
1264
  on: { type: "label", any: predicate.any, all: predicate.all, none: predicate.none },
1153
- run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(repository !== undefined && { repository }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }) },
1265
+ run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(repository !== undefined && { repository }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }), ...(backend !== undefined && { backend }) },
1154
1266
  };
1155
1267
  }
1156
1268
 
@@ -1186,10 +1298,11 @@ function normalizeComment(on, run, index, path, state) {
1186
1298
  const replicas = validateReplicas(run, at, path);
1187
1299
  const secrets = validateSecrets(run, at, path);
1188
1300
  const secretsProfile = validateSecretsProfile(run, at, path);
1301
+ const backend = validateBackend(on, run, at, path, { localWorkspace: false });
1189
1302
  const waitFor = validateWaitFor(on, run, at, path, { onType: on.type });
1190
1303
  return {
1191
1304
  on: { type: "comment", phrase: on.phrase },
1192
- run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(repository !== undefined && { repository }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }) },
1305
+ run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(repository !== undefined && { repository }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }), ...(backend !== undefined && { backend }) },
1193
1306
  };
1194
1307
  }
1195
1308
 
@@ -1252,6 +1365,7 @@ function normalizeIssue(on, run, index, path) {
1252
1365
  const replicas = validateReplicas(run, at, path);
1253
1366
  const secrets = validateSecrets(run, at, path);
1254
1367
  const secretsProfile = validateSecretsProfile(run, at, path);
1368
+ const backend = validateBackend(on, run, at, path, { localWorkspace: false });
1255
1369
  const waitFor = validateWaitFor(on, run, at, path, { onType: on.type });
1256
1370
  return {
1257
1371
  on: {
@@ -1262,7 +1376,7 @@ function normalizeIssue(on, run, index, path) {
1262
1376
  ...(number !== undefined && { number }),
1263
1377
  ...(once !== undefined && { once }),
1264
1378
  },
1265
- run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }) },
1379
+ run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }), ...(backend !== undefined && { backend }) },
1266
1380
  };
1267
1381
  }
1268
1382
 
@@ -1336,6 +1450,7 @@ function normalizePullRequest(on, run, index, path) {
1336
1450
  const replicas = validateReplicas(run, at, path);
1337
1451
  const secrets = validateSecrets(run, at, path);
1338
1452
  const secretsProfile = validateSecretsProfile(run, at, path);
1453
+ const backend = validateBackend(on, run, at, path, { localWorkspace: false });
1339
1454
  const waitFor = validateWaitFor(on, run, at, path, { onType: on.type });
1340
1455
  return {
1341
1456
  on: {
@@ -1351,7 +1466,7 @@ function normalizePullRequest(on, run, index, path) {
1351
1466
  ...(number !== undefined && { number }),
1352
1467
  ...(once !== undefined && { once }),
1353
1468
  },
1354
- run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }) },
1469
+ run: { kind: run.kind, flow: run.flow, packages, image, resume, replicas, ...(command !== undefined && { command }), ...(skillsDir !== undefined && { skillsDir }), ...(instructions !== undefined && { instructions }), ...(secrets !== undefined && { secrets }), ...(secretsProfile !== undefined && { secretsProfile }), ...(waitFor !== undefined && { waitFor }), ...(backend !== undefined && { backend }) },
1355
1470
  };
1356
1471
  }
1357
1472