@edgehero/pi-dispatch 1.8.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.8.0",
3
+ "version": "1.9.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
package/src/cli.mjs CHANGED
@@ -43,7 +43,11 @@ Config comes from the environment (see .env.example); flags override it per run.
43
43
  Prefer being walked through all of this? The operator panel's /dispatch setup does every step
44
44
  with a consent per action: pi install npm:@edgehero/pi-dispatch-admin`;
45
45
 
46
- export async function main(argv = process.argv.slice(2), env = process.env) {
46
+ // Where this command's output goes. Defaults to the real stdout, so the CLI is byte-identical; a test
47
+ // injects a collector instead of reassigning `process.stdout.write`. That matters because `node --test`
48
+ // runs each file in a child process that serialises its own results over that same stdout, so a test
49
+ // holding a replacement across an `await` swallows the runner's result frames (issue #266).
50
+ export async function main(argv = process.argv.slice(2), env = process.env, { write = (chunk) => process.stdout.write(chunk) } = {}) {
47
51
  const cmd = argv[0];
48
52
 
49
53
  if (cmd === "init") {
@@ -69,7 +73,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
69
73
  const { runGithubAppSetup } = await import("./github-app-setup.mjs");
70
74
  return runGithubAppSetup(argv.slice(2), { env });
71
75
  }
72
- process.stdout.write(`pi-dispatch setup <target> — guided credential setup\n\n targets: github\n\n pi-dispatch setup github (--webhook-url <URL> | --no-webhook) [--org <org>] [--name <appName>]\n mint GitHub App credentials via the App Manifest flow — one browser click returns the app id,\n private key, and webhook secret; every write is shown first and individually consented\n`);
76
+ write(`pi-dispatch setup <target> — guided credential setup\n\n targets: github\n\n pi-dispatch setup github (--webhook-url <URL> | --no-webhook) [--org <org>] [--name <appName>]\n mint GitHub App credentials via the App Manifest flow — one browser click returns the app id,\n private key, and webhook secret; every write is shown first and individually consented\n`);
73
77
  return argv[1] ? 1 : 0;
74
78
  }
75
79
 
@@ -144,7 +148,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
144
148
  // throws inside buildDockerRunArgs after a budget slot is reserved.
145
149
  image: values.image || undefined,
146
150
  });
147
- process.stdout.write(`queued ${jobId} — folder ${folder}\nrun \`pi-dispatch worker\` to process it.\n`);
151
+ write(`queued ${jobId} — folder ${folder}\nrun \`pi-dispatch worker\` to process it.\n`);
148
152
  } catch (error) {
149
153
  return fail(`could not reach Valkey at ${config.valkeyUrl} — is it running? (docker compose up)\n ${error.message}`);
150
154
  } finally {
@@ -208,7 +212,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
208
212
  // as "nothing happened" and walks away from a fleet with one host still spending.
209
213
  return fail(`could not ${cmd} the whole deployment at ${url}\n ${done.length > 0 ? `${cmd}d: ${done.join(", ")}` : "nothing changed"}\n failed at: ${names[done.length]}\n ${error.message}`);
210
214
  }
211
- process.stdout.write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
215
+ write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
212
216
  } else {
213
217
  // "paused" is included in the counts because jobs enqueued while paused land in the
214
218
  // `paused` list, not `wait` -- omitting it would report backlog 0 in the exact state
@@ -226,7 +230,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
226
230
  const pausedState = states.every(Boolean);
227
231
  const pausedPartial = !pausedState && states.some(Boolean);
228
232
  const out = { pausedState, ...(pausedPartial ? { pausedPartial, pausedQueues: names.filter((_, i) => states[i]) } : {}), ...counts, ...(blind ? { fleet: blind } : {}) };
229
- process.stdout.write(`${JSON.stringify(out)}\n`);
233
+ write(`${JSON.stringify(out)}\n`);
230
234
  }
231
235
  } catch (error) {
232
236
  return fail(`could not reach Valkey at ${url} — is it running? (docker compose up)\n ${error.message}`);
@@ -236,7 +240,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
236
240
  return 0;
237
241
  }
238
242
 
239
- process.stdout.write(`${USAGE}\n`);
243
+ write(`${USAGE}\n`);
240
244
  return cmd ? 1 : 0;
241
245
  }
242
246
 
@@ -50,7 +50,20 @@ export const ISOLATION_FLAGS = [
50
50
  ];
51
51
 
52
52
  /**
53
- * Build the full `docker run` argv (excluding the leading "docker").
53
+ * WHAT the box is, with no Docker vocabulary in it.
54
+ *
55
+ * Split from the argv builder below so the description of a container exists as a VALUE before it becomes
56
+ * one runtime's flags. `buildDockerRunArgs` is unchanged in name, signature and output -- it is now
57
+ * `dockerArgsFromSpec(containerSpec(opts))` -- so every caller and every assertion is untouched, and the
58
+ * only thing that is new is that the middle of that sentence can be read on its own.
59
+ *
60
+ * Mounts are structured (`{host, container, readOnly}`) rather than pre-flattened `host:container:ro`
61
+ * strings, because the flattening IS the Docker part: a runtime that does not bind-mount has to be able to
62
+ * see which host path becomes which container path, and what may be written.
63
+ *
64
+ * `dockerExtra` is named for what it is. It carries raw Docker flags (`-i -t --entrypoint bash`, a
65
+ * Linux-only `--user`), so it is the one field a non-Docker consumer must refuse rather than translate.
66
+ * Calling it `extraFlags` at the boundary would have hidden that.
54
67
  *
55
68
  * @param image pinned job image tag/digest
56
69
  * @param env the closed env map from buildContainerEnv -- passed as explicit -e NAME=VALUE
@@ -66,7 +79,7 @@ export const ISOLATION_FLAGS = [
66
79
  * docker default bridge, which is what every job did before that requirement existed
67
80
  * @param extraFlags escape hatch for a Linux-only --user uid:gid on a bind-mounted local folder
68
81
  */
69
- export function buildDockerRunArgs({
82
+ export function containerSpec({
70
83
  image,
71
84
  env,
72
85
  jobDir,
@@ -84,6 +97,56 @@ export function buildDockerRunArgs({
84
97
  if (!name) throw new Error("docker run: container name is required");
85
98
  if (!workspace) throw new Error("docker run: workspace mount is required");
86
99
 
100
+ const mounts = [];
101
+ // The WHOLE /job dir is read-only (INT-CONTAINER-JOB-INPUTS): it holds prompt.md and pi/, and
102
+ // the agent cannot rewrite any of it. /workspace is the only writable mount.
103
+ if (jobDir) mounts.push({ host: jobDir, container: "/job", readOnly: true });
104
+ mounts.push({ host: workspace, container: "/workspace", readOnly: false });
105
+ // Local jobs get a writable /outbox host bind, the same host-bind mechanism as /workspace
106
+ // (DES-WORKER-ON-HOST). github jobs pass no outboxDir, so the request channel does not exist for
107
+ // them -- an untrusted issue author cannot chain (INT-OUTBOX-CONTRACT).
108
+ if (outboxDir) mounts.push({ host: outboxDir, container: "/outbox", readOnly: false });
109
+
110
+ // This job's OWN copy of its session transcript (REQ-RESUMABLE-SESSION, INT-SESSION-STORE-CONTRACT).
111
+ // Writable, because pi appends to it as the agent works -- and per-job, exactly like jobDir, which is
112
+ // the whole reason CONST-ISOLATION-CONTAINER-PER-JOB's "none host-wide" clause still reads true. The
113
+ // shared store under PI_SESSIONS_DIR is NEVER bind-mounted: one job here would otherwise be able to
114
+ // read and rewrite every other branch's and every other repository's transcripts, which is not a
115
+ // weakening of that constraint but its inversion. Absent unless the trigger armed run.resume AND a key
116
+ // resolved, so an unarmed job's argv is byte-identical to one built before this feature existed.
117
+ if (sessionDir) mounts.push({ host: sessionDir, container: CONTAINER_SESSION_DIR, readOnly: false });
118
+
119
+ // The operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY): custom models, global skills, a global
120
+ // persona, layered UNDER each repo's own .pi/. Read-only -- it is operator-authored deploy-time config,
121
+ // the same trust class as the baked floor, but the agent still must not rewrite it. Both job kinds.
122
+ if (globalPiDir) mounts.push({ host: globalPiDir, container: CONTAINER_GLOBAL_PI_DIR, readOnly: true });
123
+
124
+ return {
125
+ image,
126
+ name,
127
+ memory,
128
+ cpus,
129
+ network,
130
+ // UNCONDITIONALLY true, and there is deliberately no parameter that can unset it. The boundary is
131
+ // not a thing a caller opts into -- CONST-ISOLATION-CONTAINER-PER-JOB is why every other flag here
132
+ // exists -- so the spec is simply unable to describe an unisolated container, and the builder below
133
+ // refuses one it is handed. A field that could be false would be a way to ask for less.
134
+ isolated: true,
135
+ mounts,
136
+ env,
137
+ dockerExtra: extraFlags,
138
+ };
139
+ }
140
+
141
+ /**
142
+ * HOW Docker spells it. The only consumer of a spec today.
143
+ */
144
+ export function dockerArgsFromSpec(spec) {
145
+ // The builder CANNOT DECLINE the boundary. `containerSpec` cannot produce anything but `true`, so this
146
+ // only ever fires on a hand-built spec -- and a hand-built spec that forgot the field is exactly the
147
+ // case that must fail loudly rather than quietly emit a container with no isolation flags at all.
148
+ if (spec?.isolated !== true) throw new Error("docker run: refusing to build an argv for a spec that is not isolated");
149
+
87
150
  // `--network` sits HERE, beside --memory and --cpus, and deliberately NOT inside ISOLATION_FLAGS.
88
151
  // That array is the LITERAL, value-free, unconditional set, and two separate places assert every member
89
152
  // of it reaches the sandbox argv *against the imported array, not a copy* (CONST-ISOLATION-CONTAINER-PER-JOB
@@ -94,41 +157,34 @@ export function buildDockerRunArgs({
94
157
  //
95
158
  // null => the flag is ABSENT, so a job argv without an egress policy is byte-identical to one built
96
159
  // before this feature existed. Same shape as the sessionDir/outboxDir/globalPiDir mounts below.
97
- const args = ["run", `--name=${name}`, ...ISOLATION_FLAGS, `--memory=${memory}`, `--cpus=${cpus}`];
98
- if (network) args.push(`--network=${network}`);
99
- args.push(...extraFlags);
160
+ const args = ["run", `--name=${spec.name}`, ...ISOLATION_FLAGS, `--memory=${spec.memory}`, `--cpus=${spec.cpus}`];
161
+ if (spec.network) args.push(`--network=${spec.network}`);
162
+ args.push(...(spec.dockerExtra ?? []));
100
163
 
101
164
  // Explicit env allowlist. Each entry is `-e NAME=VALUE`, built from the closed map -- so a
102
165
  // stray host variable cannot ride along (no bare `-e NAME` inheriting from the host, no
103
166
  // --env-file). Undefined values are skipped, never passed as an empty string.
104
- for (const [k, v] of Object.entries(env ?? {})) {
167
+ for (const [k, v] of Object.entries(spec.env ?? {})) {
105
168
  if (v === undefined || v === null) continue;
106
169
  args.push("-e", `${k}=${v}`);
107
170
  }
108
171
 
109
- // The WHOLE /job dir is read-only (INT-CONTAINER-JOB-INPUTS): it holds prompt.md and pi/, and
110
- // the agent cannot rewrite any of it. /workspace is the only writable mount.
111
- if (jobDir) args.push("-v", `${jobDir}:/job:ro`);
112
- args.push("-v", `${workspace}:/workspace`);
113
- // Local jobs get a writable /outbox host bind, the same host-bind mechanism as /workspace
114
- // (DES-WORKER-ON-HOST). github jobs pass no outboxDir, so the request channel does not exist for
115
- // them -- an untrusted issue author cannot chain (INT-OUTBOX-CONTRACT).
116
- if (outboxDir) args.push("-v", `${outboxDir}:/outbox`);
172
+ // `-v` and its value stay TWO argv elements rather than one `--volume=` token. Not cosmetic: the mount
173
+ // assertions across this suite extract mounts by adjacency (`args[i - 1] === "-v"`), so collapsing the
174
+ // pair would make those filters return nothing and turn several exact-array checks vacuously green.
175
+ for (const m of spec.mounts ?? []) args.push("-v", `${m.host}:${m.container}${m.readOnly ? ":ro" : ""}`);
117
176
 
118
- // This job's OWN copy of its session transcript (REQ-RESUMABLE-SESSION, INT-SESSION-STORE-CONTRACT).
119
- // Writable, because pi appends to it as the agent works -- and per-job, exactly like jobDir, which is
120
- // the whole reason CONST-ISOLATION-CONTAINER-PER-JOB's "none host-wide" clause still reads true. The
121
- // shared store under PI_SESSIONS_DIR is NEVER bind-mounted: one job here would otherwise be able to
122
- // read and rewrite every other branch's and every other repository's transcripts, which is not a
123
- // weakening of that constraint but its inversion. Absent unless the trigger armed run.resume AND a key
124
- // resolved, so an unarmed job's argv is byte-identical to one built before this feature existed.
125
- if (sessionDir) args.push("-v", `${sessionDir}:${CONTAINER_SESSION_DIR}`);
126
-
127
- // The operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY): custom models, global skills, a global
128
- // persona, layered UNDER each repo's own .pi/. Read-only -- it is operator-authored deploy-time config,
129
- // the same trust class as the baked floor, but the agent still must not rewrite it. Both job kinds.
130
- if (globalPiDir) args.push("-v", `${globalPiDir}:${CONTAINER_GLOBAL_PI_DIR}:ro`);
131
-
132
- args.push(image);
177
+ args.push(spec.image);
133
178
  return args;
134
179
  }
180
+
181
+ /**
182
+ * Build the full `docker run` argv (excluding the leading "docker").
183
+ *
184
+ * The public entry point, unchanged: same name, same parameters, same argv byte for byte. Kept as the
185
+ * name rather than replaced by `dockerArgsFromSpec` because `CONST-EGRESS-POLICY-IN-THE-ARGV` cites this
186
+ * symbol in its Code evidence, and because a rename would churn every call site and assertion for nothing.
187
+ */
188
+ export function buildDockerRunArgs(opts) {
189
+ return dockerArgsFromSpec(containerSpec(opts));
190
+ }
@@ -110,7 +110,7 @@ export function makeRunMirror({ redis, retentionDays, now = () => Date.now(), lo
110
110
  await bounded(redis.zremrangebyscore(RUNS_INDEX, "-inf", `(${now() - windowMs}`), timeoutMs);
111
111
  await bounded(redis.zremrangebyrank(RUNS_INDEX, 0, -indexMax - 1), timeoutMs);
112
112
  // ROLLING expiry, deliberately unlike `budget.mjs`'s set-once rule and deliberately like
113
- // `pi-dispatch:sched-stalls`. A budget window must not be pushed forward by traffic or a busy
113
+ // `pi-dispatch:sched-stalls:<schedulerId>`. A budget window must not be pushed forward by traffic or a busy
114
114
  // day never resets; an ACTIVITY index should roll with traffic, because that is what it
115
115
  // describes. A fleet that stops running jobs loses its index one window later, which is
116
116
  // correct: there is nothing left to show.
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Per-scheduler stall accounting -- the money backstop for cron (constitution.md:203-216).
2
+ * Per-scheduler stall accounting -- the money backstop for cron (CONST-RETRY-INFRA-ONLY).
3
3
  *
4
4
  * BullMQ's `maxStalledCount` does not cover scheduler jobs: `moveStalledJobsToWait` derives
5
5
  * `isRepeatableJob` from the job's `rjk` field and skips the stall-fail for a live scheduler, so a
@@ -10,16 +10,41 @@
10
10
  * Injected `redis` (ioredis-compatible), `removeJobScheduler`, and `log` keep the logic testable with
11
11
  * no queue, no bullmq import, and no real Valkey.
12
12
  *
13
- * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- constitution.md:203-216 carve-out ("BullMQ will never do this for us")
13
+ * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- CONST-RETRY-INFRA-ONLY carve-out ("BullMQ will never do this for us")
14
14
  */
15
15
 
16
- // The Redis hash of per-scheduler stall counts (field = schedulerId, value = count). Exported so the admin
17
- // panel can read it (HGETALL) for the cron drill-in without re-deriving the key string.
16
+ // The prefix every stall counter lives under, so `KEYS pi-dispatch:sched-stalls*` still shows an operator
17
+ // the whole feature -- the affordance `wait:`, `slot:` and `budget:` all assume. Exported alongside the
18
+ // builder so the admin panel and the integration teardown compose keys through one definition and cannot
19
+ // drift from the writer.
18
20
  export const STALL_KEY = "pi-dispatch:sched-stalls";
19
21
 
20
- // A rolling window: the EXPIRE is re-set on every stall, so a scheduler that stops stalling for a full
21
- // day drops back to zero. This prevents unrelated transient stalls weeks apart from accumulating into a
22
- // false teardown -- only sustained stalling inside one window trips the threshold.
22
+ /**
23
+ * One scheduler's counter. The id is VALIDATED upstream rather than hashed here: it is operator-declared in
24
+ * `triggers.json`, `triggers.mjs` already refuses a `:` in it precisely to protect this parse, and the value
25
+ * of a readable keyspace is that `GET pi-dispatch:sched-stalls:nightly` answers the question directly.
26
+ */
27
+ export const stallKey = (schedulerId) => `${STALL_KEY}:${schedulerId}`;
28
+
29
+ // ONE KEY PER SCHEDULER, so the window is per scheduler.
30
+ //
31
+ // This was one HASH with a field per scheduler and a single `EXPIRE` on the whole key, which meant any
32
+ // scheduler's stall pushed the TTL forward for EVERY scheduler's count. The window never reset on a
33
+ // deployment where anything stalled regularly, so the guard silently degraded from "sustained stalling
34
+ // inside one window" to "cumulative stalling ever": three stalls ninety days apart tore a scheduler down
35
+ // if a neighbour was stalling twice a day, and did not if the deployment was quiet. Same scheduler, same
36
+ // stalls, opposite outcome, decided by an unrelated trigger (issue #267).
37
+ //
38
+ // Per-field TTLs would have fixed it in place and are not available: `HEXPIRE` does not exist on the pinned
39
+ // `valkey/valkey:8` (verified, `ERR unknown command`, recorded under DES-HOST-REGISTRY). A key per entity is
40
+ // the only shape that gets per-entity expiry.
41
+ //
42
+ // The EXPIRE still ROLLS on every stall, deliberately, and that is not `budget.mjs`'s set-once rule being
43
+ // broken. A budget window is a CALENDAR window and must not be pushed forward by traffic or a busy day
44
+ // never resets. This is a STREAK detector -- "is this scheduler wedged right now" -- and quiet for a day
45
+ // genuinely should forget. `poll:<repo>:close-gate:<deliveryId>` is the in-repo precedent, a bounded
46
+ // consecutive-failure counter given its own key and TTL for exactly this reason: it must decay with the
47
+ // thing it measures rather than with a larger family.
23
48
  const STALL_WINDOW_SECONDS = 24 * 60 * 60;
24
49
 
25
50
  /**
@@ -45,8 +70,9 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
45
70
  return;
46
71
  }
47
72
 
48
- const count = Number(await redis.hincrby(STALL_KEY, schedulerId, 1));
49
- await redis.expire(STALL_KEY, STALL_WINDOW_SECONDS);
73
+ const key = stallKey(schedulerId);
74
+ const count = Number(await redis.incr(key));
75
+ await redis.expire(key, STALL_WINDOW_SECONDS);
50
76
 
51
77
  if (count > threshold) {
52
78
  try {
@@ -56,7 +82,7 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
56
82
  // state, not an error -- swallow it so hdel and the teardown alert still run.
57
83
  log("scheduler_teardown_remove_failed", { schedulerId, error: error?.message });
58
84
  }
59
- await redis.hdel(STALL_KEY, schedulerId);
85
+ await redis.del(key);
60
86
  // The loud log is the "alert" half of the constitution's "removeJobScheduler -- or alert".
61
87
  log("scheduler_torn_down", { schedulerId, stalls: count });
62
88
  }
package/src/start.mjs CHANGED
@@ -233,6 +233,15 @@ export function makeReaper({ log }) {
233
233
  export async function startWorker(
234
234
  env = process.env,
235
235
  {
236
+ // WHERE THE BOOT LOG BYTES GO. Defaults to the real stdout, so production is byte-identical; a test
237
+ // passes a collector instead of reassigning `process.stdout.write`.
238
+ //
239
+ // That distinction is not stylistic. `node --test` runs each file in a CHILD PROCESS that serialises
240
+ // its own results over `process.stdout`, so a test that replaces the global and holds the replacement
241
+ // across an `await` swallows the runner's result frames for whatever completes in that window. Three
242
+ // tests in `start-wiring.test.mjs` were reported as not existing at all -- no name, no count, exit 0 --
243
+ // because this function's log line went through the same channel the runner needed (issue #266).
244
+ write = (chunk) => process.stdout.write(chunk),
236
245
  makeAuth = makeGitHubAuth,
237
246
  makeHost = makeGitHubHost,
238
247
  createWorkerFn = createWorker,
@@ -263,7 +272,7 @@ export async function startWorker(
263
272
  // other module takes `log` injected, and two tests pin the KEY SET of the fields object handed to an
264
273
  // injected log (`run_record_failed`, `wait_check`); a `host` added at any call site would break them,
265
274
  // while one added inside this closure cannot reach them.
266
- const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
275
+ const log = (event, fields = {}) => write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
267
276
 
268
277
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
269
278
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
@@ -842,7 +851,7 @@ export async function startWorker(
842
851
  // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
843
852
  // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
844
853
  // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
845
- const guard = makeStallGuard({
854
+ const onStalled = makeStallGuard({
846
855
  redis,
847
856
  threshold: config.schedulerStallMax,
848
857
  // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
@@ -850,7 +859,10 @@ export async function startWorker(
850
859
  removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
851
860
  log,
852
861
  });
853
- for (const w of allWorkers) w.on("stalled", (jobId) => void guard.onStalled(jobId));
862
+ // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
863
+ // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
864
+ // so every stall threw a TypeError and the money backstop never counted one (issue #267).
865
+ for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
854
866
 
855
867
  // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
856
868
  // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile