@edgehero/pi-dispatch 1.7.0 → 1.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.7.0",
3
+ "version": "1.9.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -48,6 +48,7 @@
48
48
  "./open-browser": "./src/open-browser.mjs",
49
49
  "./git-dirty": "./src/git-dirty.mjs",
50
50
  "./queue": "./src/queue.mjs",
51
+ "./capabilities": "./src/capabilities.mjs",
51
52
  "./connection": "./src/connection.mjs",
52
53
  "./job-id": "./src/job-id.mjs",
53
54
  "./forges": "./src/forges.mjs",
@@ -66,6 +67,7 @@
66
67
  "./get-token": "./src/get-token.mjs",
67
68
  "./runtime-settings": "./src/runtime-settings.mjs",
68
69
  "./run-history": "./src/run-history.mjs",
70
+ "./run-mirror": "./src/run-mirror.mjs",
69
71
  "./sandbox": "./src/sandbox.mjs",
70
72
  "./sandbox-store": "./src/sandbox-store.mjs",
71
73
  "./subscriptions": "./src/subscriptions.mjs",
@@ -0,0 +1,179 @@
1
+ /**
2
+ * What a host can serve, and which queue a job that needs one of those things belongs on (issue #57,
3
+ * `OQ-032`).
4
+ *
5
+ * Two trigger fields bind a job to a MACHINE rather than to a repository: `run.secretsProfile` names a
6
+ * resolver the operator declared in that host's `PI_SECRET_PROFILES`, and a `run.waitFor` condition names a
7
+ * check script from its `PI_WAIT_PROFILES`. Both are refused pre-spend when the host popping the job has
8
+ * not declared them, and both refusals are RETURNED rather than thrown, so they are never retried. The
9
+ * same trigger therefore succeeds or fails depending on which worker happened to take the delivery, and it
10
+ * reads like a configuration error rather than a placement one.
11
+ *
12
+ * #57's Gap 2 exempted forge jobs on the grounds that their workspace is a fresh clone that any host can
13
+ * build. Issues #225 and #230 retracted that without saying so: a clone is portable, a resolver on one
14
+ * machine's disk is not.
15
+ *
16
+ * WHY THIS IS DECIDED AT ENQUEUE. The obvious alternative is to let any host take the job and defer it if
17
+ * it cannot serve it. That does not work here, and the reason is upstream rather than ours: BullMQ promotes
18
+ * a delayed job on EACH WORKER'S OWN CLOCK (`scripts.js` passes the client clock as the cut-off), so the
19
+ * host whose clock runs fastest wins every attempt, deterministically. If the host that cannot serve the
20
+ * job is the fast one, the job never reaches the one that can. Jitter changes when the attempt happens, not
21
+ * who wins it.
22
+ *
23
+ * THE ONE CAPABILITY DELIBERATELY NOT ROUTED is `run.resume`. A session key is `sha256(kind, repo, ref)`
24
+ * and `session-store.mjs` records that it is "not random... anyone who knows the repository and the branch
25
+ * can compute it", so publishing keys to route on them would disclose which repositories and branches a
26
+ * deployment works on, recoverable by guessing a repo name. That is more disclosing than everything else in
27
+ * the registry combined. A resume that lands on the wrong host cold-starts and says so in the record.
28
+ */
29
+
30
+ /** Class letters. Open enum: `g:` is reserved for forge credentials, the highest-value follow-on. */
31
+ export const CAP_SECRET = "s";
32
+ export const CAP_WAIT = "w";
33
+
34
+ /**
35
+ * The name charset, duplicated from `secret-profiles.mjs` and `wait-for.mjs` deliberately: this module
36
+ * imports nothing, and the two it copies already copy it from `triggers.mjs` for the same reason. What
37
+ * matters here is what the set EXCLUDES -- a comma, so the token list joins unambiguously, and a colon, so
38
+ * `<class>:<name>` decomposes at the first one.
39
+ */
40
+ const PROFILE_NAME = /^[A-Za-z0-9._-]+$/;
41
+
42
+ /**
43
+ * How fresh a host's registry row must be before a job is ROUTED to it.
44
+ *
45
+ * Not the 90s TTL, and the difference is the point. The TTL is a crash backstop: it answers "has this host
46
+ * definitely gone", and it is deliberately six missed beats so a blip cannot evict a working host from the
47
+ * panel. A routing decision needs the opposite polarity -- evidence of LIFE, not absence of expiry --
48
+ * because a job routed onto a dead host's queue sits there until that host comes back, and nothing else
49
+ * will take it. Three beats is late enough to ride out a slow beat and early enough that a stopped host
50
+ * stops attracting work long before its row expires.
51
+ */
52
+ export const ROUTE_FRESH_MS = 45_000;
53
+
54
+ /** One token, or null when the name is not one this deployment would accept. */
55
+ function token(cls, name) {
56
+ return typeof name === "string" && PROFILE_NAME.test(name) ? `${cls}:${name}` : null;
57
+ }
58
+
59
+ /**
60
+ * What THIS host can serve, from its own config, as a sorted token list.
61
+ *
62
+ * Sorted so the published string is stable: an unstable one would make the row differ every beat and any
63
+ * future fingerprint over it useless.
64
+ */
65
+ export function capabilityTokens({ secretProfiles, waitProfiles } = {}) {
66
+ const out = new Set();
67
+ for (const name of Object.keys(secretProfiles ?? {})) {
68
+ const t = token(CAP_SECRET, name);
69
+ if (t) out.add(t);
70
+ }
71
+ for (const name of Object.keys(waitProfiles ?? {})) {
72
+ const t = token(CAP_WAIT, name);
73
+ if (t) out.add(t);
74
+ }
75
+ return [...out].sort();
76
+ }
77
+
78
+ /** The registry value. Comma-joined, which the charset makes unambiguous. */
79
+ export function serializeCaps(tokens) {
80
+ return (tokens ?? []).join(",");
81
+ }
82
+
83
+ /**
84
+ * A peer's tokens, from its registry row. Peer-written, so every element is re-validated here rather than
85
+ * trusted: a row is written by another process and this one decides where money-spending work goes.
86
+ */
87
+ export function parseCaps(raw) {
88
+ const out = new Set();
89
+ for (const part of String(raw ?? "").split(",")) {
90
+ const at = part.indexOf(":");
91
+ if (at <= 0) continue;
92
+ const cls = part.slice(0, at);
93
+ const name = part.slice(at + 1);
94
+ if ((cls === CAP_SECRET || cls === CAP_WAIT) && PROFILE_NAME.test(name)) out.add(part);
95
+ }
96
+ return out;
97
+ }
98
+
99
+ /**
100
+ * What a JOB needs, from the same two fields the worker's own refusals read.
101
+ *
102
+ * `waitProfileNames`-shaped inline rather than imported, because this module is loaded by the RECEIVER and
103
+ * `wait-for.mjs` carries the whole wait engine. The extraction is three lines and the charset check below
104
+ * is the same one that module makes.
105
+ */
106
+ export function jobNeeds(job) {
107
+ const needs = new Set();
108
+ const secret = token(CAP_SECRET, job?.secretsProfile);
109
+ if (secret) needs.add(secret);
110
+ if (Array.isArray(job?.waitFor)) {
111
+ for (const condition of job.waitFor) {
112
+ const wait = token(CAP_WAIT, condition?.profile);
113
+ if (wait) needs.add(wait);
114
+ }
115
+ }
116
+ return [...needs].sort();
117
+ }
118
+
119
+ /**
120
+ * Which queue this job belongs on: a host queue name, or `null` for the shared queue.
121
+ *
122
+ * FOUR ABSTENTIONS, and each one lands on today's behaviour rather than on something new. That is what
123
+ * makes this safe to put in front of every forge delivery: the rule can only ever move a job that would
124
+ * otherwise have had a coin flip decide whether it ran.
125
+ *
126
+ * 1. The job needs nothing host-specific. The overwhelming majority of deliveries.
127
+ * 2. EVERY live host can serve it. The shared queue is then strictly better than picking one, because it
128
+ * load-balances, and it is what the docs tell operators to aim for by declaring the same profiles
129
+ * everywhere. A deployment that follows that advice is byte-identical to before.
130
+ * 3. NO host can serve it. Routing cannot help, and the shared queue produces the existing pre-spend
131
+ * refusal (`secret-profile-unknown` / `wait-profile-unknown`), which is the honest answer and already
132
+ * names what to fix. Inventing a new terminal state here would be worse than the one that exists.
133
+ * 4. No CAPABLE host has a queue of its own, or the registry could not be read. Nothing to route to.
134
+ *
135
+ * Otherwise the job goes to a capable host, chosen by hashing its jobId: deterministic, so a redelivery of
136
+ * the same job lands the same way and dedup still works, and spread, so a delivery fanned out into replicas
137
+ * does not pile every replica onto one machine.
138
+ */
139
+ export function routeForgeJob({ hosts, needs, jobId, now = () => Date.now(), freshMs = ROUTE_FRESH_MS } = {}) {
140
+ if (!Array.isArray(needs) || needs.length === 0) return null; // (1)
141
+ if (!Array.isArray(hosts) || hosts.length === 0) return null; // (4) unreadable registry, or nobody home
142
+
143
+ const live = hosts.filter((h) => {
144
+ // `staleMs` is derived by the reader; a row without one is a row we cannot date, and an undatable
145
+ // row is not evidence of life.
146
+ const stale = Number(h?.staleMs);
147
+ return Number.isFinite(stale) && stale <= freshMs;
148
+ });
149
+ if (live.length === 0) return null;
150
+
151
+ const serves = (h) => {
152
+ const caps = parseCaps(h?.caps);
153
+ return needs.every((n) => caps.has(n));
154
+ };
155
+ const capable = live.filter(serves);
156
+ if (capable.length === 0) return null; // (3)
157
+ if (capable.length === live.length) return null; // (2)
158
+
159
+ // Only a host that DECLARED a name drains a queue of its own; an undeclared one reads the shared queue
160
+ // only, so routing to it would be routing into a queue nothing drains.
161
+ const routable = capable.filter((h) => h?.routes === true || h?.routes === "true").map((h) => h?.name).filter((n) => typeof n === "string" && n !== "");
162
+ if (routable.length === 0) return null; // (4)
163
+
164
+ routable.sort();
165
+ return routable[hashIndex(String(jobId ?? ""), routable.length)];
166
+ }
167
+
168
+ /**
169
+ * A stable index from a string. FNV-1a, inline: this module imports nothing, and a cryptographic hash would
170
+ * be a strange dependency for choosing between two machines.
171
+ */
172
+ function hashIndex(text, modulo) {
173
+ let h = 0x811c9dc5;
174
+ for (let i = 0; i < text.length; i++) {
175
+ h ^= text.charCodeAt(i);
176
+ h = Math.imul(h, 0x01000193) >>> 0;
177
+ }
178
+ return h % modulo;
179
+ }
package/src/cli.mjs CHANGED
@@ -43,7 +43,11 @@ Config comes from the environment (see .env.example); flags override it per run.
43
43
  Prefer being walked through all of this? The operator panel's /dispatch setup does every step
44
44
  with a consent per action: pi install npm:@edgehero/pi-dispatch-admin`;
45
45
 
46
- export async function main(argv = process.argv.slice(2), env = process.env) {
46
+ // Where this command's output goes. Defaults to the real stdout, so the CLI is byte-identical; a test
47
+ // injects a collector instead of reassigning `process.stdout.write`. That matters because `node --test`
48
+ // runs each file in a child process that serialises its own results over that same stdout, so a test
49
+ // holding a replacement across an `await` swallows the runner's result frames (issue #266).
50
+ export async function main(argv = process.argv.slice(2), env = process.env, { write = (chunk) => process.stdout.write(chunk) } = {}) {
47
51
  const cmd = argv[0];
48
52
 
49
53
  if (cmd === "init") {
@@ -69,7 +73,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
69
73
  const { runGithubAppSetup } = await import("./github-app-setup.mjs");
70
74
  return runGithubAppSetup(argv.slice(2), { env });
71
75
  }
72
- process.stdout.write(`pi-dispatch setup <target> — guided credential setup\n\n targets: github\n\n pi-dispatch setup github (--webhook-url <URL> | --no-webhook) [--org <org>] [--name <appName>]\n mint GitHub App credentials via the App Manifest flow — one browser click returns the app id,\n private key, and webhook secret; every write is shown first and individually consented\n`);
76
+ write(`pi-dispatch setup <target> — guided credential setup\n\n targets: github\n\n pi-dispatch setup github (--webhook-url <URL> | --no-webhook) [--org <org>] [--name <appName>]\n mint GitHub App credentials via the App Manifest flow — one browser click returns the app id,\n private key, and webhook secret; every write is shown first and individually consented\n`);
73
77
  return argv[1] ? 1 : 0;
74
78
  }
75
79
 
@@ -144,7 +148,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
144
148
  // throws inside buildDockerRunArgs after a budget slot is reserved.
145
149
  image: values.image || undefined,
146
150
  });
147
- process.stdout.write(`queued ${jobId} — folder ${folder}\nrun \`pi-dispatch worker\` to process it.\n`);
151
+ write(`queued ${jobId} — folder ${folder}\nrun \`pi-dispatch worker\` to process it.\n`);
148
152
  } catch (error) {
149
153
  return fail(`could not reach Valkey at ${config.valkeyUrl} — is it running? (docker compose up)\n ${error.message}`);
150
154
  } finally {
@@ -208,7 +212,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
208
212
  // as "nothing happened" and walks away from a fleet with one host still spending.
209
213
  return fail(`could not ${cmd} the whole deployment at ${url}\n ${done.length > 0 ? `${cmd}d: ${done.join(", ")}` : "nothing changed"}\n failed at: ${names[done.length]}\n ${error.message}`);
210
214
  }
211
- process.stdout.write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
215
+ write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
212
216
  } else {
213
217
  // "paused" is included in the counts because jobs enqueued while paused land in the
214
218
  // `paused` list, not `wait` -- omitting it would report backlog 0 in the exact state
@@ -226,7 +230,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
226
230
  const pausedState = states.every(Boolean);
227
231
  const pausedPartial = !pausedState && states.some(Boolean);
228
232
  const out = { pausedState, ...(pausedPartial ? { pausedPartial, pausedQueues: names.filter((_, i) => states[i]) } : {}), ...counts, ...(blind ? { fleet: blind } : {}) };
229
- process.stdout.write(`${JSON.stringify(out)}\n`);
233
+ write(`${JSON.stringify(out)}\n`);
230
234
  }
231
235
  } catch (error) {
232
236
  return fail(`could not reach Valkey at ${url} — is it running? (docker compose up)\n ${error.message}`);
@@ -236,7 +240,7 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
236
240
  return 0;
237
241
  }
238
242
 
239
- process.stdout.write(`${USAGE}\n`);
243
+ write(`${USAGE}\n`);
240
244
  return cmd ? 1 : 0;
241
245
  }
242
246
 
@@ -50,7 +50,20 @@ export const ISOLATION_FLAGS = [
50
50
  ];
51
51
 
52
52
  /**
53
- * Build the full `docker run` argv (excluding the leading "docker").
53
+ * WHAT the box is, with no Docker vocabulary in it.
54
+ *
55
+ * Split from the argv builder below so the description of a container exists as a VALUE before it becomes
56
+ * one runtime's flags. `buildDockerRunArgs` is unchanged in name, signature and output -- it is now
57
+ * `dockerArgsFromSpec(containerSpec(opts))` -- so every caller and every assertion is untouched, and the
58
+ * only thing that is new is that the middle of that sentence can be read on its own.
59
+ *
60
+ * Mounts are structured (`{host, container, readOnly}`) rather than pre-flattened `host:container:ro`
61
+ * strings, because the flattening IS the Docker part: a runtime that does not bind-mount has to be able to
62
+ * see which host path becomes which container path, and what may be written.
63
+ *
64
+ * `dockerExtra` is named for what it is. It carries raw Docker flags (`-i -t --entrypoint bash`, a
65
+ * Linux-only `--user`), so it is the one field a non-Docker consumer must refuse rather than translate.
66
+ * Calling it `extraFlags` at the boundary would have hidden that.
54
67
  *
55
68
  * @param image pinned job image tag/digest
56
69
  * @param env the closed env map from buildContainerEnv -- passed as explicit -e NAME=VALUE
@@ -66,7 +79,7 @@ export const ISOLATION_FLAGS = [
66
79
  * docker default bridge, which is what every job did before that requirement existed
67
80
  * @param extraFlags escape hatch for a Linux-only --user uid:gid on a bind-mounted local folder
68
81
  */
69
- export function buildDockerRunArgs({
82
+ export function containerSpec({
70
83
  image,
71
84
  env,
72
85
  jobDir,
@@ -84,6 +97,56 @@ export function buildDockerRunArgs({
84
97
  if (!name) throw new Error("docker run: container name is required");
85
98
  if (!workspace) throw new Error("docker run: workspace mount is required");
86
99
 
100
+ const mounts = [];
101
+ // The WHOLE /job dir is read-only (INT-CONTAINER-JOB-INPUTS): it holds prompt.md and pi/, and
102
+ // the agent cannot rewrite any of it. /workspace is the only writable mount.
103
+ if (jobDir) mounts.push({ host: jobDir, container: "/job", readOnly: true });
104
+ mounts.push({ host: workspace, container: "/workspace", readOnly: false });
105
+ // Local jobs get a writable /outbox host bind, the same host-bind mechanism as /workspace
106
+ // (DES-WORKER-ON-HOST). github jobs pass no outboxDir, so the request channel does not exist for
107
+ // them -- an untrusted issue author cannot chain (INT-OUTBOX-CONTRACT).
108
+ if (outboxDir) mounts.push({ host: outboxDir, container: "/outbox", readOnly: false });
109
+
110
+ // This job's OWN copy of its session transcript (REQ-RESUMABLE-SESSION, INT-SESSION-STORE-CONTRACT).
111
+ // Writable, because pi appends to it as the agent works -- and per-job, exactly like jobDir, which is
112
+ // the whole reason CONST-ISOLATION-CONTAINER-PER-JOB's "none host-wide" clause still reads true. The
113
+ // shared store under PI_SESSIONS_DIR is NEVER bind-mounted: one job here would otherwise be able to
114
+ // read and rewrite every other branch's and every other repository's transcripts, which is not a
115
+ // weakening of that constraint but its inversion. Absent unless the trigger armed run.resume AND a key
116
+ // resolved, so an unarmed job's argv is byte-identical to one built before this feature existed.
117
+ if (sessionDir) mounts.push({ host: sessionDir, container: CONTAINER_SESSION_DIR, readOnly: false });
118
+
119
+ // The operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY): custom models, global skills, a global
120
+ // persona, layered UNDER each repo's own .pi/. Read-only -- it is operator-authored deploy-time config,
121
+ // the same trust class as the baked floor, but the agent still must not rewrite it. Both job kinds.
122
+ if (globalPiDir) mounts.push({ host: globalPiDir, container: CONTAINER_GLOBAL_PI_DIR, readOnly: true });
123
+
124
+ return {
125
+ image,
126
+ name,
127
+ memory,
128
+ cpus,
129
+ network,
130
+ // UNCONDITIONALLY true, and there is deliberately no parameter that can unset it. The boundary is
131
+ // not a thing a caller opts into -- CONST-ISOLATION-CONTAINER-PER-JOB is why every other flag here
132
+ // exists -- so the spec is simply unable to describe an unisolated container, and the builder below
133
+ // refuses one it is handed. A field that could be false would be a way to ask for less.
134
+ isolated: true,
135
+ mounts,
136
+ env,
137
+ dockerExtra: extraFlags,
138
+ };
139
+ }
140
+
141
+ /**
142
+ * HOW Docker spells it. The only consumer of a spec today.
143
+ */
144
+ export function dockerArgsFromSpec(spec) {
145
+ // The builder CANNOT DECLINE the boundary. `containerSpec` cannot produce anything but `true`, so this
146
+ // only ever fires on a hand-built spec -- and a hand-built spec that forgot the field is exactly the
147
+ // case that must fail loudly rather than quietly emit a container with no isolation flags at all.
148
+ if (spec?.isolated !== true) throw new Error("docker run: refusing to build an argv for a spec that is not isolated");
149
+
87
150
  // `--network` sits HERE, beside --memory and --cpus, and deliberately NOT inside ISOLATION_FLAGS.
88
151
  // That array is the LITERAL, value-free, unconditional set, and two separate places assert every member
89
152
  // of it reaches the sandbox argv *against the imported array, not a copy* (CONST-ISOLATION-CONTAINER-PER-JOB
@@ -94,41 +157,34 @@ export function buildDockerRunArgs({
94
157
  //
95
158
  // null => the flag is ABSENT, so a job argv without an egress policy is byte-identical to one built
96
159
  // before this feature existed. Same shape as the sessionDir/outboxDir/globalPiDir mounts below.
97
- const args = ["run", `--name=${name}`, ...ISOLATION_FLAGS, `--memory=${memory}`, `--cpus=${cpus}`];
98
- if (network) args.push(`--network=${network}`);
99
- args.push(...extraFlags);
160
+ const args = ["run", `--name=${spec.name}`, ...ISOLATION_FLAGS, `--memory=${spec.memory}`, `--cpus=${spec.cpus}`];
161
+ if (spec.network) args.push(`--network=${spec.network}`);
162
+ args.push(...(spec.dockerExtra ?? []));
100
163
 
101
164
  // Explicit env allowlist. Each entry is `-e NAME=VALUE`, built from the closed map -- so a
102
165
  // stray host variable cannot ride along (no bare `-e NAME` inheriting from the host, no
103
166
  // --env-file). Undefined values are skipped, never passed as an empty string.
104
- for (const [k, v] of Object.entries(env ?? {})) {
167
+ for (const [k, v] of Object.entries(spec.env ?? {})) {
105
168
  if (v === undefined || v === null) continue;
106
169
  args.push("-e", `${k}=${v}`);
107
170
  }
108
171
 
109
- // The WHOLE /job dir is read-only (INT-CONTAINER-JOB-INPUTS): it holds prompt.md and pi/, and
110
- // the agent cannot rewrite any of it. /workspace is the only writable mount.
111
- if (jobDir) args.push("-v", `${jobDir}:/job:ro`);
112
- args.push("-v", `${workspace}:/workspace`);
113
- // Local jobs get a writable /outbox host bind, the same host-bind mechanism as /workspace
114
- // (DES-WORKER-ON-HOST). github jobs pass no outboxDir, so the request channel does not exist for
115
- // them -- an untrusted issue author cannot chain (INT-OUTBOX-CONTRACT).
116
- if (outboxDir) args.push("-v", `${outboxDir}:/outbox`);
172
+ // `-v` and its value stay TWO argv elements rather than one `--volume=` token. Not cosmetic: the mount
173
+ // assertions across this suite extract mounts by adjacency (`args[i - 1] === "-v"`), so collapsing the
174
+ // pair would make those filters return nothing and turn several exact-array checks vacuously green.
175
+ for (const m of spec.mounts ?? []) args.push("-v", `${m.host}:${m.container}${m.readOnly ? ":ro" : ""}`);
117
176
 
118
- // This job's OWN copy of its session transcript (REQ-RESUMABLE-SESSION, INT-SESSION-STORE-CONTRACT).
119
- // Writable, because pi appends to it as the agent works -- and per-job, exactly like jobDir, which is
120
- // the whole reason CONST-ISOLATION-CONTAINER-PER-JOB's "none host-wide" clause still reads true. The
121
- // shared store under PI_SESSIONS_DIR is NEVER bind-mounted: one job here would otherwise be able to
122
- // read and rewrite every other branch's and every other repository's transcripts, which is not a
123
- // weakening of that constraint but its inversion. Absent unless the trigger armed run.resume AND a key
124
- // resolved, so an unarmed job's argv is byte-identical to one built before this feature existed.
125
- if (sessionDir) args.push("-v", `${sessionDir}:${CONTAINER_SESSION_DIR}`);
126
-
127
- // The operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY): custom models, global skills, a global
128
- // persona, layered UNDER each repo's own .pi/. Read-only -- it is operator-authored deploy-time config,
129
- // the same trust class as the baked floor, but the agent still must not rewrite it. Both job kinds.
130
- if (globalPiDir) args.push("-v", `${globalPiDir}:${CONTAINER_GLOBAL_PI_DIR}:ro`);
131
-
132
- args.push(image);
177
+ args.push(spec.image);
133
178
  return args;
134
179
  }
180
+
181
+ /**
182
+ * Build the full `docker run` argv (excluding the leading "docker").
183
+ *
184
+ * The public entry point, unchanged: same name, same parameters, same argv byte for byte. Kept as the
185
+ * name rather than replaced by `dockerArgsFromSpec` because `CONST-EGRESS-POLICY-IN-THE-ARGV` cites this
186
+ * symbol in its Code evidence, and because a rename would churn every call site and assertion for nothing.
187
+ */
188
+ export function buildDockerRunArgs(opts) {
189
+ return dockerArgsFromSpec(containerSpec(opts));
190
+ }
@@ -0,0 +1,221 @@
1
+ /**
2
+ * A fleet-visible copy of the run history (issue #57, Gap 3).
3
+ *
4
+ * Every worker writes its records to its own `PI_LOGS_DIR`, so on more than one machine each host's panel
5
+ * lists only the runs on its own disk. The operator sees a third of their deployment and has no way to
6
+ * know it.
7
+ *
8
+ * SHARED STORAGE IS THE OTHER ANSWER AND IT IS NOT SECOND-BEST. On a shared `PI_LOGS_DIR` the local read
9
+ * IS the merged read, with no machinery at all, and this module is redundant. It ships because that trade
10
+ * runs both ways and an operator must be allowed to decline it: sharing the directory also shares the
11
+ * PII-bearing raw `.log`, and a mount outage becomes a LOST RECORD where a Valkey outage costs only a
12
+ * fleet view. Both shapes work; `docs/multi-host.md` says which is which.
13
+ *
14
+ * THE FILE IS THE RECORD AND THIS IS A VIEW. Three things follow, and each is load-bearing:
15
+ *
16
+ * - The file is written FIRST, always. A crash between the two leaves a fleet-visible run whose durable
17
+ * source does not exist, which inverts the one claim this design rests on.
18
+ * - The mirror's TTL is never longer than the file retention window, so it can never show a run whose
19
+ * file has already been reaped. A view that outlives its source is a second source of truth, which is
20
+ * exactly what `DES-RUN-HISTORY-FLAT-FILES-NO-DB` refuses.
21
+ * - Nothing derived is stored. The bytes are the sidecar's own bytes, so there is nothing to be stale
22
+ * RELATIVE TO: a retry overwrites the same key exactly as it overwrites the same file, and cost
23
+ * classification is still computed at fold time from `subscriptions.json` rather than frozen here.
24
+ *
25
+ * WHY THE WHOLE RECORD RATHER THAN A PROJECTION. The record is PII-free BY CONSTRUCTION -- it holds no
26
+ * attacker-chosen string, which `INT-RUN-HISTORY-FILE-CONTRACT` states and `buildRecord` enforces field by
27
+ * field. Copying it whole inherits that property; a projection would re-derive it at a second serialiser,
28
+ * where the next person to add a field has to remember this file exists. It is also what the readers need:
29
+ * the cost fold, the graph and the insights view read eleven fields between them.
30
+ *
31
+ * NOT MIRRORED: the raw `.log`. It is the one artifact here that holds issue text, comment text and tool
32
+ * output, and mirroring it would move that off the machine the operator chose to keep it on. A foreign
33
+ * run's record names its host, so the panel can say where the bytes are rather than pretending there are
34
+ * none.
35
+ */
36
+
37
+ /** The index: sanitized jobId -> the run's end (or start) in millis. */
38
+ export const RUNS_INDEX = "runs:index";
39
+
40
+ /** One run's own bytes. */
41
+ export const runRecordKey = (sanitizedJobId) => `runs:rec:${sanitizedJobId}`;
42
+
43
+ /**
44
+ * The deepest any reader asks. `SCAN_WINDOW_MAX_DAYS` in the admin is 92, so a longer window would hold
45
+ * bytes nothing can request.
46
+ */
47
+ export const MIRROR_MAX_DAYS = 92;
48
+
49
+ /**
50
+ * A hard ceiling on index members, independent of the time window.
51
+ *
52
+ * The window alone does not bound memory: a deployment running thousands of jobs a day would hold a
53
+ * quarter of a million members for ninety-two days. This caps what the fleet view can cost at roughly the
54
+ * depth a panel can display, and the file on disk remains the complete history either way.
55
+ */
56
+ export const RUNS_INDEX_MAX = 5_000;
57
+
58
+ const DAY_MS = 24 * 60 * 60 * 1000;
59
+ const OP_TIMEOUT_MS = 2_000;
60
+
61
+ /**
62
+ * How long a mirrored record lives.
63
+ *
64
+ * Never longer than the operator's own retention, and never longer than what any reader asks for.
65
+ * `retentionDays: 0` means keep the files forever, which is the one case where the mirror is the shorter
66
+ * of the two, so it clamps to the reader's ceiling rather than to infinity.
67
+ */
68
+ export function mirrorWindowMs(retentionDays) {
69
+ const days = Number(retentionDays) > 0 ? Math.min(Number(retentionDays), MIRROR_MAX_DAYS) : MIRROR_MAX_DAYS;
70
+ return days * DAY_MS;
71
+ }
72
+
73
+ /**
74
+ * BullMQ's connections carry `maxRetriesPerRequest: null`, so a command against an unreachable server
75
+ * QUEUES FOREVER rather than rejecting. Every await here is bounded for that reason; an unbounded one
76
+ * would not fail the mirror, it would hang the job that was writing to it.
77
+ */
78
+ function bounded(promise, ms) {
79
+ let timer;
80
+ return Promise.race([
81
+ promise,
82
+ new Promise((_, reject) => {
83
+ timer = setTimeout(() => reject(new Error("mirror timeout")), ms);
84
+ }),
85
+ ]).finally(() => clearTimeout(timer));
86
+ }
87
+
88
+ /**
89
+ * The writer. Returns `{ mirror, close }`; `mirror` never throws and never rejects.
90
+ *
91
+ * A history blip must not fail a job that has already run and already been recorded to disk. Every failure
92
+ * here costs a row in a fleet view and nothing else, which is why the whole body is wrapped and the result
93
+ * is a boolean nobody is obliged to read.
94
+ */
95
+ export function makeRunMirror({ redis, retentionDays, now = () => Date.now(), log = () => {}, timeoutMs = OP_TIMEOUT_MS, indexMax = RUNS_INDEX_MAX } = {}) {
96
+ const windowMs = mirrorWindowMs(retentionDays);
97
+ let warned = false;
98
+
99
+ return {
100
+ async mirror(record, sanitizedJobId) {
101
+ if (!redis || !record || !sanitizedJobId) return false;
102
+ try {
103
+ const at = Date.parse(record.endedAt ?? record.startedAt ?? "");
104
+ const score = Number.isFinite(at) ? at : now();
105
+ const body = JSON.stringify(record);
106
+ await bounded(redis.set(runRecordKey(sanitizedJobId), body, "PX", windowMs), timeoutMs);
107
+ await bounded(redis.zadd(RUNS_INDEX, score, sanitizedJobId), timeoutMs);
108
+ // Trimmed by the WRITER, twice: by age, and by count. Two `ZREMRANGE`s against a run that
109
+ // took minutes is free, and it means no reader has to pay for a backlog it did not create.
110
+ await bounded(redis.zremrangebyscore(RUNS_INDEX, "-inf", `(${now() - windowMs}`), timeoutMs);
111
+ await bounded(redis.zremrangebyrank(RUNS_INDEX, 0, -indexMax - 1), timeoutMs);
112
+ // ROLLING expiry, deliberately unlike `budget.mjs`'s set-once rule and deliberately like
113
+ // `pi-dispatch:sched-stalls:<schedulerId>`. A budget window must not be pushed forward by traffic or a busy
114
+ // day never resets; an ACTIVITY index should roll with traffic, because that is what it
115
+ // describes. A fleet that stops running jobs loses its index one window later, which is
116
+ // correct: there is nothing left to show.
117
+ await bounded(redis.pexpire(RUNS_INDEX, windowMs), timeoutMs);
118
+ warned = false;
119
+ return true;
120
+ } catch (err) {
121
+ // Once per transition, not once per job: a Valkey outage during a busy hour must not turn one
122
+ // fault into a thousand log lines (`notePackageKey`'s precedent).
123
+ if (!warned) {
124
+ warned = true;
125
+ log("run_mirror_failed", { jobId: sanitizedJobId, reason: err?.message });
126
+ }
127
+ return false;
128
+ }
129
+ },
130
+ };
131
+ }
132
+
133
+ /**
134
+ * The reader. Returns `{ runs, degraded }` and never throws.
135
+ *
136
+ * `degraded` is a DISCRIMINATED channel rather than a silence, and two of its values must not collapse:
137
+ * `"off"` means the index is absent, which is what a single-host deployment and a fleet of workers still
138
+ * below the version floor both look like, while `"unreachable"` means we could not tell. A new panel
139
+ * meeting old workers has to read "off", not "error".
140
+ *
141
+ * Two round trips regardless of how many runs come back: one `ZREVRANGEBYSCORE` for the ids, one `MGET`
142
+ * for the bodies. Never one read per run.
143
+ */
144
+ export async function readMirroredRuns(redis, { limit = 50, sinceMs = 0, now = () => Date.now(), timeoutMs = OP_TIMEOUT_MS } = {}) {
145
+ if (!redis) return { runs: [], degraded: "off" };
146
+ let ids;
147
+ try {
148
+ ids = await bounded(redis.zrevrangebyscore(RUNS_INDEX, "+inf", `(${sinceMs}`, "LIMIT", 0, Math.max(1, limit)), timeoutMs);
149
+ } catch (err) {
150
+ return { runs: [], degraded: `unreachable (${err?.message ?? "?"})` };
151
+ }
152
+ if (!Array.isArray(ids) || ids.length === 0) return { runs: [], degraded: "off" };
153
+
154
+ let bodies;
155
+ try {
156
+ bodies = await bounded(redis.mget(...ids.map(runRecordKey)), timeoutMs);
157
+ } catch (err) {
158
+ return { runs: [], degraded: `unreachable (${err?.message ?? "?"})` };
159
+ }
160
+
161
+ const runs = [];
162
+ const stale = [];
163
+ for (let i = 0; i < ids.length; i++) {
164
+ const raw = bodies?.[i];
165
+ if (typeof raw !== "string" || raw === "") {
166
+ // An id whose body has expired: the per-key TTL fired and the index member outlived it. The
167
+ // READER prunes it, which is what `wait:held` does for the same shape and for the same reason --
168
+ // a writer that crashed cannot clean up after itself, and a reader is already here.
169
+ stale.push(ids[i]);
170
+ continue;
171
+ }
172
+ try {
173
+ const rec = JSON.parse(raw);
174
+ if (rec && typeof rec === "object") runs.push(rec);
175
+ } catch {
176
+ stale.push(ids[i]); // unparseable is indistinguishable from gone, and equally not showable
177
+ }
178
+ }
179
+ if (stale.length > 0) {
180
+ try {
181
+ await bounded(redis.zrem(RUNS_INDEX, ...stale), timeoutMs);
182
+ } catch {
183
+ // best-effort: a straggler in the index costs one skipped row next time, never a wrong one
184
+ }
185
+ }
186
+ return { runs, degraded: runs.length >= limit ? "truncated" : "ok" };
187
+ }
188
+
189
+ /**
190
+ * One list from two sources.
191
+ *
192
+ * DEDUP BY LATER `endedAt`, LOCAL BREAKS A TIE. Not decoration: a retry can land on a different host, so
193
+ * host A may hold attempt 0 (failed) while host B mirrored attempt 1 (completed). "Local wins" alone would
194
+ * show the stale one. Local breaking an exact tie keeps a single-host deployment reading its own files.
195
+ *
196
+ * CUT AFTER THE SORT, never before. Slicing first is the defect `held.test.mjs` already exists to prevent:
197
+ * it makes the result depend on which source happened to be longer.
198
+ */
199
+ export function mergeRuns(local, mirrored, { limit = 50 } = {}) {
200
+ const by = new Map();
201
+ const at = (r) => {
202
+ const t = Date.parse(r?.endedAt ?? r?.startedAt ?? "");
203
+ return Number.isFinite(t) ? t : -Infinity;
204
+ };
205
+ // Mirrored first, so a local record with an equal timestamp overwrites it on the second pass.
206
+ for (const r of Array.isArray(mirrored) ? mirrored : []) if (r?.jobId) by.set(r.jobId, r);
207
+ for (const r of Array.isArray(local) ? local : []) {
208
+ if (!r?.jobId) continue;
209
+ const seen = by.get(r.jobId);
210
+ if (!seen || at(r) >= at(seen)) by.set(r.jobId, r);
211
+ }
212
+ const out = [...by.values()].sort((a, b) => at(b) - at(a));
213
+ return out.slice(0, Math.max(0, limit));
214
+ }
215
+
216
+ /** The distinct hosts a merged list came from, computed from the RECORDS rather than from the mirror. */
217
+ export function hostsIn(runs) {
218
+ const names = new Set();
219
+ for (const r of runs ?? []) if (typeof r?.host === "string" && r.host !== "") names.add(r.host);
220
+ return [...names].sort();
221
+ }
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Per-scheduler stall accounting -- the money backstop for cron (constitution.md:203-216).
2
+ * Per-scheduler stall accounting -- the money backstop for cron (CONST-RETRY-INFRA-ONLY).
3
3
  *
4
4
  * BullMQ's `maxStalledCount` does not cover scheduler jobs: `moveStalledJobsToWait` derives
5
5
  * `isRepeatableJob` from the job's `rjk` field and skips the stall-fail for a live scheduler, so a
@@ -10,16 +10,41 @@
10
10
  * Injected `redis` (ioredis-compatible), `removeJobScheduler`, and `log` keep the logic testable with
11
11
  * no queue, no bullmq import, and no real Valkey.
12
12
  *
13
- * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- constitution.md:203-216 carve-out ("BullMQ will never do this for us")
13
+ * Custom: per-scheduler stall accounting; BullMQ's maxStalledCount does not cover scheduler jobs -- CONST-RETRY-INFRA-ONLY carve-out ("BullMQ will never do this for us")
14
14
  */
15
15
 
16
- // The Redis hash of per-scheduler stall counts (field = schedulerId, value = count). Exported so the admin
17
- // panel can read it (HGETALL) for the cron drill-in without re-deriving the key string.
16
+ // The prefix every stall counter lives under, so `KEYS pi-dispatch:sched-stalls*` still shows an operator
17
+ // the whole feature -- the affordance `wait:`, `slot:` and `budget:` all assume. Exported alongside the
18
+ // builder so the admin panel and the integration teardown compose keys through one definition and cannot
19
+ // drift from the writer.
18
20
  export const STALL_KEY = "pi-dispatch:sched-stalls";
19
21
 
20
- // A rolling window: the EXPIRE is re-set on every stall, so a scheduler that stops stalling for a full
21
- // day drops back to zero. This prevents unrelated transient stalls weeks apart from accumulating into a
22
- // false teardown -- only sustained stalling inside one window trips the threshold.
22
+ /**
23
+ * One scheduler's counter. The id is VALIDATED upstream rather than hashed here: it is operator-declared in
24
+ * `triggers.json`, `triggers.mjs` already refuses a `:` in it precisely to protect this parse, and the value
25
+ * of a readable keyspace is that `GET pi-dispatch:sched-stalls:nightly` answers the question directly.
26
+ */
27
+ export const stallKey = (schedulerId) => `${STALL_KEY}:${schedulerId}`;
28
+
29
+ // ONE KEY PER SCHEDULER, so the window is per scheduler.
30
+ //
31
+ // This was one HASH with a field per scheduler and a single `EXPIRE` on the whole key, which meant any
32
+ // scheduler's stall pushed the TTL forward for EVERY scheduler's count. The window never reset on a
33
+ // deployment where anything stalled regularly, so the guard silently degraded from "sustained stalling
34
+ // inside one window" to "cumulative stalling ever": three stalls ninety days apart tore a scheduler down
35
+ // if a neighbour was stalling twice a day, and did not if the deployment was quiet. Same scheduler, same
36
+ // stalls, opposite outcome, decided by an unrelated trigger (issue #267).
37
+ //
38
+ // Per-field TTLs would have fixed it in place and are not available: `HEXPIRE` does not exist on the pinned
39
+ // `valkey/valkey:8` (verified, `ERR unknown command`, recorded under DES-HOST-REGISTRY). A key per entity is
40
+ // the only shape that gets per-entity expiry.
41
+ //
42
+ // The EXPIRE still ROLLS on every stall, deliberately, and that is not `budget.mjs`'s set-once rule being
43
+ // broken. A budget window is a CALENDAR window and must not be pushed forward by traffic or a busy day
44
+ // never resets. This is a STREAK detector -- "is this scheduler wedged right now" -- and quiet for a day
45
+ // genuinely should forget. `poll:<repo>:close-gate:<deliveryId>` is the in-repo precedent, a bounded
46
+ // consecutive-failure counter given its own key and TTL for exactly this reason: it must decay with the
47
+ // thing it measures rather than with a larger family.
23
48
  const STALL_WINDOW_SECONDS = 24 * 60 * 60;
24
49
 
25
50
  /**
@@ -45,8 +70,9 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
45
70
  return;
46
71
  }
47
72
 
48
- const count = Number(await redis.hincrby(STALL_KEY, schedulerId, 1));
49
- await redis.expire(STALL_KEY, STALL_WINDOW_SECONDS);
73
+ const key = stallKey(schedulerId);
74
+ const count = Number(await redis.incr(key));
75
+ await redis.expire(key, STALL_WINDOW_SECONDS);
50
76
 
51
77
  if (count > threshold) {
52
78
  try {
@@ -56,7 +82,7 @@ export function makeStallGuard({ redis, threshold, removeJobScheduler, log }) {
56
82
  // state, not an error -- swallow it so hdel and the teardown alert still run.
57
83
  log("scheduler_teardown_remove_failed", { schedulerId, error: error?.message });
58
84
  }
59
- await redis.hdel(STALL_KEY, schedulerId);
85
+ await redis.del(key);
60
86
  // The loud log is the "alert" half of the constitution's "removeJobScheduler -- or alert".
61
87
  log("scheduler_torn_down", { schedulerId, stalls: count });
62
88
  }
package/src/start.mjs CHANGED
@@ -15,6 +15,7 @@ import { makeAzureAuth } from "./azure-auth.mjs";
15
15
  import { makeAzureHost } from "./azure-host.mjs";
16
16
  import { makeEgressPreflight } from "./egress.mjs";
17
17
  import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
18
+ import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
18
19
  import { cronFingerprint } from "./fingerprint.mjs";
19
20
  import { makeHostRegistry } from "./host-registry.mjs";
20
21
  import { makeImagePreflight } from "./image-preflight.mjs";
@@ -33,7 +34,8 @@ import { makeWaitState } from "./wait-state.mjs";
33
34
  import { hostQueueName, makeQueue } from "./queue.mjs";
34
35
  import { makeRunContainer } from "./run-container.mjs";
35
36
  import { makeSecretsResolver } from "./secrets.mjs";
36
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
37
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
38
+ import { makeRunMirror } from "./run-mirror.mjs";
37
39
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
38
40
  import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
39
41
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
@@ -231,12 +233,22 @@ export function makeReaper({ log }) {
231
233
  export async function startWorker(
232
234
  env = process.env,
233
235
  {
236
+ // WHERE THE BOOT LOG BYTES GO. Defaults to the real stdout, so production is byte-identical; a test
237
+ // passes a collector instead of reassigning `process.stdout.write`.
238
+ //
239
+ // That distinction is not stylistic. `node --test` runs each file in a CHILD PROCESS that serialises
240
+ // its own results over `process.stdout`, so a test that replaces the global and holds the replacement
241
+ // across an `await` swallows the runner's result frames for whatever completes in that window. Three
242
+ // tests in `start-wiring.test.mjs` were reported as not existing at all -- no name, no count, exit 0 --
243
+ // because this function's log line went through the same channel the runner needed (issue #266).
244
+ write = (chunk) => process.stdout.write(chunk),
234
245
  makeAuth = makeGitHubAuth,
235
246
  makeHost = makeGitHubHost,
236
247
  createWorkerFn = createWorker,
237
248
  makeReaper: makeReaperFn = makeReaper,
238
249
  makeLogSink: makeLogSinkFn = makeLogSink,
239
250
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
251
+ makeRunMirror: makeRunMirrorFn = makeRunMirror,
240
252
  makeLogReaper: makeLogReaperFn = makeLogReaper,
241
253
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
242
254
  makeRunContainer: makeRunContainerFn = makeRunContainer,
@@ -260,7 +272,7 @@ export async function startWorker(
260
272
  // other module takes `log` injected, and two tests pin the KEY SET of the fields object handed to an
261
273
  // injected log (`run_record_failed`, `wait_check`); a `host` added at any call site would break them,
262
274
  // while one added inside this closure cannot reach them.
263
- const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
275
+ const log = (event, fields = {}) => write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
264
276
 
265
277
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
266
278
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
@@ -446,10 +458,22 @@ export async function startWorker(
446
458
  // that can neither disarm nor pre-spend-check.
447
459
  const onceTriggersFile = env.PI_TRIGGERS_FILE ?? join(process.cwd(), "triggers.json");
448
460
  const disarmOnce = makeDisarmOnce({ triggersPath: onceTriggersFile, log });
461
+ // The fleet-visible copy of the run history (issue #57, Gap 3). Armed only on a deployment that declared
462
+ // a worker name: an unnamed one is a single host, its own files ARE the whole history, and a mirror
463
+ // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
464
+ // no job then issues a single extra Valkey command.
465
+ const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
449
466
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
450
467
  // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
451
468
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
452
- writeRecord(buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName }));
469
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName });
470
+ writeRecord(record);
471
+ // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
472
+ // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
473
+ // and a view that can outlive its source is a second source of truth. Not awaited, because this is
474
+ // the job's own completion path: a slow Valkey may cost a row in a panel and must never hold up a
475
+ // job that has already finished and already been written to disk. `mirror` never rejects.
476
+ void runMirror?.mirror(record, sanitizeJobId(record.jobId));
453
477
  // Strictly AFTER the durable record: "fired" means "produced a run record", and the crash
454
478
  // direction this ordering buys is the chosen one -- an armed one-shot with a record, never a
455
479
  // disarm before writeRecord RETURNED. Returned, not succeeded: the record writer swallows fs
@@ -585,6 +609,13 @@ export async function startWorker(
585
609
  // Whether this host DRAINS a queue of its own. Every worker publishes a row; only a host that declared
586
610
  // a name has somewhere for routed work to go, and a reader must not invent a queue for one that has not.
587
611
  routes: config.workerNameDeclared,
612
+ // What this host can serve that another might not (issue #57, `OQ-032`): the secret and wait profiles
613
+ // it has declared. NAMES only, never the resolver paths behind them -- a path is PII on Windows and
614
+ // operator topology everywhere, and the receiver only needs to know WHICH host, not what it runs.
615
+ //
616
+ // Recomputed per beat rather than frozen at boot, for the reason the digest is: a host that gains a
617
+ // profile on restart must start attracting that work within one beat, and one that loses it must stop.
618
+ caps: () => serializeCaps(capabilityTokens(config)),
588
619
  // The host's IANA zone, because a cron PATTERN carries none: `triggers.json` has no `tz` field and
589
620
  // BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's LOCAL time.
590
621
  // On one host that is exactly what an operator means; on two in different zones the same pattern is
@@ -820,7 +851,7 @@ export async function startWorker(
820
851
  // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
821
852
  // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
822
853
  // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
823
- const guard = makeStallGuard({
854
+ const onStalled = makeStallGuard({
824
855
  redis,
825
856
  threshold: config.schedulerStallMax,
826
857
  // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
@@ -828,7 +859,10 @@ export async function startWorker(
828
859
  removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
829
860
  log,
830
861
  });
831
- for (const w of allWorkers) w.on("stalled", (jobId) => void guard.onStalled(jobId));
862
+ // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
863
+ // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
864
+ // so every stall threw a TypeError and the money backstop never counted one (issue #267).
865
+ for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
832
866
 
833
867
  // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
834
868
  // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile