@edgehero/pi-dispatch 1.6.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -32,6 +32,13 @@ PI_CONCURRENCY=3 # how many jobs run in parallel
32
32
 
33
33
  # --- Infrastructure ---
34
34
  VALKEY_URL=redis://127.0.0.1:6379
35
+ # PI_WORKER_NAME= # what this machine calls itself. Default: your hostname, lowercased and reduced to [A-Za-z0-9._-]
36
+ # Lands on every worker log line and in every run record, and identifies this host to the others when you run more than one
37
+ # Set it if your hostname is something you would rather not have in your own run history (a laptop often carries a person's name)
38
+ # Refused at boot if it is not [A-Za-z0-9._-], does not start with a letter or digit, is over 64 characters, or ends in .json or .log
39
+ # SETTING IT TURNS ON MULTI-HOST ROUTING (docs/multi-host.md): work only this machine can do (a cron folder, a chained child,
40
+ # a manual run) is enqueued to this host's own queue instead of the shared one, and the wait-check and scoped-concurrency
41
+ # ceilings become fleet-wide instead of per process. Leave it unset on a single-machine deployment and nothing changes
35
42
  PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may name its own with "image" in triggers.json (docs/job-image.md)
36
43
  # Jobs run with --pull=never: pull or BUILD every image you name -- the worker never fetches one at job time, and doctor checks presence
37
44
  # docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.6.1",
3
+ "version": "1.7.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -58,6 +58,7 @@
58
58
  "./scoped-limits": "./src/scoped-limits.mjs",
59
59
  "./wait-for": "./src/wait-for.mjs",
60
60
  "./wait-state": "./src/wait-state.mjs",
61
+ "./host-registry": "./src/host-registry.mjs",
61
62
  "./identity": "./src/identity.mjs",
62
63
  "./gitlab-identity": "./src/gitlab-identity.mjs",
63
64
  "./forgejo-identity": "./src/forgejo-identity.mjs",
package/src/cli.mjs CHANGED
@@ -6,6 +6,9 @@ import { loadConfig } from "./config.mjs";
6
6
  import { EXIT_POLICY } from "./exit-code.mjs";
7
7
  import { gitDirty } from "./git-dirty.mjs";
8
8
 
9
+ /** How long the kill switch waits on the host registry before acting on the shared queue alone. */
10
+ const FLEET_READ_TIMEOUT_MS = 2_000;
11
+
9
12
  const USAGE = `pi-dispatch — run pi coding-agent flows on your own folders
10
13
 
11
14
  pi-dispatch init scaffold .env + triggers.json + pause-windows.json + pi-packages.json + subscriptions.json here
@@ -120,9 +123,13 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
120
123
 
121
124
  const config = loadConfig(env);
122
125
  const { parseConnection } = await import("./connection.mjs");
123
- const { makeQueue, enqueueLocalJob } = await import("./queue.mjs");
126
+ const { makeQueue, enqueueLocalJob, hostQueueName } = await import("./queue.mjs");
124
127
  // failFast: a one-shot enqueue must not hang forever if Valkey is down -- error clearly.
125
- const queue = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }));
128
+ // Onto THIS host's queue when the deployment declares a name (issue #57). The folder was checked
129
+ // against this machine's filesystem a few lines up, so this machine is the only one that can run it;
130
+ // enqueueing it where every host drains would be handing a job to a peer that has no such folder.
131
+ const hq = config.workerNameDeclared ? hostQueueName(config.workerName) : null;
132
+ const queue = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hq ? { name: hq } : {}) });
126
133
  try {
127
134
  // Absent flags stay absent (undefined) so the value resolves at job start against the
128
135
  // settings overlay/env, not a default frozen here (INT-CONFIG-OVERLAY-CONTRACT).
@@ -150,31 +157,81 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
150
157
  // The kill switch reads ONLY VALKEY_URL, not the full loadConfig -- it must work even when
151
158
  // GitHub auth is misconfigured, so an operator can always stop the queue.
152
159
  const url = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
153
- const { parseConnection } = await import("./connection.mjs");
154
- const { makeQueue } = await import("./queue.mjs");
155
- // failFast: reach the kill switch in seconds when Valkey is down, not a hang. makeQueue pins
156
- // the "pi-jobs" name -- a different name would pause an empty queue (silent no-op).
157
- const queue = makeQueue(parseConnection(url, { failFast: true }));
160
+ const { parseConnection, makeRedisClient } = await import("./connection.mjs");
161
+ const { fleetQueueNames, discoverHostQueues, unionQueueNames, makeQueue } = await import("./queue.mjs");
162
+ const { readLiveHosts } = await import("./host-registry.mjs");
163
+ // EVERY queue this deployment drains (issue #57), not just the shared one. This is the kill switch:
164
+ // pausing `pi-jobs` alone would stop forge deliveries while a named host's cron, chained children
165
+ // and manual runs kept spending -- and would print "paused" for having done it. That is the silent
166
+ // no-op the comment here already warned about for a mistyped name, arriving through a new door.
167
+ //
168
+ // Both reads fail OPEN -- between them an unreadable registry and an unreadable keyspace yield the
169
+ // shared queue alone, which is exactly what this command did before, so a Valkey blip can never make
170
+ // the kill switch refuse. But it fails open LOUDLY: a degraded read is NAMED in the output rather
171
+ // than left indistinguishable from a single-host success while a named host keeps spending.
172
+ // `readLiveHosts` RETURNS `{unreachable}` rather than rejecting, so `blind` is a branch on its
173
+ // value and the `.catch` below is only for a client that throws before it can answer.
174
+ const probe = makeRedisClient(url);
175
+ // Without this, a down Valkey dumps nine `[ioredis] Unhandled error event` traces before the one clean
176
+ // line -- the exact noise `defaultProbeValkey` exists to suppress.
177
+ probe.on?.("error", () => {});
178
+ // Both reads, concurrently, sharing one budget. The registry answers WHO IS LIVE; BullMQ's own meta
179
+ // keys answer WHICH QUEUES EXIST, and for a kill switch the second is the question that matters. A
180
+ // host whose registry writes fail for ninety seconds loses its row while its worker keeps draining,
181
+ // and a resume that misses a queue leaves it paused forever with no surface able to name it. A meta
182
+ // key outlives its worker; a registry row does not.
183
+ const [fleet, existing] = await Promise.all([
184
+ readLiveHosts(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }).catch((error) => ({ unreachable: error?.message ?? String(error) })),
185
+ discoverHostQueues(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }),
186
+ ]);
187
+ probe.disconnect?.();
188
+ const blind = fleet?.unreachable ?? null;
189
+ const names = unionQueueNames(fleetQueueNames(fleet?.hosts), existing);
190
+ // The registry being unreadable no longer means we saw one queue: the keyspace scan may well have
191
+ // found them. Report the count we ACTED on, and name the degraded read separately.
192
+ const span = `${names.length > 1 ? ` [${names.length} queues]` : ""}${blind ? ` [registry unreadable: ${blind}]` : ""}`;
193
+ const queues = [];
158
194
  try {
159
- if (cmd === "pause") {
160
- await queue.pause();
161
- process.stdout.write("paused — worker will stop taking new jobs (jobs still enqueue)\n");
162
- } else if (cmd === "resume") {
163
- await queue.resume();
164
- process.stdout.write("resumed\n");
195
+ // Constructed INSIDE the try: `makeQueue` can throw on a malformed peer-written name, and a throw
196
+ // at index k > 0 would otherwise leak the k connections already opened.
197
+ for (const name of names) queues.push(makeQueue(parseConnection(url, { failFast: true }), { name }));
198
+ if (cmd === "pause" || cmd === "resume") {
199
+ const done = [];
200
+ try {
201
+ for (const q of queues) {
202
+ await (cmd === "pause" ? q.pause() : q.resume());
203
+ done.push(q.name);
204
+ }
205
+ } catch (error) {
206
+ // A mid-loop failure leaves the deployment HALF switched. Naming what did change is the whole
207
+ // difference between an operator who knows to finish the job and one who reads "unreachable"
208
+ // as "nothing happened" and walks away from a fleet with one host still spending.
209
+ return fail(`could not ${cmd} the whole deployment at ${url}\n ${done.length > 0 ? `${cmd}d: ${done.join(", ")}` : "nothing changed"}\n failed at: ${names[done.length]}\n ${error.message}`);
210
+ }
211
+ process.stdout.write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
165
212
  } else {
166
213
  // "paused" is included in the counts because jobs enqueued while paused land in the
167
214
  // `paused` list, not `wait` -- omitting it would report backlog 0 in the exact state
168
215
  // the pause switch creates. `pausedState` (the boolean) is named apart from the
169
216
  // `paused` count `getJobCounts` returns, so the two do not collide in the output.
170
- const pausedState = await queue.isPaused();
171
- const counts = await queue.getJobCounts("waiting", "active", "paused", "delayed", "failed");
172
- process.stdout.write(`${JSON.stringify({ pausedState, ...counts })}\n`);
217
+ const states = await Promise.all(queues.map((q) => q.isPaused()));
218
+ const per = await Promise.all(queues.map((q) => q.getJobCounts("waiting", "active", "paused", "delayed", "failed")));
219
+ const counts = per.reduce((acc, c) => {
220
+ for (const [k, v] of Object.entries(c ?? {})) acc[k] = (acc[k] ?? 0) + (Number(v) || 0);
221
+ return acc;
222
+ }, {});
223
+ // Summed counts with a boolean from ONE queue would report a half-paused deployment as fully
224
+ // one or fully the other. `pausedPartial` is the third state, and the dangerous direction is
225
+ // the one it makes visible: pause ran while a host was invisible, so that host still spends.
226
+ const pausedState = states.every(Boolean);
227
+ const pausedPartial = !pausedState && states.some(Boolean);
228
+ const out = { pausedState, ...(pausedPartial ? { pausedPartial, pausedQueues: names.filter((_, i) => states[i]) } : {}), ...counts, ...(blind ? { fleet: blind } : {}) };
229
+ process.stdout.write(`${JSON.stringify(out)}\n`);
173
230
  }
174
231
  } catch (error) {
175
232
  return fail(`could not reach Valkey at ${url} — is it running? (docker compose up)\n ${error.message}`);
176
233
  } finally {
177
- await queue.close().catch(() => {});
234
+ for (const q of queues) await q.close().catch(() => {});
178
235
  }
179
236
  return 0;
180
237
  }
package/src/config.mjs CHANGED
@@ -6,6 +6,7 @@
6
6
  */
7
7
 
8
8
  import { existsSync } from "node:fs";
9
+ import { hostname } from "node:os";
9
10
  import { delimiter } from "node:path";
10
11
  import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
11
12
  import { MINTED_TOKEN_VARS } from "./forges.mjs";
@@ -212,6 +213,14 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
212
213
  const model = env.PI_MODEL ?? "claude-sonnet-4-5-20250929"; // dated snapshot; deterministic per CONST-PI-VERSION-PINNED
213
214
  return {
214
215
  valkeyUrl: env.VALKEY_URL ?? "redis://127.0.0.1:6379",
216
+ // Issue #57. What this machine calls itself: the key of its registry row, the `host` on every log
217
+ // line and run record, and the BullMQ worker name. Always populated -- a deployment that declares
218
+ // nothing still has an identity, which is what lets a fleet of two be TOLD APART before anyone has
219
+ // configured anything. `workerNameDeclared` is kept separately because "the operator named this
220
+ // machine" and "we read the hostname" are different facts, and a later slice gates a host-visible
221
+ // side effect on the first rather than the second.
222
+ workerName: workerName(env),
223
+ workerNameDeclared: Boolean(env.PI_WORKER_NAME),
215
224
  concurrency: positiveInt(env, "PI_CONCURRENCY", 3), // DES-CONCURRENCY-3
216
225
  dailyCap: positiveInt(env, "PI_DAILY_CAP", 25), // bounds container STARTS per day (money)
217
226
  weeklyCap: optionalBoundedInt(env, "PI_WEEKLY_CAP", 1), // REQ-SPEND-CAPS-MULTI-WINDOW; null = weekly window disabled
@@ -448,6 +457,80 @@ export function defaultLogsDir() {
448
457
  return `${process.env.TMPDIR ?? process.env.TEMP ?? "/tmp"}/pi-dispatch/logs`.replace(/\\/g, "/");
449
458
  }
450
459
 
460
+ /**
461
+ * What a worker may call itself (issue #57). The CHARACTER CLASS is `sanitizeJobId`'s
462
+ * (`[A-Za-z0-9._-]`), reused rather than invented so this project has one name-safe alphabet -- but that
463
+ * function is a REPLACER, not a validator, so the three rules around the class are NEW and are claimed
464
+ * as new here rather than borrowed:
465
+ *
466
+ * - a leading alphanumeric, which is what refuses `..` and a leading `-` that reads as a flag;
467
+ * - a 64-character ceiling, because the name is a Valkey key segment and a log field on every line;
468
+ * - no `.json`/`.log` tail, which is not decoration. The class contains the dot, so `prod.json` is
469
+ * otherwise a legal name -- and a later slice writes a per-host marker file into `PI_LOGS_DIR`,
470
+ * where `<something>.json` is parsed as a run record by the admin and DELETED by the log reaper.
471
+ * A name is refused here rather than escaped there, because the escape would have to be remembered
472
+ * at every site that ever composes a filename from this value.
473
+ *
474
+ * The class is `:`-free, `,`-free and `#`-free, which is what lets the name be a Valkey key segment
475
+ * UNHASHED. That is the point of validating instead of hashing (`scopeKeyPrefix` does the opposite for
476
+ * a folder path, which was never chosen for key-safety and cannot be refused): the whole value of a host
477
+ * registry is that `HGETALL host:h:mac-mini-1` is readable by a human.
478
+ */
479
+ export const WORKER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
480
+
481
+ /** True when the name would collide with the run-history filename namespace. See WORKER_NAME_RE. */
482
+ const RESERVED_NAME_TAIL = /\.(json|log)$/i;
483
+
484
+ /**
485
+ * A hostname reduced to something `WORKER_NAME_RE` accepts, for use as a DEFAULT only.
486
+ *
487
+ * Lowercased because macOS reports `Robs-Mac-Mini.local` where Linux reports `mac-mini`: two spellings
488
+ * of one machine would be two rows in the registry and two values in the run records. The `.local`
489
+ * suffix is deliberately NOT stripped -- an OS-specific suffix rule is a rule someone has to remember,
490
+ * and it costs nothing to keep.
491
+ */
492
+ export function sanitizeWorkerName(raw) {
493
+ const cleaned = String(raw ?? "")
494
+ .toLowerCase()
495
+ .replace(/[^a-z0-9._-]/g, "-")
496
+ .replace(/-{2,}/g, "-")
497
+ .replace(/^[-.]+|[-.]+$/g, "")
498
+ .slice(0, 64)
499
+ .replace(/[-.]+$/, ""); // the slice can leave a trailing separator behind
500
+ if (cleaned === "" || !WORKER_NAME_RE.test(cleaned)) return "worker";
501
+ // The reserved tail is repaired by REPLACING the dot, never by appending: a suffix on a name already at
502
+ // the 64-character ceiling would push it past, and a default that the validator would reject is a
503
+ // second, weaker alphabet arriving by the back door. `host.json` becomes `host-json`, which is the same
504
+ // length, still readable, and cannot match the tail again.
505
+ return cleaned.replace(RESERVED_NAME_TAIL, (m) => `-${m.slice(1)}`);
506
+ }
507
+
508
+ /** This machine's name, sanitized. Exported so doctor and the admin resolve it without `loadConfig`. */
509
+ export function defaultWorkerName() {
510
+ try {
511
+ return sanitizeWorkerName(hostname());
512
+ } catch {
513
+ return "worker"; // hostname() can throw on a locked-down host; a name is never worth refusing boot for
514
+ }
515
+ }
516
+
517
+ /**
518
+ * THE ASYMMETRY IS THE DESIGN. A value the operator did not choose is repaired silently; a value they
519
+ * typed is refused loudly and never quietly altered. Defaulting is a convenience, so it must not be able
520
+ * to fail; declaring is a statement, so a typo in it must not become a different machine's name.
521
+ */
522
+ function workerName(env) {
523
+ const declared = env.PI_WORKER_NAME;
524
+ if (declared === undefined || declared === "") return defaultWorkerName();
525
+ if (!WORKER_NAME_RE.test(declared)) {
526
+ throw configError(`PI_WORKER_NAME must match ${WORKER_NAME_RE.source} (letters, digits, dot, underscore, hyphen; first character alphanumeric; at most 64): ${JSON.stringify(declared)}`);
527
+ }
528
+ if (RESERVED_NAME_TAIL.test(declared)) {
529
+ throw configError(`PI_WORKER_NAME must not end in .json or .log: ${JSON.stringify(declared)} would collide with the run-history filenames in PI_LOGS_DIR`);
530
+ }
531
+ return declared;
532
+ }
533
+
451
534
  export function defaultGraphDir(env = process.env) {
452
535
  // Under the OS temp dir by default, beside logs/ and jobs/ -- the admin's graph HTML artifact
453
536
  // (issue #54) is host-side display output on the defaultLogsDir doctrine, and deliberately NOT
package/src/cron.mjs CHANGED
@@ -14,7 +14,8 @@
14
14
  */
15
15
 
16
16
  import { configError } from "./config.mjs";
17
- import { loadSchedules } from "./schedules.mjs";
17
+ import { cronFingerprint } from "./fingerprint.mjs";
18
+ import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
18
19
 
19
20
  function sentinelName(code) {
20
21
  if (code === -10) return "SchedulerJobIdCollision";
@@ -67,6 +68,99 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
67
68
  return { installed: schedules.length, removed: orphanIds.length };
68
69
  }
69
70
 
71
+ /**
72
+ * Reconcile, but only when every other LIVE worker agrees about what should be scheduled (issue #57).
73
+ *
74
+ * THE BUG THIS CLOSES. `reconcile` prunes every resident scheduler not named in THIS worker's config, so
75
+ * two workers with different triggers files each delete the other's on every boot and every file-watch
76
+ * reload. Its idempotence only ever held because one worker was the only shape, and nothing checked.
77
+ *
78
+ * WHY AGREEMENT RATHER THAN AN ELECTED OWNER. In the good case agreement is sufficient, and this module's
79
+ * own header says why: upsert is keyed by schedulerId, so when every live host agrees it does not matter
80
+ * which one reconciles or how many do. In the BAD case election is actively worse -- an elected owner
81
+ * reconciles from ITS file, so if that file is the stale one (the operator edited on the other host, or a
82
+ * compose `:ro` single-file mount pinned a dead inode, a topology `makeCheckWaitSkew` already documents)
83
+ * the fleet silently converges on the wrong set and reverts the edit with a log line that reads like
84
+ * success. That is `OQ-008`'s own verdict arriving through a new door. Agreement never picks a winner, so
85
+ * it cannot pick the wrong one, and it needs no lease because it grants no authority: the rule only ever
86
+ * WITHHOLDS a permission relative to today, which is why it cannot be a regression.
87
+ *
88
+ * The honest cost, stated rather than buried: agreement can stalemate and needs an operator, where
89
+ * election resolves automatically and possibly wrongly. For a project whose doctrine is "fail loudly, or
90
+ * fail open and say which", a stalemate that names both hosts is the right trade.
91
+ *
92
+ * PUBLISH BEFORE READ is what makes the legitimate-edit sequence race-free. An operator edits on host A;
93
+ * A's watcher fires, A publishes its new fingerprint, reads peers, sees B still on the old one and
94
+ * refuses. The operator syncs the file to B; B's watcher fires, B publishes, reads, sees A already on the
95
+ * new one, and reconciles for the whole fleet. A never has to run again -- the schedule set is global. The
96
+ * only bad interleaving would be both refusing while both are in fact current, which needs a read to see a
97
+ * stale value, and cannot happen when each side publishes synchronously before it reads.
98
+ *
99
+ * ABSENCE NEVER REFUSES. No peers, or a registry that cannot be read, both PROCEED -- which is today's
100
+ * behaviour, so a Valkey blip can never wedge a single-host deployment.
101
+ *
102
+ * BOTH HALVES ARE REFUSED, not just the prune, and the reason is not symmetry: `upsertJobScheduler` on an
103
+ * existing id is a REDEFINITION, so two hosts disagreeing about one id would flip a schedule between two
104
+ * definitions on every file change with nothing logged.
105
+ *
106
+ * A refusal RETURNS and never throws, so the caller logs it and carries on to `worker_started`: a
107
+ * divergent host must still drain the queue. Taking a host's forge capacity offline over a cron
108
+ * disagreement is the sentence this issue's own acceptance forbids.
109
+ */
110
+ export async function reconcileGated(queue, schedules, { registry, log = () => {}, reconcileFn = reconcile, tz, authored = schedules } = {}) {
111
+ // No registry wired is the same answer as a registry that cannot be read: proceed. This is what lets
112
+ // the gate be the DEFAULT on every path without a caller having to remember to arm it.
113
+ if (!registry) return await reconcileFn(queue, schedules, { log });
114
+ // THE FINGERPRINT IS OVER THE AUTHORED SET, NOT THE SERVED SUBSET, and the distinction is the whole
115
+ // reason placement and agreement can coexist. What two hosts must AGREE about is the FILE; what they
116
+ // legitimately DIFFER about is which of its triggers each one can run, because a folder lives on one
117
+ // machine. Hashing the served subset would make every correctly-configured fleet refuse itself forever:
118
+ // mini1 serves /a, mini2 serves /b, their subsets differ, and neither would ever reconcile again.
119
+ const mine = cronFingerprint(authored, { tz });
120
+ // Publishes this host's CURRENT facts rather than passing the fingerprint in. Passing it in was the
121
+ // first shape and it quietly destroyed the mechanism it depends on: the heartbeat installs `fpCron` as
122
+ // a THUNK over the live schedule ref, and a caller merging a computed string replaced that closure, so
123
+ // every later beat republished a frozen value and two hosts could drift apart again with nothing saying
124
+ // so. The thunk is installed once, at boot; this only forces it to be read NOW.
125
+ await registry.publish();
126
+ const peers = await registry.livePeers();
127
+
128
+ // `{ unreachable }` and "no peers" are different facts and the panel must keep them apart -- but for
129
+ // THIS decision they resolve the same way, because not knowing whether anyone disagrees is not knowing
130
+ // that someone does, and the rule only withholds a permission.
131
+ const others = peers?.hosts ?? [];
132
+ // An abstaining peer (cron disabled) publishes no fingerprint and is never a disagreeing party.
133
+ const opinions = others.filter((h) => typeof h.fpCron === "string" && h.fpCron !== "");
134
+ const disagreeing = opinions.filter((h) => h.fpCron !== mine);
135
+
136
+ // I cannot establish agreement with an opinion I do not have. `mine` is null only when the file could
137
+ // not be read or parsed at THIS instant while `loadSchedules` had just succeeded -- a rename's brief
138
+ // unlink window, in practice. Proceeding would prune a peer's schedulers on the strength of a
139
+ // comparison that never happened, so this refuses; refusing deletes nothing and the next watch event
140
+ // or boot re-decides. It gets its own token because "I could not read my own file" and "we disagree"
141
+ // send an operator to two different places.
142
+ if (mine === null && opinions.length > 0) {
143
+ log("cron_divergence_refused", { mine: null, reason: "own-triggers-unreadable", cronCount: schedules.length, peers: opinions.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })) });
144
+ return { refused: "own-triggers-unreadable", peers: opinions.map((h) => h.name) };
145
+ }
146
+
147
+ if (disagreeing.length > 0) {
148
+ log("cron_divergence_refused", {
149
+ mine,
150
+ cronCount: schedules.length,
151
+ // The count rides the LINE and never the RULE: it is what lets the message say "host-b reports 4
152
+ // schedules, I have 5", which is the difference between a diagnosable warning and noise.
153
+ peers: disagreeing.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })),
154
+ });
155
+ return { refused: "cron-divergence", peers: disagreeing.map((h) => h.name) };
156
+ }
157
+
158
+ // Gated on `> 0`, so a single-host deployment emits no new line at all -- the absent-when-unarmed idiom
159
+ // this repo uses for every optional field.
160
+ if (opinions.length > 0) log("cron_agreement", { peers: opinions.length });
161
+ return await reconcileFn(queue, schedules, { log });
162
+ }
163
+
70
164
  /**
71
165
  * Live-reload the cron schedulers from the (changed) triggers file: re-select the cron subset and reconcile
72
166
  * it against the resident schedulers -- an add installs, a delete prunes (reconcile already removes orphans),
@@ -75,16 +169,34 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
75
169
  * never taken down by a malformed trigger file (the OQ-008 live-edit safety). Returns `{ ok }` /
76
170
  * `{ invalid }` / `{ failed }`. `loadFn`/`reconcileFn` are injectable so the reload is unit-tested with no fs.
77
171
  */
78
- export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn = reconcile } = {}) {
172
+ export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn = reconcileGated, ref = null, registry, tz, fleet = false, authoredFn = authoredCron } = {}) {
79
173
  let schedules;
80
174
  try {
81
- schedules = loadFn(config);
175
+ schedules = loadFn(config, { fleet });
82
176
  } catch (error) {
83
177
  log("schedules_reload_invalid", { reason: error?.message ?? String(error), kept: true });
84
178
  return { invalid: error?.message ?? String(error) };
85
179
  }
180
+ // The live REF is updated before the reconcile, not after, and never on the invalid path above: the
181
+ // heartbeat fingerprints what this host currently believes, and believing the boot-time set after an
182
+ // edit is what would make two hosts' fingerprints oscillate on the beat period -- refusing or agreeing
183
+ // depending on which half of a beat a reload landed in.
184
+ if (ref) ref.current = schedules;
185
+ // The same split the boot path makes: a trigger whose folder is another host's is not this host's to
186
+ // install, and the fingerprint is computed over the SERVED set so two hosts owning different folders
187
+ // do not read each other as divergent.
188
+ const { served, unserved } = servedSchedules(schedules);
189
+ for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
86
190
  try {
87
- const r = await reconcileFn(queue, schedules, { log });
191
+ // The FILE, re-read, not `schedules` -- that is `loadSchedules`'s output, which has already replaced
192
+ // every foreign trigger with a stub and therefore differs per host by construction. Passing it here
193
+ // made every live edit on a fleet refuse, permanently, even between hosts running identical files.
194
+ const r = await reconcileFn(queue, served, { log, registry, tz, authored: authoredFn(config) });
195
+ // A refusal is NOT a reload. Wrapping it as `{ ok: true }` would log
196
+ // `schedules_reloaded {installed: undefined}` and tell an operator the edit took effect on a fleet
197
+ // where nothing was installed and nothing pruned -- the silent no-op this project refuses, arriving
198
+ // through the success path.
199
+ if (r?.refused) return r;
88
200
  log("schedules_reloaded", { installed: r.installed, removed: r.removed });
89
201
  return { ok: true, ...r };
90
202
  } catch (error) {
package/src/doctor.mjs CHANGED
@@ -49,7 +49,7 @@ import { homedir, tmpdir } from "node:os";
49
49
  import { dirname, join, delimiter } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
- import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
52
+ import { defaultSandboxDir, defaultWorkerName, globalExtensionsEnabled } from "./config.mjs";
53
53
  import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
54
54
  import { WAIT_AFTER_MAX_DEFAULT_MS, afterInstantMs, parseWaitProfiles } from "./wait-for.mjs";
55
55
  import { isForgeKind } from "./forges.mjs";
@@ -89,6 +89,7 @@ export async function runDoctor(env = process.env, deps = {}) {
89
89
  out = (s) => process.stdout.write(s),
90
90
  spawn = nodeSpawn,
91
91
  probeValkey = defaultProbeValkey,
92
+ readHosts = defaultReadHosts,
92
93
  fileExists = existsSync,
93
94
  nodeVersion = process.versions.node,
94
95
  // --fix (REQ-DEPLOYMENT-BOOTSTRAP): offer to run the exact fixes doctor already prints. The prompt
@@ -111,7 +112,7 @@ export async function runDoctor(env = process.env, deps = {}) {
111
112
  platform = process.platform,
112
113
  home = homedir(),
113
114
  } = deps;
114
- const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
115
+ const seams = { cwd, out, spawn, probeValkey, readHosts, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
115
116
 
116
117
  let checks = await collectChecks(env, seams);
117
118
  let failed = render(checks, out);
@@ -213,7 +214,7 @@ export async function defaultPromptFn(question, { input = process.stdin, output
213
214
  * a comment.
214
215
  */
215
216
  export async function collectChecks(env, seams) {
216
- const { cwd, spawn, probeValkey, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
217
+ const { cwd, spawn, probeValkey, readHosts = defaultReadHosts, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
217
218
 
218
219
  const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
219
220
  const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
@@ -829,6 +830,85 @@ export async function collectChecks(env, seams) {
829
830
  : {}),
830
831
  });
831
832
 
833
+ // --- the fleet (issue #57) -------------------------------------------------------------------------
834
+ //
835
+ // Every line here is gated on a peer actually existing, so a single-host deployment's output is
836
+ // byte-identical. And every one is a WARN rather than a failure, with one exception noted below: this
837
+ // command runs on ONE machine and must not refuse a deployment for a condition that machine cannot fix.
838
+ // This host's own image id, read through the same seam every other docker probe here uses. Only when
839
+ // the image is actually present -- an absent one is already reported above, and a second line saying
840
+ // its digest is unknown would be noise on a fault the operator has been told about.
841
+ const fleet = await readHosts(valkeyUrl);
842
+ const peers = (fleet.hosts ?? []).filter((h) => h.name !== workerNameOf(env));
843
+ // Read only when there is a peer to compare against. Every line below is gated on a peer existing, and
844
+ // the SUBPROCESS has to be too: otherwise every `doctor` run on every single-host deployment spawns an
845
+ // extra docker call whose answer nothing reads.
846
+ const imageDigest =
847
+ peers.length > 0 && imageCode === 0
848
+ ? (await runCmdCapture(spawn, "docker", ["image", "inspect", "--format={{.Id}}", jobImage])).output.trim() || null
849
+ : null;
850
+ if (peers.length > 0) {
851
+ const mine = workerNameOf(env);
852
+ checks.push({ ok: true, label: `Fleet: ${peers.length + 1} worker${peers.length === 0 ? "" : "s"} (${[mine, ...peers.map((h) => h.name)].sort().join(", ")})` });
853
+
854
+ // The one thing that is silently WRONG rather than merely undeclared. Without a declared name this
855
+ // host enqueues its own folder work to the SHARED queue, where a peer that has no such folder can
856
+ // pop it -- so the routing that makes a fleet safe is simply off, and nothing else says so.
857
+ if (!env.PI_WORKER_NAME) {
858
+ checks.push({
859
+ ok: false,
860
+ warn: true,
861
+ label: "This worker has peers but no PI_WORKER_NAME, so host routing is OFF here",
862
+ fix: "set PI_WORKER_NAME in this host's .env and restart: without it, this host's folder work is enqueued where any host can pop it, and its records carry a hostname it never chose",
863
+ });
864
+ }
865
+
866
+ // Two hosts on two builds of one tag is the failure Gap 6 names: same flow, different behaviour,
867
+ // undebuggable. A WARN and never a failure, because `{{.Id}}` is the LOCAL image id -- two
868
+ // independent builds of one Dockerfile differ, and under docker's containerd image store it is the
869
+ // manifest digest rather than the config digest, so a mixed-store fleet disagrees about identical
870
+ // content. Suspicious, never wrong.
871
+ const digests = new Set(peers.map((h) => h.imageDigest).filter(Boolean));
872
+ if (digests.size > 0 && imageDigest && !digests.has(imageDigest)) {
873
+ checks.push({
874
+ ok: false,
875
+ warn: true,
876
+ label: `Job image digest differs from ${peers.length === 1 ? "the other host" : "other hosts"}`,
877
+ fix: "rebuild or re-pull so every host runs the same image; digests are identical only when both hosts pulled one tag from one registry, so two local builds differ legitimately",
878
+ });
879
+ }
880
+
881
+ // A cron PATTERN carries no timezone and resolves in each worker's LOCAL time, so one pattern is two
882
+ // different instants on two hosts in two zones -- and the cron gate refuses that divergence rather
883
+ // than letting it drift, which is why this reads as an explanation for a refusal an operator has
884
+ // probably already met.
885
+ const zones = new Set([Intl.DateTimeFormat().resolvedOptions().timeZone, ...peers.map((h) => h.tz).filter(Boolean)]);
886
+ if (zones.size > 1) {
887
+ checks.push({
888
+ ok: false,
889
+ warn: true,
890
+ label: `Hosts disagree about the timezone (${[...zones].sort().join(", ")}), so one cron pattern is two different instants`,
891
+ fix: "set the same TZ on every host: a cron trigger carries no timezone of its own, so cron reconcile refuses while they disagree",
892
+ });
893
+ }
894
+
895
+ // Clocks. The registry's own heartbeats are the measurement, and skew matters here beyond tidiness:
896
+ // every hold clock, every TTL and the UTC day boundary the budget windows key on are read against
897
+ // whichever host is looking.
898
+ const skewed = peers.filter((h) => Number.isFinite(h.staleMs) && h.staleMs > 5 * 60_000);
899
+ if (skewed.length > 0) {
900
+ checks.push({
901
+ ok: false,
902
+ warn: true,
903
+ label: `${skewed.length} host row${skewed.length === 1 ? " is" : "s are"} stale by more than five minutes (${skewed.map((h) => h.name).join(", ")})`,
904
+ fix: "check that those workers are running and that the clocks agree -- a stale row is either a dead worker or a skewed clock, and both matter",
905
+ });
906
+ }
907
+ } else if (fleet.unreachable) {
908
+ // Said, rather than silently absent: "no peers" and "could not ask" are different facts.
909
+ checks.push({ ok: true, label: `Fleet: could not read the host registry (${fleet.unreachable})` });
910
+ }
911
+
832
912
  const keys = PROVIDER_KEYS[provider] ?? [`${provider.toUpperCase()}_API_KEY`];
833
913
  let keyOk = keys.some((k) => (env[k] ?? "").trim().length > 0);
834
914
  let keyNote = "";
@@ -2212,6 +2292,36 @@ function parseGhTokenScopes(output) {
2212
2292
  * error handler is attached, so a down Valkey is reported as one ✗ line — not the ioredis stack traces
2213
2293
  * a BullMQ Queue's internal client would dump. Reuses `parseConnection`'s fail-fast options (cli.mjs:88).
2214
2294
  */
2295
+ /**
2296
+ * The fleet's registry rows (issue #57), through a fail-fast client that is always disconnected.
2297
+ *
2298
+ * A SEAM rather than a direct import so the multi-host checks are testable with no Valkey at all, which
2299
+ * is the posture every other network-touching check here already takes. Never throws: a fleet this
2300
+ * command cannot see is a fleet it says nothing about, not a doctor that fails.
2301
+ */
2302
+ /** This host's name as the worker computes it, so doctor and the worker cannot disagree about who "I" am. */
2303
+ function workerNameOf(env) {
2304
+ return env.PI_WORKER_NAME || defaultWorkerName();
2305
+ }
2306
+
2307
+ async function defaultReadHosts(url) {
2308
+ try {
2309
+ const { Redis } = await import("ioredis");
2310
+ const { parseConnection } = await import("./connection.mjs");
2311
+ const { readLiveHosts } = await import("./host-registry.mjs");
2312
+ const client = new Redis({ ...parseConnection(url, { failFast: true }), lazyConnect: true });
2313
+ client.on("error", () => {});
2314
+ try {
2315
+ await client.connect();
2316
+ return await readLiveHosts(client);
2317
+ } finally {
2318
+ client.disconnect();
2319
+ }
2320
+ } catch (err) {
2321
+ return { unreachable: err?.message ?? "registry unreadable" };
2322
+ }
2323
+ }
2324
+
2215
2325
  async function defaultProbeValkey(url) {
2216
2326
  const { Redis } = await import("ioredis");
2217
2327
  const { parseConnection } = await import("./connection.mjs");
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Canonical fingerprints of configuration, so two hosts can find out whether they agree (issue #57).
3
+ *
4
+ * Pure, and importing nothing but `node:crypto`: a fingerprint must be computable in a tier-1 test with
5
+ * no queue, no Valkey and no filesystem, exactly as `parseTriggers` is.
6
+ *
7
+ * WHY A HASH RATHER THAN THE VALUE. Two of these travel through the host registry, whose content rule
8
+ * refuses paths outright (`INT-HOST-REGISTRY-CONTRACT`) -- and a cron schedule set legitimately contains
9
+ * `run.folder`, `run.task` and secret NAMES. Hashing is what makes them admissible: the registry carries
10
+ * proof of agreement rather than the thing agreed on. This is the inverse of `scopeKeyPrefix`'s argument,
11
+ * which hashes a scope because it may contain `:` and `/`; here the reason is disclosure, not syntax.
12
+ */
13
+
14
+ import { createHash } from "node:crypto";
15
+
16
+ /**
17
+ * A stable 16-hex digest of any JSON-able value.
18
+ *
19
+ * Object keys are sorted RECURSIVELY, and that is load-bearing rather than tidy. `normalizeCronSchedule`
20
+ * builds its `data` object as a literal, so its key ORDER is a property of the worker's source: two hosts
21
+ * mid-upgrade would otherwise canonicalise the same file differently and refuse each other for the whole
22
+ * rollout. Sorting removes the spurious disagreement while leaving the real one -- a genuinely new field
23
+ * still changes the hash, which is correct, because the stored repeatable's data really did change.
24
+ *
25
+ * Sixteen hex, the `scopeKeyPrefix` and `localJobId` idiom, because this is compared and displayed rather
26
+ * than used as a security boundary.
27
+ */
28
+ export function fingerprint(value) {
29
+ return createHash("sha256").update(canonical(value)).digest("hex").slice(0, 16);
30
+ }
31
+
32
+ function canonical(value) {
33
+ if (value === null || typeof value !== "object") return JSON.stringify(value ?? null);
34
+ if (Array.isArray(value)) return `[${value.map(canonical).join(",")}]`;
35
+ const keys = Object.keys(value).sort();
36
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonical(value[k])}`).join(",")}}`;
37
+ }
38
+
39
+ /**
40
+ * The fingerprint of a worker's cron schedule set, or `null` when this worker has no opinion.
41
+ *
42
+ * ABSTAIN VERSUS OPINE IS THE SUBTLE PART, and conflating the two would leave the bug this gate exists to
43
+ * close. `loadSchedules` returns `[]` for two different states:
44
+ *
45
+ * - `PI_TRIGGERS_FILE` unset, which means CRON IS DISABLED on this host. Such a worker has no view of
46
+ * what should be scheduled, so it must never be able to disagree with one that does. It ABSTAINS.
47
+ * - a triggers file that is present and declares zero cron entries. That is an OPINION -- "there should
48
+ * be no schedulers" -- and it is this bug's purest form: today, deleting the last cron trigger on one
49
+ * host prunes the whole fleet's schedulers through the file-watch path.
50
+ *
51
+ * So `null` in means abstain (`null` out); an empty ARRAY is a real fingerprint.
52
+ *
53
+ * The `tz` rides the hash because a cron PATTERN carries no timezone: `triggers.json` has no `tz` field on
54
+ * a cron entry, and BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's
55
+ * LOCAL system time. On one host that is exactly what an operator means; on two hosts in different zones
56
+ * the same pattern is two different instants, with nothing anywhere saying so. Including the zone makes
57
+ * that a disagreement the gate can see. (`pause-windows.json` has carried an explicit `tz` since it
58
+ * shipped and is already fleet-correct; the asymmetry is why this one needs stating.)
59
+ *
60
+ * Fingerprinted over the NORMALIZED schedules rather than the file's bytes, deliberately. Bytes diverge on
61
+ * whitespace, on key order, and on every webhook trigger the worker does not own -- so two hosts differing
62
+ * only in a `label` rule would freeze cron forever over a difference that cannot affect it. The normalized
63
+ * set diverges exactly when the reconcile INPUTS diverge, which is the property that makes reconcile
64
+ * idempotent in the first place.
65
+ */
66
+ export function cronFingerprint(schedules, { tz } = {}) {
67
+ // Takes the AUTHORED shape (`schedules.mjs -> authoredCron`), never the placement-resolved one.
68
+ if (schedules === null || schedules === undefined) return null;
69
+ return fingerprint({
70
+ tz: tz ?? "",
71
+ // EVERY field of an authored entry, projected explicitly. An earlier shape listed the keys of a
72
+ // NORMALIZED schedule (`name`, `data`, `opts`), which are all `undefined` on an authored one -- so
73
+ // the hash saw only the id and the pattern, and an operator changing `run.folder`, `run.flow` or
74
+ // `run.image` on one host and not the other passed the gate silently. That broke this module's own
75
+ // stated invariant, that the set diverges exactly when the reconcile inputs diverge.
76
+ schedules: schedules.map((s) => ({ schedulerId: s.schedulerId, pattern: s.pattern, run: s.run })),
77
+ });
78
+ }