@edgehero/pi-dispatch 1.6.1 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +7 -0
- package/package.json +2 -1
- package/src/cli.mjs +74 -17
- package/src/config.mjs +83 -0
- package/src/cron.mjs +116 -4
- package/src/doctor.mjs +113 -3
- package/src/fingerprint.mjs +78 -0
- package/src/fleet-lease.mjs +179 -0
- package/src/host-registry.mjs +279 -0
- package/src/image-preflight.mjs +8 -3
- package/src/index.mjs +196 -51
- package/src/queue.mjs +130 -2
- package/src/run-history.mjs +23 -1
- package/src/schedules.mjs +67 -4
- package/src/start.mjs +217 -27
package/.env.example
CHANGED
|
@@ -32,6 +32,13 @@ PI_CONCURRENCY=3 # how many jobs run in parallel
|
|
|
32
32
|
|
|
33
33
|
# --- Infrastructure ---
|
|
34
34
|
VALKEY_URL=redis://127.0.0.1:6379
|
|
35
|
+
# PI_WORKER_NAME= # what this machine calls itself. Default: your hostname, lowercased and reduced to [A-Za-z0-9._-]
|
|
36
|
+
# Lands on every worker log line and in every run record, and identifies this host to the others when you run more than one
|
|
37
|
+
# Set it if your hostname is something you would rather not have in your own run history (a laptop often carries a person's name)
|
|
38
|
+
# Refused at boot if it is not [A-Za-z0-9._-], does not start with a letter or digit, is over 64 characters, or ends in .json or .log
|
|
39
|
+
# SETTING IT TURNS ON MULTI-HOST ROUTING (docs/multi-host.md): work only this machine can do (a cron folder, a chained child,
|
|
40
|
+
# a manual run) is enqueued to this host's own queue instead of the shared one, and the wait-check and scoped-concurrency
|
|
41
|
+
# ceilings become fleet-wide instead of per process. Leave it unset on a single-machine deployment and nothing changes
|
|
35
42
|
PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may name its own with "image" in triggers.json (docs/job-image.md)
|
|
36
43
|
# Jobs run with --pull=never: pull or BUILD every image you name -- the worker never fetches one at job time, and doctor checks presence
|
|
37
44
|
# docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.7.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
|
@@ -58,6 +58,7 @@
|
|
|
58
58
|
"./scoped-limits": "./src/scoped-limits.mjs",
|
|
59
59
|
"./wait-for": "./src/wait-for.mjs",
|
|
60
60
|
"./wait-state": "./src/wait-state.mjs",
|
|
61
|
+
"./host-registry": "./src/host-registry.mjs",
|
|
61
62
|
"./identity": "./src/identity.mjs",
|
|
62
63
|
"./gitlab-identity": "./src/gitlab-identity.mjs",
|
|
63
64
|
"./forgejo-identity": "./src/forgejo-identity.mjs",
|
package/src/cli.mjs
CHANGED
|
@@ -6,6 +6,9 @@ import { loadConfig } from "./config.mjs";
|
|
|
6
6
|
import { EXIT_POLICY } from "./exit-code.mjs";
|
|
7
7
|
import { gitDirty } from "./git-dirty.mjs";
|
|
8
8
|
|
|
9
|
+
/** How long the kill switch waits on the host registry before acting on the shared queue alone. */
|
|
10
|
+
const FLEET_READ_TIMEOUT_MS = 2_000;
|
|
11
|
+
|
|
9
12
|
const USAGE = `pi-dispatch — run pi coding-agent flows on your own folders
|
|
10
13
|
|
|
11
14
|
pi-dispatch init scaffold .env + triggers.json + pause-windows.json + pi-packages.json + subscriptions.json here
|
|
@@ -120,9 +123,13 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
|
|
|
120
123
|
|
|
121
124
|
const config = loadConfig(env);
|
|
122
125
|
const { parseConnection } = await import("./connection.mjs");
|
|
123
|
-
const { makeQueue, enqueueLocalJob } = await import("./queue.mjs");
|
|
126
|
+
const { makeQueue, enqueueLocalJob, hostQueueName } = await import("./queue.mjs");
|
|
124
127
|
// failFast: a one-shot enqueue must not hang forever if Valkey is down -- error clearly.
|
|
125
|
-
|
|
128
|
+
// Onto THIS host's queue when the deployment declares a name (issue #57). The folder was checked
|
|
129
|
+
// against this machine's filesystem a few lines up, so this machine is the only one that can run it;
|
|
130
|
+
// enqueueing it where every host drains would be handing a job to a peer that has no such folder.
|
|
131
|
+
const hq = config.workerNameDeclared ? hostQueueName(config.workerName) : null;
|
|
132
|
+
const queue = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hq ? { name: hq } : {}) });
|
|
126
133
|
try {
|
|
127
134
|
// Absent flags stay absent (undefined) so the value resolves at job start against the
|
|
128
135
|
// settings overlay/env, not a default frozen here (INT-CONFIG-OVERLAY-CONTRACT).
|
|
@@ -150,31 +157,81 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
|
|
|
150
157
|
// The kill switch reads ONLY VALKEY_URL, not the full loadConfig -- it must work even when
|
|
151
158
|
// GitHub auth is misconfigured, so an operator can always stop the queue.
|
|
152
159
|
const url = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
|
|
153
|
-
const { parseConnection } = await import("./connection.mjs");
|
|
154
|
-
const { makeQueue } = await import("./queue.mjs");
|
|
155
|
-
|
|
156
|
-
//
|
|
157
|
-
|
|
160
|
+
const { parseConnection, makeRedisClient } = await import("./connection.mjs");
|
|
161
|
+
const { fleetQueueNames, discoverHostQueues, unionQueueNames, makeQueue } = await import("./queue.mjs");
|
|
162
|
+
const { readLiveHosts } = await import("./host-registry.mjs");
|
|
163
|
+
// EVERY queue this deployment drains (issue #57), not just the shared one. This is the kill switch:
|
|
164
|
+
// pausing `pi-jobs` alone would stop forge deliveries while a named host's cron, chained children
|
|
165
|
+
// and manual runs kept spending -- and would print "paused" for having done it. That is the silent
|
|
166
|
+
// no-op the comment here already warned about for a mistyped name, arriving through a new door.
|
|
167
|
+
//
|
|
168
|
+
// Both reads fail OPEN -- between them an unreadable registry and an unreadable keyspace yield the
|
|
169
|
+
// shared queue alone, which is exactly what this command did before, so a Valkey blip can never make
|
|
170
|
+
// the kill switch refuse. But it fails open LOUDLY: a degraded read is NAMED in the output rather
|
|
171
|
+
// than left indistinguishable from a single-host success while a named host keeps spending.
|
|
172
|
+
// `readLiveHosts` RETURNS `{unreachable}` rather than rejecting, so `blind` is a branch on its
|
|
173
|
+
// value and the `.catch` below is only for a client that throws before it can answer.
|
|
174
|
+
const probe = makeRedisClient(url);
|
|
175
|
+
// Without this, a down Valkey dumps nine `[ioredis] Unhandled error event` traces before the one clean
|
|
176
|
+
// line -- the exact noise `defaultProbeValkey` exists to suppress.
|
|
177
|
+
probe.on?.("error", () => {});
|
|
178
|
+
// Both reads, concurrently, sharing one budget. The registry answers WHO IS LIVE; BullMQ's own meta
|
|
179
|
+
// keys answer WHICH QUEUES EXIST, and for a kill switch the second is the question that matters. A
|
|
180
|
+
// host whose registry writes fail for ninety seconds loses its row while its worker keeps draining,
|
|
181
|
+
// and a resume that misses a queue leaves it paused forever with no surface able to name it. A meta
|
|
182
|
+
// key outlives its worker; a registry row does not.
|
|
183
|
+
const [fleet, existing] = await Promise.all([
|
|
184
|
+
readLiveHosts(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }).catch((error) => ({ unreachable: error?.message ?? String(error) })),
|
|
185
|
+
discoverHostQueues(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }),
|
|
186
|
+
]);
|
|
187
|
+
probe.disconnect?.();
|
|
188
|
+
const blind = fleet?.unreachable ?? null;
|
|
189
|
+
const names = unionQueueNames(fleetQueueNames(fleet?.hosts), existing);
|
|
190
|
+
// The registry being unreadable no longer means we saw one queue: the keyspace scan may well have
|
|
191
|
+
// found them. Report the count we ACTED on, and name the degraded read separately.
|
|
192
|
+
const span = `${names.length > 1 ? ` [${names.length} queues]` : ""}${blind ? ` [registry unreadable: ${blind}]` : ""}`;
|
|
193
|
+
const queues = [];
|
|
158
194
|
try {
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
195
|
+
// Constructed INSIDE the try: `makeQueue` can throw on a malformed peer-written name, and a throw
|
|
196
|
+
// at index k > 0 would otherwise leak the k connections already opened.
|
|
197
|
+
for (const name of names) queues.push(makeQueue(parseConnection(url, { failFast: true }), { name }));
|
|
198
|
+
if (cmd === "pause" || cmd === "resume") {
|
|
199
|
+
const done = [];
|
|
200
|
+
try {
|
|
201
|
+
for (const q of queues) {
|
|
202
|
+
await (cmd === "pause" ? q.pause() : q.resume());
|
|
203
|
+
done.push(q.name);
|
|
204
|
+
}
|
|
205
|
+
} catch (error) {
|
|
206
|
+
// A mid-loop failure leaves the deployment HALF switched. Naming what did change is the whole
|
|
207
|
+
// difference between an operator who knows to finish the job and one who reads "unreachable"
|
|
208
|
+
// as "nothing happened" and walks away from a fleet with one host still spending.
|
|
209
|
+
return fail(`could not ${cmd} the whole deployment at ${url}\n ${done.length > 0 ? `${cmd}d: ${done.join(", ")}` : "nothing changed"}\n failed at: ${names[done.length]}\n ${error.message}`);
|
|
210
|
+
}
|
|
211
|
+
process.stdout.write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
|
|
165
212
|
} else {
|
|
166
213
|
// "paused" is included in the counts because jobs enqueued while paused land in the
|
|
167
214
|
// `paused` list, not `wait` -- omitting it would report backlog 0 in the exact state
|
|
168
215
|
// the pause switch creates. `pausedState` (the boolean) is named apart from the
|
|
169
216
|
// `paused` count `getJobCounts` returns, so the two do not collide in the output.
|
|
170
|
-
const
|
|
171
|
-
const
|
|
172
|
-
|
|
217
|
+
const states = await Promise.all(queues.map((q) => q.isPaused()));
|
|
218
|
+
const per = await Promise.all(queues.map((q) => q.getJobCounts("waiting", "active", "paused", "delayed", "failed")));
|
|
219
|
+
const counts = per.reduce((acc, c) => {
|
|
220
|
+
for (const [k, v] of Object.entries(c ?? {})) acc[k] = (acc[k] ?? 0) + (Number(v) || 0);
|
|
221
|
+
return acc;
|
|
222
|
+
}, {});
|
|
223
|
+
// Summed counts with a boolean from ONE queue would report a half-paused deployment as fully
|
|
224
|
+
// one or fully the other. `pausedPartial` is the third state, and the dangerous direction is
|
|
225
|
+
// the one it makes visible: pause ran while a host was invisible, so that host still spends.
|
|
226
|
+
const pausedState = states.every(Boolean);
|
|
227
|
+
const pausedPartial = !pausedState && states.some(Boolean);
|
|
228
|
+
const out = { pausedState, ...(pausedPartial ? { pausedPartial, pausedQueues: names.filter((_, i) => states[i]) } : {}), ...counts, ...(blind ? { fleet: blind } : {}) };
|
|
229
|
+
process.stdout.write(`${JSON.stringify(out)}\n`);
|
|
173
230
|
}
|
|
174
231
|
} catch (error) {
|
|
175
232
|
return fail(`could not reach Valkey at ${url} — is it running? (docker compose up)\n ${error.message}`);
|
|
176
233
|
} finally {
|
|
177
|
-
await
|
|
234
|
+
for (const q of queues) await q.close().catch(() => {});
|
|
178
235
|
}
|
|
179
236
|
return 0;
|
|
180
237
|
}
|
package/src/config.mjs
CHANGED
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import { existsSync } from "node:fs";
|
|
9
|
+
import { hostname } from "node:os";
|
|
9
10
|
import { delimiter } from "node:path";
|
|
10
11
|
import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
|
|
11
12
|
import { MINTED_TOKEN_VARS } from "./forges.mjs";
|
|
@@ -212,6 +213,14 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
|
|
|
212
213
|
const model = env.PI_MODEL ?? "claude-sonnet-4-5-20250929"; // dated snapshot; deterministic per CONST-PI-VERSION-PINNED
|
|
213
214
|
return {
|
|
214
215
|
valkeyUrl: env.VALKEY_URL ?? "redis://127.0.0.1:6379",
|
|
216
|
+
// Issue #57. What this machine calls itself: the key of its registry row, the `host` on every log
|
|
217
|
+
// line and run record, and the BullMQ worker name. Always populated -- a deployment that declares
|
|
218
|
+
// nothing still has an identity, which is what lets a fleet of two be TOLD APART before anyone has
|
|
219
|
+
// configured anything. `workerNameDeclared` is kept separately because "the operator named this
|
|
220
|
+
// machine" and "we read the hostname" are different facts, and a later slice gates a host-visible
|
|
221
|
+
// side effect on the first rather than the second.
|
|
222
|
+
workerName: workerName(env),
|
|
223
|
+
workerNameDeclared: Boolean(env.PI_WORKER_NAME),
|
|
215
224
|
concurrency: positiveInt(env, "PI_CONCURRENCY", 3), // DES-CONCURRENCY-3
|
|
216
225
|
dailyCap: positiveInt(env, "PI_DAILY_CAP", 25), // bounds container STARTS per day (money)
|
|
217
226
|
weeklyCap: optionalBoundedInt(env, "PI_WEEKLY_CAP", 1), // REQ-SPEND-CAPS-MULTI-WINDOW; null = weekly window disabled
|
|
@@ -448,6 +457,80 @@ export function defaultLogsDir() {
|
|
|
448
457
|
return `${process.env.TMPDIR ?? process.env.TEMP ?? "/tmp"}/pi-dispatch/logs`.replace(/\\/g, "/");
|
|
449
458
|
}
|
|
450
459
|
|
|
460
|
+
/**
|
|
461
|
+
* What a worker may call itself (issue #57). The CHARACTER CLASS is `sanitizeJobId`'s
|
|
462
|
+
* (`[A-Za-z0-9._-]`), reused rather than invented so this project has one name-safe alphabet -- but that
|
|
463
|
+
* function is a REPLACER, not a validator, so the three rules around the class are NEW and are claimed
|
|
464
|
+
* as new here rather than borrowed:
|
|
465
|
+
*
|
|
466
|
+
* - a leading alphanumeric, which is what refuses `..` and a leading `-` that reads as a flag;
|
|
467
|
+
* - a 64-character ceiling, because the name is a Valkey key segment and a log field on every line;
|
|
468
|
+
* - no `.json`/`.log` tail, which is not decoration. The class contains the dot, so `prod.json` is
|
|
469
|
+
* otherwise a legal name -- and a later slice writes a per-host marker file into `PI_LOGS_DIR`,
|
|
470
|
+
* where `<something>.json` is parsed as a run record by the admin and DELETED by the log reaper.
|
|
471
|
+
* A name is refused here rather than escaped there, because the escape would have to be remembered
|
|
472
|
+
* at every site that ever composes a filename from this value.
|
|
473
|
+
*
|
|
474
|
+
* The class is `:`-free, `,`-free and `#`-free, which is what lets the name be a Valkey key segment
|
|
475
|
+
* UNHASHED. That is the point of validating instead of hashing (`scopeKeyPrefix` does the opposite for
|
|
476
|
+
* a folder path, which was never chosen for key-safety and cannot be refused): the whole value of a host
|
|
477
|
+
* registry is that `HGETALL host:h:mac-mini-1` is readable by a human.
|
|
478
|
+
*/
|
|
479
|
+
export const WORKER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
|
|
480
|
+
|
|
481
|
+
/** True when the name would collide with the run-history filename namespace. See WORKER_NAME_RE. */
|
|
482
|
+
const RESERVED_NAME_TAIL = /\.(json|log)$/i;
|
|
483
|
+
|
|
484
|
+
/**
|
|
485
|
+
* A hostname reduced to something `WORKER_NAME_RE` accepts, for use as a DEFAULT only.
|
|
486
|
+
*
|
|
487
|
+
* Lowercased because macOS reports `Robs-Mac-Mini.local` where Linux reports `mac-mini`: two spellings
|
|
488
|
+
* of one machine would be two rows in the registry and two values in the run records. The `.local`
|
|
489
|
+
* suffix is deliberately NOT stripped -- an OS-specific suffix rule is a rule someone has to remember,
|
|
490
|
+
* and it costs nothing to keep.
|
|
491
|
+
*/
|
|
492
|
+
export function sanitizeWorkerName(raw) {
|
|
493
|
+
const cleaned = String(raw ?? "")
|
|
494
|
+
.toLowerCase()
|
|
495
|
+
.replace(/[^a-z0-9._-]/g, "-")
|
|
496
|
+
.replace(/-{2,}/g, "-")
|
|
497
|
+
.replace(/^[-.]+|[-.]+$/g, "")
|
|
498
|
+
.slice(0, 64)
|
|
499
|
+
.replace(/[-.]+$/, ""); // the slice can leave a trailing separator behind
|
|
500
|
+
if (cleaned === "" || !WORKER_NAME_RE.test(cleaned)) return "worker";
|
|
501
|
+
// The reserved tail is repaired by REPLACING the dot, never by appending: a suffix on a name already at
|
|
502
|
+
// the 64-character ceiling would push it past, and a default that the validator would reject is a
|
|
503
|
+
// second, weaker alphabet arriving by the back door. `host.json` becomes `host-json`, which is the same
|
|
504
|
+
// length, still readable, and cannot match the tail again.
|
|
505
|
+
return cleaned.replace(RESERVED_NAME_TAIL, (m) => `-${m.slice(1)}`);
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
/** This machine's name, sanitized. Exported so doctor and the admin resolve it without `loadConfig`. */
|
|
509
|
+
export function defaultWorkerName() {
|
|
510
|
+
try {
|
|
511
|
+
return sanitizeWorkerName(hostname());
|
|
512
|
+
} catch {
|
|
513
|
+
return "worker"; // hostname() can throw on a locked-down host; a name is never worth refusing boot for
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
/**
|
|
518
|
+
* THE ASYMMETRY IS THE DESIGN. A value the operator did not choose is repaired silently; a value they
|
|
519
|
+
* typed is refused loudly and never quietly altered. Defaulting is a convenience, so it must not be able
|
|
520
|
+
* to fail; declaring is a statement, so a typo in it must not become a different machine's name.
|
|
521
|
+
*/
|
|
522
|
+
function workerName(env) {
|
|
523
|
+
const declared = env.PI_WORKER_NAME;
|
|
524
|
+
if (declared === undefined || declared === "") return defaultWorkerName();
|
|
525
|
+
if (!WORKER_NAME_RE.test(declared)) {
|
|
526
|
+
throw configError(`PI_WORKER_NAME must match ${WORKER_NAME_RE.source} (letters, digits, dot, underscore, hyphen; first character alphanumeric; at most 64): ${JSON.stringify(declared)}`);
|
|
527
|
+
}
|
|
528
|
+
if (RESERVED_NAME_TAIL.test(declared)) {
|
|
529
|
+
throw configError(`PI_WORKER_NAME must not end in .json or .log: ${JSON.stringify(declared)} would collide with the run-history filenames in PI_LOGS_DIR`);
|
|
530
|
+
}
|
|
531
|
+
return declared;
|
|
532
|
+
}
|
|
533
|
+
|
|
451
534
|
export function defaultGraphDir(env = process.env) {
|
|
452
535
|
// Under the OS temp dir by default, beside logs/ and jobs/ -- the admin's graph HTML artifact
|
|
453
536
|
// (issue #54) is host-side display output on the defaultLogsDir doctrine, and deliberately NOT
|
package/src/cron.mjs
CHANGED
|
@@ -14,7 +14,8 @@
|
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
16
|
import { configError } from "./config.mjs";
|
|
17
|
-
import {
|
|
17
|
+
import { cronFingerprint } from "./fingerprint.mjs";
|
|
18
|
+
import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
|
|
18
19
|
|
|
19
20
|
function sentinelName(code) {
|
|
20
21
|
if (code === -10) return "SchedulerJobIdCollision";
|
|
@@ -67,6 +68,99 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
|
|
|
67
68
|
return { installed: schedules.length, removed: orphanIds.length };
|
|
68
69
|
}
|
|
69
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Reconcile, but only when every other LIVE worker agrees about what should be scheduled (issue #57).
|
|
73
|
+
*
|
|
74
|
+
* THE BUG THIS CLOSES. `reconcile` prunes every resident scheduler not named in THIS worker's config, so
|
|
75
|
+
* two workers with different triggers files each delete the other's on every boot and every file-watch
|
|
76
|
+
* reload. Its idempotence only ever held because one worker was the only shape, and nothing checked.
|
|
77
|
+
*
|
|
78
|
+
* WHY AGREEMENT RATHER THAN AN ELECTED OWNER. In the good case agreement is sufficient, and this module's
|
|
79
|
+
* own header says why: upsert is keyed by schedulerId, so when every live host agrees it does not matter
|
|
80
|
+
* which one reconciles or how many do. In the BAD case election is actively worse -- an elected owner
|
|
81
|
+
* reconciles from ITS file, so if that file is the stale one (the operator edited on the other host, or a
|
|
82
|
+
* compose `:ro` single-file mount pinned a dead inode, a topology `makeCheckWaitSkew` already documents)
|
|
83
|
+
* the fleet silently converges on the wrong set and reverts the edit with a log line that reads like
|
|
84
|
+
* success. That is `OQ-008`'s own verdict arriving through a new door. Agreement never picks a winner, so
|
|
85
|
+
* it cannot pick the wrong one, and it needs no lease because it grants no authority: the rule only ever
|
|
86
|
+
* WITHHOLDS a permission relative to today, which is why it cannot be a regression.
|
|
87
|
+
*
|
|
88
|
+
* The honest cost, stated rather than buried: agreement can stalemate and needs an operator, where
|
|
89
|
+
* election resolves automatically and possibly wrongly. For a project whose doctrine is "fail loudly, or
|
|
90
|
+
* fail open and say which", a stalemate that names both hosts is the right trade.
|
|
91
|
+
*
|
|
92
|
+
* PUBLISH BEFORE READ is what makes the legitimate-edit sequence race-free. An operator edits on host A;
|
|
93
|
+
* A's watcher fires, A publishes its new fingerprint, reads peers, sees B still on the old one and
|
|
94
|
+
* refuses. The operator syncs the file to B; B's watcher fires, B publishes, reads, sees A already on the
|
|
95
|
+
* new one, and reconciles for the whole fleet. A never has to run again -- the schedule set is global. The
|
|
96
|
+
* only bad interleaving would be both refusing while both are in fact current, which needs a read to see a
|
|
97
|
+
* stale value, and cannot happen when each side publishes synchronously before it reads.
|
|
98
|
+
*
|
|
99
|
+
* ABSENCE NEVER REFUSES. No peers, or a registry that cannot be read, both PROCEED -- which is today's
|
|
100
|
+
* behaviour, so a Valkey blip can never wedge a single-host deployment.
|
|
101
|
+
*
|
|
102
|
+
* BOTH HALVES ARE REFUSED, not just the prune, and the reason is not symmetry: `upsertJobScheduler` on an
|
|
103
|
+
* existing id is a REDEFINITION, so two hosts disagreeing about one id would flip a schedule between two
|
|
104
|
+
* definitions on every file change with nothing logged.
|
|
105
|
+
*
|
|
106
|
+
* A refusal RETURNS and never throws, so the caller logs it and carries on to `worker_started`: a
|
|
107
|
+
* divergent host must still drain the queue. Taking a host's forge capacity offline over a cron
|
|
108
|
+
* disagreement is the sentence this issue's own acceptance forbids.
|
|
109
|
+
*/
|
|
110
|
+
export async function reconcileGated(queue, schedules, { registry, log = () => {}, reconcileFn = reconcile, tz, authored = schedules } = {}) {
|
|
111
|
+
// No registry wired is the same answer as a registry that cannot be read: proceed. This is what lets
|
|
112
|
+
// the gate be the DEFAULT on every path without a caller having to remember to arm it.
|
|
113
|
+
if (!registry) return await reconcileFn(queue, schedules, { log });
|
|
114
|
+
// THE FINGERPRINT IS OVER THE AUTHORED SET, NOT THE SERVED SUBSET, and the distinction is the whole
|
|
115
|
+
// reason placement and agreement can coexist. What two hosts must AGREE about is the FILE; what they
|
|
116
|
+
// legitimately DIFFER about is which of its triggers each one can run, because a folder lives on one
|
|
117
|
+
// machine. Hashing the served subset would make every correctly-configured fleet refuse itself forever:
|
|
118
|
+
// mini1 serves /a, mini2 serves /b, their subsets differ, and neither would ever reconcile again.
|
|
119
|
+
const mine = cronFingerprint(authored, { tz });
|
|
120
|
+
// Publishes this host's CURRENT facts rather than passing the fingerprint in. Passing it in was the
|
|
121
|
+
// first shape and it quietly destroyed the mechanism it depends on: the heartbeat installs `fpCron` as
|
|
122
|
+
// a THUNK over the live schedule ref, and a caller merging a computed string replaced that closure, so
|
|
123
|
+
// every later beat republished a frozen value and two hosts could drift apart again with nothing saying
|
|
124
|
+
// so. The thunk is installed once, at boot; this only forces it to be read NOW.
|
|
125
|
+
await registry.publish();
|
|
126
|
+
const peers = await registry.livePeers();
|
|
127
|
+
|
|
128
|
+
// `{ unreachable }` and "no peers" are different facts and the panel must keep them apart -- but for
|
|
129
|
+
// THIS decision they resolve the same way, because not knowing whether anyone disagrees is not knowing
|
|
130
|
+
// that someone does, and the rule only withholds a permission.
|
|
131
|
+
const others = peers?.hosts ?? [];
|
|
132
|
+
// An abstaining peer (cron disabled) publishes no fingerprint and is never a disagreeing party.
|
|
133
|
+
const opinions = others.filter((h) => typeof h.fpCron === "string" && h.fpCron !== "");
|
|
134
|
+
const disagreeing = opinions.filter((h) => h.fpCron !== mine);
|
|
135
|
+
|
|
136
|
+
// I cannot establish agreement with an opinion I do not have. `mine` is null only when the file could
|
|
137
|
+
// not be read or parsed at THIS instant while `loadSchedules` had just succeeded -- a rename's brief
|
|
138
|
+
// unlink window, in practice. Proceeding would prune a peer's schedulers on the strength of a
|
|
139
|
+
// comparison that never happened, so this refuses; refusing deletes nothing and the next watch event
|
|
140
|
+
// or boot re-decides. It gets its own token because "I could not read my own file" and "we disagree"
|
|
141
|
+
// send an operator to two different places.
|
|
142
|
+
if (mine === null && opinions.length > 0) {
|
|
143
|
+
log("cron_divergence_refused", { mine: null, reason: "own-triggers-unreadable", cronCount: schedules.length, peers: opinions.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })) });
|
|
144
|
+
return { refused: "own-triggers-unreadable", peers: opinions.map((h) => h.name) };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
if (disagreeing.length > 0) {
|
|
148
|
+
log("cron_divergence_refused", {
|
|
149
|
+
mine,
|
|
150
|
+
cronCount: schedules.length,
|
|
151
|
+
// The count rides the LINE and never the RULE: it is what lets the message say "host-b reports 4
|
|
152
|
+
// schedules, I have 5", which is the difference between a diagnosable warning and noise.
|
|
153
|
+
peers: disagreeing.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })),
|
|
154
|
+
});
|
|
155
|
+
return { refused: "cron-divergence", peers: disagreeing.map((h) => h.name) };
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// Gated on `> 0`, so a single-host deployment emits no new line at all -- the absent-when-unarmed idiom
|
|
159
|
+
// this repo uses for every optional field.
|
|
160
|
+
if (opinions.length > 0) log("cron_agreement", { peers: opinions.length });
|
|
161
|
+
return await reconcileFn(queue, schedules, { log });
|
|
162
|
+
}
|
|
163
|
+
|
|
70
164
|
/**
|
|
71
165
|
* Live-reload the cron schedulers from the (changed) triggers file: re-select the cron subset and reconcile
|
|
72
166
|
* it against the resident schedulers -- an add installs, a delete prunes (reconcile already removes orphans),
|
|
@@ -75,16 +169,34 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
|
|
|
75
169
|
* never taken down by a malformed trigger file (the OQ-008 live-edit safety). Returns `{ ok }` /
|
|
76
170
|
* `{ invalid }` / `{ failed }`. `loadFn`/`reconcileFn` are injectable so the reload is unit-tested with no fs.
|
|
77
171
|
*/
|
|
78
|
-
export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn =
|
|
172
|
+
export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn = reconcileGated, ref = null, registry, tz, fleet = false, authoredFn = authoredCron } = {}) {
|
|
79
173
|
let schedules;
|
|
80
174
|
try {
|
|
81
|
-
schedules = loadFn(config);
|
|
175
|
+
schedules = loadFn(config, { fleet });
|
|
82
176
|
} catch (error) {
|
|
83
177
|
log("schedules_reload_invalid", { reason: error?.message ?? String(error), kept: true });
|
|
84
178
|
return { invalid: error?.message ?? String(error) };
|
|
85
179
|
}
|
|
180
|
+
// The live REF is updated before the reconcile, not after, and never on the invalid path above: the
|
|
181
|
+
// heartbeat fingerprints what this host currently believes, and believing the boot-time set after an
|
|
182
|
+
// edit is what would make two hosts' fingerprints oscillate on the beat period -- refusing or agreeing
|
|
183
|
+
// depending on which half of a beat a reload landed in.
|
|
184
|
+
if (ref) ref.current = schedules;
|
|
185
|
+
// The same split the boot path makes: a trigger whose folder is another host's is not this host's to
|
|
186
|
+
// install, and the fingerprint is computed over the SERVED set so two hosts owning different folders
|
|
187
|
+
// do not read each other as divergent.
|
|
188
|
+
const { served, unserved } = servedSchedules(schedules);
|
|
189
|
+
for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
|
|
86
190
|
try {
|
|
87
|
-
|
|
191
|
+
// The FILE, re-read, not `schedules` -- that is `loadSchedules`'s output, which has already replaced
|
|
192
|
+
// every foreign trigger with a stub and therefore differs per host by construction. Passing it here
|
|
193
|
+
// made every live edit on a fleet refuse, permanently, even between hosts running identical files.
|
|
194
|
+
const r = await reconcileFn(queue, served, { log, registry, tz, authored: authoredFn(config) });
|
|
195
|
+
// A refusal is NOT a reload. Wrapping it as `{ ok: true }` would log
|
|
196
|
+
// `schedules_reloaded {installed: undefined}` and tell an operator the edit took effect on a fleet
|
|
197
|
+
// where nothing was installed and nothing pruned -- the silent no-op this project refuses, arriving
|
|
198
|
+
// through the success path.
|
|
199
|
+
if (r?.refused) return r;
|
|
88
200
|
log("schedules_reloaded", { installed: r.installed, removed: r.removed });
|
|
89
201
|
return { ok: true, ...r };
|
|
90
202
|
} catch (error) {
|
package/src/doctor.mjs
CHANGED
|
@@ -49,7 +49,7 @@ import { homedir, tmpdir } from "node:os";
|
|
|
49
49
|
import { dirname, join, delimiter } from "node:path";
|
|
50
50
|
import { fileURLToPath } from "node:url";
|
|
51
51
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
52
|
-
import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
|
|
52
|
+
import { defaultSandboxDir, defaultWorkerName, globalExtensionsEnabled } from "./config.mjs";
|
|
53
53
|
import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
|
|
54
54
|
import { WAIT_AFTER_MAX_DEFAULT_MS, afterInstantMs, parseWaitProfiles } from "./wait-for.mjs";
|
|
55
55
|
import { isForgeKind } from "./forges.mjs";
|
|
@@ -89,6 +89,7 @@ export async function runDoctor(env = process.env, deps = {}) {
|
|
|
89
89
|
out = (s) => process.stdout.write(s),
|
|
90
90
|
spawn = nodeSpawn,
|
|
91
91
|
probeValkey = defaultProbeValkey,
|
|
92
|
+
readHosts = defaultReadHosts,
|
|
92
93
|
fileExists = existsSync,
|
|
93
94
|
nodeVersion = process.versions.node,
|
|
94
95
|
// --fix (REQ-DEPLOYMENT-BOOTSTRAP): offer to run the exact fixes doctor already prints. The prompt
|
|
@@ -111,7 +112,7 @@ export async function runDoctor(env = process.env, deps = {}) {
|
|
|
111
112
|
platform = process.platform,
|
|
112
113
|
home = homedir(),
|
|
113
114
|
} = deps;
|
|
114
|
-
const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
|
|
115
|
+
const seams = { cwd, out, spawn, probeValkey, readHosts, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
|
|
115
116
|
|
|
116
117
|
let checks = await collectChecks(env, seams);
|
|
117
118
|
let failed = render(checks, out);
|
|
@@ -213,7 +214,7 @@ export async function defaultPromptFn(question, { input = process.stdin, output
|
|
|
213
214
|
* a comment.
|
|
214
215
|
*/
|
|
215
216
|
export async function collectChecks(env, seams) {
|
|
216
|
-
const { cwd, spawn, probeValkey, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
|
|
217
|
+
const { cwd, spawn, probeValkey, readHosts = defaultReadHosts, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
|
|
217
218
|
|
|
218
219
|
const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
|
|
219
220
|
const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
|
|
@@ -829,6 +830,85 @@ export async function collectChecks(env, seams) {
|
|
|
829
830
|
: {}),
|
|
830
831
|
});
|
|
831
832
|
|
|
833
|
+
// --- the fleet (issue #57) -------------------------------------------------------------------------
|
|
834
|
+
//
|
|
835
|
+
// Every line here is gated on a peer actually existing, so a single-host deployment's output is
|
|
836
|
+
// byte-identical. And every one is a WARN rather than a failure, with one exception noted below: this
|
|
837
|
+
// command runs on ONE machine and must not refuse a deployment for a condition that machine cannot fix.
|
|
838
|
+
// This host's own image id, read through the same seam every other docker probe here uses. Only when
|
|
839
|
+
// the image is actually present -- an absent one is already reported above, and a second line saying
|
|
840
|
+
// its digest is unknown would be noise on a fault the operator has been told about.
|
|
841
|
+
const fleet = await readHosts(valkeyUrl);
|
|
842
|
+
const peers = (fleet.hosts ?? []).filter((h) => h.name !== workerNameOf(env));
|
|
843
|
+
// Read only when there is a peer to compare against. Every line below is gated on a peer existing, and
|
|
844
|
+
// the SUBPROCESS has to be too: otherwise every `doctor` run on every single-host deployment spawns an
|
|
845
|
+
// extra docker call whose answer nothing reads.
|
|
846
|
+
const imageDigest =
|
|
847
|
+
peers.length > 0 && imageCode === 0
|
|
848
|
+
? (await runCmdCapture(spawn, "docker", ["image", "inspect", "--format={{.Id}}", jobImage])).output.trim() || null
|
|
849
|
+
: null;
|
|
850
|
+
if (peers.length > 0) {
|
|
851
|
+
const mine = workerNameOf(env);
|
|
852
|
+
checks.push({ ok: true, label: `Fleet: ${peers.length + 1} worker${peers.length === 0 ? "" : "s"} (${[mine, ...peers.map((h) => h.name)].sort().join(", ")})` });
|
|
853
|
+
|
|
854
|
+
// The one thing that is silently WRONG rather than merely undeclared. Without a declared name this
|
|
855
|
+
// host enqueues its own folder work to the SHARED queue, where a peer that has no such folder can
|
|
856
|
+
// pop it -- so the routing that makes a fleet safe is simply off, and nothing else says so.
|
|
857
|
+
if (!env.PI_WORKER_NAME) {
|
|
858
|
+
checks.push({
|
|
859
|
+
ok: false,
|
|
860
|
+
warn: true,
|
|
861
|
+
label: "This worker has peers but no PI_WORKER_NAME, so host routing is OFF here",
|
|
862
|
+
fix: "set PI_WORKER_NAME in this host's .env and restart: without it, this host's folder work is enqueued where any host can pop it, and its records carry a hostname it never chose",
|
|
863
|
+
});
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
// Two hosts on two builds of one tag is the failure Gap 6 names: same flow, different behaviour,
|
|
867
|
+
// undebuggable. A WARN and never a failure, because `{{.Id}}` is the LOCAL image id -- two
|
|
868
|
+
// independent builds of one Dockerfile differ, and under docker's containerd image store it is the
|
|
869
|
+
// manifest digest rather than the config digest, so a mixed-store fleet disagrees about identical
|
|
870
|
+
// content. Suspicious, never wrong.
|
|
871
|
+
const digests = new Set(peers.map((h) => h.imageDigest).filter(Boolean));
|
|
872
|
+
if (digests.size > 0 && imageDigest && !digests.has(imageDigest)) {
|
|
873
|
+
checks.push({
|
|
874
|
+
ok: false,
|
|
875
|
+
warn: true,
|
|
876
|
+
label: `Job image digest differs from ${peers.length === 1 ? "the other host" : "other hosts"}`,
|
|
877
|
+
fix: "rebuild or re-pull so every host runs the same image; digests are identical only when both hosts pulled one tag from one registry, so two local builds differ legitimately",
|
|
878
|
+
});
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
// A cron PATTERN carries no timezone and resolves in each worker's LOCAL time, so one pattern is two
|
|
882
|
+
// different instants on two hosts in two zones -- and the cron gate refuses that divergence rather
|
|
883
|
+
// than letting it drift, which is why this reads as an explanation for a refusal an operator has
|
|
884
|
+
// probably already met.
|
|
885
|
+
const zones = new Set([Intl.DateTimeFormat().resolvedOptions().timeZone, ...peers.map((h) => h.tz).filter(Boolean)]);
|
|
886
|
+
if (zones.size > 1) {
|
|
887
|
+
checks.push({
|
|
888
|
+
ok: false,
|
|
889
|
+
warn: true,
|
|
890
|
+
label: `Hosts disagree about the timezone (${[...zones].sort().join(", ")}), so one cron pattern is two different instants`,
|
|
891
|
+
fix: "set the same TZ on every host: a cron trigger carries no timezone of its own, so cron reconcile refuses while they disagree",
|
|
892
|
+
});
|
|
893
|
+
}
|
|
894
|
+
|
|
895
|
+
// Clocks. The registry's own heartbeats are the measurement, and skew matters here beyond tidiness:
|
|
896
|
+
// every hold clock, every TTL and the UTC day boundary the budget windows key on are read against
|
|
897
|
+
// whichever host is looking.
|
|
898
|
+
const skewed = peers.filter((h) => Number.isFinite(h.staleMs) && h.staleMs > 5 * 60_000);
|
|
899
|
+
if (skewed.length > 0) {
|
|
900
|
+
checks.push({
|
|
901
|
+
ok: false,
|
|
902
|
+
warn: true,
|
|
903
|
+
label: `${skewed.length} host row${skewed.length === 1 ? " is" : "s are"} stale by more than five minutes (${skewed.map((h) => h.name).join(", ")})`,
|
|
904
|
+
fix: "check that those workers are running and that the clocks agree -- a stale row is either a dead worker or a skewed clock, and both matter",
|
|
905
|
+
});
|
|
906
|
+
}
|
|
907
|
+
} else if (fleet.unreachable) {
|
|
908
|
+
// Said, rather than silently absent: "no peers" and "could not ask" are different facts.
|
|
909
|
+
checks.push({ ok: true, label: `Fleet: could not read the host registry (${fleet.unreachable})` });
|
|
910
|
+
}
|
|
911
|
+
|
|
832
912
|
const keys = PROVIDER_KEYS[provider] ?? [`${provider.toUpperCase()}_API_KEY`];
|
|
833
913
|
let keyOk = keys.some((k) => (env[k] ?? "").trim().length > 0);
|
|
834
914
|
let keyNote = "";
|
|
@@ -2212,6 +2292,36 @@ function parseGhTokenScopes(output) {
|
|
|
2212
2292
|
* error handler is attached, so a down Valkey is reported as one ✗ line — not the ioredis stack traces
|
|
2213
2293
|
* a BullMQ Queue's internal client would dump. Reuses `parseConnection`'s fail-fast options (cli.mjs:88).
|
|
2214
2294
|
*/
|
|
2295
|
+
/**
|
|
2296
|
+
* The fleet's registry rows (issue #57), through a fail-fast client that is always disconnected.
|
|
2297
|
+
*
|
|
2298
|
+
* A SEAM rather than a direct import so the multi-host checks are testable with no Valkey at all, which
|
|
2299
|
+
* is the posture every other network-touching check here already takes. Never throws: a fleet this
|
|
2300
|
+
* command cannot see is a fleet it says nothing about, not a doctor that fails.
|
|
2301
|
+
*/
|
|
2302
|
+
/** This host's name as the worker computes it, so doctor and the worker cannot disagree about who "I" am. */
|
|
2303
|
+
function workerNameOf(env) {
|
|
2304
|
+
return env.PI_WORKER_NAME || defaultWorkerName();
|
|
2305
|
+
}
|
|
2306
|
+
|
|
2307
|
+
async function defaultReadHosts(url) {
|
|
2308
|
+
try {
|
|
2309
|
+
const { Redis } = await import("ioredis");
|
|
2310
|
+
const { parseConnection } = await import("./connection.mjs");
|
|
2311
|
+
const { readLiveHosts } = await import("./host-registry.mjs");
|
|
2312
|
+
const client = new Redis({ ...parseConnection(url, { failFast: true }), lazyConnect: true });
|
|
2313
|
+
client.on("error", () => {});
|
|
2314
|
+
try {
|
|
2315
|
+
await client.connect();
|
|
2316
|
+
return await readLiveHosts(client);
|
|
2317
|
+
} finally {
|
|
2318
|
+
client.disconnect();
|
|
2319
|
+
}
|
|
2320
|
+
} catch (err) {
|
|
2321
|
+
return { unreachable: err?.message ?? "registry unreadable" };
|
|
2322
|
+
}
|
|
2323
|
+
}
|
|
2324
|
+
|
|
2215
2325
|
async function defaultProbeValkey(url) {
|
|
2216
2326
|
const { Redis } = await import("ioredis");
|
|
2217
2327
|
const { parseConnection } = await import("./connection.mjs");
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical fingerprints of configuration, so two hosts can find out whether they agree (issue #57).
|
|
3
|
+
*
|
|
4
|
+
* Pure, and importing nothing but `node:crypto`: a fingerprint must be computable in a tier-1 test with
|
|
5
|
+
* no queue, no Valkey and no filesystem, exactly as `parseTriggers` is.
|
|
6
|
+
*
|
|
7
|
+
* WHY A HASH RATHER THAN THE VALUE. Two of these travel through the host registry, whose content rule
|
|
8
|
+
* refuses paths outright (`INT-HOST-REGISTRY-CONTRACT`) -- and a cron schedule set legitimately contains
|
|
9
|
+
* `run.folder`, `run.task` and secret NAMES. Hashing is what makes them admissible: the registry carries
|
|
10
|
+
* proof of agreement rather than the thing agreed on. This is the inverse of `scopeKeyPrefix`'s argument,
|
|
11
|
+
* which hashes a scope because it may contain `:` and `/`; here the reason is disclosure, not syntax.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { createHash } from "node:crypto";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* A stable 16-hex digest of any JSON-able value.
|
|
18
|
+
*
|
|
19
|
+
* Object keys are sorted RECURSIVELY, and that is load-bearing rather than tidy. `normalizeCronSchedule`
|
|
20
|
+
* builds its `data` object as a literal, so its key ORDER is a property of the worker's source: two hosts
|
|
21
|
+
* mid-upgrade would otherwise canonicalise the same file differently and refuse each other for the whole
|
|
22
|
+
* rollout. Sorting removes the spurious disagreement while leaving the real one -- a genuinely new field
|
|
23
|
+
* still changes the hash, which is correct, because the stored repeatable's data really did change.
|
|
24
|
+
*
|
|
25
|
+
* Sixteen hex, the `scopeKeyPrefix` and `localJobId` idiom, because this is compared and displayed rather
|
|
26
|
+
* than used as a security boundary.
|
|
27
|
+
*/
|
|
28
|
+
export function fingerprint(value) {
|
|
29
|
+
return createHash("sha256").update(canonical(value)).digest("hex").slice(0, 16);
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function canonical(value) {
|
|
33
|
+
if (value === null || typeof value !== "object") return JSON.stringify(value ?? null);
|
|
34
|
+
if (Array.isArray(value)) return `[${value.map(canonical).join(",")}]`;
|
|
35
|
+
const keys = Object.keys(value).sort();
|
|
36
|
+
return `{${keys.map((k) => `${JSON.stringify(k)}:${canonical(value[k])}`).join(",")}}`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* The fingerprint of a worker's cron schedule set, or `null` when this worker has no opinion.
|
|
41
|
+
*
|
|
42
|
+
* ABSTAIN VERSUS OPINE IS THE SUBTLE PART, and conflating the two would leave the bug this gate exists to
|
|
43
|
+
* close. `loadSchedules` returns `[]` for two different states:
|
|
44
|
+
*
|
|
45
|
+
* - `PI_TRIGGERS_FILE` unset, which means CRON IS DISABLED on this host. Such a worker has no view of
|
|
46
|
+
* what should be scheduled, so it must never be able to disagree with one that does. It ABSTAINS.
|
|
47
|
+
* - a triggers file that is present and declares zero cron entries. That is an OPINION -- "there should
|
|
48
|
+
* be no schedulers" -- and it is this bug's purest form: today, deleting the last cron trigger on one
|
|
49
|
+
* host prunes the whole fleet's schedulers through the file-watch path.
|
|
50
|
+
*
|
|
51
|
+
* So `null` in means abstain (`null` out); an empty ARRAY is a real fingerprint.
|
|
52
|
+
*
|
|
53
|
+
* The `tz` rides the hash because a cron PATTERN carries no timezone: `triggers.json` has no `tz` field on
|
|
54
|
+
* a cron entry, and BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's
|
|
55
|
+
* LOCAL system time. On one host that is exactly what an operator means; on two hosts in different zones
|
|
56
|
+
* the same pattern is two different instants, with nothing anywhere saying so. Including the zone makes
|
|
57
|
+
* that a disagreement the gate can see. (`pause-windows.json` has carried an explicit `tz` since it
|
|
58
|
+
* shipped and is already fleet-correct; the asymmetry is why this one needs stating.)
|
|
59
|
+
*
|
|
60
|
+
* Fingerprinted over the NORMALIZED schedules rather than the file's bytes, deliberately. Bytes diverge on
|
|
61
|
+
* whitespace, on key order, and on every webhook trigger the worker does not own -- so two hosts differing
|
|
62
|
+
* only in a `label` rule would freeze cron forever over a difference that cannot affect it. The normalized
|
|
63
|
+
* set diverges exactly when the reconcile INPUTS diverge, which is the property that makes reconcile
|
|
64
|
+
* idempotent in the first place.
|
|
65
|
+
*/
|
|
66
|
+
export function cronFingerprint(schedules, { tz } = {}) {
|
|
67
|
+
// Takes the AUTHORED shape (`schedules.mjs -> authoredCron`), never the placement-resolved one.
|
|
68
|
+
if (schedules === null || schedules === undefined) return null;
|
|
69
|
+
return fingerprint({
|
|
70
|
+
tz: tz ?? "",
|
|
71
|
+
// EVERY field of an authored entry, projected explicitly. An earlier shape listed the keys of a
|
|
72
|
+
// NORMALIZED schedule (`name`, `data`, `opts`), which are all `undefined` on an authored one -- so
|
|
73
|
+
// the hash saw only the id and the pattern, and an operator changing `run.folder`, `run.flow` or
|
|
74
|
+
// `run.image` on one host and not the other passed the gate silently. That broke this module's own
|
|
75
|
+
// stated invariant, that the set diverges exactly when the reconcile inputs diverge.
|
|
76
|
+
schedules: schedules.map((s) => ({ schedulerId: s.schedulerId, pattern: s.pattern, run: s.run })),
|
|
77
|
+
});
|
|
78
|
+
}
|