@edgehero/pi-dispatch 1.6.1 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +7 -0
- package/package.json +4 -1
- package/src/capabilities.mjs +179 -0
- package/src/cli.mjs +74 -17
- package/src/config.mjs +83 -0
- package/src/cron.mjs +116 -4
- package/src/doctor.mjs +113 -3
- package/src/fingerprint.mjs +78 -0
- package/src/fleet-lease.mjs +179 -0
- package/src/host-registry.mjs +279 -0
- package/src/image-preflight.mjs +8 -3
- package/src/index.mjs +196 -51
- package/src/queue.mjs +130 -2
- package/src/run-history.mjs +23 -1
- package/src/run-mirror.mjs +221 -0
- package/src/schedules.mjs +67 -4
- package/src/start.mjs +240 -28
package/.env.example
CHANGED
|
@@ -32,6 +32,13 @@ PI_CONCURRENCY=3 # how many jobs run in parallel
|
|
|
32
32
|
|
|
33
33
|
# --- Infrastructure ---
|
|
34
34
|
VALKEY_URL=redis://127.0.0.1:6379
|
|
35
|
+
# PI_WORKER_NAME= # what this machine calls itself. Default: your hostname, lowercased and reduced to [A-Za-z0-9._-]
|
|
36
|
+
# Lands on every worker log line and in every run record, and identifies this host to the others when you run more than one
|
|
37
|
+
# Set it if your hostname is something you would rather not have in your own run history (a laptop often carries a person's name)
|
|
38
|
+
# Refused at boot if it is not [A-Za-z0-9._-], does not start with a letter or digit, is over 64 characters, or ends in .json or .log
|
|
39
|
+
# SETTING IT TURNS ON MULTI-HOST ROUTING (docs/multi-host.md): work only this machine can do (a cron folder, a chained child,
|
|
40
|
+
# a manual run) is enqueued to this host's own queue instead of the shared one, and the wait-check and scoped-concurrency
|
|
41
|
+
# ceilings become fleet-wide instead of per process. Leave it unset on a single-machine deployment and nothing changes
|
|
35
42
|
PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may name its own with "image" in triggers.json (docs/job-image.md)
|
|
36
43
|
# Jobs run with --pull=never: pull or BUILD every image you name -- the worker never fetches one at job time, and doctor checks presence
|
|
37
44
|
# docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.8.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
|
@@ -48,6 +48,7 @@
|
|
|
48
48
|
"./open-browser": "./src/open-browser.mjs",
|
|
49
49
|
"./git-dirty": "./src/git-dirty.mjs",
|
|
50
50
|
"./queue": "./src/queue.mjs",
|
|
51
|
+
"./capabilities": "./src/capabilities.mjs",
|
|
51
52
|
"./connection": "./src/connection.mjs",
|
|
52
53
|
"./job-id": "./src/job-id.mjs",
|
|
53
54
|
"./forges": "./src/forges.mjs",
|
|
@@ -58,6 +59,7 @@
|
|
|
58
59
|
"./scoped-limits": "./src/scoped-limits.mjs",
|
|
59
60
|
"./wait-for": "./src/wait-for.mjs",
|
|
60
61
|
"./wait-state": "./src/wait-state.mjs",
|
|
62
|
+
"./host-registry": "./src/host-registry.mjs",
|
|
61
63
|
"./identity": "./src/identity.mjs",
|
|
62
64
|
"./gitlab-identity": "./src/gitlab-identity.mjs",
|
|
63
65
|
"./forgejo-identity": "./src/forgejo-identity.mjs",
|
|
@@ -65,6 +67,7 @@
|
|
|
65
67
|
"./get-token": "./src/get-token.mjs",
|
|
66
68
|
"./runtime-settings": "./src/runtime-settings.mjs",
|
|
67
69
|
"./run-history": "./src/run-history.mjs",
|
|
70
|
+
"./run-mirror": "./src/run-mirror.mjs",
|
|
68
71
|
"./sandbox": "./src/sandbox.mjs",
|
|
69
72
|
"./sandbox-store": "./src/sandbox-store.mjs",
|
|
70
73
|
"./subscriptions": "./src/subscriptions.mjs",
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a host can serve, and which queue a job that needs one of those things belongs on (issue #57,
|
|
3
|
+
* `OQ-032`).
|
|
4
|
+
*
|
|
5
|
+
* Two trigger fields bind a job to a MACHINE rather than to a repository: `run.secretsProfile` names a
|
|
6
|
+
* resolver the operator declared in that host's `PI_SECRET_PROFILES`, and a `run.waitFor` condition names a
|
|
7
|
+
* check script from its `PI_WAIT_PROFILES`. Both are refused pre-spend when the host popping the job has
|
|
8
|
+
* not declared them, and both refusals are RETURNED rather than thrown, so they are never retried. The
|
|
9
|
+
* same trigger therefore succeeds or fails depending on which worker happened to take the delivery, and it
|
|
10
|
+
* reads like a configuration error rather than a placement one.
|
|
11
|
+
*
|
|
12
|
+
* #57's Gap 2 exempted forge jobs on the grounds that their workspace is a fresh clone that any host can
|
|
13
|
+
* build. Issues #225 and #230 retracted that without saying so: a clone is portable, a resolver on one
|
|
14
|
+
* machine's disk is not.
|
|
15
|
+
*
|
|
16
|
+
* WHY THIS IS DECIDED AT ENQUEUE. The obvious alternative is to let any host take the job and defer it if
|
|
17
|
+
* it cannot serve it. That does not work here, and the reason is upstream rather than ours: BullMQ promotes
|
|
18
|
+
* a delayed job on EACH WORKER'S OWN CLOCK (`scripts.js` passes the client clock as the cut-off), so the
|
|
19
|
+
* host whose clock runs fastest wins every attempt, deterministically. If the host that cannot serve the
|
|
20
|
+
* job is the fast one, the job never reaches the one that can. Jitter changes when the attempt happens, not
|
|
21
|
+
* who wins it.
|
|
22
|
+
*
|
|
23
|
+
* THE ONE CAPABILITY DELIBERATELY NOT ROUTED is `run.resume`. A session key is `sha256(kind, repo, ref)`
|
|
24
|
+
* and `session-store.mjs` records that it is "not random... anyone who knows the repository and the branch
|
|
25
|
+
* can compute it", so publishing keys to route on them would disclose which repositories and branches a
|
|
26
|
+
* deployment works on, recoverable by guessing a repo name. That is more disclosing than everything else in
|
|
27
|
+
* the registry combined. A resume that lands on the wrong host cold-starts and says so in the record.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
/** Class letters. Open enum: `g:` is reserved for forge credentials, the highest-value follow-on. */
|
|
31
|
+
export const CAP_SECRET = "s";
|
|
32
|
+
export const CAP_WAIT = "w";
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* The name charset, duplicated from `secret-profiles.mjs` and `wait-for.mjs` deliberately: this module
|
|
36
|
+
* imports nothing, and the two it copies already copy it from `triggers.mjs` for the same reason. What
|
|
37
|
+
* matters here is what the set EXCLUDES -- a comma, so the token list joins unambiguously, and a colon, so
|
|
38
|
+
* `<class>:<name>` decomposes at the first one.
|
|
39
|
+
*/
|
|
40
|
+
const PROFILE_NAME = /^[A-Za-z0-9._-]+$/;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* How fresh a host's registry row must be before a job is ROUTED to it.
|
|
44
|
+
*
|
|
45
|
+
* Not the 90s TTL, and the difference is the point. The TTL is a crash backstop: it answers "has this host
|
|
46
|
+
* definitely gone", and it is deliberately six missed beats so a blip cannot evict a working host from the
|
|
47
|
+
* panel. A routing decision needs the opposite polarity -- evidence of LIFE, not absence of expiry --
|
|
48
|
+
* because a job routed onto a dead host's queue sits there until that host comes back, and nothing else
|
|
49
|
+
* will take it. Three beats is late enough to ride out a slow beat and early enough that a stopped host
|
|
50
|
+
* stops attracting work long before its row expires.
|
|
51
|
+
*/
|
|
52
|
+
export const ROUTE_FRESH_MS = 45_000;
|
|
53
|
+
|
|
54
|
+
/** One token, or null when the name is not one this deployment would accept. */
|
|
55
|
+
function token(cls, name) {
|
|
56
|
+
return typeof name === "string" && PROFILE_NAME.test(name) ? `${cls}:${name}` : null;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* What THIS host can serve, from its own config, as a sorted token list.
|
|
61
|
+
*
|
|
62
|
+
* Sorted so the published string is stable: an unstable one would make the row differ every beat and any
|
|
63
|
+
* future fingerprint over it useless.
|
|
64
|
+
*/
|
|
65
|
+
export function capabilityTokens({ secretProfiles, waitProfiles } = {}) {
|
|
66
|
+
const out = new Set();
|
|
67
|
+
for (const name of Object.keys(secretProfiles ?? {})) {
|
|
68
|
+
const t = token(CAP_SECRET, name);
|
|
69
|
+
if (t) out.add(t);
|
|
70
|
+
}
|
|
71
|
+
for (const name of Object.keys(waitProfiles ?? {})) {
|
|
72
|
+
const t = token(CAP_WAIT, name);
|
|
73
|
+
if (t) out.add(t);
|
|
74
|
+
}
|
|
75
|
+
return [...out].sort();
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** The registry value. Comma-joined, which the charset makes unambiguous. */
|
|
79
|
+
export function serializeCaps(tokens) {
|
|
80
|
+
return (tokens ?? []).join(",");
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* A peer's tokens, from its registry row. Peer-written, so every element is re-validated here rather than
|
|
85
|
+
* trusted: a row is written by another process and this one decides where money-spending work goes.
|
|
86
|
+
*/
|
|
87
|
+
export function parseCaps(raw) {
|
|
88
|
+
const out = new Set();
|
|
89
|
+
for (const part of String(raw ?? "").split(",")) {
|
|
90
|
+
const at = part.indexOf(":");
|
|
91
|
+
if (at <= 0) continue;
|
|
92
|
+
const cls = part.slice(0, at);
|
|
93
|
+
const name = part.slice(at + 1);
|
|
94
|
+
if ((cls === CAP_SECRET || cls === CAP_WAIT) && PROFILE_NAME.test(name)) out.add(part);
|
|
95
|
+
}
|
|
96
|
+
return out;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* What a JOB needs, from the same two fields the worker's own refusals read.
|
|
101
|
+
*
|
|
102
|
+
* `waitProfileNames`-shaped inline rather than imported, because this module is loaded by the RECEIVER and
|
|
103
|
+
* `wait-for.mjs` carries the whole wait engine. The extraction is three lines and the charset check below
|
|
104
|
+
* is the same one that module makes.
|
|
105
|
+
*/
|
|
106
|
+
export function jobNeeds(job) {
|
|
107
|
+
const needs = new Set();
|
|
108
|
+
const secret = token(CAP_SECRET, job?.secretsProfile);
|
|
109
|
+
if (secret) needs.add(secret);
|
|
110
|
+
if (Array.isArray(job?.waitFor)) {
|
|
111
|
+
for (const condition of job.waitFor) {
|
|
112
|
+
const wait = token(CAP_WAIT, condition?.profile);
|
|
113
|
+
if (wait) needs.add(wait);
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
return [...needs].sort();
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Which queue this job belongs on: a host queue name, or `null` for the shared queue.
|
|
121
|
+
*
|
|
122
|
+
* FOUR ABSTENTIONS, and each one lands on today's behaviour rather than on something new. That is what
|
|
123
|
+
* makes this safe to put in front of every forge delivery: the rule can only ever move a job that would
|
|
124
|
+
* otherwise have had a coin flip decide whether it ran.
|
|
125
|
+
*
|
|
126
|
+
* 1. The job needs nothing host-specific. The overwhelming majority of deliveries.
|
|
127
|
+
* 2. EVERY live host can serve it. The shared queue is then strictly better than picking one, because it
|
|
128
|
+
* load-balances, and it is what the docs tell operators to aim for by declaring the same profiles
|
|
129
|
+
* everywhere. A deployment that follows that advice is byte-identical to before.
|
|
130
|
+
* 3. NO host can serve it. Routing cannot help, and the shared queue produces the existing pre-spend
|
|
131
|
+
* refusal (`secret-profile-unknown` / `wait-profile-unknown`), which is the honest answer and already
|
|
132
|
+
* names what to fix. Inventing a new terminal state here would be worse than the one that exists.
|
|
133
|
+
* 4. No CAPABLE host has a queue of its own, or the registry could not be read. Nothing to route to.
|
|
134
|
+
*
|
|
135
|
+
* Otherwise the job goes to a capable host, chosen by hashing its jobId: deterministic, so a redelivery of
|
|
136
|
+
* the same job lands the same way and dedup still works, and spread, so a delivery fanned out into replicas
|
|
137
|
+
* does not pile every replica onto one machine.
|
|
138
|
+
*/
|
|
139
|
+
export function routeForgeJob({ hosts, needs, jobId, now = () => Date.now(), freshMs = ROUTE_FRESH_MS } = {}) {
|
|
140
|
+
if (!Array.isArray(needs) || needs.length === 0) return null; // (1)
|
|
141
|
+
if (!Array.isArray(hosts) || hosts.length === 0) return null; // (4) unreadable registry, or nobody home
|
|
142
|
+
|
|
143
|
+
const live = hosts.filter((h) => {
|
|
144
|
+
// `staleMs` is derived by the reader; a row without one is a row we cannot date, and an undatable
|
|
145
|
+
// row is not evidence of life.
|
|
146
|
+
const stale = Number(h?.staleMs);
|
|
147
|
+
return Number.isFinite(stale) && stale <= freshMs;
|
|
148
|
+
});
|
|
149
|
+
if (live.length === 0) return null;
|
|
150
|
+
|
|
151
|
+
const serves = (h) => {
|
|
152
|
+
const caps = parseCaps(h?.caps);
|
|
153
|
+
return needs.every((n) => caps.has(n));
|
|
154
|
+
};
|
|
155
|
+
const capable = live.filter(serves);
|
|
156
|
+
if (capable.length === 0) return null; // (3)
|
|
157
|
+
if (capable.length === live.length) return null; // (2)
|
|
158
|
+
|
|
159
|
+
// Only a host that DECLARED a name drains a queue of its own; an undeclared one reads the shared queue
|
|
160
|
+
// only, so routing to it would be routing into a queue nothing drains.
|
|
161
|
+
const routable = capable.filter((h) => h?.routes === true || h?.routes === "true").map((h) => h?.name).filter((n) => typeof n === "string" && n !== "");
|
|
162
|
+
if (routable.length === 0) return null; // (4)
|
|
163
|
+
|
|
164
|
+
routable.sort();
|
|
165
|
+
return routable[hashIndex(String(jobId ?? ""), routable.length)];
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* A stable index from a string. FNV-1a, inline: this module imports nothing, and a cryptographic hash would
|
|
170
|
+
* be a strange dependency for choosing between two machines.
|
|
171
|
+
*/
|
|
172
|
+
function hashIndex(text, modulo) {
|
|
173
|
+
let h = 0x811c9dc5;
|
|
174
|
+
for (let i = 0; i < text.length; i++) {
|
|
175
|
+
h ^= text.charCodeAt(i);
|
|
176
|
+
h = Math.imul(h, 0x01000193) >>> 0;
|
|
177
|
+
}
|
|
178
|
+
return h % modulo;
|
|
179
|
+
}
|
package/src/cli.mjs
CHANGED
|
@@ -6,6 +6,9 @@ import { loadConfig } from "./config.mjs";
|
|
|
6
6
|
import { EXIT_POLICY } from "./exit-code.mjs";
|
|
7
7
|
import { gitDirty } from "./git-dirty.mjs";
|
|
8
8
|
|
|
9
|
+
/** How long the kill switch waits on the host registry before acting on the shared queue alone. */
|
|
10
|
+
const FLEET_READ_TIMEOUT_MS = 2_000;
|
|
11
|
+
|
|
9
12
|
const USAGE = `pi-dispatch — run pi coding-agent flows on your own folders
|
|
10
13
|
|
|
11
14
|
pi-dispatch init scaffold .env + triggers.json + pause-windows.json + pi-packages.json + subscriptions.json here
|
|
@@ -120,9 +123,13 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
|
|
|
120
123
|
|
|
121
124
|
const config = loadConfig(env);
|
|
122
125
|
const { parseConnection } = await import("./connection.mjs");
|
|
123
|
-
const { makeQueue, enqueueLocalJob } = await import("./queue.mjs");
|
|
126
|
+
const { makeQueue, enqueueLocalJob, hostQueueName } = await import("./queue.mjs");
|
|
124
127
|
// failFast: a one-shot enqueue must not hang forever if Valkey is down -- error clearly.
|
|
125
|
-
|
|
128
|
+
// Onto THIS host's queue when the deployment declares a name (issue #57). The folder was checked
|
|
129
|
+
// against this machine's filesystem a few lines up, so this machine is the only one that can run it;
|
|
130
|
+
// enqueueing it where every host drains would be handing a job to a peer that has no such folder.
|
|
131
|
+
const hq = config.workerNameDeclared ? hostQueueName(config.workerName) : null;
|
|
132
|
+
const queue = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hq ? { name: hq } : {}) });
|
|
126
133
|
try {
|
|
127
134
|
// Absent flags stay absent (undefined) so the value resolves at job start against the
|
|
128
135
|
// settings overlay/env, not a default frozen here (INT-CONFIG-OVERLAY-CONTRACT).
|
|
@@ -150,31 +157,81 @@ export async function main(argv = process.argv.slice(2), env = process.env) {
|
|
|
150
157
|
// The kill switch reads ONLY VALKEY_URL, not the full loadConfig -- it must work even when
|
|
151
158
|
// GitHub auth is misconfigured, so an operator can always stop the queue.
|
|
152
159
|
const url = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
|
|
153
|
-
const { parseConnection } = await import("./connection.mjs");
|
|
154
|
-
const { makeQueue } = await import("./queue.mjs");
|
|
155
|
-
|
|
156
|
-
//
|
|
157
|
-
|
|
160
|
+
const { parseConnection, makeRedisClient } = await import("./connection.mjs");
|
|
161
|
+
const { fleetQueueNames, discoverHostQueues, unionQueueNames, makeQueue } = await import("./queue.mjs");
|
|
162
|
+
const { readLiveHosts } = await import("./host-registry.mjs");
|
|
163
|
+
// EVERY queue this deployment drains (issue #57), not just the shared one. This is the kill switch:
|
|
164
|
+
// pausing `pi-jobs` alone would stop forge deliveries while a named host's cron, chained children
|
|
165
|
+
// and manual runs kept spending -- and would print "paused" for having done it. That is the silent
|
|
166
|
+
// no-op the comment here already warned about for a mistyped name, arriving through a new door.
|
|
167
|
+
//
|
|
168
|
+
// Both reads fail OPEN -- between them an unreadable registry and an unreadable keyspace yield the
|
|
169
|
+
// shared queue alone, which is exactly what this command did before, so a Valkey blip can never make
|
|
170
|
+
// the kill switch refuse. But it fails open LOUDLY: a degraded read is NAMED in the output rather
|
|
171
|
+
// than left indistinguishable from a single-host success while a named host keeps spending.
|
|
172
|
+
// `readLiveHosts` RETURNS `{unreachable}` rather than rejecting, so `blind` is a branch on its
|
|
173
|
+
// value and the `.catch` below is only for a client that throws before it can answer.
|
|
174
|
+
const probe = makeRedisClient(url);
|
|
175
|
+
// Without this, a down Valkey dumps nine `[ioredis] Unhandled error event` traces before the one clean
|
|
176
|
+
// line -- the exact noise `defaultProbeValkey` exists to suppress.
|
|
177
|
+
probe.on?.("error", () => {});
|
|
178
|
+
// Both reads, concurrently, sharing one budget. The registry answers WHO IS LIVE; BullMQ's own meta
|
|
179
|
+
// keys answer WHICH QUEUES EXIST, and for a kill switch the second is the question that matters. A
|
|
180
|
+
// host whose registry writes fail for ninety seconds loses its row while its worker keeps draining,
|
|
181
|
+
// and a resume that misses a queue leaves it paused forever with no surface able to name it. A meta
|
|
182
|
+
// key outlives its worker; a registry row does not.
|
|
183
|
+
const [fleet, existing] = await Promise.all([
|
|
184
|
+
readLiveHosts(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }).catch((error) => ({ unreachable: error?.message ?? String(error) })),
|
|
185
|
+
discoverHostQueues(probe, { timeoutMs: FLEET_READ_TIMEOUT_MS }),
|
|
186
|
+
]);
|
|
187
|
+
probe.disconnect?.();
|
|
188
|
+
const blind = fleet?.unreachable ?? null;
|
|
189
|
+
const names = unionQueueNames(fleetQueueNames(fleet?.hosts), existing);
|
|
190
|
+
// The registry being unreadable no longer means we saw one queue: the keyspace scan may well have
|
|
191
|
+
// found them. Report the count we ACTED on, and name the degraded read separately.
|
|
192
|
+
const span = `${names.length > 1 ? ` [${names.length} queues]` : ""}${blind ? ` [registry unreadable: ${blind}]` : ""}`;
|
|
193
|
+
const queues = [];
|
|
158
194
|
try {
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
195
|
+
// Constructed INSIDE the try: `makeQueue` can throw on a malformed peer-written name, and a throw
|
|
196
|
+
// at index k > 0 would otherwise leak the k connections already opened.
|
|
197
|
+
for (const name of names) queues.push(makeQueue(parseConnection(url, { failFast: true }), { name }));
|
|
198
|
+
if (cmd === "pause" || cmd === "resume") {
|
|
199
|
+
const done = [];
|
|
200
|
+
try {
|
|
201
|
+
for (const q of queues) {
|
|
202
|
+
await (cmd === "pause" ? q.pause() : q.resume());
|
|
203
|
+
done.push(q.name);
|
|
204
|
+
}
|
|
205
|
+
} catch (error) {
|
|
206
|
+
// A mid-loop failure leaves the deployment HALF switched. Naming what did change is the whole
|
|
207
|
+
// difference between an operator who knows to finish the job and one who reads "unreachable"
|
|
208
|
+
// as "nothing happened" and walks away from a fleet with one host still spending.
|
|
209
|
+
return fail(`could not ${cmd} the whole deployment at ${url}\n ${done.length > 0 ? `${cmd}d: ${done.join(", ")}` : "nothing changed"}\n failed at: ${names[done.length]}\n ${error.message}`);
|
|
210
|
+
}
|
|
211
|
+
process.stdout.write(cmd === "pause" ? `paused — worker will stop taking new jobs (jobs still enqueue)${span}\n` : `resumed${span}\n`);
|
|
165
212
|
} else {
|
|
166
213
|
// "paused" is included in the counts because jobs enqueued while paused land in the
|
|
167
214
|
// `paused` list, not `wait` -- omitting it would report backlog 0 in the exact state
|
|
168
215
|
// the pause switch creates. `pausedState` (the boolean) is named apart from the
|
|
169
216
|
// `paused` count `getJobCounts` returns, so the two do not collide in the output.
|
|
170
|
-
const
|
|
171
|
-
const
|
|
172
|
-
|
|
217
|
+
const states = await Promise.all(queues.map((q) => q.isPaused()));
|
|
218
|
+
const per = await Promise.all(queues.map((q) => q.getJobCounts("waiting", "active", "paused", "delayed", "failed")));
|
|
219
|
+
const counts = per.reduce((acc, c) => {
|
|
220
|
+
for (const [k, v] of Object.entries(c ?? {})) acc[k] = (acc[k] ?? 0) + (Number(v) || 0);
|
|
221
|
+
return acc;
|
|
222
|
+
}, {});
|
|
223
|
+
// Summed counts with a boolean from ONE queue would report a half-paused deployment as fully
|
|
224
|
+
// one or fully the other. `pausedPartial` is the third state, and the dangerous direction is
|
|
225
|
+
// the one it makes visible: pause ran while a host was invisible, so that host still spends.
|
|
226
|
+
const pausedState = states.every(Boolean);
|
|
227
|
+
const pausedPartial = !pausedState && states.some(Boolean);
|
|
228
|
+
const out = { pausedState, ...(pausedPartial ? { pausedPartial, pausedQueues: names.filter((_, i) => states[i]) } : {}), ...counts, ...(blind ? { fleet: blind } : {}) };
|
|
229
|
+
process.stdout.write(`${JSON.stringify(out)}\n`);
|
|
173
230
|
}
|
|
174
231
|
} catch (error) {
|
|
175
232
|
return fail(`could not reach Valkey at ${url} — is it running? (docker compose up)\n ${error.message}`);
|
|
176
233
|
} finally {
|
|
177
|
-
await
|
|
234
|
+
for (const q of queues) await q.close().catch(() => {});
|
|
178
235
|
}
|
|
179
236
|
return 0;
|
|
180
237
|
}
|
package/src/config.mjs
CHANGED
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import { existsSync } from "node:fs";
|
|
9
|
+
import { hostname } from "node:os";
|
|
9
10
|
import { delimiter } from "node:path";
|
|
10
11
|
import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
|
|
11
12
|
import { MINTED_TOKEN_VARS } from "./forges.mjs";
|
|
@@ -212,6 +213,14 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
|
|
|
212
213
|
const model = env.PI_MODEL ?? "claude-sonnet-4-5-20250929"; // dated snapshot; deterministic per CONST-PI-VERSION-PINNED
|
|
213
214
|
return {
|
|
214
215
|
valkeyUrl: env.VALKEY_URL ?? "redis://127.0.0.1:6379",
|
|
216
|
+
// Issue #57. What this machine calls itself: the key of its registry row, the `host` on every log
|
|
217
|
+
// line and run record, and the BullMQ worker name. Always populated -- a deployment that declares
|
|
218
|
+
// nothing still has an identity, which is what lets a fleet of two be TOLD APART before anyone has
|
|
219
|
+
// configured anything. `workerNameDeclared` is kept separately because "the operator named this
|
|
220
|
+
// machine" and "we read the hostname" are different facts, and a later slice gates a host-visible
|
|
221
|
+
// side effect on the first rather than the second.
|
|
222
|
+
workerName: workerName(env),
|
|
223
|
+
workerNameDeclared: Boolean(env.PI_WORKER_NAME),
|
|
215
224
|
concurrency: positiveInt(env, "PI_CONCURRENCY", 3), // DES-CONCURRENCY-3
|
|
216
225
|
dailyCap: positiveInt(env, "PI_DAILY_CAP", 25), // bounds container STARTS per day (money)
|
|
217
226
|
weeklyCap: optionalBoundedInt(env, "PI_WEEKLY_CAP", 1), // REQ-SPEND-CAPS-MULTI-WINDOW; null = weekly window disabled
|
|
@@ -448,6 +457,80 @@ export function defaultLogsDir() {
|
|
|
448
457
|
return `${process.env.TMPDIR ?? process.env.TEMP ?? "/tmp"}/pi-dispatch/logs`.replace(/\\/g, "/");
|
|
449
458
|
}
|
|
450
459
|
|
|
460
|
+
/**
|
|
461
|
+
* What a worker may call itself (issue #57). The CHARACTER CLASS is `sanitizeJobId`'s
|
|
462
|
+
* (`[A-Za-z0-9._-]`), reused rather than invented so this project has one name-safe alphabet -- but that
|
|
463
|
+
* function is a REPLACER, not a validator, so the three rules around the class are NEW and are claimed
|
|
464
|
+
* as new here rather than borrowed:
|
|
465
|
+
*
|
|
466
|
+
* - a leading alphanumeric, which is what refuses `..` and a leading `-` that reads as a flag;
|
|
467
|
+
* - a 64-character ceiling, because the name is a Valkey key segment and a log field on every line;
|
|
468
|
+
* - no `.json`/`.log` tail, which is not decoration. The class contains the dot, so `prod.json` is
|
|
469
|
+
* otherwise a legal name -- and a later slice writes a per-host marker file into `PI_LOGS_DIR`,
|
|
470
|
+
* where `<something>.json` is parsed as a run record by the admin and DELETED by the log reaper.
|
|
471
|
+
* A name is refused here rather than escaped there, because the escape would have to be remembered
|
|
472
|
+
* at every site that ever composes a filename from this value.
|
|
473
|
+
*
|
|
474
|
+
* The class is `:`-free, `,`-free and `#`-free, which is what lets the name be a Valkey key segment
|
|
475
|
+
* UNHASHED. That is the point of validating instead of hashing (`scopeKeyPrefix` does the opposite for
|
|
476
|
+
* a folder path, which was never chosen for key-safety and cannot be refused): the whole value of a host
|
|
477
|
+
* registry is that `HGETALL host:h:mac-mini-1` is readable by a human.
|
|
478
|
+
*/
|
|
479
|
+
export const WORKER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;
|
|
480
|
+
|
|
481
|
+
/** True when the name would collide with the run-history filename namespace. See WORKER_NAME_RE. */
|
|
482
|
+
const RESERVED_NAME_TAIL = /\.(json|log)$/i;
|
|
483
|
+
|
|
484
|
+
/**
|
|
485
|
+
* A hostname reduced to something `WORKER_NAME_RE` accepts, for use as a DEFAULT only.
|
|
486
|
+
*
|
|
487
|
+
* Lowercased because macOS reports `Robs-Mac-Mini.local` where Linux reports `mac-mini`: two spellings
|
|
488
|
+
* of one machine would be two rows in the registry and two values in the run records. The `.local`
|
|
489
|
+
* suffix is deliberately NOT stripped -- an OS-specific suffix rule is a rule someone has to remember,
|
|
490
|
+
* and it costs nothing to keep.
|
|
491
|
+
*/
|
|
492
|
+
export function sanitizeWorkerName(raw) {
|
|
493
|
+
const cleaned = String(raw ?? "")
|
|
494
|
+
.toLowerCase()
|
|
495
|
+
.replace(/[^a-z0-9._-]/g, "-")
|
|
496
|
+
.replace(/-{2,}/g, "-")
|
|
497
|
+
.replace(/^[-.]+|[-.]+$/g, "")
|
|
498
|
+
.slice(0, 64)
|
|
499
|
+
.replace(/[-.]+$/, ""); // the slice can leave a trailing separator behind
|
|
500
|
+
if (cleaned === "" || !WORKER_NAME_RE.test(cleaned)) return "worker";
|
|
501
|
+
// The reserved tail is repaired by REPLACING the dot, never by appending: a suffix on a name already at
|
|
502
|
+
// the 64-character ceiling would push it past, and a default that the validator would reject is a
|
|
503
|
+
// second, weaker alphabet arriving by the back door. `host.json` becomes `host-json`, which is the same
|
|
504
|
+
// length, still readable, and cannot match the tail again.
|
|
505
|
+
return cleaned.replace(RESERVED_NAME_TAIL, (m) => `-${m.slice(1)}`);
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
/** This machine's name, sanitized. Exported so doctor and the admin resolve it without `loadConfig`. */
|
|
509
|
+
export function defaultWorkerName() {
|
|
510
|
+
try {
|
|
511
|
+
return sanitizeWorkerName(hostname());
|
|
512
|
+
} catch {
|
|
513
|
+
return "worker"; // hostname() can throw on a locked-down host; a name is never worth refusing boot for
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
/**
|
|
518
|
+
* THE ASYMMETRY IS THE DESIGN. A value the operator did not choose is repaired silently; a value they
|
|
519
|
+
* typed is refused loudly and never quietly altered. Defaulting is a convenience, so it must not be able
|
|
520
|
+
* to fail; declaring is a statement, so a typo in it must not become a different machine's name.
|
|
521
|
+
*/
|
|
522
|
+
function workerName(env) {
|
|
523
|
+
const declared = env.PI_WORKER_NAME;
|
|
524
|
+
if (declared === undefined || declared === "") return defaultWorkerName();
|
|
525
|
+
if (!WORKER_NAME_RE.test(declared)) {
|
|
526
|
+
throw configError(`PI_WORKER_NAME must match ${WORKER_NAME_RE.source} (letters, digits, dot, underscore, hyphen; first character alphanumeric; at most 64): ${JSON.stringify(declared)}`);
|
|
527
|
+
}
|
|
528
|
+
if (RESERVED_NAME_TAIL.test(declared)) {
|
|
529
|
+
throw configError(`PI_WORKER_NAME must not end in .json or .log: ${JSON.stringify(declared)} would collide with the run-history filenames in PI_LOGS_DIR`);
|
|
530
|
+
}
|
|
531
|
+
return declared;
|
|
532
|
+
}
|
|
533
|
+
|
|
451
534
|
export function defaultGraphDir(env = process.env) {
|
|
452
535
|
// Under the OS temp dir by default, beside logs/ and jobs/ -- the admin's graph HTML artifact
|
|
453
536
|
// (issue #54) is host-side display output on the defaultLogsDir doctrine, and deliberately NOT
|
package/src/cron.mjs
CHANGED
|
@@ -14,7 +14,8 @@
|
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
16
|
import { configError } from "./config.mjs";
|
|
17
|
-
import {
|
|
17
|
+
import { cronFingerprint } from "./fingerprint.mjs";
|
|
18
|
+
import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
|
|
18
19
|
|
|
19
20
|
function sentinelName(code) {
|
|
20
21
|
if (code === -10) return "SchedulerJobIdCollision";
|
|
@@ -67,6 +68,99 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
|
|
|
67
68
|
return { installed: schedules.length, removed: orphanIds.length };
|
|
68
69
|
}
|
|
69
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Reconcile, but only when every other LIVE worker agrees about what should be scheduled (issue #57).
|
|
73
|
+
*
|
|
74
|
+
* THE BUG THIS CLOSES. `reconcile` prunes every resident scheduler not named in THIS worker's config, so
|
|
75
|
+
* two workers with different triggers files each delete the other's on every boot and every file-watch
|
|
76
|
+
* reload. Its idempotence only ever held because one worker was the only shape, and nothing checked.
|
|
77
|
+
*
|
|
78
|
+
* WHY AGREEMENT RATHER THAN AN ELECTED OWNER. In the good case agreement is sufficient, and this module's
|
|
79
|
+
* own header says why: upsert is keyed by schedulerId, so when every live host agrees it does not matter
|
|
80
|
+
* which one reconciles or how many do. In the BAD case election is actively worse -- an elected owner
|
|
81
|
+
* reconciles from ITS file, so if that file is the stale one (the operator edited on the other host, or a
|
|
82
|
+
* compose `:ro` single-file mount pinned a dead inode, a topology `makeCheckWaitSkew` already documents)
|
|
83
|
+
* the fleet silently converges on the wrong set and reverts the edit with a log line that reads like
|
|
84
|
+
* success. That is `OQ-008`'s own verdict arriving through a new door. Agreement never picks a winner, so
|
|
85
|
+
* it cannot pick the wrong one, and it needs no lease because it grants no authority: the rule only ever
|
|
86
|
+
* WITHHOLDS a permission relative to today, which is why it cannot be a regression.
|
|
87
|
+
*
|
|
88
|
+
* The honest cost, stated rather than buried: agreement can stalemate and needs an operator, where
|
|
89
|
+
* election resolves automatically and possibly wrongly. For a project whose doctrine is "fail loudly, or
|
|
90
|
+
* fail open and say which", a stalemate that names both hosts is the right trade.
|
|
91
|
+
*
|
|
92
|
+
* PUBLISH BEFORE READ is what makes the legitimate-edit sequence race-free. An operator edits on host A;
|
|
93
|
+
* A's watcher fires, A publishes its new fingerprint, reads peers, sees B still on the old one and
|
|
94
|
+
* refuses. The operator syncs the file to B; B's watcher fires, B publishes, reads, sees A already on the
|
|
95
|
+
* new one, and reconciles for the whole fleet. A never has to run again -- the schedule set is global. The
|
|
96
|
+
* only bad interleaving would be both refusing while both are in fact current, which needs a read to see a
|
|
97
|
+
* stale value, and cannot happen when each side publishes synchronously before it reads.
|
|
98
|
+
*
|
|
99
|
+
* ABSENCE NEVER REFUSES. No peers, or a registry that cannot be read, both PROCEED -- which is today's
|
|
100
|
+
* behaviour, so a Valkey blip can never wedge a single-host deployment.
|
|
101
|
+
*
|
|
102
|
+
* BOTH HALVES ARE REFUSED, not just the prune, and the reason is not symmetry: `upsertJobScheduler` on an
|
|
103
|
+
* existing id is a REDEFINITION, so two hosts disagreeing about one id would flip a schedule between two
|
|
104
|
+
* definitions on every file change with nothing logged.
|
|
105
|
+
*
|
|
106
|
+
* A refusal RETURNS and never throws, so the caller logs it and carries on to `worker_started`: a
|
|
107
|
+
* divergent host must still drain the queue. Taking a host's forge capacity offline over a cron
|
|
108
|
+
* disagreement is the sentence this issue's own acceptance forbids.
|
|
109
|
+
*/
|
|
110
|
+
export async function reconcileGated(queue, schedules, { registry, log = () => {}, reconcileFn = reconcile, tz, authored = schedules } = {}) {
|
|
111
|
+
// No registry wired is the same answer as a registry that cannot be read: proceed. This is what lets
|
|
112
|
+
// the gate be the DEFAULT on every path without a caller having to remember to arm it.
|
|
113
|
+
if (!registry) return await reconcileFn(queue, schedules, { log });
|
|
114
|
+
// THE FINGERPRINT IS OVER THE AUTHORED SET, NOT THE SERVED SUBSET, and the distinction is the whole
|
|
115
|
+
// reason placement and agreement can coexist. What two hosts must AGREE about is the FILE; what they
|
|
116
|
+
// legitimately DIFFER about is which of its triggers each one can run, because a folder lives on one
|
|
117
|
+
// machine. Hashing the served subset would make every correctly-configured fleet refuse itself forever:
|
|
118
|
+
// mini1 serves /a, mini2 serves /b, their subsets differ, and neither would ever reconcile again.
|
|
119
|
+
const mine = cronFingerprint(authored, { tz });
|
|
120
|
+
// Publishes this host's CURRENT facts rather than passing the fingerprint in. Passing it in was the
|
|
121
|
+
// first shape and it quietly destroyed the mechanism it depends on: the heartbeat installs `fpCron` as
|
|
122
|
+
// a THUNK over the live schedule ref, and a caller merging a computed string replaced that closure, so
|
|
123
|
+
// every later beat republished a frozen value and two hosts could drift apart again with nothing saying
|
|
124
|
+
// so. The thunk is installed once, at boot; this only forces it to be read NOW.
|
|
125
|
+
await registry.publish();
|
|
126
|
+
const peers = await registry.livePeers();
|
|
127
|
+
|
|
128
|
+
// `{ unreachable }` and "no peers" are different facts and the panel must keep them apart -- but for
|
|
129
|
+
// THIS decision they resolve the same way, because not knowing whether anyone disagrees is not knowing
|
|
130
|
+
// that someone does, and the rule only withholds a permission.
|
|
131
|
+
const others = peers?.hosts ?? [];
|
|
132
|
+
// An abstaining peer (cron disabled) publishes no fingerprint and is never a disagreeing party.
|
|
133
|
+
const opinions = others.filter((h) => typeof h.fpCron === "string" && h.fpCron !== "");
|
|
134
|
+
const disagreeing = opinions.filter((h) => h.fpCron !== mine);
|
|
135
|
+
|
|
136
|
+
// I cannot establish agreement with an opinion I do not have. `mine` is null only when the file could
|
|
137
|
+
// not be read or parsed at THIS instant while `loadSchedules` had just succeeded -- a rename's brief
|
|
138
|
+
// unlink window, in practice. Proceeding would prune a peer's schedulers on the strength of a
|
|
139
|
+
// comparison that never happened, so this refuses; refusing deletes nothing and the next watch event
|
|
140
|
+
// or boot re-decides. It gets its own token because "I could not read my own file" and "we disagree"
|
|
141
|
+
// send an operator to two different places.
|
|
142
|
+
if (mine === null && opinions.length > 0) {
|
|
143
|
+
log("cron_divergence_refused", { mine: null, reason: "own-triggers-unreadable", cronCount: schedules.length, peers: opinions.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })) });
|
|
144
|
+
return { refused: "own-triggers-unreadable", peers: opinions.map((h) => h.name) };
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
if (disagreeing.length > 0) {
|
|
148
|
+
log("cron_divergence_refused", {
|
|
149
|
+
mine,
|
|
150
|
+
cronCount: schedules.length,
|
|
151
|
+
// The count rides the LINE and never the RULE: it is what lets the message say "host-b reports 4
|
|
152
|
+
// schedules, I have 5", which is the difference between a diagnosable warning and noise.
|
|
153
|
+
peers: disagreeing.map((h) => ({ host: h.name, fpCron: h.fpCron, cronCount: Number(h.cronCount) || 0 })),
|
|
154
|
+
});
|
|
155
|
+
return { refused: "cron-divergence", peers: disagreeing.map((h) => h.name) };
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// Gated on `> 0`, so a single-host deployment emits no new line at all -- the absent-when-unarmed idiom
|
|
159
|
+
// this repo uses for every optional field.
|
|
160
|
+
if (opinions.length > 0) log("cron_agreement", { peers: opinions.length });
|
|
161
|
+
return await reconcileFn(queue, schedules, { log });
|
|
162
|
+
}
|
|
163
|
+
|
|
70
164
|
/**
|
|
71
165
|
* Live-reload the cron schedulers from the (changed) triggers file: re-select the cron subset and reconcile
|
|
72
166
|
* it against the resident schedulers -- an add installs, a delete prunes (reconcile already removes orphans),
|
|
@@ -75,16 +169,34 @@ export async function reconcile(queue, schedules, { log = () => {} } = {}) {
|
|
|
75
169
|
* never taken down by a malformed trigger file (the OQ-008 live-edit safety). Returns `{ ok }` /
|
|
76
170
|
* `{ invalid }` / `{ failed }`. `loadFn`/`reconcileFn` are injectable so the reload is unit-tested with no fs.
|
|
77
171
|
*/
|
|
78
|
-
export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn =
|
|
172
|
+
export async function reloadSchedules(config, queue, { log = () => {}, loadFn = loadSchedules, reconcileFn = reconcileGated, ref = null, registry, tz, fleet = false, authoredFn = authoredCron } = {}) {
|
|
79
173
|
let schedules;
|
|
80
174
|
try {
|
|
81
|
-
schedules = loadFn(config);
|
|
175
|
+
schedules = loadFn(config, { fleet });
|
|
82
176
|
} catch (error) {
|
|
83
177
|
log("schedules_reload_invalid", { reason: error?.message ?? String(error), kept: true });
|
|
84
178
|
return { invalid: error?.message ?? String(error) };
|
|
85
179
|
}
|
|
180
|
+
// The live REF is updated before the reconcile, not after, and never on the invalid path above: the
|
|
181
|
+
// heartbeat fingerprints what this host currently believes, and believing the boot-time set after an
|
|
182
|
+
// edit is what would make two hosts' fingerprints oscillate on the beat period -- refusing or agreeing
|
|
183
|
+
// depending on which half of a beat a reload landed in.
|
|
184
|
+
if (ref) ref.current = schedules;
|
|
185
|
+
// The same split the boot path makes: a trigger whose folder is another host's is not this host's to
|
|
186
|
+
// install, and the fingerprint is computed over the SERVED set so two hosts owning different folders
|
|
187
|
+
// do not read each other as divergent.
|
|
188
|
+
const { served, unserved } = servedSchedules(schedules);
|
|
189
|
+
for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
|
|
86
190
|
try {
|
|
87
|
-
|
|
191
|
+
// The FILE, re-read, not `schedules` -- that is `loadSchedules`'s output, which has already replaced
|
|
192
|
+
// every foreign trigger with a stub and therefore differs per host by construction. Passing it here
|
|
193
|
+
// made every live edit on a fleet refuse, permanently, even between hosts running identical files.
|
|
194
|
+
const r = await reconcileFn(queue, served, { log, registry, tz, authored: authoredFn(config) });
|
|
195
|
+
// A refusal is NOT a reload. Wrapping it as `{ ok: true }` would log
|
|
196
|
+
// `schedules_reloaded {installed: undefined}` and tell an operator the edit took effect on a fleet
|
|
197
|
+
// where nothing was installed and nothing pruned -- the silent no-op this project refuses, arriving
|
|
198
|
+
// through the success path.
|
|
199
|
+
if (r?.refused) return r;
|
|
88
200
|
log("schedules_reloaded", { installed: r.installed, removed: r.removed });
|
|
89
201
|
return { ok: true, ...r };
|
|
90
202
|
} catch (error) {
|