@edgehero/pi-dispatch 1.6.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +7 -0
- package/package.json +2 -1
- package/src/cli.mjs +74 -17
- package/src/config.mjs +83 -0
- package/src/cron.mjs +116 -4
- package/src/doctor.mjs +113 -3
- package/src/fingerprint.mjs +78 -0
- package/src/fleet-lease.mjs +179 -0
- package/src/host-registry.mjs +279 -0
- package/src/image-preflight.mjs +8 -3
- package/src/index.mjs +219 -65
- package/src/queue.mjs +130 -2
- package/src/run-history.mjs +98 -4
- package/src/schedules.mjs +67 -4
- package/src/start.mjs +217 -27
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fleet-wide leases for the two bounds that stopped meaning what they say when a second host appeared
|
|
3
|
+
* (issue #57).
|
|
4
|
+
*
|
|
5
|
+
* `PI_WAIT_CHECK_SLOTS` and a `scoped-limits.json` row's `concurrent` are both enforced by
|
|
6
|
+
* `makeInFlight()`, an in-process `Map`. One worker per docker daemon made that correct; two hosts make
|
|
7
|
+
* it a bound that MULTIPLIES BY THE OPERATOR'S DEPLOYMENT SHAPE, which is not a bound. Four hosts with
|
|
8
|
+
* `PI_WAIT_CHECK_SLOTS=1` run four concurrent checks against one Jira; four hosts with
|
|
9
|
+
* `{"scope":"acme/web","concurrent":1}` run four paid containers on one repository.
|
|
10
|
+
*
|
|
11
|
+
* WHAT THIS OWES `OQ-008` AND `DES-CONCURRENCY-3`. Both refuse a Redis-held in-flight count, in terms:
|
|
12
|
+
* "a Redis-held count would survive a crash WRONGLY -- a claim for a container the reaper just killed,
|
|
13
|
+
* demanding TTL/heartbeat machinery, a second source of truth about what is running". Every clause of
|
|
14
|
+
* that is about a CONTAINER, and the two leases here answer it differently:
|
|
15
|
+
*
|
|
16
|
+
* The CHECK lease claims no container. It claims a subprocess this same process spawned, bounded by
|
|
17
|
+
* `PI_WAIT_CHECK_TIMEOUT_MS`, holding no folder and spending no money. Three properties invert. What a
|
|
18
|
+
* stale claim costs: a container claim is a folder mutex nobody holds and a job that never runs, while
|
|
19
|
+
* a check claim is one check deferred by at most the TTL. What contradicts it: the reaper is a second
|
|
20
|
+
* source of truth about containers and runs at boot with authority, while nothing enumerates,
|
|
21
|
+
* inspects or reaps a check. And how long it can be wrong: a container has no natural expiry, while a
|
|
22
|
+
* check has a hard, configured, small timeout, so the TTL is DERIVED rather than guessed.
|
|
23
|
+
*
|
|
24
|
+
* The SCOPE claim really is for a container, so the refusal lands squarely -- and the answer is the
|
|
25
|
+
* boot reaper. It establishes, at boot, that this host holds no `pi-job-*` containers, so a claim
|
|
26
|
+
* whose value names this host is a claim for a container that no longer exists, and deleting it is not
|
|
27
|
+
* a second source of truth: it is the SAME source writing down what it just established. Making the
|
|
28
|
+
* reaper the claim's owner removes the contradiction rather than arguing around it.
|
|
29
|
+
*
|
|
30
|
+
* N INDEPENDENT KEYS, NEVER ONE COUNTER. A counter with one TTL loses every claim when it expires and
|
|
31
|
+
* leaks a permanent `+1` on a crash; N `SET NX PX` keys mean a lost release costs exactly one slot for
|
|
32
|
+
* exactly the TTL and never the whole semaphore. `wait:key:<dedupId>` is the in-repo precedent.
|
|
33
|
+
*
|
|
34
|
+
* And a property the in-process map does not have: RELEASE IS IDEMPOTENT here, because it deletes only a
|
|
35
|
+
* key whose value is still ours. `makeInFlight().release` clamps at zero but a double release on a
|
|
36
|
+
* `concurrent: 2` scope frees the other holder's slot.
|
|
37
|
+
*/
|
|
38
|
+
|
|
39
|
+
/** Slot keys. Both live under a prefix an operator can see whole with one `KEYS`. */
|
|
40
|
+
export const checkSlotKey = (i) => `wait:check:${i}`;
|
|
41
|
+
export const scopeSlotKey = (hash, i) => `slot:s:${hash}:${i}`;
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Build a lease over `slots` numbered keys.
|
|
45
|
+
*
|
|
46
|
+
* `holder` is `<workerName>#<jobId>`: the worker-name charset excludes `#`, so the value decomposes and
|
|
47
|
+
* the boot sweep can recognise its own claims without a second index to keep in step.
|
|
48
|
+
*
|
|
49
|
+
* FAIL OPEN, and in the GRANTING direction. A Valkey fault must never be able to wedge every wait in a
|
|
50
|
+
* deployment or stop every scoped job; the in-process bound is still there underneath, so failing open
|
|
51
|
+
* degrades the fleet bound to the per-host one, which is exactly the behaviour before this existed.
|
|
52
|
+
*/
|
|
53
|
+
export function makeFleetLease({ redis, holderPrefix, keyFor, ttlMs, now = () => Date.now(), log = () => {}, timeoutMs = 2_000 }) {
|
|
54
|
+
// Bounded for `host-registry.mjs`'s reason, which applies to every module sharing this client:
|
|
55
|
+
// `maxRetriesPerRequest: null` makes a command against an unreachable server QUEUE rather than reject,
|
|
56
|
+
// so a try/catch around it catches nothing and an outage would hang the gate rather than fail it.
|
|
57
|
+
const bounded = (p) =>
|
|
58
|
+
new Promise((resolve, reject) => {
|
|
59
|
+
const t = setTimeout(() => reject(new Error("lease timeout")), timeoutMs);
|
|
60
|
+
Promise.resolve(p).then(
|
|
61
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
62
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
63
|
+
);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
/**
|
|
68
|
+
* Take one of `slots`, or `null` when they are all held. Returns a handle whose `release` and
|
|
69
|
+
* `refresh` act only on the key this call actually won.
|
|
70
|
+
*
|
|
71
|
+
* Probing starts at `hash(id) mod slots` rather than at 0. Without the rotation every host tries
|
|
72
|
+
* index 0 first, so a host can sit behind a busy slot while a free one exists two along -- a
|
|
73
|
+
* starvation that looks exactly like the capacity shortage the bound is meant to report.
|
|
74
|
+
*/
|
|
75
|
+
async acquire(id, { slots, keyArgs = [], ttlMs: perCall } = {}) {
|
|
76
|
+
if (!Number.isFinite(slots) || slots < 1) return { ok: true, release: async () => {}, refresh: async () => true };
|
|
77
|
+
const holder = `${holderPrefix}#${id}`;
|
|
78
|
+
const start = Math.abs(hashCode(String(id))) % slots;
|
|
79
|
+
for (let n = 0; n < slots; n++) {
|
|
80
|
+
const key = keyFor(...keyArgs, (start + n) % slots);
|
|
81
|
+
try {
|
|
82
|
+
const won = await bounded(redis.set(key, holder, "PX", perCall ?? ttlMs, "NX"));
|
|
83
|
+
if (!won) continue;
|
|
84
|
+
return {
|
|
85
|
+
ok: true,
|
|
86
|
+
key,
|
|
87
|
+
async release() {
|
|
88
|
+
try {
|
|
89
|
+
// Release-if-MINE, which is what makes this idempotent where the in-process map is
|
|
90
|
+
// not: a double release cannot free another holder's slot, because the second call
|
|
91
|
+
// finds a value that is no longer ours.
|
|
92
|
+
if ((await bounded(redis.get(key))) === holder) await bounded(redis.del(key));
|
|
93
|
+
} catch {
|
|
94
|
+
// The TTL is the backstop. A lost release costs one slot for one TTL.
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
async refresh(nextMs = ttlMs) {
|
|
98
|
+
try {
|
|
99
|
+
if ((await bounded(redis.get(key))) !== holder) return false;
|
|
100
|
+
await bounded(redis.pexpire(key, nextMs));
|
|
101
|
+
return true;
|
|
102
|
+
} catch {
|
|
103
|
+
return true; // a blip is not a reason to believe we lost a slot we hold
|
|
104
|
+
}
|
|
105
|
+
},
|
|
106
|
+
};
|
|
107
|
+
} catch (err) {
|
|
108
|
+
// Granting is the safe direction: the in-process bound is still underneath, so this
|
|
109
|
+
// degrades the fleet-wide ceiling to the per-host one rather than to nothing.
|
|
110
|
+
log("fleet_lease_unavailable", { reason: err?.message });
|
|
111
|
+
return { ok: true, degraded: true, release: async () => {}, refresh: async () => true };
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
return null;
|
|
115
|
+
},
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** A small, stable, non-cryptographic spread for the slot rotation. Not a key, so not a digest. */
|
|
120
|
+
function hashCode(s) {
|
|
121
|
+
let h = 0;
|
|
122
|
+
for (let i = 0; i < s.length; i++) h = (h * 31 + s.charCodeAt(i)) | 0;
|
|
123
|
+
return h;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Delete every scope slot this host still claims, at boot, right after the container reaper.
|
|
128
|
+
*
|
|
129
|
+
* THIS IS THE ANSWER TO `OQ-008`, and its correctness rests entirely on one precondition: the reaper
|
|
130
|
+
* must have ENUMERATED. `makeReaper` catches its own `docker ps` failure and logs `reaper_skipped`, and
|
|
131
|
+
* on that path nothing was listed and nothing reaped -- so this host has NOT established that it holds
|
|
132
|
+
* no containers, and its claims may be for containers that are still running. Sweeping then would free
|
|
133
|
+
* slots for another host to start more alongside them, which is a money overrun rather than a tidy-up.
|
|
134
|
+
* `makeInFlight`'s own escape ("a state where no NEW container can start either") does not transfer,
|
|
135
|
+
* because the sweep frees slots for a DIFFERENT machine.
|
|
136
|
+
*
|
|
137
|
+
* Driven by config rather than by a scan: `scoped-limits.json` enumerates every scope that can carry a
|
|
138
|
+
* claim and `concurrent` bounds the index, so this is `sum(concurrent)` GETs -- typically under twenty.
|
|
139
|
+
* No `KEYS`, no `SCAN`, and no index set to leak.
|
|
140
|
+
*/
|
|
141
|
+
export function makeScopeClaimSweeper({ redis, workerName, limits, log = () => {}, timeoutMs = 2_000 }) {
|
|
142
|
+
return async function sweep({ reaped }) {
|
|
143
|
+
if (!reaped) {
|
|
144
|
+
log("scope_claims_sweep_skipped", { reason: "reaper-skipped" });
|
|
145
|
+
return { swept: 0, skipped: true };
|
|
146
|
+
}
|
|
147
|
+
const prefix = `${workerName}#`;
|
|
148
|
+
let swept = 0;
|
|
149
|
+
for (const row of limits ?? []) {
|
|
150
|
+
const n = Number(row?.concurrent);
|
|
151
|
+
if (!Number.isFinite(n) || n < 1 || !row?.hash) continue;
|
|
152
|
+
for (let i = 0; i < n; i++) {
|
|
153
|
+
const key = scopeSlotKey(row.hash, i);
|
|
154
|
+
try {
|
|
155
|
+
const held = await withTimeout(redis.get(key), timeoutMs);
|
|
156
|
+
if (typeof held === "string" && held.startsWith(prefix)) {
|
|
157
|
+
await withTimeout(redis.del(key), timeoutMs);
|
|
158
|
+
swept++;
|
|
159
|
+
}
|
|
160
|
+
} catch {
|
|
161
|
+
// Best-effort by contract: this is an OPTIMISATION over the TTL, never the mechanism, so a
|
|
162
|
+
// fault here costs at most one TTL of a stale claim and must never block boot.
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
if (swept > 0) log("scope_claims_swept", { count: swept });
|
|
167
|
+
return { swept, skipped: false };
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
function withTimeout(p, ms) {
|
|
172
|
+
return new Promise((resolve, reject) => {
|
|
173
|
+
const t = setTimeout(() => reject(new Error("lease timeout")), ms);
|
|
174
|
+
Promise.resolve(p).then(
|
|
175
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
176
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
177
|
+
);
|
|
178
|
+
});
|
|
179
|
+
}
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The `host:` keyspace: which workers are alive, and what each one is (issue #57).
|
|
3
|
+
*
|
|
4
|
+
* Every gap in the multi-host issue wants the same missing fact -- WHICH HOSTS EXIST AND WHAT CAN EACH
|
|
5
|
+
* ONE DO -- so one structure serves all of them rather than each growing its own. This module is that
|
|
6
|
+
* structure and nothing else. It carries no configuration and no authority: every row is one host's
|
|
7
|
+
* SELF-DESCRIPTION, published by that host, and no reader may write another's.
|
|
8
|
+
*
|
|
9
|
+
* host:live SET of worker names -- an index the READER prunes, never a source of truth
|
|
10
|
+
* host:h:<name> HASH of that host's self-description, PEXPIRE'd on every beat
|
|
11
|
+
*
|
|
12
|
+
* WHY THIS IS NOT THE REDIS STATE `OQ-008` AND `DES-CONCURRENCY-3` REFUSED. That refusal is about a
|
|
13
|
+
* claim whose truth-maker lives on the host while the claim lives in Redis -- an in-flight count
|
|
14
|
+
* asserting a container the boot reaper had just killed, with no way for the two to notice they
|
|
15
|
+
* disagree. Here the truth-maker IS the host process, and the refresh IS the claim: when the process
|
|
16
|
+
* dies the claim stops being renewed and expires on its own. There is no reaper to contradict and no
|
|
17
|
+
* second authority to drift from.
|
|
18
|
+
*
|
|
19
|
+
* The sharper test, and the one to apply to anything added here later: DELETE THE WHOLE `host:*`
|
|
20
|
+
* KEYSPACE WHILE THE FLEET RUNS, and every host must behave exactly as it does today. That holds
|
|
21
|
+
* because nothing here is consulted to decide anything a single-host deployment decides differently --
|
|
22
|
+
* a reader that cannot read treats absence as "no peers", which is v1.6.1's behaviour. A Redis-side
|
|
23
|
+
* toggle fails that test: deleting it loses the operator's edit. Anything that would fail it does not
|
|
24
|
+
* belong in this keyspace.
|
|
25
|
+
*
|
|
26
|
+
* THE CONTENT RULE, which is a contract and not a style note: names, integers and digests. Never a
|
|
27
|
+
* path, never a URL with credentials, never a repository name, never operator free text. This is
|
|
28
|
+
* `targetFor`'s `local:<basename>` discipline applied to a Valkey value, and it earns its strictness
|
|
29
|
+
* from the reader rather than the writer -- the panel is where an operator screenshots, and doctor
|
|
30
|
+
* prints these rows. A value that must be carried but cannot satisfy the rule is HASHED before it gets
|
|
31
|
+
* here (`scopeKeyPrefix`'s idiom), never abbreviated.
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
/** The index. A SET cannot expire its members, so the leak is handled by the reader, as `wait:held` does. */
|
|
35
|
+
export const HOST_SET = "host:live";
|
|
36
|
+
|
|
37
|
+
/** One host's row. `h:` is a sub-namespace letter under one prefix, as `budget:` uses `w:`/`m:`/`t:`/`s:`. */
|
|
38
|
+
export const hostKey = (name) => `host:h:${name}`;
|
|
39
|
+
|
|
40
|
+
/** How often a live worker republishes. Cheap: three writes, off every job path. */
|
|
41
|
+
export const HOST_BEAT_MS = 15_000;
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* How long a row outlives its last beat. SIX missed beats, not one or two: a heartbeat competes with the
|
|
45
|
+
* event loop of a process whose whole job is spawning containers, so a single late beat is ordinary and
|
|
46
|
+
* must not evict a healthy host. Ninety seconds is short enough that a crashed host stops being counted
|
|
47
|
+
* within one panel refresh cycle of an operator noticing anything at all.
|
|
48
|
+
*/
|
|
49
|
+
export const HOST_TTL_MS = 90_000;
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Build the registry accessor. `redis` is the client the budget and the wait state already share.
|
|
53
|
+
*
|
|
54
|
+
* EVERY METHOD FAILS OPEN, and the direction is the design: a registry fault must be able to cost a
|
|
55
|
+
* panel row and must never be able to invent a refusal. Nothing here is on the paid path, so there is no
|
|
56
|
+
* case where throwing would be more correct than continuing.
|
|
57
|
+
*/
|
|
58
|
+
/**
|
|
59
|
+
* How long any single registry command may take before it is treated as a failure.
|
|
60
|
+
*
|
|
61
|
+
* THIS BOUND IS THE WHOLE FAIL-OPEN MECHANISM, and getting it wrong is subtle enough to be worth the
|
|
62
|
+
* paragraph. `makeRedisClient` sets `maxRetriesPerRequest: null`, which BullMQ's blocking connections
|
|
63
|
+
* require and which means a command issued against an unreachable server does NOT reject -- it QUEUES,
|
|
64
|
+
* forever. So a `try/catch` around it catches nothing: the failure mode of this client is a HANG, and a
|
|
65
|
+
* hang is not an exception. Every `await` in this module therefore goes through `bounded`, which converts
|
|
66
|
+
* "never answers" into "answered no" so the catch below can do its job.
|
|
67
|
+
*/
|
|
68
|
+
const REGISTRY_OP_TIMEOUT_MS = 2_000;
|
|
69
|
+
|
|
70
|
+
/** Reject rather than wait forever. See REGISTRY_OP_TIMEOUT_MS for why this is not optional here. */
|
|
71
|
+
function bounded(promise, ms = REGISTRY_OP_TIMEOUT_MS) {
|
|
72
|
+
return new Promise((resolve, reject) => {
|
|
73
|
+
// NOT unref'd, deliberately. An unref'd timer does not fire when nothing else is holding the event
|
|
74
|
+
// loop, so the bound would silently stop existing in exactly the situation it is for -- a shutdown
|
|
75
|
+
// where the hung command is the last thing running. The cost is that a pending call can delay exit
|
|
76
|
+
// by at most this timeout, which is the point: bounded, and short.
|
|
77
|
+
const t = setTimeout(() => reject(new Error("registry timeout")), ms);
|
|
78
|
+
Promise.resolve(promise).then(
|
|
79
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
80
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
81
|
+
);
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs = HOST_TTL_MS, log = () => {}, timeoutMs = REGISTRY_OP_TIMEOUT_MS }) {
|
|
86
|
+
let timer = null;
|
|
87
|
+
let facts = {};
|
|
88
|
+
// Set the moment `close` is entered, so a beat already in flight cannot re-HSET the row AFTER the DEL
|
|
89
|
+
// and recreate the very ghost `close` exists to remove.
|
|
90
|
+
let closed = false;
|
|
91
|
+
// Logged once per TRANSITION rather than once per beat: at four beats a minute, a Valkey outage would
|
|
92
|
+
// otherwise drown the log it is meant to serve. `notePackageKey` sets this precedent and states it.
|
|
93
|
+
let reachable = true;
|
|
94
|
+
|
|
95
|
+
const write = async (fields) => {
|
|
96
|
+
const key = hostKey(name);
|
|
97
|
+
const flat = [];
|
|
98
|
+
// A fact may be a VALUE or a THUNK. A thunk is what lets a fact that changes without a restart --
|
|
99
|
+
// the cron fingerprint after a live triggers-file edit, the live concurrency after an overlay
|
|
100
|
+
// change -- be re-read on every beat rather than frozen at the call that started the heartbeat.
|
|
101
|
+
// Without it a peer would compare against what this host believed at boot, and two hosts would see
|
|
102
|
+
// each other's fingerprint oscillate on the beat period after any edit.
|
|
103
|
+
for (const [k, v] of Object.entries(fields)) {
|
|
104
|
+
let resolved;
|
|
105
|
+
try {
|
|
106
|
+
resolved = typeof v === "function" ? v() : v;
|
|
107
|
+
} catch {
|
|
108
|
+
resolved = ""; // a fact that cannot be computed is absent, never a reason to skip the beat
|
|
109
|
+
}
|
|
110
|
+
const value = String(resolved ?? "");
|
|
111
|
+
// THE CONTENT RULE, ENFORCED HERE rather than asserted in a test. A test can only check the fields
|
|
112
|
+
// it happens to publish itself, so it could never catch the day someone adds `logsDir` at a call
|
|
113
|
+
// site. This can: a value that is path-shaped is DROPPED and named, loudly, once.
|
|
114
|
+
//
|
|
115
|
+
// The rule is mechanical rather than a blanket ban on `/`, because one legitimate value contains
|
|
116
|
+
// one: an IANA zone is `Europe/Amsterdam`. What no admissible value has is a filesystem ROOT -- a
|
|
117
|
+
// leading separator, a backslash, or a drive letter.
|
|
118
|
+
if (/^[/\\]|\\|^[A-Za-z]:[/\\]/.test(value)) {
|
|
119
|
+
log("host_registry_field_refused", { host: name, field: k, reason: "path-shaped" });
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
flat.push(k, value);
|
|
123
|
+
}
|
|
124
|
+
await bounded(redis.hset(key, ...flat), timeoutMs);
|
|
125
|
+
// PEXPIRE ON EVERY BEAT, which REVERSES this project's stated TTL rule ("set the TTL only when the
|
|
126
|
+
// key is first created, so a long window cannot push its expiry forward" -- budget.mjs). The
|
|
127
|
+
// reversal is correct because the OBJECT is different, and the distinction is worth keeping:
|
|
128
|
+
//
|
|
129
|
+
// a COUNTER whose TTL refreshes on every increment stops being a window, which is why the budget
|
|
130
|
+
// keys set it once and why a busy day must not postpone its own reset;
|
|
131
|
+
//
|
|
132
|
+
// a LEASE's expiry IS the liveness claim, so refreshing it is not a leak, it is the mechanism.
|
|
133
|
+
//
|
|
134
|
+
// The in-repo precedent is lease-shaped and two files away: `wait-state.hold` re-PEXPIREs the
|
|
135
|
+
// supersede lease `wait:key:<dedupId>` -- and only after checking the lease is still ours, which is
|
|
136
|
+
// the same ownership check this key gets for free by being named after its only writer.
|
|
137
|
+
await bounded(redis.pexpire(key, ttlMs), timeoutMs);
|
|
138
|
+
await bounded(redis.sadd(HOST_SET, name), timeoutMs);
|
|
139
|
+
};
|
|
140
|
+
|
|
141
|
+
// The beat currently in flight, so `close` can DRAIN before it deletes. A `closed` flag checked at the
|
|
142
|
+
// top of `beat` is not enough on its own: a beat that had already passed that check would land its
|
|
143
|
+
// HSET after the DEL and recreate the very ghost `close` exists to remove.
|
|
144
|
+
let inFlight = null;
|
|
145
|
+
|
|
146
|
+
const beat = async (fields = facts) => {
|
|
147
|
+
if (closed) return;
|
|
148
|
+
facts = { ...facts, ...fields };
|
|
149
|
+
try {
|
|
150
|
+
inFlight = write({ ...facts, name, beatAt: now() });
|
|
151
|
+
await inFlight;
|
|
152
|
+
if (!reachable) {
|
|
153
|
+
reachable = true;
|
|
154
|
+
log("host_registry_restored", { host: name });
|
|
155
|
+
}
|
|
156
|
+
} catch (err) {
|
|
157
|
+
if (reachable) {
|
|
158
|
+
reachable = false;
|
|
159
|
+
log("host_registry_unreachable", { host: name, reason: err?.message });
|
|
160
|
+
}
|
|
161
|
+
} finally {
|
|
162
|
+
inFlight = null;
|
|
163
|
+
}
|
|
164
|
+
};
|
|
165
|
+
|
|
166
|
+
return {
|
|
167
|
+
/** This worker's own name, so a caller never re-derives it from config and gets a different answer. */
|
|
168
|
+
self: () => name,
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* Flush this host's current facts NOW, for a caller that must be visible before it reads peers.
|
|
172
|
+
*
|
|
173
|
+
* Takes NO fields, deliberately. It used to merge them, and that quietly destroyed the thunk
|
|
174
|
+
* mechanism it depends on: a caller passing `fpCron` as a computed STRING replaced the closure the
|
|
175
|
+
* heartbeat installed, so every later beat published a frozen value and two hosts' fingerprints
|
|
176
|
+
* could drift apart again. A caller that wants a fact re-read on every beat installs a thunk once,
|
|
177
|
+
* at `start`; a caller that wants it published NOW calls this.
|
|
178
|
+
*/
|
|
179
|
+
async publish() {
|
|
180
|
+
await beat();
|
|
181
|
+
},
|
|
182
|
+
|
|
183
|
+
/**
|
|
184
|
+
* Every live host EXCEPT this one. Separate from `readLiveHosts` because "who else is there" is the
|
|
185
|
+
* question every caller actually has, and excluding self at the one place stops each of them doing it
|
|
186
|
+
* differently -- and stops a stale self-row, written by a previous process of this same host, being
|
|
187
|
+
* read as a peer that disagrees with the process that is running now.
|
|
188
|
+
*/
|
|
189
|
+
async livePeers() {
|
|
190
|
+
const res = await readLiveHosts(redis, { now, timeoutMs });
|
|
191
|
+
if (res.unreachable) return res;
|
|
192
|
+
return { hosts: res.hosts.filter((h) => h.name !== name) };
|
|
193
|
+
},
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
|
|
197
|
+
* `setTimeout` -- and `.unref()`'d so it can never hold the process open, which is the posture the
|
|
198
|
+
* three `fs.watch` watchers already take. `stop` is registered as an extraCloser beside the runtime
|
|
199
|
+
* queue, so a clean shutdown clears it before `process.exit`.
|
|
200
|
+
*/
|
|
201
|
+
async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
|
|
202
|
+
if (closed || timer) return; // a second start would leak the first interval
|
|
203
|
+
await beat({ ...fields, startedAt: now() });
|
|
204
|
+
if (closed) return; // close() landed while the first beat was in flight
|
|
205
|
+
timer = setInterval(() => void beat(), intervalMs);
|
|
206
|
+
timer.unref?.();
|
|
207
|
+
},
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* Leave. A clean shutdown DELETES the row rather than letting it expire, so the TTL only ever
|
|
211
|
+
* covers a crash and a rolling restart does not leave a ghost peer behind for ninety seconds.
|
|
212
|
+
* `wait-state.release` takes the same posture for the same reason.
|
|
213
|
+
*/
|
|
214
|
+
async close() {
|
|
215
|
+
if (closed) return; // idempotent, and the flag is also what stops an in-flight beat resurrecting the row
|
|
216
|
+
closed = true;
|
|
217
|
+
if (timer) clearInterval(timer);
|
|
218
|
+
timer = null;
|
|
219
|
+
// Drain before deleting, bounded like everything else here: an unbounded wait on a beat that is
|
|
220
|
+
// itself hung would be the shutdown hang this module's timeout exists to prevent.
|
|
221
|
+
await bounded(inFlight ?? Promise.resolve(), timeoutMs).catch(() => {});
|
|
222
|
+
try {
|
|
223
|
+
await bounded(redis.del(hostKey(name)), timeoutMs);
|
|
224
|
+
await bounded(redis.srem(HOST_SET, name), timeoutMs);
|
|
225
|
+
} catch {
|
|
226
|
+
// Fail open: the TTL is the backstop, and a ghost row costs a panel line, never a decision.
|
|
227
|
+
}
|
|
228
|
+
},
|
|
229
|
+
};
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Every live host's row, with the index pruned as it is read.
|
|
234
|
+
*
|
|
235
|
+
* THE INDEX LEAK IS HANDLED HERE, BY THE READER, and that is a decision rather than an oversight: a SET
|
|
236
|
+
* cannot expire its members, so a host that dies leaves one behind. The alternative -- no index, and a
|
|
237
|
+
* `SCAN host:h:*` instead -- was measured and refused for the `wait:held` keyspace on exactly this
|
|
238
|
+
* shape: the panel reads every second, and a keyspace scan walks HARDEST on deployments with nothing to
|
|
239
|
+
* show, because the early exit never fires. So the index exists, a member whose hash is gone is stale by
|
|
240
|
+
* definition, and the reader removes it in passing.
|
|
241
|
+
*
|
|
242
|
+
* `{ unreachable }` and `[]` are DIFFERENT ANSWERS and must never be collapsed by a caller, even where
|
|
243
|
+
* one treats them alike: "there are no other hosts" and "I could not find out" differ, and a panel that
|
|
244
|
+
* renders the second as the first tells an operator their fleet is gone when Valkey merely blinked.
|
|
245
|
+
*/
|
|
246
|
+
export async function readLiveHosts(redis, { now = () => Date.now(), timeoutMs = REGISTRY_OP_TIMEOUT_MS } = {}) {
|
|
247
|
+
let names;
|
|
248
|
+
try {
|
|
249
|
+
names = await bounded(redis.smembers(HOST_SET), timeoutMs);
|
|
250
|
+
// Inside the try, not after it: a client that answers with something non-iterable would otherwise
|
|
251
|
+
// throw straight out of a function whose contract is that every method fails open.
|
|
252
|
+
if (names !== null && names !== undefined && !Array.isArray(names)) throw new Error("smembers did not return a list");
|
|
253
|
+
} catch (err) {
|
|
254
|
+
return { unreachable: err?.message ?? "registry unreadable" };
|
|
255
|
+
}
|
|
256
|
+
const hosts = [];
|
|
257
|
+
for (const member of names ?? []) {
|
|
258
|
+
try {
|
|
259
|
+
const row = await bounded(redis.hgetall(hostKey(member)), timeoutMs);
|
|
260
|
+
if (!row || Object.keys(row).length === 0) {
|
|
261
|
+
await bounded(redis.srem(HOST_SET, member), timeoutMs).catch(() => {});
|
|
262
|
+
continue;
|
|
263
|
+
}
|
|
264
|
+
const beatAt = typeof row.beatAt === "string" && row.beatAt.trim() !== "" ? Number(row.beatAt) : NaN;
|
|
265
|
+
hosts.push({
|
|
266
|
+
...row,
|
|
267
|
+
name: row.name || member,
|
|
268
|
+
// Derived rather than stored, so the panel can say "stale 2m" about a row that still lives.
|
|
269
|
+
// A row whose clock is AHEAD of ours reads as 0 rather than negative: the difference is the
|
|
270
|
+
// skew between two machines, which is its own signal and not this field's to report.
|
|
271
|
+
staleMs: Number.isFinite(beatAt) ? Math.max(0, now() - beatAt) : null,
|
|
272
|
+
});
|
|
273
|
+
} catch {
|
|
274
|
+
// One unreadable row degrades that row, never the listing.
|
|
275
|
+
}
|
|
276
|
+
}
|
|
277
|
+
hosts.sort((a, b) => String(a.name).localeCompare(String(b.name)));
|
|
278
|
+
return { hosts };
|
|
279
|
+
}
|
package/src/image-preflight.mjs
CHANGED
|
@@ -70,7 +70,7 @@ export function makeImagePreflight({ image, spawnFn = spawn }) {
|
|
|
70
70
|
// become the push race it exists to avoid, with nothing in the run record saying so.
|
|
71
71
|
const probe = await runDocker(spawnFn, ["image", "inspect", `--format={{.Id}}${FIELD_SEP}${PI_VERSION_TEMPLATE}${FIELD_SEP}${FORGES_TEMPLATE}${FIELD_SEP}${CAPABILITIES_TEMPLATE}`, wanted], true);
|
|
72
72
|
if (probe.code === 0) {
|
|
73
|
-
const [piVersion, forges, capabilities] = parseLabels(probe.stdout);
|
|
73
|
+
const [piVersion, forges, capabilities, imageDigest] = parseLabels(probe.stdout);
|
|
74
74
|
const kind = job?.kind;
|
|
75
75
|
// Absent label => ALLOW. The polarity matters and is the opposite of what "declare your
|
|
76
76
|
// capabilities" suggests: every operator-built image predating this label (OQ-012) declares
|
|
@@ -96,7 +96,7 @@ export function makeImagePreflight({ image, spawnFn = spawn }) {
|
|
|
96
96
|
if (job?.command !== undefined && !(capabilities ?? []).includes("commands")) {
|
|
97
97
|
return { commandUnsupported: wanted, declared: capabilities ?? [] };
|
|
98
98
|
}
|
|
99
|
-
return { ok: true, image: wanted, piVersion };
|
|
99
|
+
return { ok: true, image: wanted, piVersion, imageDigest };
|
|
100
100
|
}
|
|
101
101
|
if ((await runDocker(spawnFn, ["info"])).code === 0) return { missing: wanted };
|
|
102
102
|
return { unavailable: wanted };
|
|
@@ -135,10 +135,15 @@ function parseLabels(stdout) {
|
|
|
135
135
|
// the safe answer on every field. On `capabilities` "safe" means the caller refuses a replica job,
|
|
136
136
|
// which is the same direction a genuinely unlabelled image goes.
|
|
137
137
|
const parts = String(stdout ?? "").trim().split(FIELD_SEP);
|
|
138
|
+
// Field 0 is `{{.Id}}`, the image's own digest. It has been fetched on every job since this format
|
|
139
|
+
// string had four fields and was thrown away until issue #57, which needs it to answer "are these two
|
|
140
|
+
// hosts running the same image?" -- a question `OQ-012` records as unanswerable and which turns out to
|
|
141
|
+
// cost nothing to answer, because the inspect that would have asked it already runs.
|
|
142
|
+
const imageDigest = label(parts[0]);
|
|
138
143
|
const piVersion = label(parts[1]);
|
|
139
144
|
// A label that is present but parses to nothing usable is treated as ABSENT rather than as an empty
|
|
140
145
|
// list -- on `forges` the latter would refuse every job on an image whose label was merely malformed.
|
|
141
|
-
return [piVersion, list(parts[2]), list(parts[3])];
|
|
146
|
+
return [piVersion, list(parts[2]), list(parts[3]), imageDigest];
|
|
142
147
|
}
|
|
143
148
|
|
|
144
149
|
/** One comma-separated label value as a non-empty array, or `null` when it declares nothing usable. */
|