@edgehero/pi-dispatch 1.6.1 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +7 -0
- package/package.json +4 -1
- package/src/capabilities.mjs +179 -0
- package/src/cli.mjs +74 -17
- package/src/config.mjs +83 -0
- package/src/cron.mjs +116 -4
- package/src/doctor.mjs +113 -3
- package/src/fingerprint.mjs +78 -0
- package/src/fleet-lease.mjs +179 -0
- package/src/host-registry.mjs +279 -0
- package/src/image-preflight.mjs +8 -3
- package/src/index.mjs +196 -51
- package/src/queue.mjs +130 -2
- package/src/run-history.mjs +23 -1
- package/src/run-mirror.mjs +221 -0
- package/src/schedules.mjs +67 -4
- package/src/start.mjs +240 -28
package/src/doctor.mjs
CHANGED
|
@@ -49,7 +49,7 @@ import { homedir, tmpdir } from "node:os";
|
|
|
49
49
|
import { dirname, join, delimiter } from "node:path";
|
|
50
50
|
import { fileURLToPath } from "node:url";
|
|
51
51
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
52
|
-
import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
|
|
52
|
+
import { defaultSandboxDir, defaultWorkerName, globalExtensionsEnabled } from "./config.mjs";
|
|
53
53
|
import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
|
|
54
54
|
import { WAIT_AFTER_MAX_DEFAULT_MS, afterInstantMs, parseWaitProfiles } from "./wait-for.mjs";
|
|
55
55
|
import { isForgeKind } from "./forges.mjs";
|
|
@@ -89,6 +89,7 @@ export async function runDoctor(env = process.env, deps = {}) {
|
|
|
89
89
|
out = (s) => process.stdout.write(s),
|
|
90
90
|
spawn = nodeSpawn,
|
|
91
91
|
probeValkey = defaultProbeValkey,
|
|
92
|
+
readHosts = defaultReadHosts,
|
|
92
93
|
fileExists = existsSync,
|
|
93
94
|
nodeVersion = process.versions.node,
|
|
94
95
|
// --fix (REQ-DEPLOYMENT-BOOTSTRAP): offer to run the exact fixes doctor already prints. The prompt
|
|
@@ -111,7 +112,7 @@ export async function runDoctor(env = process.env, deps = {}) {
|
|
|
111
112
|
platform = process.platform,
|
|
112
113
|
home = homedir(),
|
|
113
114
|
} = deps;
|
|
114
|
-
const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
|
|
115
|
+
const seams = { cwd, out, spawn, probeValkey, readHosts, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
|
|
115
116
|
|
|
116
117
|
let checks = await collectChecks(env, seams);
|
|
117
118
|
let failed = render(checks, out);
|
|
@@ -213,7 +214,7 @@ export async function defaultPromptFn(question, { input = process.stdin, output
|
|
|
213
214
|
* a comment.
|
|
214
215
|
*/
|
|
215
216
|
export async function collectChecks(env, seams) {
|
|
216
|
-
const { cwd, spawn, probeValkey, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
|
|
217
|
+
const { cwd, spawn, probeValkey, readHosts = defaultReadHosts, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
|
|
217
218
|
|
|
218
219
|
const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
|
|
219
220
|
const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
|
|
@@ -829,6 +830,85 @@ export async function collectChecks(env, seams) {
|
|
|
829
830
|
: {}),
|
|
830
831
|
});
|
|
831
832
|
|
|
833
|
+
// --- the fleet (issue #57) -------------------------------------------------------------------------
|
|
834
|
+
//
|
|
835
|
+
// Every line here is gated on a peer actually existing, so a single-host deployment's output is
|
|
836
|
+
// byte-identical. And every one is a WARN rather than a failure, with one exception noted below: this
|
|
837
|
+
// command runs on ONE machine and must not refuse a deployment for a condition that machine cannot fix.
|
|
838
|
+
// This host's own image id, read through the same seam every other docker probe here uses. Only when
|
|
839
|
+
// the image is actually present -- an absent one is already reported above, and a second line saying
|
|
840
|
+
// its digest is unknown would be noise on a fault the operator has been told about.
|
|
841
|
+
const fleet = await readHosts(valkeyUrl);
|
|
842
|
+
const peers = (fleet.hosts ?? []).filter((h) => h.name !== workerNameOf(env));
|
|
843
|
+
// Read only when there is a peer to compare against. Every line below is gated on a peer existing, and
|
|
844
|
+
// the SUBPROCESS has to be too: otherwise every `doctor` run on every single-host deployment spawns an
|
|
845
|
+
// extra docker call whose answer nothing reads.
|
|
846
|
+
const imageDigest =
|
|
847
|
+
peers.length > 0 && imageCode === 0
|
|
848
|
+
? (await runCmdCapture(spawn, "docker", ["image", "inspect", "--format={{.Id}}", jobImage])).output.trim() || null
|
|
849
|
+
: null;
|
|
850
|
+
if (peers.length > 0) {
|
|
851
|
+
const mine = workerNameOf(env);
|
|
852
|
+
checks.push({ ok: true, label: `Fleet: ${peers.length + 1} worker${peers.length === 0 ? "" : "s"} (${[mine, ...peers.map((h) => h.name)].sort().join(", ")})` });
|
|
853
|
+
|
|
854
|
+
// The one thing that is silently WRONG rather than merely undeclared. Without a declared name this
|
|
855
|
+
// host enqueues its own folder work to the SHARED queue, where a peer that has no such folder can
|
|
856
|
+
// pop it -- so the routing that makes a fleet safe is simply off, and nothing else says so.
|
|
857
|
+
if (!env.PI_WORKER_NAME) {
|
|
858
|
+
checks.push({
|
|
859
|
+
ok: false,
|
|
860
|
+
warn: true,
|
|
861
|
+
label: "This worker has peers but no PI_WORKER_NAME, so host routing is OFF here",
|
|
862
|
+
fix: "set PI_WORKER_NAME in this host's .env and restart: without it, this host's folder work is enqueued where any host can pop it, and its records carry a hostname it never chose",
|
|
863
|
+
});
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
// Two hosts on two builds of one tag is the failure Gap 6 names: same flow, different behaviour,
|
|
867
|
+
// undebuggable. A WARN and never a failure, because `{{.Id}}` is the LOCAL image id -- two
|
|
868
|
+
// independent builds of one Dockerfile differ, and under docker's containerd image store it is the
|
|
869
|
+
// manifest digest rather than the config digest, so a mixed-store fleet disagrees about identical
|
|
870
|
+
// content. Suspicious, never wrong.
|
|
871
|
+
const digests = new Set(peers.map((h) => h.imageDigest).filter(Boolean));
|
|
872
|
+
if (digests.size > 0 && imageDigest && !digests.has(imageDigest)) {
|
|
873
|
+
checks.push({
|
|
874
|
+
ok: false,
|
|
875
|
+
warn: true,
|
|
876
|
+
label: `Job image digest differs from ${peers.length === 1 ? "the other host" : "other hosts"}`,
|
|
877
|
+
fix: "rebuild or re-pull so every host runs the same image; digests are identical only when both hosts pulled one tag from one registry, so two local builds differ legitimately",
|
|
878
|
+
});
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
// A cron PATTERN carries no timezone and resolves in each worker's LOCAL time, so one pattern is two
|
|
882
|
+
// different instants on two hosts in two zones -- and the cron gate refuses that divergence rather
|
|
883
|
+
// than letting it drift, which is why this reads as an explanation for a refusal an operator has
|
|
884
|
+
// probably already met.
|
|
885
|
+
const zones = new Set([Intl.DateTimeFormat().resolvedOptions().timeZone, ...peers.map((h) => h.tz).filter(Boolean)]);
|
|
886
|
+
if (zones.size > 1) {
|
|
887
|
+
checks.push({
|
|
888
|
+
ok: false,
|
|
889
|
+
warn: true,
|
|
890
|
+
label: `Hosts disagree about the timezone (${[...zones].sort().join(", ")}), so one cron pattern is two different instants`,
|
|
891
|
+
fix: "set the same TZ on every host: a cron trigger carries no timezone of its own, so cron reconcile refuses while they disagree",
|
|
892
|
+
});
|
|
893
|
+
}
|
|
894
|
+
|
|
895
|
+
// Clocks. The registry's own heartbeats are the measurement, and skew matters here beyond tidiness:
|
|
896
|
+
// every hold clock, every TTL and the UTC day boundary the budget windows key on are read against
|
|
897
|
+
// whichever host is looking.
|
|
898
|
+
const skewed = peers.filter((h) => Number.isFinite(h.staleMs) && h.staleMs > 5 * 60_000);
|
|
899
|
+
if (skewed.length > 0) {
|
|
900
|
+
checks.push({
|
|
901
|
+
ok: false,
|
|
902
|
+
warn: true,
|
|
903
|
+
label: `${skewed.length} host row${skewed.length === 1 ? " is" : "s are"} stale by more than five minutes (${skewed.map((h) => h.name).join(", ")})`,
|
|
904
|
+
fix: "check that those workers are running and that the clocks agree -- a stale row is either a dead worker or a skewed clock, and both matter",
|
|
905
|
+
});
|
|
906
|
+
}
|
|
907
|
+
} else if (fleet.unreachable) {
|
|
908
|
+
// Said, rather than silently absent: "no peers" and "could not ask" are different facts.
|
|
909
|
+
checks.push({ ok: true, label: `Fleet: could not read the host registry (${fleet.unreachable})` });
|
|
910
|
+
}
|
|
911
|
+
|
|
832
912
|
const keys = PROVIDER_KEYS[provider] ?? [`${provider.toUpperCase()}_API_KEY`];
|
|
833
913
|
let keyOk = keys.some((k) => (env[k] ?? "").trim().length > 0);
|
|
834
914
|
let keyNote = "";
|
|
@@ -2212,6 +2292,36 @@ function parseGhTokenScopes(output) {
|
|
|
2212
2292
|
* error handler is attached, so a down Valkey is reported as one ✗ line — not the ioredis stack traces
|
|
2213
2293
|
* a BullMQ Queue's internal client would dump. Reuses `parseConnection`'s fail-fast options (cli.mjs:88).
|
|
2214
2294
|
*/
|
|
2295
|
+
/**
|
|
2296
|
+
* The fleet's registry rows (issue #57), through a fail-fast client that is always disconnected.
|
|
2297
|
+
*
|
|
2298
|
+
* A SEAM rather than a direct import so the multi-host checks are testable with no Valkey at all, which
|
|
2299
|
+
* is the posture every other network-touching check here already takes. Never throws: a fleet this
|
|
2300
|
+
* command cannot see is a fleet it says nothing about, not a doctor that fails.
|
|
2301
|
+
*/
|
|
2302
|
+
/** This host's name as the worker computes it, so doctor and the worker cannot disagree about who "I" am. */
|
|
2303
|
+
function workerNameOf(env) {
|
|
2304
|
+
return env.PI_WORKER_NAME || defaultWorkerName();
|
|
2305
|
+
}
|
|
2306
|
+
|
|
2307
|
+
async function defaultReadHosts(url) {
|
|
2308
|
+
try {
|
|
2309
|
+
const { Redis } = await import("ioredis");
|
|
2310
|
+
const { parseConnection } = await import("./connection.mjs");
|
|
2311
|
+
const { readLiveHosts } = await import("./host-registry.mjs");
|
|
2312
|
+
const client = new Redis({ ...parseConnection(url, { failFast: true }), lazyConnect: true });
|
|
2313
|
+
client.on("error", () => {});
|
|
2314
|
+
try {
|
|
2315
|
+
await client.connect();
|
|
2316
|
+
return await readLiveHosts(client);
|
|
2317
|
+
} finally {
|
|
2318
|
+
client.disconnect();
|
|
2319
|
+
}
|
|
2320
|
+
} catch (err) {
|
|
2321
|
+
return { unreachable: err?.message ?? "registry unreadable" };
|
|
2322
|
+
}
|
|
2323
|
+
}
|
|
2324
|
+
|
|
2215
2325
|
async function defaultProbeValkey(url) {
|
|
2216
2326
|
const { Redis } = await import("ioredis");
|
|
2217
2327
|
const { parseConnection } = await import("./connection.mjs");
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Canonical fingerprints of configuration, so two hosts can find out whether they agree (issue #57).
|
|
3
|
+
*
|
|
4
|
+
* Pure, and importing nothing but `node:crypto`: a fingerprint must be computable in a tier-1 test with
|
|
5
|
+
* no queue, no Valkey and no filesystem, exactly as `parseTriggers` is.
|
|
6
|
+
*
|
|
7
|
+
* WHY A HASH RATHER THAN THE VALUE. Two of these travel through the host registry, whose content rule
|
|
8
|
+
* refuses paths outright (`INT-HOST-REGISTRY-CONTRACT`) -- and a cron schedule set legitimately contains
|
|
9
|
+
* `run.folder`, `run.task` and secret NAMES. Hashing is what makes them admissible: the registry carries
|
|
10
|
+
* proof of agreement rather than the thing agreed on. This is the inverse of `scopeKeyPrefix`'s argument,
|
|
11
|
+
* which hashes a scope because it may contain `:` and `/`; here the reason is disclosure, not syntax.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { createHash } from "node:crypto";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* A stable 16-hex digest of any JSON-able value.
|
|
18
|
+
*
|
|
19
|
+
* Object keys are sorted RECURSIVELY, and that is load-bearing rather than tidy. `normalizeCronSchedule`
|
|
20
|
+
* builds its `data` object as a literal, so its key ORDER is a property of the worker's source: two hosts
|
|
21
|
+
* mid-upgrade would otherwise canonicalise the same file differently and refuse each other for the whole
|
|
22
|
+
* rollout. Sorting removes the spurious disagreement while leaving the real one -- a genuinely new field
|
|
23
|
+
* still changes the hash, which is correct, because the stored repeatable's data really did change.
|
|
24
|
+
*
|
|
25
|
+
* Sixteen hex, the `scopeKeyPrefix` and `localJobId` idiom, because this is compared and displayed rather
|
|
26
|
+
* than used as a security boundary.
|
|
27
|
+
*/
|
|
28
|
+
export function fingerprint(value) {
|
|
29
|
+
return createHash("sha256").update(canonical(value)).digest("hex").slice(0, 16);
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function canonical(value) {
|
|
33
|
+
if (value === null || typeof value !== "object") return JSON.stringify(value ?? null);
|
|
34
|
+
if (Array.isArray(value)) return `[${value.map(canonical).join(",")}]`;
|
|
35
|
+
const keys = Object.keys(value).sort();
|
|
36
|
+
return `{${keys.map((k) => `${JSON.stringify(k)}:${canonical(value[k])}`).join(",")}}`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* The fingerprint of a worker's cron schedule set, or `null` when this worker has no opinion.
|
|
41
|
+
*
|
|
42
|
+
* ABSTAIN VERSUS OPINE IS THE SUBTLE PART, and conflating the two would leave the bug this gate exists to
|
|
43
|
+
* close. `loadSchedules` returns `[]` for two different states:
|
|
44
|
+
*
|
|
45
|
+
* - `PI_TRIGGERS_FILE` unset, which means CRON IS DISABLED on this host. Such a worker has no view of
|
|
46
|
+
* what should be scheduled, so it must never be able to disagree with one that does. It ABSTAINS.
|
|
47
|
+
* - a triggers file that is present and declares zero cron entries. That is an OPINION -- "there should
|
|
48
|
+
* be no schedulers" -- and it is this bug's purest form: today, deleting the last cron trigger on one
|
|
49
|
+
* host prunes the whole fleet's schedulers through the file-watch path.
|
|
50
|
+
*
|
|
51
|
+
* So `null` in means abstain (`null` out); an empty ARRAY is a real fingerprint.
|
|
52
|
+
*
|
|
53
|
+
* The `tz` rides the hash because a cron PATTERN carries no timezone: `triggers.json` has no `tz` field on
|
|
54
|
+
* a cron entry, and BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's
|
|
55
|
+
* LOCAL system time. On one host that is exactly what an operator means; on two hosts in different zones
|
|
56
|
+
* the same pattern is two different instants, with nothing anywhere saying so. Including the zone makes
|
|
57
|
+
* that a disagreement the gate can see. (`pause-windows.json` has carried an explicit `tz` since it
|
|
58
|
+
* shipped and is already fleet-correct; the asymmetry is why this one needs stating.)
|
|
59
|
+
*
|
|
60
|
+
* Fingerprinted over the NORMALIZED schedules rather than the file's bytes, deliberately. Bytes diverge on
|
|
61
|
+
* whitespace, on key order, and on every webhook trigger the worker does not own -- so two hosts differing
|
|
62
|
+
* only in a `label` rule would freeze cron forever over a difference that cannot affect it. The normalized
|
|
63
|
+
* set diverges exactly when the reconcile INPUTS diverge, which is the property that makes reconcile
|
|
64
|
+
* idempotent in the first place.
|
|
65
|
+
*/
|
|
66
|
+
export function cronFingerprint(schedules, { tz } = {}) {
|
|
67
|
+
// Takes the AUTHORED shape (`schedules.mjs -> authoredCron`), never the placement-resolved one.
|
|
68
|
+
if (schedules === null || schedules === undefined) return null;
|
|
69
|
+
return fingerprint({
|
|
70
|
+
tz: tz ?? "",
|
|
71
|
+
// EVERY field of an authored entry, projected explicitly. An earlier shape listed the keys of a
|
|
72
|
+
// NORMALIZED schedule (`name`, `data`, `opts`), which are all `undefined` on an authored one -- so
|
|
73
|
+
// the hash saw only the id and the pattern, and an operator changing `run.folder`, `run.flow` or
|
|
74
|
+
// `run.image` on one host and not the other passed the gate silently. That broke this module's own
|
|
75
|
+
// stated invariant, that the set diverges exactly when the reconcile inputs diverge.
|
|
76
|
+
schedules: schedules.map((s) => ({ schedulerId: s.schedulerId, pattern: s.pattern, run: s.run })),
|
|
77
|
+
});
|
|
78
|
+
}
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fleet-wide leases for the two bounds that stopped meaning what they say when a second host appeared
|
|
3
|
+
* (issue #57).
|
|
4
|
+
*
|
|
5
|
+
* `PI_WAIT_CHECK_SLOTS` and a `scoped-limits.json` row's `concurrent` are both enforced by
|
|
6
|
+
* `makeInFlight()`, an in-process `Map`. One worker per docker daemon made that correct; two hosts make
|
|
7
|
+
* it a bound that MULTIPLIES BY THE OPERATOR'S DEPLOYMENT SHAPE, which is not a bound. Four hosts with
|
|
8
|
+
* `PI_WAIT_CHECK_SLOTS=1` run four concurrent checks against one Jira; four hosts with
|
|
9
|
+
* `{"scope":"acme/web","concurrent":1}` run four paid containers on one repository.
|
|
10
|
+
*
|
|
11
|
+
* WHAT THIS OWES `OQ-008` AND `DES-CONCURRENCY-3`. Both refuse a Redis-held in-flight count, in terms:
|
|
12
|
+
* "a Redis-held count would survive a crash WRONGLY -- a claim for a container the reaper just killed,
|
|
13
|
+
* demanding TTL/heartbeat machinery, a second source of truth about what is running". Every clause of
|
|
14
|
+
* that is about a CONTAINER, and the two leases here answer it differently:
|
|
15
|
+
*
|
|
16
|
+
* The CHECK lease claims no container. It claims a subprocess this same process spawned, bounded by
|
|
17
|
+
* `PI_WAIT_CHECK_TIMEOUT_MS`, holding no folder and spending no money. Three properties invert. What a
|
|
18
|
+
* stale claim costs: a container claim is a folder mutex nobody holds and a job that never runs, while
|
|
19
|
+
* a check claim is one check deferred by at most the TTL. What contradicts it: the reaper is a second
|
|
20
|
+
* source of truth about containers and runs at boot with authority, while nothing enumerates,
|
|
21
|
+
* inspects or reaps a check. And how long it can be wrong: a container has no natural expiry, while a
|
|
22
|
+
* check has a hard, configured, small timeout, so the TTL is DERIVED rather than guessed.
|
|
23
|
+
*
|
|
24
|
+
* The SCOPE claim really is for a container, so the refusal lands squarely -- and the answer is the
|
|
25
|
+
* boot reaper. It establishes, at boot, that this host holds no `pi-job-*` containers, so a claim
|
|
26
|
+
* whose value names this host is a claim for a container that no longer exists, and deleting it is not
|
|
27
|
+
* a second source of truth: it is the SAME source writing down what it just established. Making the
|
|
28
|
+
* reaper the claim's owner removes the contradiction rather than arguing around it.
|
|
29
|
+
*
|
|
30
|
+
* N INDEPENDENT KEYS, NEVER ONE COUNTER. A counter with one TTL loses every claim when it expires and
|
|
31
|
+
* leaks a permanent `+1` on a crash; N `SET NX PX` keys mean a lost release costs exactly one slot for
|
|
32
|
+
* exactly the TTL and never the whole semaphore. `wait:key:<dedupId>` is the in-repo precedent.
|
|
33
|
+
*
|
|
34
|
+
* And a property the in-process map does not have: RELEASE IS IDEMPOTENT here, because it deletes only a
|
|
35
|
+
* key whose value is still ours. `makeInFlight().release` clamps at zero but a double release on a
|
|
36
|
+
* `concurrent: 2` scope frees the other holder's slot.
|
|
37
|
+
*/
|
|
38
|
+
|
|
39
|
+
/** Slot keys. Both live under a prefix an operator can see whole with one `KEYS`. */
|
|
40
|
+
export const checkSlotKey = (i) => `wait:check:${i}`;
|
|
41
|
+
export const scopeSlotKey = (hash, i) => `slot:s:${hash}:${i}`;
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Build a lease over `slots` numbered keys.
|
|
45
|
+
*
|
|
46
|
+
* `holder` is `<workerName>#<jobId>`: the worker-name charset excludes `#`, so the value decomposes and
|
|
47
|
+
* the boot sweep can recognise its own claims without a second index to keep in step.
|
|
48
|
+
*
|
|
49
|
+
* FAIL OPEN, and in the GRANTING direction. A Valkey fault must never be able to wedge every wait in a
|
|
50
|
+
* deployment or stop every scoped job; the in-process bound is still there underneath, so failing open
|
|
51
|
+
* degrades the fleet bound to the per-host one, which is exactly the behaviour before this existed.
|
|
52
|
+
*/
|
|
53
|
+
export function makeFleetLease({ redis, holderPrefix, keyFor, ttlMs, now = () => Date.now(), log = () => {}, timeoutMs = 2_000 }) {
|
|
54
|
+
// Bounded for `host-registry.mjs`'s reason, which applies to every module sharing this client:
|
|
55
|
+
// `maxRetriesPerRequest: null` makes a command against an unreachable server QUEUE rather than reject,
|
|
56
|
+
// so a try/catch around it catches nothing and an outage would hang the gate rather than fail it.
|
|
57
|
+
const bounded = (p) =>
|
|
58
|
+
new Promise((resolve, reject) => {
|
|
59
|
+
const t = setTimeout(() => reject(new Error("lease timeout")), timeoutMs);
|
|
60
|
+
Promise.resolve(p).then(
|
|
61
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
62
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
63
|
+
);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
/**
|
|
68
|
+
* Take one of `slots`, or `null` when they are all held. Returns a handle whose `release` and
|
|
69
|
+
* `refresh` act only on the key this call actually won.
|
|
70
|
+
*
|
|
71
|
+
* Probing starts at `hash(id) mod slots` rather than at 0. Without the rotation every host tries
|
|
72
|
+
* index 0 first, so a host can sit behind a busy slot while a free one exists two along -- a
|
|
73
|
+
* starvation that looks exactly like the capacity shortage the bound is meant to report.
|
|
74
|
+
*/
|
|
75
|
+
async acquire(id, { slots, keyArgs = [], ttlMs: perCall } = {}) {
|
|
76
|
+
if (!Number.isFinite(slots) || slots < 1) return { ok: true, release: async () => {}, refresh: async () => true };
|
|
77
|
+
const holder = `${holderPrefix}#${id}`;
|
|
78
|
+
const start = Math.abs(hashCode(String(id))) % slots;
|
|
79
|
+
for (let n = 0; n < slots; n++) {
|
|
80
|
+
const key = keyFor(...keyArgs, (start + n) % slots);
|
|
81
|
+
try {
|
|
82
|
+
const won = await bounded(redis.set(key, holder, "PX", perCall ?? ttlMs, "NX"));
|
|
83
|
+
if (!won) continue;
|
|
84
|
+
return {
|
|
85
|
+
ok: true,
|
|
86
|
+
key,
|
|
87
|
+
async release() {
|
|
88
|
+
try {
|
|
89
|
+
// Release-if-MINE, which is what makes this idempotent where the in-process map is
|
|
90
|
+
// not: a double release cannot free another holder's slot, because the second call
|
|
91
|
+
// finds a value that is no longer ours.
|
|
92
|
+
if ((await bounded(redis.get(key))) === holder) await bounded(redis.del(key));
|
|
93
|
+
} catch {
|
|
94
|
+
// The TTL is the backstop. A lost release costs one slot for one TTL.
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
async refresh(nextMs = ttlMs) {
|
|
98
|
+
try {
|
|
99
|
+
if ((await bounded(redis.get(key))) !== holder) return false;
|
|
100
|
+
await bounded(redis.pexpire(key, nextMs));
|
|
101
|
+
return true;
|
|
102
|
+
} catch {
|
|
103
|
+
return true; // a blip is not a reason to believe we lost a slot we hold
|
|
104
|
+
}
|
|
105
|
+
},
|
|
106
|
+
};
|
|
107
|
+
} catch (err) {
|
|
108
|
+
// Granting is the safe direction: the in-process bound is still underneath, so this
|
|
109
|
+
// degrades the fleet-wide ceiling to the per-host one rather than to nothing.
|
|
110
|
+
log("fleet_lease_unavailable", { reason: err?.message });
|
|
111
|
+
return { ok: true, degraded: true, release: async () => {}, refresh: async () => true };
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
return null;
|
|
115
|
+
},
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** A small, stable, non-cryptographic spread for the slot rotation. Not a key, so not a digest. */
|
|
120
|
+
function hashCode(s) {
|
|
121
|
+
let h = 0;
|
|
122
|
+
for (let i = 0; i < s.length; i++) h = (h * 31 + s.charCodeAt(i)) | 0;
|
|
123
|
+
return h;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Delete every scope slot this host still claims, at boot, right after the container reaper.
|
|
128
|
+
*
|
|
129
|
+
* THIS IS THE ANSWER TO `OQ-008`, and its correctness rests entirely on one precondition: the reaper
|
|
130
|
+
* must have ENUMERATED. `makeReaper` catches its own `docker ps` failure and logs `reaper_skipped`, and
|
|
131
|
+
* on that path nothing was listed and nothing reaped -- so this host has NOT established that it holds
|
|
132
|
+
* no containers, and its claims may be for containers that are still running. Sweeping then would free
|
|
133
|
+
* slots for another host to start more alongside them, which is a money overrun rather than a tidy-up.
|
|
134
|
+
* `makeInFlight`'s own escape ("a state where no NEW container can start either") does not transfer,
|
|
135
|
+
* because the sweep frees slots for a DIFFERENT machine.
|
|
136
|
+
*
|
|
137
|
+
* Driven by config rather than by a scan: `scoped-limits.json` enumerates every scope that can carry a
|
|
138
|
+
* claim and `concurrent` bounds the index, so this is `sum(concurrent)` GETs -- typically under twenty.
|
|
139
|
+
* No `KEYS`, no `SCAN`, and no index set to leak.
|
|
140
|
+
*/
|
|
141
|
+
export function makeScopeClaimSweeper({ redis, workerName, limits, log = () => {}, timeoutMs = 2_000 }) {
|
|
142
|
+
return async function sweep({ reaped }) {
|
|
143
|
+
if (!reaped) {
|
|
144
|
+
log("scope_claims_sweep_skipped", { reason: "reaper-skipped" });
|
|
145
|
+
return { swept: 0, skipped: true };
|
|
146
|
+
}
|
|
147
|
+
const prefix = `${workerName}#`;
|
|
148
|
+
let swept = 0;
|
|
149
|
+
for (const row of limits ?? []) {
|
|
150
|
+
const n = Number(row?.concurrent);
|
|
151
|
+
if (!Number.isFinite(n) || n < 1 || !row?.hash) continue;
|
|
152
|
+
for (let i = 0; i < n; i++) {
|
|
153
|
+
const key = scopeSlotKey(row.hash, i);
|
|
154
|
+
try {
|
|
155
|
+
const held = await withTimeout(redis.get(key), timeoutMs);
|
|
156
|
+
if (typeof held === "string" && held.startsWith(prefix)) {
|
|
157
|
+
await withTimeout(redis.del(key), timeoutMs);
|
|
158
|
+
swept++;
|
|
159
|
+
}
|
|
160
|
+
} catch {
|
|
161
|
+
// Best-effort by contract: this is an OPTIMISATION over the TTL, never the mechanism, so a
|
|
162
|
+
// fault here costs at most one TTL of a stale claim and must never block boot.
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
if (swept > 0) log("scope_claims_swept", { count: swept });
|
|
167
|
+
return { swept, skipped: false };
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
function withTimeout(p, ms) {
|
|
172
|
+
return new Promise((resolve, reject) => {
|
|
173
|
+
const t = setTimeout(() => reject(new Error("lease timeout")), ms);
|
|
174
|
+
Promise.resolve(p).then(
|
|
175
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
176
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
177
|
+
);
|
|
178
|
+
});
|
|
179
|
+
}
|