@edgehero/pi-dispatch 1.6.1 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/doctor.mjs CHANGED
@@ -49,7 +49,7 @@ import { homedir, tmpdir } from "node:os";
49
49
  import { dirname, join, delimiter } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
- import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
52
+ import { defaultSandboxDir, defaultWorkerName, globalExtensionsEnabled } from "./config.mjs";
53
53
  import { canonicalScope, parseScopedLimits } from "./scoped-limits.mjs";
54
54
  import { WAIT_AFTER_MAX_DEFAULT_MS, afterInstantMs, parseWaitProfiles } from "./wait-for.mjs";
55
55
  import { isForgeKind } from "./forges.mjs";
@@ -89,6 +89,7 @@ export async function runDoctor(env = process.env, deps = {}) {
89
89
  out = (s) => process.stdout.write(s),
90
90
  spawn = nodeSpawn,
91
91
  probeValkey = defaultProbeValkey,
92
+ readHosts = defaultReadHosts,
92
93
  fileExists = existsSync,
93
94
  nodeVersion = process.versions.node,
94
95
  // --fix (REQ-DEPLOYMENT-BOOTSTRAP): offer to run the exact fixes doctor already prints. The prompt
@@ -111,7 +112,7 @@ export async function runDoctor(env = process.env, deps = {}) {
111
112
  platform = process.platform,
112
113
  home = homedir(),
113
114
  } = deps;
114
- const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
115
+ const seams = { cwd, out, spawn, probeValkey, readHosts, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
115
116
 
116
117
  let checks = await collectChecks(env, seams);
117
118
  let failed = render(checks, out);
@@ -213,7 +214,7 @@ export async function defaultPromptFn(question, { input = process.stdin, output
213
214
  * a comment.
214
215
  */
215
216
  export async function collectChecks(env, seams) {
216
- const { cwd, spawn, probeValkey, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
217
+ const { cwd, spawn, probeValkey, readHosts = defaultReadHosts, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
217
218
 
218
219
  const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
219
220
  const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
@@ -829,6 +830,85 @@ export async function collectChecks(env, seams) {
829
830
  : {}),
830
831
  });
831
832
 
833
+ // --- the fleet (issue #57) -------------------------------------------------------------------------
834
+ //
835
+ // Every line here is gated on a peer actually existing, so a single-host deployment's output is
836
+ // byte-identical. And every one is a WARN rather than a failure, with one exception noted below: this
837
+ // command runs on ONE machine and must not refuse a deployment for a condition that machine cannot fix.
838
+ // This host's own image id, read through the same seam every other docker probe here uses. Only when
839
+ // the image is actually present -- an absent one is already reported above, and a second line saying
840
+ // its digest is unknown would be noise on a fault the operator has been told about.
841
+ const fleet = await readHosts(valkeyUrl);
842
+ const peers = (fleet.hosts ?? []).filter((h) => h.name !== workerNameOf(env));
843
+ // Read only when there is a peer to compare against. Every line below is gated on a peer existing, and
844
+ // the SUBPROCESS has to be too: otherwise every `doctor` run on every single-host deployment spawns an
845
+ // extra docker call whose answer nothing reads.
846
+ const imageDigest =
847
+ peers.length > 0 && imageCode === 0
848
+ ? (await runCmdCapture(spawn, "docker", ["image", "inspect", "--format={{.Id}}", jobImage])).output.trim() || null
849
+ : null;
850
+ if (peers.length > 0) {
851
+ const mine = workerNameOf(env);
852
+ checks.push({ ok: true, label: `Fleet: ${peers.length + 1} worker${peers.length === 0 ? "" : "s"} (${[mine, ...peers.map((h) => h.name)].sort().join(", ")})` });
853
+
854
+ // The one thing that is silently WRONG rather than merely undeclared. Without a declared name this
855
+ // host enqueues its own folder work to the SHARED queue, where a peer that has no such folder can
856
+ // pop it -- so the routing that makes a fleet safe is simply off, and nothing else says so.
857
+ if (!env.PI_WORKER_NAME) {
858
+ checks.push({
859
+ ok: false,
860
+ warn: true,
861
+ label: "This worker has peers but no PI_WORKER_NAME, so host routing is OFF here",
862
+ fix: "set PI_WORKER_NAME in this host's .env and restart: without it, this host's folder work is enqueued where any host can pop it, and its records carry a hostname it never chose",
863
+ });
864
+ }
865
+
866
+ // Two hosts on two builds of one tag is the failure Gap 6 names: same flow, different behaviour,
867
+ // undebuggable. A WARN and never a failure, because `{{.Id}}` is the LOCAL image id -- two
868
+ // independent builds of one Dockerfile differ, and under docker's containerd image store it is the
869
+ // manifest digest rather than the config digest, so a mixed-store fleet disagrees about identical
870
+ // content. Suspicious, never wrong.
871
+ const digests = new Set(peers.map((h) => h.imageDigest).filter(Boolean));
872
+ if (digests.size > 0 && imageDigest && !digests.has(imageDigest)) {
873
+ checks.push({
874
+ ok: false,
875
+ warn: true,
876
+ label: `Job image digest differs from ${peers.length === 1 ? "the other host" : "other hosts"}`,
877
+ fix: "rebuild or re-pull so every host runs the same image; digests are identical only when both hosts pulled one tag from one registry, so two local builds differ legitimately",
878
+ });
879
+ }
880
+
881
+ // A cron PATTERN carries no timezone and resolves in each worker's LOCAL time, so one pattern is two
882
+ // different instants on two hosts in two zones -- and the cron gate refuses that divergence rather
883
+ // than letting it drift, which is why this reads as an explanation for a refusal an operator has
884
+ // probably already met.
885
+ const zones = new Set([Intl.DateTimeFormat().resolvedOptions().timeZone, ...peers.map((h) => h.tz).filter(Boolean)]);
886
+ if (zones.size > 1) {
887
+ checks.push({
888
+ ok: false,
889
+ warn: true,
890
+ label: `Hosts disagree about the timezone (${[...zones].sort().join(", ")}), so one cron pattern is two different instants`,
891
+ fix: "set the same TZ on every host: a cron trigger carries no timezone of its own, so cron reconcile refuses while they disagree",
892
+ });
893
+ }
894
+
895
+ // Clocks. The registry's own heartbeats are the measurement, and skew matters here beyond tidiness:
896
+ // every hold clock, every TTL and the UTC day boundary the budget windows key on are read against
897
+ // whichever host is looking.
898
+ const skewed = peers.filter((h) => Number.isFinite(h.staleMs) && h.staleMs > 5 * 60_000);
899
+ if (skewed.length > 0) {
900
+ checks.push({
901
+ ok: false,
902
+ warn: true,
903
+ label: `${skewed.length} host row${skewed.length === 1 ? " is" : "s are"} stale by more than five minutes (${skewed.map((h) => h.name).join(", ")})`,
904
+ fix: "check that those workers are running and that the clocks agree -- a stale row is either a dead worker or a skewed clock, and both matter",
905
+ });
906
+ }
907
+ } else if (fleet.unreachable) {
908
+ // Said, rather than silently absent: "no peers" and "could not ask" are different facts.
909
+ checks.push({ ok: true, label: `Fleet: could not read the host registry (${fleet.unreachable})` });
910
+ }
911
+
832
912
  const keys = PROVIDER_KEYS[provider] ?? [`${provider.toUpperCase()}_API_KEY`];
833
913
  let keyOk = keys.some((k) => (env[k] ?? "").trim().length > 0);
834
914
  let keyNote = "";
@@ -2212,6 +2292,36 @@ function parseGhTokenScopes(output) {
2212
2292
  * error handler is attached, so a down Valkey is reported as one ✗ line — not the ioredis stack traces
2213
2293
  * a BullMQ Queue's internal client would dump. Reuses `parseConnection`'s fail-fast options (cli.mjs:88).
2214
2294
  */
2295
+ /**
2296
+ * The fleet's registry rows (issue #57), through a fail-fast client that is always disconnected.
2297
+ *
2298
+ * A SEAM rather than a direct import so the multi-host checks are testable with no Valkey at all, which
2299
+ * is the posture every other network-touching check here already takes. Never throws: a fleet this
2300
+ * command cannot see is a fleet it says nothing about, not a doctor that fails.
2301
+ */
2302
+ /** This host's name as the worker computes it, so doctor and the worker cannot disagree about who "I" am. */
2303
+ function workerNameOf(env) {
2304
+ return env.PI_WORKER_NAME || defaultWorkerName();
2305
+ }
2306
+
2307
+ async function defaultReadHosts(url) {
2308
+ try {
2309
+ const { Redis } = await import("ioredis");
2310
+ const { parseConnection } = await import("./connection.mjs");
2311
+ const { readLiveHosts } = await import("./host-registry.mjs");
2312
+ const client = new Redis({ ...parseConnection(url, { failFast: true }), lazyConnect: true });
2313
+ client.on("error", () => {});
2314
+ try {
2315
+ await client.connect();
2316
+ return await readLiveHosts(client);
2317
+ } finally {
2318
+ client.disconnect();
2319
+ }
2320
+ } catch (err) {
2321
+ return { unreachable: err?.message ?? "registry unreadable" };
2322
+ }
2323
+ }
2324
+
2215
2325
  async function defaultProbeValkey(url) {
2216
2326
  const { Redis } = await import("ioredis");
2217
2327
  const { parseConnection } = await import("./connection.mjs");
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Canonical fingerprints of configuration, so two hosts can find out whether they agree (issue #57).
3
+ *
4
+ * Pure, and importing nothing but `node:crypto`: a fingerprint must be computable in a tier-1 test with
5
+ * no queue, no Valkey and no filesystem, exactly as `parseTriggers` is.
6
+ *
7
+ * WHY A HASH RATHER THAN THE VALUE. Two of these travel through the host registry, whose content rule
8
+ * refuses paths outright (`INT-HOST-REGISTRY-CONTRACT`) -- and a cron schedule set legitimately contains
9
+ * `run.folder`, `run.task` and secret NAMES. Hashing is what makes them admissible: the registry carries
10
+ * proof of agreement rather than the thing agreed on. This is the inverse of `scopeKeyPrefix`'s argument,
11
+ * which hashes a scope because it may contain `:` and `/`; here the reason is disclosure, not syntax.
12
+ */
13
+
14
+ import { createHash } from "node:crypto";
15
+
16
+ /**
17
+ * A stable 16-hex digest of any JSON-able value.
18
+ *
19
+ * Object keys are sorted RECURSIVELY, and that is load-bearing rather than tidy. `normalizeCronSchedule`
20
+ * builds its `data` object as a literal, so its key ORDER is a property of the worker's source: two hosts
21
+ * mid-upgrade would otherwise canonicalise the same file differently and refuse each other for the whole
22
+ * rollout. Sorting removes the spurious disagreement while leaving the real one -- a genuinely new field
23
+ * still changes the hash, which is correct, because the stored repeatable's data really did change.
24
+ *
25
+ * Sixteen hex, the `scopeKeyPrefix` and `localJobId` idiom, because this is compared and displayed rather
26
+ * than used as a security boundary.
27
+ */
28
+ export function fingerprint(value) {
29
+ return createHash("sha256").update(canonical(value)).digest("hex").slice(0, 16);
30
+ }
31
+
32
+ function canonical(value) {
33
+ if (value === null || typeof value !== "object") return JSON.stringify(value ?? null);
34
+ if (Array.isArray(value)) return `[${value.map(canonical).join(",")}]`;
35
+ const keys = Object.keys(value).sort();
36
+ return `{${keys.map((k) => `${JSON.stringify(k)}:${canonical(value[k])}`).join(",")}}`;
37
+ }
38
+
39
+ /**
40
+ * The fingerprint of a worker's cron schedule set, or `null` when this worker has no opinion.
41
+ *
42
+ * ABSTAIN VERSUS OPINE IS THE SUBTLE PART, and conflating the two would leave the bug this gate exists to
43
+ * close. `loadSchedules` returns `[]` for two different states:
44
+ *
45
+ * - `PI_TRIGGERS_FILE` unset, which means CRON IS DISABLED on this host. Such a worker has no view of
46
+ * what should be scheduled, so it must never be able to disagree with one that does. It ABSTAINS.
47
+ * - a triggers file that is present and declares zero cron entries. That is an OPINION -- "there should
48
+ * be no schedulers" -- and it is this bug's purest form: today, deleting the last cron trigger on one
49
+ * host prunes the whole fleet's schedulers through the file-watch path.
50
+ *
51
+ * So `null` in means abstain (`null` out); an empty ARRAY is a real fingerprint.
52
+ *
53
+ * The `tz` rides the hash because a cron PATTERN carries no timezone: `triggers.json` has no `tz` field on
54
+ * a cron entry, and BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's
55
+ * LOCAL system time. On one host that is exactly what an operator means; on two hosts in different zones
56
+ * the same pattern is two different instants, with nothing anywhere saying so. Including the zone makes
57
+ * that a disagreement the gate can see. (`pause-windows.json` has carried an explicit `tz` since it
58
+ * shipped and is already fleet-correct; the asymmetry is why this one needs stating.)
59
+ *
60
+ * Fingerprinted over the NORMALIZED schedules rather than the file's bytes, deliberately. Bytes diverge on
61
+ * whitespace, on key order, and on every webhook trigger the worker does not own -- so two hosts differing
62
+ * only in a `label` rule would freeze cron forever over a difference that cannot affect it. The normalized
63
+ * set diverges exactly when the reconcile INPUTS diverge, which is the property that makes reconcile
64
+ * idempotent in the first place.
65
+ */
66
+ export function cronFingerprint(schedules, { tz } = {}) {
67
+ // Takes the AUTHORED shape (`schedules.mjs -> authoredCron`), never the placement-resolved one.
68
+ if (schedules === null || schedules === undefined) return null;
69
+ return fingerprint({
70
+ tz: tz ?? "",
71
+ // EVERY field of an authored entry, projected explicitly. An earlier shape listed the keys of a
72
+ // NORMALIZED schedule (`name`, `data`, `opts`), which are all `undefined` on an authored one -- so
73
+ // the hash saw only the id and the pattern, and an operator changing `run.folder`, `run.flow` or
74
+ // `run.image` on one host and not the other passed the gate silently. That broke this module's own
75
+ // stated invariant, that the set diverges exactly when the reconcile inputs diverge.
76
+ schedules: schedules.map((s) => ({ schedulerId: s.schedulerId, pattern: s.pattern, run: s.run })),
77
+ });
78
+ }
@@ -0,0 +1,179 @@
1
+ /**
2
+ * Fleet-wide leases for the two bounds that stopped meaning what they say when a second host appeared
3
+ * (issue #57).
4
+ *
5
+ * `PI_WAIT_CHECK_SLOTS` and a `scoped-limits.json` row's `concurrent` are both enforced by
6
+ * `makeInFlight()`, an in-process `Map`. One worker per docker daemon made that correct; two hosts make
7
+ * it a bound that MULTIPLIES BY THE OPERATOR'S DEPLOYMENT SHAPE, which is not a bound. Four hosts with
8
+ * `PI_WAIT_CHECK_SLOTS=1` run four concurrent checks against one Jira; four hosts with
9
+ * `{"scope":"acme/web","concurrent":1}` run four paid containers on one repository.
10
+ *
11
+ * WHAT THIS OWES `OQ-008` AND `DES-CONCURRENCY-3`. Both refuse a Redis-held in-flight count, in terms:
12
+ * "a Redis-held count would survive a crash WRONGLY -- a claim for a container the reaper just killed,
13
+ * demanding TTL/heartbeat machinery, a second source of truth about what is running". Every clause of
14
+ * that is about a CONTAINER, and the two leases here answer it differently:
15
+ *
16
+ * The CHECK lease claims no container. It claims a subprocess this same process spawned, bounded by
17
+ * `PI_WAIT_CHECK_TIMEOUT_MS`, holding no folder and spending no money. Three properties invert. What a
18
+ * stale claim costs: a container claim is a folder mutex nobody holds and a job that never runs, while
19
+ * a check claim is one check deferred by at most the TTL. What contradicts it: the reaper is a second
20
+ * source of truth about containers and runs at boot with authority, while nothing enumerates,
21
+ * inspects or reaps a check. And how long it can be wrong: a container has no natural expiry, while a
22
+ * check has a hard, configured, small timeout, so the TTL is DERIVED rather than guessed.
23
+ *
24
+ * The SCOPE claim really is for a container, so the refusal lands squarely -- and the answer is the
25
+ * boot reaper. It establishes, at boot, that this host holds no `pi-job-*` containers, so a claim
26
+ * whose value names this host is a claim for a container that no longer exists, and deleting it is not
27
+ * a second source of truth: it is the SAME source writing down what it just established. Making the
28
+ * reaper the claim's owner removes the contradiction rather than arguing around it.
29
+ *
30
+ * N INDEPENDENT KEYS, NEVER ONE COUNTER. A counter with one TTL loses every claim when it expires and
31
+ * leaks a permanent `+1` on a crash; N `SET NX PX` keys mean a lost release costs exactly one slot for
32
+ * exactly the TTL and never the whole semaphore. `wait:key:<dedupId>` is the in-repo precedent.
33
+ *
34
+ * And a property the in-process map does not have: RELEASE IS IDEMPOTENT here, because it deletes only a
35
+ * key whose value is still ours. `makeInFlight().release` clamps at zero but a double release on a
36
+ * `concurrent: 2` scope frees the other holder's slot.
37
+ */
38
+
39
+ /** Slot keys. Both live under a prefix an operator can see whole with one `KEYS`. */
40
+ export const checkSlotKey = (i) => `wait:check:${i}`;
41
+ export const scopeSlotKey = (hash, i) => `slot:s:${hash}:${i}`;
42
+
43
+ /**
44
+ * Build a lease over `slots` numbered keys.
45
+ *
46
+ * `holder` is `<workerName>#<jobId>`: the worker-name charset excludes `#`, so the value decomposes and
47
+ * the boot sweep can recognise its own claims without a second index to keep in step.
48
+ *
49
+ * FAIL OPEN, and in the GRANTING direction. A Valkey fault must never be able to wedge every wait in a
50
+ * deployment or stop every scoped job; the in-process bound is still there underneath, so failing open
51
+ * degrades the fleet bound to the per-host one, which is exactly the behaviour before this existed.
52
+ */
53
+ export function makeFleetLease({ redis, holderPrefix, keyFor, ttlMs, now = () => Date.now(), log = () => {}, timeoutMs = 2_000 }) {
54
+ // Bounded for `host-registry.mjs`'s reason, which applies to every module sharing this client:
55
+ // `maxRetriesPerRequest: null` makes a command against an unreachable server QUEUE rather than reject,
56
+ // so a try/catch around it catches nothing and an outage would hang the gate rather than fail it.
57
+ const bounded = (p) =>
58
+ new Promise((resolve, reject) => {
59
+ const t = setTimeout(() => reject(new Error("lease timeout")), timeoutMs);
60
+ Promise.resolve(p).then(
61
+ (v) => (clearTimeout(t), resolve(v)),
62
+ (e) => (clearTimeout(t), reject(e)),
63
+ );
64
+ });
65
+
66
+ return {
67
+ /**
68
+ * Take one of `slots`, or `null` when they are all held. Returns a handle whose `release` and
69
+ * `refresh` act only on the key this call actually won.
70
+ *
71
+ * Probing starts at `hash(id) mod slots` rather than at 0. Without the rotation every host tries
72
+ * index 0 first, so a host can sit behind a busy slot while a free one exists two along -- a
73
+ * starvation that looks exactly like the capacity shortage the bound is meant to report.
74
+ */
75
+ async acquire(id, { slots, keyArgs = [], ttlMs: perCall } = {}) {
76
+ if (!Number.isFinite(slots) || slots < 1) return { ok: true, release: async () => {}, refresh: async () => true };
77
+ const holder = `${holderPrefix}#${id}`;
78
+ const start = Math.abs(hashCode(String(id))) % slots;
79
+ for (let n = 0; n < slots; n++) {
80
+ const key = keyFor(...keyArgs, (start + n) % slots);
81
+ try {
82
+ const won = await bounded(redis.set(key, holder, "PX", perCall ?? ttlMs, "NX"));
83
+ if (!won) continue;
84
+ return {
85
+ ok: true,
86
+ key,
87
+ async release() {
88
+ try {
89
+ // Release-if-MINE, which is what makes this idempotent where the in-process map is
90
+ // not: a double release cannot free another holder's slot, because the second call
91
+ // finds a value that is no longer ours.
92
+ if ((await bounded(redis.get(key))) === holder) await bounded(redis.del(key));
93
+ } catch {
94
+ // The TTL is the backstop. A lost release costs one slot for one TTL.
95
+ }
96
+ },
97
+ async refresh(nextMs = ttlMs) {
98
+ try {
99
+ if ((await bounded(redis.get(key))) !== holder) return false;
100
+ await bounded(redis.pexpire(key, nextMs));
101
+ return true;
102
+ } catch {
103
+ return true; // a blip is not a reason to believe we lost a slot we hold
104
+ }
105
+ },
106
+ };
107
+ } catch (err) {
108
+ // Granting is the safe direction: the in-process bound is still underneath, so this
109
+ // degrades the fleet-wide ceiling to the per-host one rather than to nothing.
110
+ log("fleet_lease_unavailable", { reason: err?.message });
111
+ return { ok: true, degraded: true, release: async () => {}, refresh: async () => true };
112
+ }
113
+ }
114
+ return null;
115
+ },
116
+ };
117
+ }
118
+
119
+ /** A small, stable, non-cryptographic spread for the slot rotation. Not a key, so not a digest. */
120
+ function hashCode(s) {
121
+ let h = 0;
122
+ for (let i = 0; i < s.length; i++) h = (h * 31 + s.charCodeAt(i)) | 0;
123
+ return h;
124
+ }
125
+
126
+ /**
127
+ * Delete every scope slot this host still claims, at boot, right after the container reaper.
128
+ *
129
+ * THIS IS THE ANSWER TO `OQ-008`, and its correctness rests entirely on one precondition: the reaper
130
+ * must have ENUMERATED. `makeReaper` catches its own `docker ps` failure and logs `reaper_skipped`, and
131
+ * on that path nothing was listed and nothing reaped -- so this host has NOT established that it holds
132
+ * no containers, and its claims may be for containers that are still running. Sweeping then would free
133
+ * slots for another host to start more alongside them, which is a money overrun rather than a tidy-up.
134
+ * `makeInFlight`'s own escape ("a state where no NEW container can start either") does not transfer,
135
+ * because the sweep frees slots for a DIFFERENT machine.
136
+ *
137
+ * Driven by config rather than by a scan: `scoped-limits.json` enumerates every scope that can carry a
138
+ * claim and `concurrent` bounds the index, so this is `sum(concurrent)` GETs -- typically under twenty.
139
+ * No `KEYS`, no `SCAN`, and no index set to leak.
140
+ */
141
+ export function makeScopeClaimSweeper({ redis, workerName, limits, log = () => {}, timeoutMs = 2_000 }) {
142
+ return async function sweep({ reaped }) {
143
+ if (!reaped) {
144
+ log("scope_claims_sweep_skipped", { reason: "reaper-skipped" });
145
+ return { swept: 0, skipped: true };
146
+ }
147
+ const prefix = `${workerName}#`;
148
+ let swept = 0;
149
+ for (const row of limits ?? []) {
150
+ const n = Number(row?.concurrent);
151
+ if (!Number.isFinite(n) || n < 1 || !row?.hash) continue;
152
+ for (let i = 0; i < n; i++) {
153
+ const key = scopeSlotKey(row.hash, i);
154
+ try {
155
+ const held = await withTimeout(redis.get(key), timeoutMs);
156
+ if (typeof held === "string" && held.startsWith(prefix)) {
157
+ await withTimeout(redis.del(key), timeoutMs);
158
+ swept++;
159
+ }
160
+ } catch {
161
+ // Best-effort by contract: this is an OPTIMISATION over the TTL, never the mechanism, so a
162
+ // fault here costs at most one TTL of a stale claim and must never block boot.
163
+ }
164
+ }
165
+ }
166
+ if (swept > 0) log("scope_claims_swept", { count: swept });
167
+ return { swept, skipped: false };
168
+ };
169
+ }
170
+
171
+ function withTimeout(p, ms) {
172
+ return new Promise((resolve, reject) => {
173
+ const t = setTimeout(() => reject(new Error("lease timeout")), ms);
174
+ Promise.resolve(p).then(
175
+ (v) => (clearTimeout(t), resolve(v)),
176
+ (e) => (clearTimeout(t), reject(e)),
177
+ );
178
+ });
179
+ }