@edgehero/pi-dispatch 3.1.0 → 4.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +38 -0
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +599 -16
- package/src/host-budget.mjs +736 -0
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/provider-steering.mjs +1 -0
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +2 -1
|
@@ -0,0 +1,736 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The host budget (issue #596, phase 2, DES-HOST-BUDGET): how much memory and CPU this host's running jobs may hold
|
|
3
|
+
* together, and the one ledger that keeps their sizes inside it.
|
|
4
|
+
*
|
|
5
|
+
* A LEAF beside `job-size.mjs`, importing nothing else of this project's: `config.mjs` validates the four settings at
|
|
6
|
+
* boot with `hostBudgetSettings`, and `scoped-limits.mjs` imports `config.mjs`, so a project row is matched here by its
|
|
7
|
+
* scope string (`project:<id>`) the way `job-size.mjs` matches it.
|
|
8
|
+
*
|
|
9
|
+
* WHY A LEDGER AND NOT A COUNT. `PI_CONCURRENCY` counts containers, and a count cannot tell one 20g job from five 2g
|
|
10
|
+
* ones. The budget sums the SIZE each running job was started at (`memMiB`, `cpuCenti`, integers, so a sum is exact),
|
|
11
|
+
* keyed by job id, so a release is idempotent: the job that took a hold gives back exactly what it took, once, whatever
|
|
12
|
+
* the limits file says by then. `PI_CONCURRENCY` stays an upper bound on the count, and it is judged HERE as the
|
|
13
|
+
* budget's third dimension (one per job, `admit`), so a hold keeps a job slot too; whichever binds first, binds.
|
|
14
|
+
*
|
|
15
|
+
* WHY HOLDS. A deferred job goes to the back of the delayed set, so a 20g job behind a stream of 2g jobs would wait
|
|
16
|
+
* forever: every time 2g frees, a 2g job takes it. A HOLD is room the budget keeps for a waiting job while it is away:
|
|
17
|
+
* - tier 1, the oldest waiter of each project that runs fewer than its `minJobs` here (the soft minimum);
|
|
18
|
+
* - tier 2, the single oldest waiter of all.
|
|
19
|
+
* A job is admitted only if what runs, plus its own size, plus every hold RANKED ABOVE IT, fits in both memory and CPU,
|
|
20
|
+
* and its project stays within its `hostShare`. A hold that is not the job's own only obstacle is SUSPENDED (another
|
|
21
|
+
* gate deferred the job: a full scope, a busy endpoint, a pause window), so a job never keeps room it could not use. A
|
|
22
|
+
* hold whose job has not come back is VERIFIED after `HOLD_VERIFY_AFTER_MS` against the queue (`job.getState()`) and
|
|
23
|
+
* dropped when the job is active elsewhere, finished or gone; on a Valkey error it is dropped once its job has been away
|
|
24
|
+
* for `HOLD_DROP_ON_ERROR_MS`.
|
|
25
|
+
*
|
|
26
|
+
* WHY ADMISSION IS SYNCHRONOUS. `gate` reads the ledger, decides and writes it with no `await` between, so on Node's one
|
|
27
|
+
* thread two pickups on one host can never both see the same free room. The facts the budget is computed from are read
|
|
28
|
+
* OFF the decision (`refresh`, on the tick), never inside it.
|
|
29
|
+
*
|
|
30
|
+
* A JOB WHOSE STOP FAILED keeps its hold (`orphan`): the container may still be running and still be using what its
|
|
31
|
+
* size promised. The hold is given back only when the runtime confirms the container is gone (`sweep`, on the tick). A
|
|
32
|
+
* worker that restarts kills every job container it can (the boot reaper), then SEEDS the ledger with every one still
|
|
33
|
+
* listed (`survivors`), so a container the reaper could not remove is still counted; until a venue's listing is read,
|
|
34
|
+
* the gate admits nothing on that venue.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import { formatCpus, formatMemory, parseCpus, parseMemory } from "./job-size.mjs";
|
|
38
|
+
|
|
39
|
+
/** The four settings, env only for this release (never the settings overlay: a budget must not move under running jobs). */
|
|
40
|
+
export const HOST_BUDGET_KEYS = Object.freeze({
|
|
41
|
+
memory: "PI_HOST_MEMORY_BUDGET",
|
|
42
|
+
cpus: "PI_HOST_CPU_BUDGET",
|
|
43
|
+
reserveMemory: "PI_HOST_RESERVE_MEMORY",
|
|
44
|
+
reserveCpus: "PI_HOST_RESERVE_CPUS",
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* How often a job the budget deferred asks again. Its own value, like every re-check cadence, because nothing records WHY
|
|
49
|
+
* a job sits in the delayed set and the wake instant is the only evidence there is: distinct from the scope re-check
|
|
50
|
+
* (5 s), the endpoint re-check (7 s), the wait throttle floor (11 s) and the supersede re-ask (15 s). A test keeps them
|
|
51
|
+
* apart. Short, because a hold keeps the room meanwhile: the wait is the latency of a freed slot, not a queue position.
|
|
52
|
+
*/
|
|
53
|
+
export const BUDGET_RECHECK_MS = 9_000;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* How long a job on the SHARED queue whose size can never fit THIS host waits before another pickup asks again (gate
|
|
57
|
+
* round 1 of phase 2). Longer than every re-check above, because the answer is not "soon" but "on another
|
|
58
|
+
* host": the job goes back to the shared queue for a host it fits on. It is never refused for the fleet: a registry
|
|
59
|
+
* row is absent while its host restarts and a timed-out read drops a row, so "no live host fits" is not a verdict the
|
|
60
|
+
* registry can give. Doctor names a project that fits no live host instead.
|
|
61
|
+
*/
|
|
62
|
+
export const NEVER_FITS_RECHECK_MS = 60_000;
|
|
63
|
+
|
|
64
|
+
/** A hold whose job has not been back for this long is checked against the queue. */
|
|
65
|
+
export const HOLD_VERIFY_AFTER_MS = 15_000;
|
|
66
|
+
/** A hold that cannot be checked (Valkey does not answer) is dropped once its job has been away this long. */
|
|
67
|
+
export const HOLD_DROP_ON_ERROR_MS = 120_000;
|
|
68
|
+
/**
|
|
69
|
+
* How long ONE queue read of a hold's job (`getState`) may take (gate round 1 of phase 2). The worker's client
|
|
70
|
+
* is built with `maxRetriesPerRequest: null`, so a command against an unreachable server queues forever rather than
|
|
71
|
+
* rejecting: unbounded, one hung read held the whole verify, the 120 s drop never came, and every later tick piled up
|
|
72
|
+
* behind it. A read past this bound is an unanswered read, the error branch.
|
|
73
|
+
*/
|
|
74
|
+
export const HOLD_STATE_READ_BOUND_MS = 5_000;
|
|
75
|
+
/** How often the budget re-reads its facts, verifies stale holds and sweeps orphans. Off every job path. */
|
|
76
|
+
export const HOST_BUDGET_TICK_MS = 5_000;
|
|
77
|
+
|
|
78
|
+
/** The `auto` memory reserve: 10% of the host's memory, at least 1 GiB and at most 4 GiB. */
|
|
79
|
+
export const RESERVE_MEMORY_MIN_MIB = 1024;
|
|
80
|
+
export const RESERVE_MEMORY_MAX_MIB = 4096;
|
|
81
|
+
/** The `auto` CPU reserve: one CPU on a host with at least this many, else none. */
|
|
82
|
+
export const RESERVE_CPUS_FROM = 4;
|
|
83
|
+
|
|
84
|
+
/** The largest CPU budget a setting may name: the runtimes' own CPU count ceiling (`daemon-facts.mjs` `cpuCount`). */
|
|
85
|
+
const BUDGET_CPUS_CEILING_CENTI = 4096 * 100;
|
|
86
|
+
/** The largest memory budget or reserve a setting may name: 64 TiB, `daemon-facts.mjs` `memoryMiB`'s ceiling. */
|
|
87
|
+
const BUDGET_MEMORY_CEILING_MIB = 64 * 1024 * 1024;
|
|
88
|
+
|
|
89
|
+
const PROJECT_ROW_PREFIX = "project:";
|
|
90
|
+
|
|
91
|
+
/** A memory setting (`64g`, `1536m`) in MiB, or 0 for `"0"` where `zero` allows it. Throws an Error naming the rule. */
|
|
92
|
+
function parseBudgetMemory(value, { zero = false } = {}) {
|
|
93
|
+
if (zero && value === "0") return 0;
|
|
94
|
+
if (typeof value !== "string" || !/^[1-9]\d{0,8}[mg]$/.test(value)) {
|
|
95
|
+
throw new Error(`a memory amount is a whole number of megabytes or gigabytes such as "1536m" or "64g"${zero ? ", or 0" : ""} (got ${JSON.stringify(value)})`);
|
|
96
|
+
}
|
|
97
|
+
const mib = Number(value.slice(0, -1)) * (value.endsWith("g") ? 1024 : 1);
|
|
98
|
+
if (mib > BUDGET_MEMORY_CEILING_MIB) throw new Error(`a memory amount must be at most ${BUDGET_MEMORY_CEILING_MIB / 1024}g (got ${JSON.stringify(value)})`);
|
|
99
|
+
return mib;
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** A CPU setting (`3.5`, `16`) in hundredths, or 0 for `"0"` where `zero` allows it. Throws an Error naming the rule. */
|
|
103
|
+
function parseBudgetCpus(value, { zero = false } = {}) {
|
|
104
|
+
if (typeof value !== "string" || !/^(?:0|[1-9]\d{0,5})(?:\.\d{1,2})?$/.test(value)) {
|
|
105
|
+
throw new Error(`a CPU amount is a number with at most two decimals such as 3.5 or 16${zero ? ", or 0" : ""} (got ${JSON.stringify(value)})`);
|
|
106
|
+
}
|
|
107
|
+
const [whole, fraction = ""] = value.split(".");
|
|
108
|
+
const centi = Number(whole) * 100 + Number(fraction.padEnd(2, "0"));
|
|
109
|
+
if (centi === 0 && !zero) throw new Error(`a CPU amount must be above 0 (got ${JSON.stringify(value)})`);
|
|
110
|
+
if (centi > BUDGET_CPUS_CEILING_CENTI) throw new Error(`a CPU amount must be at most ${BUDGET_CPUS_CEILING_CENTI / 100} (got ${JSON.stringify(value)})`);
|
|
111
|
+
return centi;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* The four settings as the worker runs them, validated (issue #596): `{ memory, cpus, reserveMemory, reserveCpus }`, each
|
|
116
|
+
* `{ mode: "auto" }`, `{ mode: "off" }` (the two budgets only) or `{ mode: "value", memMiB | cpuCenti }`. Unset or empty
|
|
117
|
+
* is `auto`. Throws an Error naming the key and the rule for anything else, and for a budget VALUE below one job of the
|
|
118
|
+
* deployment's default size (`jobDefault`, `{ memMiB, cpuCenti }`): such a budget would refuse every default-size job,
|
|
119
|
+
* so it is a configuration error at boot rather than a refusal per job. `auto` never falls below that size by
|
|
120
|
+
* construction (`computeHostBudget`), which is decision 3 of the issue.
|
|
121
|
+
*
|
|
122
|
+
* A reserve is what `auto` leaves for the host itself (the OS, the egress proxy, a Valkey, the worker). It applies only
|
|
123
|
+
* to `auto`: a budget given as a value IS the budget.
|
|
124
|
+
*/
|
|
125
|
+
export function hostBudgetSettings(env = {}, jobDefault = { memMiB: 4096, cpuCenti: 200 }) {
|
|
126
|
+
const read = (key, parse) => {
|
|
127
|
+
const raw = env?.[key];
|
|
128
|
+
if (raw === undefined || raw === null || String(raw).trim() === "" || String(raw).trim() === "auto") return { mode: "auto" };
|
|
129
|
+
const text = String(raw).trim();
|
|
130
|
+
try {
|
|
131
|
+
return parse(text);
|
|
132
|
+
} catch (error) {
|
|
133
|
+
throw new Error(`${key}: ${error.message}`);
|
|
134
|
+
}
|
|
135
|
+
};
|
|
136
|
+
const memory = read(HOST_BUDGET_KEYS.memory, (text) => (text === "off" ? { mode: "off" } : { mode: "value", memMiB: parseBudgetMemory(text) }));
|
|
137
|
+
const cpus = read(HOST_BUDGET_KEYS.cpus, (text) => (text === "off" ? { mode: "off" } : { mode: "value", cpuCenti: parseBudgetCpus(text) }));
|
|
138
|
+
const reserveMemory = read(HOST_BUDGET_KEYS.reserveMemory, (text) => ({ mode: "value", memMiB: parseBudgetMemory(text, { zero: true }) }));
|
|
139
|
+
const reserveCpus = read(HOST_BUDGET_KEYS.reserveCpus, (text) => ({ mode: "value", cpuCenti: parseBudgetCpus(text, { zero: true }) }));
|
|
140
|
+
if (memory.mode === "value" && memory.memMiB < jobDefault.memMiB) {
|
|
141
|
+
throw new Error(`${HOST_BUDGET_KEYS.memory}: ${formatMemory(memory.memMiB)} is below one job of the default size (${formatMemory(jobDefault.memMiB)}, PI_JOB_MEMORY), so every such job would be refused; raise it, lower PI_JOB_MEMORY, or use auto`);
|
|
142
|
+
}
|
|
143
|
+
if (cpus.mode === "value" && cpus.cpuCenti < jobDefault.cpuCenti) {
|
|
144
|
+
throw new Error(`${HOST_BUDGET_KEYS.cpus}: ${formatCpus(cpus.cpuCenti)} is below one job of the default size (${formatCpus(jobDefault.cpuCenti)} CPUs, PI_JOB_CPUS), so every such job would be refused; raise it, lower PI_JOB_CPUS, or use auto`);
|
|
145
|
+
}
|
|
146
|
+
return { memory, cpus, reserveMemory, reserveCpus };
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** The smaller of the known values (a number), or null when none is known. */
|
|
150
|
+
function smallerKnown(...values) {
|
|
151
|
+
const known = values.filter((v) => Number.isSafeInteger(v) && v > 0);
|
|
152
|
+
return known.length === 0 ? null : Math.min(...known);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* The budget this host's facts give under the settings: `{ memMiB, cpuCenti, detail }`. Each budget is an integer,
|
|
157
|
+
* `Infinity` for `off`, or `null` when `auto` has no fact to work from (the runtime did not answer): the gate then fails
|
|
158
|
+
* OPEN on that dimension and says so (`host_budget_unknown`), the posture `--cpus` takes when the CPU count is unknown.
|
|
159
|
+
*
|
|
160
|
+
* `facts`: `{ memTotalMiB, hostCpus, userMemMiB, userCpuCenti }`, the runtime's own memory and CPU count
|
|
161
|
+
* (`daemon-facts.mjs`) and, on rootless Podman, the user service's `memory.max` and `cpu.max` (`readUserServiceLimits`).
|
|
162
|
+
* `auto` is the smaller of those, minus the reserve, and never below one job of the default size (`jobDefault`):
|
|
163
|
+
* `floored` says when that floor, not the host, set the number.
|
|
164
|
+
*/
|
|
165
|
+
export function computeHostBudget(settings, facts = {}, jobDefault = { memMiB: 4096, cpuCenti: 200 }) {
|
|
166
|
+
const detail = { memMode: settings.memory.mode, cpuMode: settings.cpus.mode, memTotalMiB: null, memReserveMiB: null, memFloored: false, cpuTotalCenti: null, cpuReserveCenti: null, cpuFloored: false };
|
|
167
|
+
let memMiB = null;
|
|
168
|
+
if (settings.memory.mode === "off") memMiB = Infinity;
|
|
169
|
+
else if (settings.memory.mode === "value") memMiB = settings.memory.memMiB;
|
|
170
|
+
else {
|
|
171
|
+
const total = smallerKnown(facts.memTotalMiB, facts.userMemMiB);
|
|
172
|
+
if (total !== null) {
|
|
173
|
+
const reserve = settings.reserveMemory.mode === "value" ? settings.reserveMemory.memMiB : Math.min(RESERVE_MEMORY_MAX_MIB, Math.max(RESERVE_MEMORY_MIN_MIB, Math.floor(total / 10)));
|
|
174
|
+
detail.memTotalMiB = total;
|
|
175
|
+
detail.memReserveMiB = reserve;
|
|
176
|
+
detail.memFloored = total - reserve < jobDefault.memMiB;
|
|
177
|
+
memMiB = Math.max(total - reserve, jobDefault.memMiB);
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
let cpuCenti = null;
|
|
181
|
+
if (settings.cpus.mode === "off") cpuCenti = Infinity;
|
|
182
|
+
else if (settings.cpus.mode === "value") cpuCenti = settings.cpus.cpuCenti;
|
|
183
|
+
else {
|
|
184
|
+
const total = smallerKnown(Number.isSafeInteger(facts.hostCpus) ? facts.hostCpus * 100 : null, facts.userCpuCenti);
|
|
185
|
+
if (total !== null) {
|
|
186
|
+
const reserve = settings.reserveCpus.mode === "value" ? settings.reserveCpus.cpuCenti : total >= RESERVE_CPUS_FROM * 100 ? 100 : 0;
|
|
187
|
+
detail.cpuTotalCenti = total;
|
|
188
|
+
detail.cpuReserveCenti = reserve;
|
|
189
|
+
detail.cpuFloored = total - reserve < jobDefault.cpuCenti;
|
|
190
|
+
cpuCenti = Math.max(total - reserve, jobDefault.cpuCenti);
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
return { memMiB, cpuCenti, detail };
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* The cgroup directories that bound a rootless account's containers, outermost first: the account's slice and its
|
|
198
|
+
* systemd user manager's service (`user@<uid>.service`), under which rootless Podman's containers live. Either may carry
|
|
199
|
+
* a `memory.max` or `cpu.max` an administrator set (`systemctl set-property`), and both bound every container below.
|
|
200
|
+
*/
|
|
201
|
+
export function userServiceCgroupDirs(uid) {
|
|
202
|
+
return [`/sys/fs/cgroup/user.slice/user-${uid}.slice`, `/sys/fs/cgroup/user.slice/user-${uid}.slice/user@${uid}.service`];
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** A `memory.max` file's value in MiB (rounded down), or null for `max`, an empty file or anything else. */
|
|
206
|
+
export function parseMemoryMax(text) {
|
|
207
|
+
const value = String(text ?? "").trim();
|
|
208
|
+
if (!/^[0-9]{1,20}$/.test(value)) return null;
|
|
209
|
+
const bytes = Number(value);
|
|
210
|
+
return Number.isSafeInteger(bytes) && bytes >= 1048576 ? Math.floor(bytes / 1048576) : null;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/** A `cpu.max` file's quota in hundredths of a CPU (rounded down), or null for `max <period>` or anything else. */
|
|
214
|
+
export function parseCpuMax(text) {
|
|
215
|
+
const match = /^([0-9]{1,15}) ([0-9]{1,15})$/.exec(String(text ?? "").trim());
|
|
216
|
+
if (!match) return null;
|
|
217
|
+
const quota = Number(match[1]);
|
|
218
|
+
const period = Number(match[2]);
|
|
219
|
+
if (period <= 0 || quota <= 0) return null;
|
|
220
|
+
const centi = Math.floor((quota * 100) / period);
|
|
221
|
+
return centi > 0 ? centi : null;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/**
|
|
225
|
+
* The smallest `memory.max` and `cpu.max` over a rootless account's slice and user service (`userServiceCgroupDirs`), as
|
|
226
|
+
* `{ userMemMiB, userCpuCenti }`, each null where no file is readable or none sets a bound. Best effort and synchronous:
|
|
227
|
+
* an unreadable file is no bound rather than an error, so a host without cgroup v2 or with another layout keeps the
|
|
228
|
+
* runtime's own numbers. `readFile` is injected (`fs.readFileSync` with "utf8").
|
|
229
|
+
*/
|
|
230
|
+
export function readUserServiceLimits({ uid, readFile }) {
|
|
231
|
+
let userMemMiB = null;
|
|
232
|
+
let userCpuCenti = null;
|
|
233
|
+
if (!Number.isSafeInteger(uid) || uid < 0 || typeof readFile !== "function") return { userMemMiB, userCpuCenti };
|
|
234
|
+
const read = (path) => {
|
|
235
|
+
try {
|
|
236
|
+
return readFile(path);
|
|
237
|
+
} catch {
|
|
238
|
+
return null;
|
|
239
|
+
}
|
|
240
|
+
};
|
|
241
|
+
for (const dir of userServiceCgroupDirs(uid)) {
|
|
242
|
+
userMemMiB = smallerKnown(userMemMiB, parseMemoryMax(read(`${dir}/memory.max`)));
|
|
243
|
+
userCpuCenti = smallerKnown(userCpuCenti, parseCpuMax(read(`${dir}/cpu.max`)));
|
|
244
|
+
}
|
|
245
|
+
return { userMemMiB, userCpuCenti };
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* A project's two budget knobs from a limits snapshot: `{ hostShare, minJobs, size }`, hostShare a percentage or null,
|
|
250
|
+
* minJobs an integer or 0, `size` the row's `{ memory, cpus }` strings as written (for doctor). Nothing for no row.
|
|
251
|
+
*/
|
|
252
|
+
export function projectBudgetRow(limits, project) {
|
|
253
|
+
if (typeof project !== "string" || !Array.isArray(limits)) return { hostShare: null, minJobs: 0 };
|
|
254
|
+
const row = limits.find((l) => l?.scope === `${PROJECT_ROW_PREFIX}${project}`) ?? null;
|
|
255
|
+
return {
|
|
256
|
+
hostShare: Number.isSafeInteger(row?.hostShare) ? row.hostShare : null,
|
|
257
|
+
minJobs: Number.isSafeInteger(row?.minJobs) && row.minJobs > 0 ? row.minJobs : 0,
|
|
258
|
+
};
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/** `x` fits in `budget` (null is unknown and `Infinity` is off: both fit). */
|
|
262
|
+
function fits(x, budget) {
|
|
263
|
+
return budget === null || budget === Infinity || x <= budget;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/** `x` fits in `share` percent of `budget`, in integers (null share is no share). */
|
|
267
|
+
function withinShare(x, budget, share) {
|
|
268
|
+
return share === null || budget === null || budget === Infinity || x * 100 <= share * budget;
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
function sumOf(entries) {
|
|
272
|
+
let memMiB = 0;
|
|
273
|
+
let cpuCenti = 0;
|
|
274
|
+
for (const e of entries) {
|
|
275
|
+
memMiB += e.memMiB;
|
|
276
|
+
cpuCenti += e.cpuCenti;
|
|
277
|
+
}
|
|
278
|
+
return { memMiB, cpuCenti };
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/** Oldest first; the id breaks a tie, so the order is total and two hosts would rank one set the same way. */
|
|
282
|
+
function byAge(a, b) {
|
|
283
|
+
return a.firstAt - b.firstAt || (a.id < b.id ? -1 : a.id > b.id ? 1 : 0);
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/**
|
|
287
|
+
* The holds, in rank order (PURE): tier 1, the oldest unsuspended waiter of each project that runs fewer than its
|
|
288
|
+
* `minJobs` here (oldest first among them), then tier 2, the single oldest unsuspended waiter of all when it is not
|
|
289
|
+
* already in tier 1. `running` counts every ledger entry of a project, orphans included: an orphan's container may run.
|
|
290
|
+
*/
|
|
291
|
+
export function rankHolds(waiters, ledger, minJobsOf = () => 0) {
|
|
292
|
+
const active = [...waiters].filter((w) => !w.suspended).sort(byAge);
|
|
293
|
+
const running = new Map();
|
|
294
|
+
for (const e of ledger) if (e.project) running.set(e.project, (running.get(e.project) ?? 0) + 1);
|
|
295
|
+
const tier1 = [];
|
|
296
|
+
const seen = new Set();
|
|
297
|
+
for (const w of active) {
|
|
298
|
+
if (!w.project || seen.has(w.project)) continue;
|
|
299
|
+
seen.add(w.project);
|
|
300
|
+
const min = minJobsOf(w.project);
|
|
301
|
+
if (min > 0 && (running.get(w.project) ?? 0) < min) tier1.push(w);
|
|
302
|
+
}
|
|
303
|
+
const oldest = active[0];
|
|
304
|
+
return oldest && !tier1.includes(oldest) ? [...tier1, oldest] : tier1;
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* THE ADMISSION RULE (PURE): may `ask` (`{ id, project, memMiB, cpuCenti }`) start now? `{ ok: true }`, or `{ ok: false,
|
|
309
|
+
* why }` with `why` `share` (its project would hold more than its `hostShare` of the budget) or `budget` (what runs, plus
|
|
310
|
+
* `ask`, plus every hold ranked above it, does not fit in memory, in CPU or in the job COUNT). `holds` is `rankHolds`'
|
|
311
|
+
* order; a job that holds no hold has every hold above it. The share is judged first, so a job its own project's share
|
|
312
|
+
* stops is not reported as waiting for the budget (and so does not hold room it could not use).
|
|
313
|
+
*
|
|
314
|
+
* THE COUNT is the third dimension (gate round 1 of phase 2): every running job, the ask and every hold above
|
|
315
|
+
* it is ONE, against `budget.count` (the live `PI_CONCURRENCY`; null or absent is no count bound). It was a separate
|
|
316
|
+
* host slot taken BEFORE this gate, and that starved a big job: a full host slot deferred it at the slot, so its hold
|
|
317
|
+
* was suspended, and every small job that ended was replaced at once by the next one from the host queue. Inside the
|
|
318
|
+
* budget the oldest waiter's hold keeps a count slot as well as its memory and CPU, so the next free slot is its own.
|
|
319
|
+
* `hostShare` does not apply to the count: it is a share of the machine's memory and CPU, not of its job slots.
|
|
320
|
+
*/
|
|
321
|
+
export function admit({ budget, ledger, holds, ask, shareOf = () => null }) {
|
|
322
|
+
const share = ask.project ? shareOf(ask.project) : null;
|
|
323
|
+
if (share !== null) {
|
|
324
|
+
const mine = sumOf(ledger.filter((e) => e.project === ask.project));
|
|
325
|
+
if (!withinShare(mine.memMiB + ask.memMiB, budget.memMiB, share) || !withinShare(mine.cpuCenti + ask.cpuCenti, budget.cpuCenti, share)) return { ok: false, why: "share" };
|
|
326
|
+
}
|
|
327
|
+
const rank = holds.findIndex((h) => h.id === ask.id);
|
|
328
|
+
const counted = [...ledger, ask, ...(rank === -1 ? holds : holds.slice(0, rank))];
|
|
329
|
+
const need = sumOf(counted);
|
|
330
|
+
const count = Number.isSafeInteger(budget.count) ? budget.count : null;
|
|
331
|
+
if (!fits(need.memMiB, budget.memMiB) || !fits(need.cpuCenti, budget.cpuCenti) || !fits(counted.length, count)) return { ok: false, why: "budget" };
|
|
332
|
+
return { ok: true };
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/**
|
|
336
|
+
* Whether a size can EVER start on a host with this budget (PURE): null when it can, `host` when it is larger than the
|
|
337
|
+
* budget in memory or CPU, `share` when it fits the budget but not its project's `hostShare` of it. An unknown or `off`
|
|
338
|
+
* budget never refuses.
|
|
339
|
+
*/
|
|
340
|
+
export function neverFits(size, budget, share = null) {
|
|
341
|
+
if (!fits(size.memMiB, budget.memMiB) || !fits(size.cpuCenti, budget.cpuCenti)) return "host";
|
|
342
|
+
if (!withinShare(size.memMiB, budget.memMiB, share) || !withinShare(size.cpuCenti, budget.cpuCenti, share)) return "share";
|
|
343
|
+
return null;
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
/** A budget as the registry row carries it (INT-HOST-REGISTRY-CONTRACT): an integer as text, `off`, or "" for unknown. */
|
|
347
|
+
export function budgetField(value) {
|
|
348
|
+
return value === Infinity ? "off" : Number.isSafeInteger(value) ? String(value) : "";
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
/** A registry row's published budget back: `{ memMiB, cpuCenti }`, each an integer, `Infinity` (`off`) or null. */
|
|
352
|
+
export function publishedBudget(row) {
|
|
353
|
+
const read = (v) => (v === "off" ? Infinity : typeof v === "string" && /^[0-9]{1,15}$/.test(v) ? Number(v) : null);
|
|
354
|
+
return { memMiB: read(row?.budgetMemMiB), cpuCenti: read(row?.budgetCpuCenti) };
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* The largest size any running job here could have been started at, for a surviving job container that carries no size
|
|
359
|
+
* label (one started by a worker from before the labels): the larger of the default size and every project row's size,
|
|
360
|
+
* in each dimension, and never more than the budget (`budget`, `{ memMiB, cpuCenti }`, each capped only when it is an
|
|
361
|
+
* integer). Pessimistic on purpose: a guess too small lets the next job overcommit the host, a guess too large only
|
|
362
|
+
* delays one until the sweep sees the container gone. CAPPED (gate round 2 of phase 2): a project row larger
|
|
363
|
+
* than the budget (a size that never fits) would otherwise count a survivor as more than the whole host, which no
|
|
364
|
+
* container here could hold and which the ledger's sums then carried into doctor and the registry.
|
|
365
|
+
*/
|
|
366
|
+
export function pessimisticSize(limits, jobDefault, budget = {}) {
|
|
367
|
+
let memMiB = jobDefault.memMiB;
|
|
368
|
+
let cpuCenti = jobDefault.cpuCenti;
|
|
369
|
+
for (const row of Array.isArray(limits) ? limits : []) {
|
|
370
|
+
if (typeof row?.scope !== "string" || !row.scope.startsWith(PROJECT_ROW_PREFIX)) continue;
|
|
371
|
+
try {
|
|
372
|
+
if (typeof row.memory === "string") memMiB = Math.max(memMiB, parseMemory(row.memory));
|
|
373
|
+
} catch {
|
|
374
|
+
// a row the loader refused never reaches here; a bad one is no size
|
|
375
|
+
}
|
|
376
|
+
try {
|
|
377
|
+
if (row.cpus !== null && row.cpus !== undefined) cpuCenti = Math.max(cpuCenti, parseCpus(row.cpus));
|
|
378
|
+
} catch {
|
|
379
|
+
// as above
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
const cap = (value, limit) => (Number.isSafeInteger(limit) && limit > 0 ? Math.min(value, limit) : value);
|
|
383
|
+
return { memMiB: cap(memMiB, budget?.memMiB), cpuCenti: cap(cpuCenti, budget?.cpuCenti) };
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* `promise`'s answer, or a rejection once `ms` have passed; the timer is cleared when the answer comes. NOT unref'd: the
|
|
388
|
+
* wait it bounds is the tick's own, and an unref'd timer beside a read that never settles leaves the event loop with
|
|
389
|
+
* nothing to run, so the verify would never return (the shutdown path exits the process either way).
|
|
390
|
+
*/
|
|
391
|
+
function bounded(promise, ms) {
|
|
392
|
+
let timer;
|
|
393
|
+
return Promise.race([
|
|
394
|
+
Promise.resolve(promise).finally(() => clearTimeout(timer)),
|
|
395
|
+
new Promise((_resolve, reject) => {
|
|
396
|
+
timer = setTimeout(() => reject(new Error(`no answer in ${ms} ms`)), ms);
|
|
397
|
+
}),
|
|
398
|
+
]);
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* The host budget's state and its four doors (`gate`, `suspend`, `release`, `orphan`), built ONCE per worker
|
|
403
|
+
* (`createWorker`) and shared by both of its processors, so one host keeps one ledger whatever queue a job came from.
|
|
404
|
+
*
|
|
405
|
+
* settings `hostBudgetSettings`' answer
|
|
406
|
+
* jobDefault the deployment's default size, `{ memMiB, cpuCenti }`, under which `auto` never falls
|
|
407
|
+
* readFacts async () => `{ memTotalMiB, hostCpus, userMemMiB, userCpuCenti }`, the runtime's answers (cached by
|
|
408
|
+
* their own readers); read by `refresh`, never by `gate`
|
|
409
|
+
* scopedLimits () => the live limits snapshot, for the holds' `minJobs` and the shares when no pickup snapshot is given
|
|
410
|
+
* countLimit () => the live `PI_CONCURRENCY` (an integer), the budget's third dimension (`admit`); an answer that
|
|
411
|
+
* is not an integer is no bound for that gate. REQUIRED (gate round 2 of phase 2): a default of
|
|
412
|
+
* "no bound" let a wiring that forgot it drop the count silently, so a missing one throws here
|
|
413
|
+
* containerGone async (name, venue) => true (the runtime says it is gone), false (it runs), null (could not ask)
|
|
414
|
+
* survivors the job containers each blessed venue still lists after the boot reaper, each `{ name, venue, memMiB,
|
|
415
|
+
* cpuCenti }` (a size label absent is null): an OBJECT of one async lister per venue name (`{ local,
|
|
416
|
+
* podman }`), each THROWING when its venue cannot be listed, or one async function for the whole host.
|
|
417
|
+
* Null for a wiring without one (nothing to seed)
|
|
418
|
+
* defaultVenue the venue a job that names none runs on (`config.defaultBackend`), for the per-venue seed
|
|
419
|
+
* onRefresh (budget, facts) => anything, after every refresh, NOT awaited and never allowed to throw into it: the
|
|
420
|
+
* CPU reserve (`cpu-reserve.mjs`) keeps the jobs' parent cgroup's quota at the budget from here, so a
|
|
421
|
+
* slow `systemctl` or helper container delays no refresh, no gate and no pickup
|
|
422
|
+
*
|
|
423
|
+
* ONE PICKUP, ONE TICKET (gate round 1 of phase 2). The same job id can reach the gate twice on one host: a
|
|
424
|
+
* scheduled job whose lock lapsed while its first attempt still runs is moved back to wait by BullMQ's stall check and
|
|
425
|
+
* picked up again here, and its record does not exist yet. The ledger is keyed by job id, so the second attempt used to
|
|
426
|
+
* overwrite the first's entry and its release then freed the hold of a container that still ran. Now `enter` hands each
|
|
427
|
+
* pickup a ticket, the entry carries the ticket of the pickup that took it, `release` and `orphan` act only on the entry
|
|
428
|
+
* their own ticket took, and the gate DEFERS a job whose id the ledger already holds (`running-here`, running or orphan)
|
|
429
|
+
* without making it a waiter: what that id promised is still in use until its own pickup gives it back.
|
|
430
|
+
*
|
|
431
|
+
* SEEDED AT BOOT. The ledger lives in process memory, so a worker that restarts starts empty while a job
|
|
432
|
+
* container the boot reaper could not remove may still run. `survivors` lists what remains on every blessed venue and
|
|
433
|
+
* each is seeded as an ORPHAN from its `pi.dispatch.mem` and `pi.dispatch.cpu` labels (`pessimisticSize` without
|
|
434
|
+
* them, capped at the budget, so the first refresh is awaited before the first listing); the sweep gives each back once
|
|
435
|
+
* the runtime says it is gone. Until a venue's listing has been read once the gate ADMITS NOTHING ON THAT VENUE
|
|
436
|
+
* (`unseeded`, fail closed, said once per streak and venue as `host_budget_seed_unread`), and every tick asks again: an
|
|
437
|
+
* empty ledger beside containers nobody counted is the overcommit the budget exists to refuse.
|
|
438
|
+
*
|
|
439
|
+
* PER VENUE (gate round 2 of phase 2). One listing for the whole host made a venue that cannot be listed stop
|
|
440
|
+
* every OTHER venue's jobs too: `PI_BACKENDS=local,podman` with Podman absent or down answered `unseeded` for every
|
|
441
|
+
* docker job, forever. A job is told its venue (`gate`'s `venue`, the default when it names none), and only its own
|
|
442
|
+
* venue's unread listing stops it. A job whose venue is not known (null with no `defaultVenue`) is stopped by any.
|
|
443
|
+
*
|
|
444
|
+
* A SURVIVOR'S NAME IS TAKEN. A seeded survivor is keyed `container:<name>`, not by a job id, so the id check
|
|
445
|
+
* alone admitted the job whose container that is (`pi-job-<id>`, retried here after a restart), and its `docker run`
|
|
446
|
+
* then created a container the sweep took for the survivor and removed. The gate now defers `running-here` any job
|
|
447
|
+
* whose container name an orphan carries, and the sweep never asks about a name a live entry carries.
|
|
448
|
+
*/
|
|
449
|
+
export function makeHostBudget({ settings, jobDefault = { memMiB: 4096, cpuCenti: 200 }, readFacts = async () => ({}), scopedLimits = () => [], countLimit, containerGone = async () => null, survivors = null, defaultVenue = null, onRefresh = () => {}, now = () => Date.now(), log = () => {}, stateReadBoundMs = HOLD_STATE_READ_BOUND_MS }) {
|
|
450
|
+
if (typeof countLimit !== "function") throw new TypeError("makeHostBudget: countLimit is required (the live PI_CONCURRENCY, the budget's third dimension)");
|
|
451
|
+
let budget = { memMiB: null, cpuCenti: null };
|
|
452
|
+
let detail = null;
|
|
453
|
+
let unknownSaid = "";
|
|
454
|
+
/** jobId (or `container:<name>` for a seeded survivor) -> { id, project, memMiB, cpuCenti, at, ticket, name, orphan: null | { name, venue, since } } */
|
|
455
|
+
const ledger = new Map();
|
|
456
|
+
/** jobId -> { id, project, memMiB, cpuCenti, firstAt, lastAt, checkedAt, suspended, state } */
|
|
457
|
+
const waiters = new Map();
|
|
458
|
+
/** jobId -> how many of this host's pickups are handling it right now, so a hold of an `active` job is not mistaken for elsewhere. */
|
|
459
|
+
const handling = new Map();
|
|
460
|
+
let tickets = 0;
|
|
461
|
+
// The venues whose boot listing is not read yet, each with its lister: `*` for a single whole-host listing.
|
|
462
|
+
const WHOLE_HOST = "*";
|
|
463
|
+
const listers = new Map(typeof survivors === "function" ? [[WHOLE_HOST, survivors]] : survivors && typeof survivors === "object" ? Object.entries(survivors).filter(([, fn]) => typeof fn === "function") : []);
|
|
464
|
+
const unseeded = new Set(listers.keys());
|
|
465
|
+
const seedSaid = new Set();
|
|
466
|
+
/** Whether a job on `venue` (null: the default) must wait for an unread listing. */
|
|
467
|
+
const blockedBySeed = (venue) => {
|
|
468
|
+
if (unseeded.size === 0) return false;
|
|
469
|
+
if (unseeded.has(WHOLE_HOST)) return true;
|
|
470
|
+
const v = venue ?? defaultVenue;
|
|
471
|
+
return v === null || v === undefined ? true : unseeded.has(v);
|
|
472
|
+
};
|
|
473
|
+
const rulesOf = (limits) => ({
|
|
474
|
+
shareOf: (p) => projectBudgetRow(limits, p).hostShare,
|
|
475
|
+
minJobsOf: (p) => projectBudgetRow(limits, p).minJobs,
|
|
476
|
+
});
|
|
477
|
+
const countNow = () => {
|
|
478
|
+
let limit = null;
|
|
479
|
+
try {
|
|
480
|
+
limit = countLimit();
|
|
481
|
+
} catch {
|
|
482
|
+
limit = null;
|
|
483
|
+
}
|
|
484
|
+
return Number.isSafeInteger(limit) && limit >= 0 ? limit : null;
|
|
485
|
+
};
|
|
486
|
+
|
|
487
|
+
const refresh = async () => {
|
|
488
|
+
let facts = {};
|
|
489
|
+
try {
|
|
490
|
+
facts = (await readFacts()) ?? {};
|
|
491
|
+
} catch {
|
|
492
|
+
facts = {};
|
|
493
|
+
}
|
|
494
|
+
const next = computeHostBudget(settings, facts, jobDefault);
|
|
495
|
+
budget = { memMiB: next.memMiB, cpuCenti: next.cpuCenti };
|
|
496
|
+
detail = next.detail;
|
|
497
|
+
for (const entry of ledger.values()) {
|
|
498
|
+
if (entry.guess) Object.assign(entry, pessimisticSize([], entry.guess, budget));
|
|
499
|
+
}
|
|
500
|
+
const unknown = [budget.memMiB === null ? "memory" : null, budget.cpuCenti === null ? "cpus" : null].filter(Boolean).join(",");
|
|
501
|
+
if (unknown !== unknownSaid) {
|
|
502
|
+
unknownSaid = unknown;
|
|
503
|
+
if (unknown !== "") log("host_budget_unknown", { unknown, reason: "the runtime gave no memory or CPU count, so no job is held back on it" });
|
|
504
|
+
else log("host_budget", { memMiB: budgetField(budget.memMiB), cpuCenti: budgetField(budget.cpuCenti), memFloored: detail.memFloored, cpuFloored: detail.cpuFloored });
|
|
505
|
+
}
|
|
506
|
+
try {
|
|
507
|
+
Promise.resolve(onRefresh({ ...budget }, facts)).catch(() => {});
|
|
508
|
+
} catch {
|
|
509
|
+
// a hook that throws synchronously is the hook's own failure, never the budget's
|
|
510
|
+
}
|
|
511
|
+
return budget;
|
|
512
|
+
};
|
|
513
|
+
|
|
514
|
+
/** Reads every venue's listing still unread, each on its own: one venue that cannot be listed leaves the others read. */
|
|
515
|
+
const seedVenue = async (venue) => {
|
|
516
|
+
let listed;
|
|
517
|
+
try {
|
|
518
|
+
listed = await listers.get(venue)();
|
|
519
|
+
if (!Array.isArray(listed)) throw new Error("not a listing");
|
|
520
|
+
} catch {
|
|
521
|
+
if (!seedSaid.has(venue)) {
|
|
522
|
+
seedSaid.add(venue);
|
|
523
|
+
log("host_budget_seed_unread", { venue: venue === WHOLE_HOST ? null : venue, reason: "the job containers left from before this worker started could not be listed, so no job on this venue is admitted until they are" });
|
|
524
|
+
}
|
|
525
|
+
return;
|
|
526
|
+
}
|
|
527
|
+
// The guess is kept uncapped on the entry and capped at the budget in force, here and again on every refresh, so
|
|
528
|
+
// the order of the first refresh and the first listing does not decide the size.
|
|
529
|
+
const guess = pessimisticSize(scopedLimits(), jobDefault);
|
|
530
|
+
const t = now();
|
|
531
|
+
for (const c of listed) {
|
|
532
|
+
if (typeof c?.name !== "string" || c.name === "") continue;
|
|
533
|
+
const id = `container:${c.name}`;
|
|
534
|
+
const labelled = Number.isSafeInteger(c.memMiB) && c.memMiB > 0 && Number.isSafeInteger(c.cpuCenti) && c.cpuCenti > 0;
|
|
535
|
+
const size = labelled ? { memMiB: c.memMiB, cpuCenti: c.cpuCenti } : pessimisticSize([], guess, budget);
|
|
536
|
+
ledger.set(id, { id, project: null, memMiB: size.memMiB, cpuCenti: size.cpuCenti, at: t, ticket: null, name: c.name, orphan: { name: c.name, venue: c.venue ?? null, since: t }, guess: labelled ? null : guess });
|
|
537
|
+
log("host_budget_seeded", { memMiB: size.memMiB, cpuCenti: size.cpuCenti, labelled });
|
|
538
|
+
}
|
|
539
|
+
unseeded.delete(venue);
|
|
540
|
+
if (seedSaid.has(venue)) log("host_budget_seed_read", { venue: venue === WHOLE_HOST ? null : venue, seeded: listed.length });
|
|
541
|
+
seedSaid.delete(venue);
|
|
542
|
+
};
|
|
543
|
+
const seed = async () => {
|
|
544
|
+
await Promise.all([...unseeded].map(seedVenue));
|
|
545
|
+
return unseeded.size === 0;
|
|
546
|
+
};
|
|
547
|
+
const holdsNow = (limits) => rankHolds(waiters.values(), [...ledger.values()], rulesOf(limits).minJobsOf);
|
|
548
|
+
|
|
549
|
+
/** Drops the holds of waiters whose job is not coming back (see the header). Every queue read is bounded. */
|
|
550
|
+
const verify = async () => {
|
|
551
|
+
const t = now();
|
|
552
|
+
for (const w of [...waiters.values()]) {
|
|
553
|
+
if (t - Math.max(w.lastAt, w.checkedAt) < HOLD_VERIFY_AFTER_MS) continue;
|
|
554
|
+
let state;
|
|
555
|
+
try {
|
|
556
|
+
if (typeof w.state !== "function") throw new Error("no state reader");
|
|
557
|
+
state = await bounded(w.state(), stateReadBoundMs);
|
|
558
|
+
} catch {
|
|
559
|
+
// Judged at the clock NOW, not at the tick's start: a bounded read still took its time.
|
|
560
|
+
if (now() - w.lastAt >= HOLD_DROP_ON_ERROR_MS && waiters.get(w.id) === w) {
|
|
561
|
+
waiters.delete(w.id);
|
|
562
|
+
log("host_budget_hold_dropped", { jobId: w.id, because: "unverifiable" });
|
|
563
|
+
}
|
|
564
|
+
continue;
|
|
565
|
+
}
|
|
566
|
+
// The waiter may have come back (or been replaced) while the state was read: only the one read is judged.
|
|
567
|
+
if (waiters.get(w.id) !== w) continue;
|
|
568
|
+
const gone = state === "completed" || state === "failed" || state === "unknown" || (state === "active" && !handling.has(w.id));
|
|
569
|
+
if (gone) {
|
|
570
|
+
waiters.delete(w.id);
|
|
571
|
+
log("host_budget_hold_dropped", { jobId: w.id, because: state === "active" ? "active-elsewhere" : state });
|
|
572
|
+
} else {
|
|
573
|
+
waiters.set(w.id, { ...w, checkedAt: t });
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
};
|
|
577
|
+
|
|
578
|
+
/**
|
|
579
|
+
* Gives back an orphan's hold once the runtime says its container is gone. NEVER asks about a name a live (admitted,
|
|
580
|
+
* not orphaned) entry carries: the runtime's answer about that name is the live job's container, and asking
|
|
581
|
+
* removes one not yet started. The gate keeps that from arising; this is the sweep not relying on it.
|
|
582
|
+
*/
|
|
583
|
+
const sweep = async () => {
|
|
584
|
+
for (const entry of [...ledger.values()]) {
|
|
585
|
+
if (!entry.orphan) continue;
|
|
586
|
+
if (typeof entry.orphan.name === "string" && [...ledger.values()].some((e) => !e.orphan && e.name === entry.orphan.name)) continue;
|
|
587
|
+
let gone = null;
|
|
588
|
+
try {
|
|
589
|
+
gone = await containerGone(entry.orphan.name, entry.orphan.venue);
|
|
590
|
+
} catch {
|
|
591
|
+
gone = null;
|
|
592
|
+
}
|
|
593
|
+
if (gone === true && ledger.get(entry.id) === entry) {
|
|
594
|
+
ledger.delete(entry.id);
|
|
595
|
+
log("host_budget_orphan_released", { jobId: entry.id, heldForMs: now() - entry.orphan.since });
|
|
596
|
+
}
|
|
597
|
+
}
|
|
598
|
+
};
|
|
599
|
+
|
|
600
|
+
// ONE IN FLIGHT PER PIECE: a tick that finds a piece still running from an earlier tick skips that piece
|
|
601
|
+
// rather than starting a second, and the pieces never wait on one another, so a slow facts read or queue read cannot
|
|
602
|
+
// pile ticks up behind it or keep the sweep from giving an orphan's room back.
|
|
603
|
+
const running = new Map();
|
|
604
|
+
const once = (key, fn) => {
|
|
605
|
+
if (running.has(key)) return running.get(key);
|
|
606
|
+
const p = Promise.resolve()
|
|
607
|
+
.then(fn)
|
|
608
|
+
.catch(() => {})
|
|
609
|
+
.finally(() => running.delete(key));
|
|
610
|
+
running.set(key, p);
|
|
611
|
+
return p;
|
|
612
|
+
};
|
|
613
|
+
// The first refresh BEFORE the first listing (the tick may still list first; a guessed size is re-capped either way).
|
|
614
|
+
const ready = refresh().then((first) => once("seed", seed).then(() => first));
|
|
615
|
+
|
|
616
|
+
return {
|
|
617
|
+
ready,
|
|
618
|
+
refresh,
|
|
619
|
+
/** The budget in force: `{ memMiB, cpuCenti }`, each an integer, `Infinity` (off) or null (unknown). */
|
|
620
|
+
current: () => ({ ...budget }),
|
|
621
|
+
detail: () => detail,
|
|
622
|
+
/** null when `size` can start here some day, else `host` or `share` (`neverFits`). */
|
|
623
|
+
neverFits: (size, project, limits = scopedLimits()) => neverFits(size, budget, project ? rulesOf(limits).shareOf(project) : null),
|
|
624
|
+
/** The project's `hostShare` in the given snapshot, for a refusal's record. */
|
|
625
|
+
shareOf: (project, limits = scopedLimits()) => (project ? rulesOf(limits).shareOf(project) : null),
|
|
626
|
+
/** Marks a job as inside one of this host's processors (`enter`, which hands the pickup its TICKET) or no longer (`leave`). */
|
|
627
|
+
enter(id) {
|
|
628
|
+
handling.set(id, (handling.get(id) ?? 0) + 1);
|
|
629
|
+
tickets += 1;
|
|
630
|
+
return tickets;
|
|
631
|
+
},
|
|
632
|
+
leave(id) {
|
|
633
|
+
const n = (handling.get(id) ?? 0) - 1;
|
|
634
|
+
if (n > 0) handling.set(id, n);
|
|
635
|
+
else handling.delete(id);
|
|
636
|
+
},
|
|
637
|
+
/**
|
|
638
|
+
* THE GATE, synchronous: `{ admitted: true }` with the hold taken under `ticket`, or `{ admitted: false, why }`:
|
|
639
|
+
* `unseeded` (the boot listing of the job's `venue` is not read yet) and `running-here` (the ledger already holds
|
|
640
|
+
* this id, or an orphan carries the container `name` this pickup will use) make no waiter; `budget` and `share`
|
|
641
|
+
* keep (or make) the job a waiter. `getState` is the job's own, kept for `verify`.
|
|
642
|
+
*/
|
|
643
|
+
gate({ id, ticket = null, project = null, size, venue = null, name = null, getState = null, limits = scopedLimits() }) {
|
|
644
|
+
if (blockedBySeed(venue)) return { admitted: false, why: "unseeded", rank: -1 };
|
|
645
|
+
if (ledger.has(id)) return { admitted: false, why: "running-here", rank: -1 };
|
|
646
|
+
if (name !== null && [...ledger.values()].some((e) => e.orphan?.name === name)) return { admitted: false, why: "running-here", rank: -1 };
|
|
647
|
+
const t = now();
|
|
648
|
+
const was = waiters.get(id);
|
|
649
|
+
const ask = { id, project, memMiB: size.memMiB, cpuCenti: size.cpuCenti, firstAt: was?.firstAt ?? t };
|
|
650
|
+
const rules = rulesOf(limits);
|
|
651
|
+
// The asking job is ranked as the waiter it is (or would be), so a tier 1 job arriving now outranks an old tier 2
|
|
652
|
+
// hold, which is what `minJobs` promises, and an old waiter keeps its rank across its own deferrals.
|
|
653
|
+
const candidates = new Map(waiters);
|
|
654
|
+
candidates.set(id, { ...ask, suspended: false });
|
|
655
|
+
const holds = rankHolds(candidates.values(), [...ledger.values()], rules.minJobsOf);
|
|
656
|
+
const verdict = admit({ budget: { ...budget, count: countNow() }, ledger: [...ledger.values()], holds, ask, shareOf: rules.shareOf });
|
|
657
|
+
if (verdict.ok) {
|
|
658
|
+
waiters.delete(id);
|
|
659
|
+
ledger.set(id, { id, project, memMiB: size.memMiB, cpuCenti: size.cpuCenti, at: t, ticket, name, orphan: null });
|
|
660
|
+
return { admitted: true };
|
|
661
|
+
}
|
|
662
|
+
// A share-stopped job is a waiter that holds nothing: the budget is not its obstacle.
|
|
663
|
+
waiters.set(id, { ...ask, lastAt: t, checkedAt: t, suspended: verdict.why === "share", state: typeof getState === "function" ? getState : was?.state ?? null });
|
|
664
|
+
return { admitted: false, why: verdict.why, rank: holds.findIndex((h) => h.id === id) };
|
|
665
|
+
},
|
|
666
|
+
/** Another gate deferred the job: its hold (if any) keeps its age and stops counting until it comes back. */
|
|
667
|
+
suspend(id) {
|
|
668
|
+
const w = waiters.get(id);
|
|
669
|
+
if (w) waiters.set(id, { ...w, suspended: true, lastAt: now() });
|
|
670
|
+
},
|
|
671
|
+
/** The job ended without waiting for the budget again (refused, failed, ran): its waiter is dropped. */
|
|
672
|
+
forget(id) {
|
|
673
|
+
waiters.delete(id);
|
|
674
|
+
},
|
|
675
|
+
/**
|
|
676
|
+
* Gives back a running job's hold. IDEMPOTENT: true only the first time, never for an orphan, and only for the
|
|
677
|
+
* entry this pickup's `ticket` took (another pickup of the same id gives back nothing of it).
|
|
678
|
+
*/
|
|
679
|
+
release(id, { ticket = null } = {}) {
|
|
680
|
+
const entry = ledger.get(id);
|
|
681
|
+
if (!entry || entry.orphan || entry.ticket !== ticket) return false;
|
|
682
|
+
ledger.delete(id);
|
|
683
|
+
return true;
|
|
684
|
+
},
|
|
685
|
+
/** The job's container may outlive it (its stop did not take): the hold stays until `sweep` sees the container gone. */
|
|
686
|
+
orphan(id, { name = null, venue = null, ticket = null } = {}) {
|
|
687
|
+
const entry = ledger.get(id);
|
|
688
|
+
if (!entry || entry.orphan || entry.ticket !== ticket) return false;
|
|
689
|
+
ledger.set(id, { ...entry, orphan: { name, venue, since: now() } });
|
|
690
|
+
log("host_budget_orphan", { jobId: id, memMiB: entry.memMiB, cpuCenti: entry.cpuCenti });
|
|
691
|
+
return true;
|
|
692
|
+
},
|
|
693
|
+
verify,
|
|
694
|
+
sweep,
|
|
695
|
+
/** Reads each venue's boot listing again while it has not been read (`unseeded`), one read in flight with the tick's. True once every one has. */
|
|
696
|
+
seed: () => once("seed", seed).then(() => unseeded.size === 0),
|
|
697
|
+
/** One tick: facts, the boot listing while unread, the stale holds and the orphans, each on its own. Never throws. */
|
|
698
|
+
tick() {
|
|
699
|
+
return Promise.all([once("refresh", refresh), once("seed", seed), once("verify", verify), once("sweep", sweep)]).then(() => undefined);
|
|
700
|
+
},
|
|
701
|
+
/**
|
|
702
|
+
* What the registry row and doctor show: the budget, what runs (orphans included), what the holds keep, and counts.
|
|
703
|
+
* Integers only, `seeded` (whether every venue's boot listing has been read) and `unseeded` (the venues whose has
|
|
704
|
+
* not, sorted; `*` for a whole-host listing).
|
|
705
|
+
*/
|
|
706
|
+
snapshot(limits = scopedLimits()) {
|
|
707
|
+
const entries = [...ledger.values()];
|
|
708
|
+
const used = sumOf(entries);
|
|
709
|
+
const holds = holdsNow(limits);
|
|
710
|
+
const held = sumOf(holds);
|
|
711
|
+
return {
|
|
712
|
+
memMiB: budget.memMiB,
|
|
713
|
+
cpuCenti: budget.cpuCenti,
|
|
714
|
+
usedMemMiB: used.memMiB,
|
|
715
|
+
usedCpuCenti: used.cpuCenti,
|
|
716
|
+
heldMemMiB: held.memMiB,
|
|
717
|
+
heldCpuCenti: held.cpuCenti,
|
|
718
|
+
running: entries.length,
|
|
719
|
+
orphans: entries.filter((e) => e.orphan).length,
|
|
720
|
+
holds: holds.length,
|
|
721
|
+
waiters: waiters.size,
|
|
722
|
+
seeded: unseeded.size === 0,
|
|
723
|
+
unseeded: [...unseeded].sort(),
|
|
724
|
+
};
|
|
725
|
+
},
|
|
726
|
+
/** Test and doctor seams: copies, never the live maps. */
|
|
727
|
+
entries: () => [...ledger.values()].map((e) => ({ ...e })),
|
|
728
|
+
waiting: () => [...waiters.values()].map(({ state: _s, ...w }) => ({ ...w })),
|
|
729
|
+
};
|
|
730
|
+
}
|
|
731
|
+
|
|
732
|
+
/** The largest memory and CPU sizes a project may run at on a host with this budget and share, for doctor. */
|
|
733
|
+
export function largestFit(budget, share = null) {
|
|
734
|
+
const cap = (b) => (b === null || b === Infinity ? b : share === null ? b : Math.floor((share * b) / 100));
|
|
735
|
+
return { memMiB: cap(budget.memMiB), cpuCenti: cap(budget.cpuCenti) };
|
|
736
|
+
}
|