@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,344 @@
1
+ /**
2
+ * The aggregate CPU reserve (issue #596, phase 2, DES-HOST-BUDGET): ONE parent cgroup for every job container, with the
3
+ * host's CPU budget as its quota, so all jobs TOGETHER leave the host's reserve free.
4
+ *
5
+ * WHY A PARENT AND NOT A FLAG PER JOB. `--cpus` is a quota per container and the quotas do not sum: two busy jobs on a
6
+ * 4-core host with `--cpus=3` used 4.06 cores (measured). The ledger in `host-budget.mjs` bounds the jobs' SIZES
7
+ * together, not what they USE. A parent cgroup with a quota bounds the use: with `pidispatch.slice` at the budget
8
+ * (NCPU-1), three busy jobs held 3.00-3.06 of 4 cores and 12.95-13.02 of 14 on all five venues of the lab, and a
9
+ * busy neighbour outside kept its core (round-size/pr3/lab/FACTS.md, Q2).
10
+ *
11
+ * THE PARENT ALONE ALREADY HELPS, which is why it is ALWAYS on a job's argv, quota or not. Inside a parent, a job's
12
+ * weight competes only with its sibling jobs; the parent competes with the egress proxy, Valkey and the host's services
13
+ * as ONE group of the default weight 100. Without it a job of weight 10000 (`--cpu-shares=262144`) left a single-threaded
14
+ * proxy-like container 0.01 of a core; inside a weight-100 parent with no quota at all the proxy kept 0.98-1.00 (Q3, Q4).
15
+ * The runtime creates a missing parent silently with no quota (every venue, exit 0, measured), so the flag never fails
16
+ * a run; nothing can be learned from a run either, which is why the quota is READ back here rather than assumed.
17
+ *
18
+ * ONE NAME, NO DASH. `--cgroup-parent=pidispatch.slice` is accepted by every driver measured (under systemd it is a
19
+ * slice, under Docker Desktop's cgroupfs driver the directory `/sys/fs/cgroup/pidispatch.slice`). A dash nests:
20
+ * `pd-jobs.slice` lands in `/pd.slice/pd-jobs.slice/`, where a quota put on the name the operator typed would be on
21
+ * another cgroup than the one the jobs share.
22
+ *
23
+ * WHO CAN SET THE QUOTA, per venue (Q1), and therefore what this module does:
24
+ * - rootless Podman (`user-systemd`): the worker's own account, through its systemd user manager, which delegates
25
+ * `cpu`: `systemctl --user set-property pidispatch.slice CPUQuota=<budget*100>%`. Persistent (a drop-in under
26
+ * `~/.config/systemd/user.control`), so a reboot keeps it; re-applied whenever the budget changes and read back.
27
+ * - Docker with the cgroupfs driver, Docker Desktop's (`helper`): kernel state in the runtime's VM, writable from a
28
+ * one-shot container that runs as uid 0 with every capability dropped and only the parent's own directory mounted.
29
+ * NOT persistent: a Docker Desktop restart drops it, so it is re-read on the facts' own cadence and re-applied.
30
+ * - Docker with the systemd driver and rootful Podman (`system-systemd`): root only (`Interactive authentication
31
+ * required.` for anyone else), so the worker only READS `systemctl show -P CPUQuotaPerSecUSec pidispatch.slice`
32
+ * and doctor prints the one operator command. A direct `cpu.max` write (root, or the docker group through a bind
33
+ * mount) works until the next `systemctl daemon-reload`, which every package upgrade runs, resets it: rejected.
34
+ *
35
+ * FAILS OPEN, AND SAYS WHICH. Every runtime command is bounded (`RESERVE_STEP_TIMEOUT_MS`) and runs off every job path
36
+ * (on the budget's refresh), so a hung `systemctl` or a slow daemon never holds a pickup. A quota that could not be set
37
+ * or read leaves the parent without one: the jobs still share the parent (and so still cannot starve the proxy), the
38
+ * reserve across them is not held, the worker logs `cpu_reserve_fail_open` with a named reason and doctor warns.
39
+ *
40
+ * NO AGGREGATE `memory.max`. Measured: a parent memory limit makes the KERNEL choose the victim by size, and it killed
41
+ * the innocent 300 MiB job while the one that grew lived (Q6). Admission by the ledger stays the memory bound; an
42
+ * aggregate memory limit is an operator-only backstop, documented, never set here.
43
+ */
44
+
45
+ import { parseCpuMax } from "./host-budget.mjs";
46
+ import { CGROUP_PARENT } from "./job-size.mjs";
47
+
48
+ // The parent's name and the parentless shares cap live in `job-size.mjs`, the leaf `container-spec.mjs` may import (its
49
+ // leaf property is what `packages.mjs` rests on); re-exported here, where they are explained.
50
+ export { CGROUP_PARENT, PARENTLESS_SHARES_MAX } from "./job-size.mjs";
51
+
52
+ /** Each runtime command's bound: a `systemctl` call or a one-shot container's whole life. */
53
+ export const RESERVE_STEP_TIMEOUT_MS = 15_000;
54
+
55
+ /**
56
+ * How often a quota already read back as held is read again: the cadence the runtime's own facts are re-read at
57
+ * (`JOB_USER_FACTS_MAX_AGE_MS`, ten minutes), so a Docker Desktop restart that dropped the helper's write is found on
58
+ * the same clock that finds its new CPU count. Restated here rather than imported, to keep this module a leaf; a test
59
+ * holds the two equal.
60
+ */
61
+ export const RESERVE_RECHECK_MS = 10 * 60_000;
62
+
63
+ /** How soon a quota NOT held (unset, unreadable, a failed write) is tried again. */
64
+ export const RESERVE_RETRY_MS = 60_000;
65
+
66
+ /** The CFS period every quota here is written in, in microseconds (the kernel's and systemd's default). */
67
+ export const QUOTA_PERIOD_US = 100_000;
68
+
69
+ /** systemd's property for a CPU budget in hundredths: one CPU is `100%`, so the hundredths ARE the percentage. */
70
+ export function cpuQuotaProperty(cpuCenti) {
71
+ return `CPUQuota=${cpuCenti}%`;
72
+ }
73
+
74
+ /** The `cpu.max` line for a CPU budget in hundredths: `<quota> 100000`, integers only (1 CPU = 100000 us per period). */
75
+ export function cpuMaxLine(cpuCenti) {
76
+ return `${cpuCenti * (QUOTA_PERIOD_US / 100)} ${QUOTA_PERIOD_US}`;
77
+ }
78
+
79
+ /** The command an operator runs, once, as root, on a venue whose quota only root can set. */
80
+ export function operatorQuotaCommand(cpuCenti) {
81
+ return cpuCenti === null ? `sudo systemctl set-property ${CGROUP_PARENT} CPUQuota=` : `sudo systemctl set-property ${CGROUP_PARENT} ${cpuQuotaProperty(cpuCenti)}`;
82
+ }
83
+
84
+ /** The rootless venue's own command (no root), as the worker runs it: `null` clears the quota. */
85
+ export function userQuotaCommand(cpuCenti) {
86
+ return `systemctl --user set-property ${CGROUP_PARENT} ${cpuCenti === null ? "CPUQuota=" : cpuQuotaProperty(cpuCenti)}`;
87
+ }
88
+
89
+ /** `systemctl` argv (no leading binary) that sets, or with `null` clears, the user manager's quota on the parent. */
90
+ export function userSetPropertyArgs(cpuCenti) {
91
+ return ["--user", "set-property", CGROUP_PARENT, cpuCenti === null ? "CPUQuota=" : cpuQuotaProperty(cpuCenti)];
92
+ }
93
+
94
+ /** `systemctl` argv that reads the parent's quota, from the user manager (`user: true`) or the system one. */
95
+ export function showQuotaArgs({ user }) {
96
+ return [...(user ? ["--user"] : []), "show", "-P", "CPUQuotaPerSecUSec", CGROUP_PARENT];
97
+ }
98
+
99
+ /** The host path of the parent under the cgroupfs driver. The ONLY `/sys/fs/cgroup` path the helper ever mounts. */
100
+ export const HELPER_PARENT_PATH = `/sys/fs/cgroup/${CGROUP_PARENT}`;
101
+
102
+ /**
103
+ * The one-shot helper's `docker run` argv (no leading binary): with `cpuCenti` it WRITES the quota (`null` writes `max`,
104
+ * no quota), without (`read: true`) it only reads `cpu.max` through a read-only mount.
105
+ *
106
+ * Every flag is a reason. uid 0 because writing a knob of an existing cgroup needs it even with every capability
107
+ * dropped (uid 1000 got `Permission denied`, measured); `--cap-drop=ALL` because creating a cgroup needs
108
+ * CAP_DAC_OVERRIDE and this must not be able to; `--network=none`, `--read-only`, `no-new-privileges`; `--pull=never`
109
+ * and an image the worker already pins (the job image), never a tag fetched for this (CONST-PI-VERSION-PINNED);
110
+ * `--cgroup-parent` so the helper itself is a job-parented container and the parent exists before the mount resolves.
111
+ * The mount is the parent's own directory and nothing else: a bind source docker does not find it CREATES, and under
112
+ * `/sys/fs/cgroup` that mkdir makes a cgroup, so any other path (a typo) would leave a stray one. The value is built
113
+ * from an integer, so nothing an operator or a job wrote reaches the shell.
114
+ */
115
+ export function helperArgs({ image, cpuCenti = undefined, read = false }) {
116
+ if (typeof image !== "string" || image === "" || image.startsWith("-")) throw new Error(`cpu reserve: refusing a helper image ${JSON.stringify(image)}`);
117
+ const write = !read;
118
+ if (write && cpuCenti !== null && !(Number.isSafeInteger(cpuCenti) && cpuCenti > 0)) throw new Error(`cpu reserve: refusing a quota that is not a positive integer of hundredths: ${JSON.stringify(cpuCenti)}`);
119
+ const line = write ? (cpuCenti === null ? `max ${QUOTA_PERIOD_US}` : cpuMaxLine(cpuCenti)) : null;
120
+ return [
121
+ "run",
122
+ "--rm",
123
+ "--pull=never",
124
+ "--user=0:0",
125
+ "--cap-drop=ALL",
126
+ "--security-opt",
127
+ "no-new-privileges",
128
+ "--network=none",
129
+ "--read-only",
130
+ `--cgroup-parent=${CGROUP_PARENT}`,
131
+ "-v",
132
+ `${HELPER_PARENT_PATH}:/p${write ? "" : ":ro"}`,
133
+ "--entrypoint",
134
+ write ? "sh" : "cat",
135
+ image,
136
+ ...(write ? ["-c", `echo "${line}" > /p/cpu.max`] : ["/p/cpu.max"]),
137
+ ];
138
+ }
139
+
140
+ /**
141
+ * A systemd timespan as `systemctl show` prints `CPUQuotaPerSecUSec`, in hundredths of a CPU: `{ cpuCenti }`, with
142
+ * `cpuCenti: null` for `infinity` (no quota), or null when the text is not a timespan. `3s` is 300% (measured), and
143
+ * systemd prints a fraction either as `1.500000s` or `1s 500ms` depending on its version, so every unit and both forms
144
+ * are read. A quota below a hundredth of a CPU rounds down to nothing and is no fact.
145
+ */
146
+ export function parseQuotaPerSec(text) {
147
+ const value = String(text ?? "").trim();
148
+ if (value === "infinity") return { cpuCenti: null };
149
+ if (!/^(?:\d{1,12}(?:\.\d{1,9})?(?:us|µs|μs|ms|s|min|h)\s*)+$/.test(value)) return null;
150
+ const unit = { us: 1, µs: 1, μs: 1, ms: 1_000, s: 1_000_000, min: 60_000_000, h: 3_600_000_000 };
151
+ let usec = 0;
152
+ for (const m of value.matchAll(/(\d{1,12}(?:\.\d{1,9})?)(us|µs|μs|ms|s|min|h)/g)) usec += Number(m[1]) * unit[m[2]];
153
+ const centi = Math.floor(Math.round(usec) / 10_000);
154
+ return Number.isSafeInteger(centi) && centi > 0 ? { cpuCenti: centi } : null;
155
+ }
156
+
157
+ /** A `cpu.max` read as `{ cpuCenti }` (`null` for `max <period>`, no quota), or null when it is not a `cpu.max` line. */
158
+ export function parseCpuMaxRead(text) {
159
+ const value = String(text ?? "").trim();
160
+ if (/^max [0-9]{1,15}$/.test(value)) return { cpuCenti: null };
161
+ const centi = parseCpuMax(value);
162
+ return centi === null ? null : { cpuCenti: centi };
163
+ }
164
+
165
+ /**
166
+ * Whether a container on this runtime is started under the parent (`CGROUP_PARENT`) or without one (`null`), from the
167
+ * runtime's own facts. Without only where Podman says it uses a cgroup manager other than `systemd`: there libpod puts a
168
+ * rootless container with a named parent at `/<parent>/libpod-<id>` from the hierarchy's ROOT (source, not measured), a
169
+ * directory a rootless account cannot create, so a job would fail to start. Every other answer keeps the parent,
170
+ * including no answer (a podman venue whose `info` did not answer runs no job anyway). Docker's drivers both take it.
171
+ */
172
+ export function cgroupParentFor(runtime) {
173
+ if (runtime?.podman === true && typeof runtime.cgroupManager === "string" && runtime.cgroupManager !== "systemd") return null;
174
+ return CGROUP_PARENT;
175
+ }
176
+
177
+ /**
178
+ * HOW the quota is kept on one venue (PURE), from that venue's facts: `{ venue, parent, method, why }`.
179
+ * `parent` false when jobs run without the parent (`cgroupParentFor`): their shares are then capped instead
180
+ * `method` `user-systemd` | `helper` | `system-systemd` | null (the quota is not kept here; `why` says which)
181
+ * `why` a fixed token: `podman-cgroupfs`, `cgroup-v1`, `remote-daemon`, `rootless-docker`, `driver-unknown`,
182
+ * `podman-rootful-remote` or null
183
+ * `facts` is the venue's parsed info: for `local` the `docker info` facts (`parseDaemonFacts`, with `cgroupDriver` and
184
+ * `cgroupVersion`), for `podman` the `podman info` read (`parsePodmanInfo`). `endpointLocal` is the docker endpoint's
185
+ * observed locality, and `platform` this process's: a systemd read on THIS host says nothing about a remote daemon's.
186
+ */
187
+ export function reservePlan({ venue, facts, endpointLocal = false, platform = process.platform }) {
188
+ const plan = (parent, method, why) => ({ venue, parent, method, why });
189
+ if (venue === "podman") {
190
+ if (cgroupParentFor({ podman: true, cgroupManager: facts?.cgroupManager }) === null) return plan(false, null, "podman-cgroupfs");
191
+ if (facts?.cgroupVersion && facts.cgroupVersion !== "v2") return plan(true, null, "cgroup-v1");
192
+ if (facts?.rootless === true) return plan(true, "user-systemd", null);
193
+ return plan(true, null, "podman-rootful-remote");
194
+ }
195
+ if (facts?.cgroupVersion && facts.cgroupVersion !== "v2") return plan(true, null, "cgroup-v1");
196
+ if (facts?.rootless === true) return plan(true, null, "rootless-docker");
197
+ if (facts?.cgroupDriver === "cgroupfs") return plan(true, "helper", null);
198
+ if (facts?.cgroupDriver === "systemd") return endpointLocal === true && platform === "linux" ? plan(true, "system-systemd", null) : plan(true, null, "remote-daemon");
199
+ return plan(true, null, "driver-unknown");
200
+ }
201
+
202
+ /** A runtime command's failure as a fixed token, never its stderr (which can carry paths and endpoints). */
203
+ export function stepFailure(result, bin) {
204
+ // Two runner shapes: `execDockerBounded`'s `error`, and doctor's `liveRunVia`'s `ended` (`timeout` or `error`).
205
+ if (result?.error?.timedOut || result?.error?.killed || result?.ended === "timeout") return "timeout";
206
+ if (result?.ended === "error") return "spawn-failed";
207
+ if (result?.error?.code === "ENOENT") return `${bin}-not-found`;
208
+ if (typeof result?.error?.code === "string") return `spawn-${result.error.code.toLowerCase()}`;
209
+ if (Number.isSafeInteger(result?.code) && result.code !== 0) return `exit-${result.code}`;
210
+ if (result?.error) return "spawn-failed";
211
+ return null;
212
+ }
213
+
214
+ /**
215
+ * Read the parent's quota on a venue: `{ ok: true, cpuCenti }` (`null` is no quota) or `{ ok: false, reason }`.
216
+ * `run(bin, args, { timeoutMs })` resolves `{ code, stdout, error }` and never rejects (`execDockerBounded`'s shape).
217
+ */
218
+ export async function readQuota(plan, { run, image, bin = plan.venue === "podman" ? "podman" : "docker", timeoutMs = RESERVE_STEP_TIMEOUT_MS }) {
219
+ let result;
220
+ let parse;
221
+ let used;
222
+ if (plan.method === "user-systemd" || plan.method === "system-systemd") {
223
+ used = "systemctl";
224
+ result = await run("systemctl", showQuotaArgs({ user: plan.method === "user-systemd" }), { timeoutMs }).catch((error) => ({ code: null, stdout: "", error }));
225
+ parse = parseQuotaPerSec;
226
+ } else if (plan.method === "helper") {
227
+ used = bin;
228
+ result = await run(bin, helperArgs({ image, read: true }), { timeoutMs }).catch((error) => ({ code: null, stdout: "", error }));
229
+ parse = parseCpuMaxRead;
230
+ } else {
231
+ return { ok: false, reason: plan.why ?? "unmanaged" };
232
+ }
233
+ const failed = stepFailure(result, used);
234
+ if (failed) return { ok: false, reason: failed };
235
+ const read = parse(result?.stdout);
236
+ return read === null ? { ok: false, reason: "unparseable" } : { ok: true, cpuCenti: read.cpuCenti };
237
+ }
238
+
239
+ /** Set (or with `null` clear) the parent's quota where this worker may: `{ ok: true }` or `{ ok: false, reason }`. */
240
+ export async function writeQuota(plan, cpuCenti, { run, image, bin = plan.venue === "podman" ? "podman" : "docker", timeoutMs = RESERVE_STEP_TIMEOUT_MS }) {
241
+ let result;
242
+ let used;
243
+ if (plan.method === "user-systemd") {
244
+ used = "systemctl";
245
+ result = await run("systemctl", userSetPropertyArgs(cpuCenti), { timeoutMs }).catch((error) => ({ code: null, stdout: "", error }));
246
+ } else if (plan.method === "helper") {
247
+ used = bin;
248
+ result = await run(bin, helperArgs({ image, cpuCenti }), { timeoutMs }).catch((error) => ({ code: null, stdout: "", error }));
249
+ } else {
250
+ return { ok: false, reason: plan.method === "system-systemd" ? "needs-root" : (plan.why ?? "unmanaged") };
251
+ }
252
+ const failed = stepFailure(result, used);
253
+ return failed ? { ok: false, reason: failed } : { ok: true };
254
+ }
255
+
256
+ /**
257
+ * The reserve's state per venue, kept in step with the budget (`sync`, on every budget refresh; never on a job path).
258
+ *
259
+ * `sync({ cpuCenti, plans })`: `cpuCenti` is the budget in force (an integer, `Infinity` for off, null for unknown) and
260
+ * `plans` the venues' `reservePlan`s. Per venue, the quota WANTED is the budget, or no quota when the budget is off
261
+ * (the worker clears what it set: an old persistent drop-in would otherwise keep capping jobs at a budget the operator
262
+ * turned off). A venue is read again when the wanted quota or its method changed, when a held quota is
263
+ * `RESERVE_RECHECK_MS` old, and when one not held is `RESERVE_RETRY_MS` old. Where the worker may write, a quota that
264
+ * reads back wrong is written and read back once more. One sync at a time: a sync asked for while one runs is dropped
265
+ * (the next refresh asks again), so slow commands cannot pile up.
266
+ *
267
+ * States (`status`): `held` (the quota read back is the one wanted), `unset` / `differs` (read back, not as wanted:
268
+ * on `system-systemd` the operator's command is the fix), `unreadable`, `unmanaged` (no method here, `why`),
269
+ * `no-parent`, `budget-unknown`. Logged once per change: `cpu_reserve` when held, `cpu_reserve_fail_open` otherwise.
270
+ */
271
+ /**
272
+ * What a reserve that is not held MEANS for the jobs, one sentence per case. `differs` is two different truths, and the
273
+ * generic "no quota held" line was the opposite of one of them (the executing review of the aggregate reserve): a quota
274
+ * IS held there, the wrong one. With the budget off (`wantCenti` null) and the operator's quota still set, jobs together
275
+ * are capped to a budget the operator turned off, which is what doctor says too.
276
+ */
277
+ export function failOpenSaid(state) {
278
+ if (state.status === "no-parent") return "jobs run without the parent, their CPU weight capped at 1024: a fair share for the proxy and Valkey, not a reserve";
279
+ if (state.status === "differs" && Number.isSafeInteger(state.quotaCenti)) {
280
+ const held = `${state.quotaCenti / 100} CPUs`;
281
+ if (state.wantCenti === null) return `the CPU budget is off, but the parent still holds a quota of ${held}, so jobs together are still capped to it until it is cleared`;
282
+ return `jobs run under the parent held to a quota of ${held}, not the CPU budget of ${state.wantCenti / 100}: jobs together are capped there, and the reserve kept is not the one configured`;
283
+ }
284
+ return "jobs run under the parent with no quota held across them: the proxy and Valkey are not starved, the host's CPU reserve is not kept";
285
+ }
286
+
287
+ export function makeCpuReserve({ run, image, now = () => Date.now(), log = () => {} }) {
288
+ const states = new Map();
289
+ const said = new Map();
290
+ let running = false;
291
+
292
+ const record = (venue, state) => {
293
+ states.set(venue, state);
294
+ const key = `${state.status}|${state.method}|${state.wantCenti}|${state.quotaCenti}|${state.reason}`;
295
+ if (said.get(venue) === key) return;
296
+ said.set(venue, key);
297
+ const fields = { venue, method: state.method, status: state.status, wantCenti: state.wantCenti ?? "none", quotaCenti: state.quotaCenti === undefined ? "unread" : (state.quotaCenti ?? "none"), reason: state.reason ?? "" };
298
+ if (state.status === "held") log("cpu_reserve", fields);
299
+ else log("cpu_reserve_fail_open", { ...fields, failOpen: failOpenSaid(state) });
300
+ };
301
+
302
+ const one = async (plan, cpuCenti) => {
303
+ const base = { venue: plan.venue, method: plan.method, wantCenti: undefined, quotaCenti: undefined, reason: plan.why, checkedAt: now() };
304
+ if (!plan.parent) return record(plan.venue, { ...base, status: "no-parent" });
305
+ if (cpuCenti === null || cpuCenti === undefined) return record(plan.venue, { ...base, status: "budget-unknown", reason: "budget-unknown" });
306
+ const want = cpuCenti === Infinity ? null : cpuCenti;
307
+ if (!plan.method) return record(plan.venue, { ...base, wantCenti: want, status: "unmanaged" });
308
+ const prev = states.get(plan.venue);
309
+ const t = now();
310
+ const fresh = prev && prev.wantCenti === want && prev.method === plan.method && t - prev.checkedAt < (prev.status === "held" ? RESERVE_RECHECK_MS : RESERVE_RETRY_MS);
311
+ if (fresh) return undefined;
312
+ const seen = await readQuota(plan, { run, image });
313
+ if (seen.ok && seen.cpuCenti === want) return record(plan.venue, { ...base, wantCenti: want, quotaCenti: seen.cpuCenti, status: "held", reason: null, checkedAt: now() });
314
+ if (plan.method === "system-systemd") {
315
+ return record(plan.venue, { ...base, wantCenti: want, quotaCenti: seen.ok ? seen.cpuCenti : undefined, status: !seen.ok ? "unreadable" : seen.cpuCenti === null ? "unset" : "differs", reason: seen.ok ? "needs-root" : seen.reason, checkedAt: now() });
316
+ }
317
+ const wrote = await writeQuota(plan, want, { run, image });
318
+ const back = await readQuota(plan, { run, image });
319
+ if (wrote.ok && back.ok && back.cpuCenti === want) return record(plan.venue, { ...base, wantCenti: want, quotaCenti: back.cpuCenti, status: "held", reason: null, checkedAt: now() });
320
+ const status = !back.ok ? "unreadable" : back.cpuCenti === null ? "unset" : "differs";
321
+ return record(plan.venue, { ...base, wantCenti: want, quotaCenti: back.ok ? back.cpuCenti : undefined, status, reason: !wrote.ok ? `write-${wrote.reason}` : !back.ok ? `read-${back.reason}` : "not-applied", checkedAt: now() });
322
+ };
323
+
324
+ return {
325
+ async sync({ cpuCenti, plans = [] }) {
326
+ if (running) return false;
327
+ running = true;
328
+ try {
329
+ for (const plan of plans) {
330
+ try {
331
+ await one(plan, cpuCenti);
332
+ } catch (error) {
333
+ record(plan.venue, { venue: plan.venue, method: plan.method, wantCenti: undefined, quotaCenti: undefined, status: "unreadable", reason: `threw-${String(error?.code ?? "error").toLowerCase().replace(/[^a-z0-9-]/g, "").slice(0, 24) || "error"}`, checkedAt: now() });
334
+ }
335
+ }
336
+ } finally {
337
+ running = false;
338
+ }
339
+ return true;
340
+ },
341
+ /** Copies of each venue's state, for the registry and tests. */
342
+ states: () => [...states.values()].map((s) => ({ ...s })),
343
+ };
344
+ }
@@ -78,6 +78,14 @@ export function parseDaemonFacts(output) {
78
78
  // listener (source): `unix://...` or `tcp://...` for a running service, the bare default path in-process.
79
79
  remoteSocketPath: isUnixSocketPath(path) ? path : null,
80
80
  serverVersion: displayVersion(body.version?.Version),
81
+ // Issue #596: the runtime's own CPU count, which sets every job's `--cpus` ceiling (`hostCpuCeiling`).
82
+ hostCpus: cpuCount(body.host.cpus),
83
+ // Issue #596, phase 2: the memory the runtime's host has, in MiB, for the host budget's `auto` (`host-budget.mjs`).
84
+ memTotalMiB: memoryMiB(body.host.memTotal),
85
+ // Issue #596, phase 2: how this runtime manages cgroups, which decides how the jobs' parent cgroup gets its
86
+ // quota (`cpu-reserve.mjs` `reservePlan`). Podman's own words: `host.cgroupManager`, `host.cgroupVersion`.
87
+ cgroupDriver: cgroupWord(body.host.cgroupManager),
88
+ cgroupVersion: cgroupVersionOf(body.host.cgroupVersion),
81
89
  } };
82
90
  }
83
91
  if (typeof body.ServerVersion === "string" && body.ServerVersion !== "" && (typeof body.OperatingSystem === "string" || Array.isArray(body.SecurityOptions))) {
@@ -98,6 +106,24 @@ export function parseDaemonFacts(output) {
98
106
  serviceIsRemote: null,
99
107
  remoteSocketPath: null,
100
108
  serverVersion: displayVersion(body.ServerVersion),
109
+ // Issue #596: the daemon's CPU count (`NCPU`, the VM's on Docker Desktop, measured 14 there and 4 on Ubuntu),
110
+ // which sets every job's `--cpus` ceiling. Docker refuses a `--cpus` above it, so the worker's own count is
111
+ // never used in its place.
112
+ hostCpus: cpuCount(body.NCPU),
113
+ // Issue #596, phase 2: the daemon's memory (`MemTotal`, bytes; the VM's on Docker Desktop), in MiB, for the host
114
+ // budget's `auto` (`host-budget.mjs`). The worker's own `os.totalmem()` is never used in its place, for NCPU's reason.
115
+ memTotalMiB: memoryMiB(body.MemTotal),
116
+ // Issue #596: whether the daemon can bound swap (`SwapLimit`). Where it is false, `--memory-swap` cannot be
117
+ // enforced and a job may swap past its memory; doctor warns. null on Podman (its compat value is not read,
118
+ // for `bounds`' reason) and when the key is absent.
119
+ swapLimit: podman || typeof body.SwapLimit !== "boolean" ? null : body.SwapLimit,
120
+ // Issue #596: whether the daemon can set a CPU weight (`CPUShares`). Where it is false, Docker drops
121
+ // `--cpu-shares` with a client warning and the job has no weight. null on Podman and when absent, as above.
122
+ cpuShares: podman || typeof body.CPUShares !== "boolean" ? null : body.CPUShares,
123
+ // Issue #596, phase 2: `CgroupDriver` (`cgroupfs` on Docker Desktop, `systemd` on a systemd host, measured) and
124
+ // `CgroupVersion` (`2`), which decide how the jobs' parent cgroup gets its quota (`cpu-reserve.mjs`).
125
+ cgroupDriver: cgroupWord(body.CgroupDriver),
126
+ cgroupVersion: cgroupVersionOf(body.CgroupVersion),
101
127
  } };
102
128
  }
103
129
  }
@@ -113,6 +139,32 @@ export function displayVersion(value) {
113
139
  return typeof value === "string" && /^[0-9A-Za-z.+~_-]{1,40}$/.test(value) ? value : null;
114
140
  }
115
141
 
142
+ /** A cgroup driver or manager's name: a short lower-case word, else null (no fact). */
143
+ function cgroupWord(value) {
144
+ return typeof value === "string" && /^[a-z][a-z0-9_-]{0,31}$/.test(value) ? value : null;
145
+ }
146
+
147
+ /** A cgroup version as `v1` or `v2`, from Docker's `2` or Podman's `v2`, else null. */
148
+ function cgroupVersionOf(value) {
149
+ const text = typeof value === "string" ? value : "";
150
+ return /^v?[12]$/.test(text) ? `v${text.replace(/^v/, "")}` : null;
151
+ }
152
+
153
+ /** A runtime's CPU count: a whole number from 1 to 4096, else null (no fact rather than a guess). */
154
+ function cpuCount(value) {
155
+ return Number.isSafeInteger(value) && value >= 1 && value <= 4096 ? value : null;
156
+ }
157
+
158
+ /**
159
+ * A runtime's memory in whole MiB (rounded down), from its byte count: a safe integer of at least 64 MiB and at most
160
+ * 64 TiB, else null. A host below 64 MiB runs no job at all, so a smaller value is a parse fault rather than a fact.
161
+ */
162
+ export function memoryMiB(bytes) {
163
+ if (!Number.isSafeInteger(bytes) || bytes < 64 * 1048576) return null;
164
+ const mib = Math.floor(bytes / 1048576);
165
+ return mib <= 64 * 1024 * 1024 ? mib : null;
166
+ }
167
+
116
168
  function isUnixSocketPath(path) {
117
169
  return typeof path === "string" && (path.startsWith("/") || path.startsWith("unix:///"));
118
170
  }
@@ -166,5 +218,11 @@ export function parsePodmanInfo(stdout) {
166
218
  // measured on 5.8.1 and 4.9.3), under which Podman 5 records this account's rootless network helper. Parsed as
167
219
  // graphRoot is: an absolute path with no control character, else no fact (and the live network check refuses).
168
220
  runRoot: typeof body.store?.runRoot === "string" && body.store.runRoot.length <= 4096 && /^\/[^\u0000-\u001f\u007f]*$/.test(body.store.runRoot) ? body.store.runRoot : null,
221
+ // Issue #596: the host's CPU count as Podman reports it (`host.cpus`, measured 4 on both lab VMs), which sets every
222
+ // podman job's `--cpus` ceiling. A rootless account's own `cpu.max` may be lower; the host budget reads it (`host-budget.mjs`).
223
+ hostCpus: cpuCount(host.cpus),
224
+ // Issue #596, phase 2: the host's memory as Podman reports it (`host.memTotal`, bytes), in MiB, for the host budget's
225
+ // `auto`. A rootless account's own `memory.max` may be lower; `host-budget.mjs` reads that beside it.
226
+ memTotalMiB: memoryMiB(host.memTotal),
169
227
  };
170
228
  }
@@ -21,7 +21,8 @@
21
21
  // row, and this move is recorded in its own row rather than by leaving that pointer to rot.
22
22
  export { containerSpec, CONTAINER_GLOBAL_PI_DIR, CONTAINER_SESSION_DIR, CONTAINER_SESSION_FILE } from "./container-spec.mjs";
23
23
  import { isAbsolute, relative } from "node:path";
24
- import { assertCidFile, assertJobUser, containerSpec } from "./container-spec.mjs";
24
+ import { SIZE_LABEL_CPU, SIZE_LABEL_MEM, assertCidFile, assertJobUser, containerSpec } from "./container-spec.mjs";
25
+ import { CGROUP_PARENT, CPU_SHARES_MAX, CPU_SHARES_MIN } from "./job-size.mjs";
25
26
 
26
27
  /** The fixed isolation flags. Not configurable -- these ARE the boundary. */
27
28
  export const ISOLATION_FLAGS = [
@@ -39,8 +40,10 @@ export const ISOLATION_FLAGS = [
39
40
  "--cap-drop=ALL", // pi would otherwise inherit the launching user's capabilities
40
41
  "--security-opt",
41
42
  "no-new-privileges",
42
- "--pids-limit=512", // bound a fork bomb (UNVERIFIED figure; measured headroom ~4.5x, see spec)
43
- "--shm-size=1g", // Chromium OOMs on the default 64MB /dev/shm; NOT --ipc=host (shares host ns)
43
+ "--pids-limit=512", // bound a fork bomb (UNVERIFIED figure; measured headroom ~4.5x, see spec). FIXED, never per size
44
+ // `--shm-size` LEFT this array in issue #596: it is now min(1g, memory/2) of the job's size, a value, and this array
45
+ // is the literal, value-free set (see `--network` in `argsFromSpec`). It is emitted beside `--memory` instead, still
46
+ // on every argv, and still NOT `--ipc=host` (Chromium OOMs on the default 64MB /dev/shm; host IPC shares the host ns).
44
47
  ];
45
48
 
46
49
  /**
@@ -85,8 +88,9 @@ export const ISOLATION_FLAGS = [
85
88
  * `DOCKER_EXTRA_ALLOWED`, or it is refused. The deny-list stays for its named reasons and its tests.
86
89
  */
87
90
  export const DOCKER_EXTRA_FORBIDDEN = [
88
- // Each of the seven logical flags in ISOLATION_FLAGS, and the argv member beside them that the worker's
89
- // own machinery reads back. A flag missing from here is a flag the array cannot defend.
91
+ // Each of the six logical flags in ISOLATION_FLAGS, `--shm-size` (in that array until issue #596 made it a size),
92
+ // and the argv member beside them that the worker's own machinery reads back. A flag missing from here is a flag the
93
+ // array cannot defend.
90
94
  "--rm", // `--rm=false` leaves the container behind; verified against docker 27.4.0. `ephemeral` rests on it.
91
95
  "--init",
92
96
  "--shm-size",
@@ -105,6 +109,31 @@ export const DOCKER_EXTRA_FORBIDDEN = [
105
109
  "--memory",
106
110
  "-m",
107
111
  "--cpus",
112
+ // The job's size (issue #596). Every flag that sets a CPU or memory bound, a weight or an OOM preference: a repeat
113
+ // would win last and give one job another job's promise. `-c` is `--cpu-shares`' short form.
114
+ "--cpu-shares",
115
+ "-c",
116
+ "--cpu-quota",
117
+ "--cpu-period",
118
+ "--cpu-rt-period",
119
+ "--cpu-rt-runtime",
120
+ "--cpuset-cpus",
121
+ "--cpuset-mems",
122
+ "--memory-reservation",
123
+ "--memory-swappiness",
124
+ "--kernel-memory",
125
+ "--blkio-weight",
126
+ "--blkio-weight-device",
127
+ "--device-read-bps",
128
+ "--device-write-bps",
129
+ "--device-read-iops",
130
+ "--device-write-iops",
131
+ "--oom-score-adj",
132
+ // Issue #596, phase 2: the size labels doctor holds the host budget against; a second `--label` would win last and
133
+ // make a container read as another size.
134
+ "--label",
135
+ "-l",
136
+ "--label-file",
108
137
  "--network",
109
138
  "--net",
110
139
  "--pull",
@@ -118,6 +147,8 @@ export const DOCKER_EXTRA_FORBIDDEN = [
118
147
  "--ulimit",
119
148
  "--sysctl",
120
149
  "--group-add",
150
+ // Issue #596, phase 2: the parent every job shares, whose quota keeps the host's CPU reserve across all jobs. A second
151
+ // `--cgroup-parent` would win last and run the job outside it, beyond the quota and back beside the proxy's weight.
121
152
  "--cgroup-parent",
122
153
  "--device-cgroup-rule",
123
154
  "--gpus",
@@ -234,6 +265,11 @@ function argsFromSpec(spec, { userns }) {
234
265
  // Re-checked here for a hand-built spec, the same reason `isolated` is.
235
266
  assertJobUser(spec.user);
236
267
  assertCidFile(spec.cidFile);
268
+ assertSizing(spec);
269
+ // A hand-built spec may only name the one parent, or none (`undefined` is a spec from before the field: none).
270
+ if (spec.cgroupParent !== undefined && spec.cgroupParent !== null && spec.cgroupParent !== CGROUP_PARENT) {
271
+ throw new Error(`docker run: refusing a spec whose cgroup parent is not ${CGROUP_PARENT}: ${JSON.stringify(spec.cgroupParent)}`);
272
+ }
237
273
 
238
274
  // `--network` sits HERE, beside --memory and --cpus, and deliberately NOT inside ISOLATION_FLAGS.
239
275
  // That array is the LITERAL, value-free, unconditional set, and two separate places assert every member
@@ -245,7 +281,23 @@ function argsFromSpec(spec, { userns }) {
245
281
  //
246
282
  // null => the flag is ABSENT, so a job argv without an egress policy is byte-identical to one built
247
283
  // before this feature existed. Same shape as the sessionDir/outboxDir/globalPiDir mounts below.
248
- const args = ["run", `--name=${spec.name}`, ...ISOLATION_FLAGS, `--memory=${spec.memory}`, `--cpus=${spec.cpus}`];
284
+ // Issue #596: the job's size. `--memory-swap` equal to `--memory`, so no swap beyond it; `--cpus` the host ceiling,
285
+ // absent when the runtime's CPU count is unknown (`hostCpuCeiling` says why that fails open); `--cpu-shares` the job's
286
+ // weight; `--shm-size` min(1g, memory/2).
287
+ const args = ["run", `--name=${spec.name}`, ...ISOLATION_FLAGS, `--memory=${spec.memory}`, `--memory-swap=${spec.memorySwap}`];
288
+ if (spec.cpus !== null) args.push(`--cpus=${spec.cpus}`);
289
+ args.push(`--cpu-shares=${spec.cpuShares}`, `--shm-size=${spec.shmSize}`);
290
+ // THE AGGREGATE CPU RESERVE (issue #596, phase 2). `--cpus` above bounds ONE job; what keeps the host's reserve free
291
+ // across ALL of them is the one parent cgroup every job container is started under, whose quota is the host's CPU
292
+ // budget (`cpu-reserve.mjs` says who sets it per venue). Emitted on every container this builder makes, docker and
293
+ // podman alike, quota or not: inside the parent a job's weight competes only with sibling jobs, so even an unset
294
+ // quota keeps a large job from starving the egress proxy and Valkey, which are NOT built here and never carry it.
295
+ // A missing parent is created silently by the runtime, so the flag never fails a run (measured on all five venues).
296
+ // Absent only where the runtime cannot take one (`cgroupParentFor`), and then the spec's shares are capped instead.
297
+ // `--cgroup-parent` is refused in `dockerExtra`, in both spellings, so nothing else can move a job out of it.
298
+ if (spec.cgroupParent !== null && spec.cgroupParent !== undefined) args.push(`--cgroup-parent=${spec.cgroupParent}`);
299
+ // Issue #596, phase 2: the size labels (`SIZE_LABEL_MEM`, `SIZE_LABEL_CPU`), integers from the validated size.
300
+ for (const [key, value] of Object.entries(spec.labels ?? {})) args.push(`--label=${key}=${value}`);
249
301
  if (spec.network) args.push(`--network=${spec.network}`);
250
302
  else if (userns) args.push("--network=private");
251
303
  // null => ABSENT, so a job the image's own USER runs has an argv byte-identical to one built before issue #341.
@@ -283,6 +335,31 @@ function argsFromSpec(spec, { userns }) {
283
335
  return args;
284
336
  }
285
337
 
338
+ /**
339
+ * The size fields of a spec, re-checked for a hand-built one (issue #596): `memory` and `shmSize` in the spelling
340
+ * `formatMemory` writes, `memorySwap` EQUAL to `memory` (a spec that allows swap is refused, never repaired), `cpus` a
341
+ * positive number of CPUs with at most two decimals (a CPU budget may be fractional, issue #596 phase 2) or null,
342
+ * `cpuShares` an integer in the runtime's range, and `labels` absent or exactly the two size labels with integer values.
343
+ * A spec from before these fields is refused rather than given `--memory-swap=undefined`.
344
+ */
345
+ function labelsOk(labels) {
346
+ if (labels === undefined) return true;
347
+ if (labels === null || typeof labels !== "object") return false;
348
+ const keys = Object.keys(labels);
349
+ return keys.length === 2 && keys.includes(SIZE_LABEL_MEM) && keys.includes(SIZE_LABEL_CPU) && keys.every((k) => typeof labels[k] === "string" && /^[1-9]\d{0,8}$/.test(labels[k]));
350
+ }
351
+
352
+ function assertSizing(spec) {
353
+ const memoryish = (v) => typeof v === "string" && /^[1-9]\d{0,6}[mg]$/.test(v);
354
+ const ok = memoryish(spec.memory)
355
+ && spec.memorySwap === spec.memory
356
+ && memoryish(spec.shmSize)
357
+ && (spec.cpus === null || (typeof spec.cpus === "string" && /^(?:0|[1-9]\d{0,5})(?:\.\d{1,2})?$/.test(spec.cpus) && Number(spec.cpus) > 0))
358
+ && labelsOk(spec.labels)
359
+ && Number.isSafeInteger(spec.cpuShares) && spec.cpuShares >= CPU_SHARES_MIN && spec.cpuShares <= CPU_SHARES_MAX;
360
+ if (!ok) throw new Error(`docker run: refusing a spec whose size fields are not containerSpec's (memory, memorySwap equal to it, cpus, cpuShares, shmSize): ${JSON.stringify({ memory: spec.memory ?? null, memorySwap: spec.memorySwap ?? null, cpus: spec.cpus ?? null, cpuShares: spec.cpuShares ?? null, shmSize: spec.shmSize ?? null })}`);
361
+ }
362
+
286
363
  /**
287
364
  * Build the full `docker run` argv (excluding the leading "docker").
288
365
  *