@edgehero/pi-dispatch 3.0.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +39 -1
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +631 -21
- package/src/env-allowlist.mjs +23 -5
- package/src/host-budget.mjs +736 -0
- package/src/host-pi.mjs +1 -1
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/model-catalog.mjs +1 -1
- package/src/model-endpoints.mjs +22 -0
- package/src/models-json.mjs +10 -4
- package/src/output-cap.mjs +3 -3
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/reserved-env.mjs +3 -2
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +8 -5
package/src/job-size.mjs
ADDED
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How big one job's container is (issue #596, phase 1): its memory and its CPU weight, per project.
|
|
3
|
+
*
|
|
4
|
+
* A LEAF, importing nothing, for the reason `container-spec.mjs` is one: that module builds every container's argv
|
|
5
|
+
* from a size, and `packages.mjs` relies on it importing nothing heavy. So the project row is matched here by its
|
|
6
|
+
* scope string (`project:<id>`) rather than through `scoped-limits.mjs`; a test holds the two spellings equal.
|
|
7
|
+
*
|
|
8
|
+
* A size is held as TWO INTEGERS, never as the strings an operator writes: `memMiB` (mebibytes) and `cpuCenti`
|
|
9
|
+
* (hundredths of a CPU). Integers so a size compares, sums and records exactly (the host budget, `host-budget.mjs`, adds them
|
|
10
|
+
* up), and so the one spelling the runtime is handed is derived from them (`formatMemory`), never passed through.
|
|
11
|
+
*
|
|
12
|
+
* WHAT A SIZE BECOMES (`containerSizing`, read by `containerSpec`):
|
|
13
|
+
* - `--memory=<size>` and `--memory-swap=<size>`, equal, so a job gets NO swap beyond its memory. Docker and Podman
|
|
14
|
+
* give a container swap equal to its memory by default; with the two equal `memory.swap.max` reads 0 on every
|
|
15
|
+
* venue measured, rootless Podman included (lab, issue #596). A job that needs more memory needs a bigger size,
|
|
16
|
+
* not a slower one.
|
|
17
|
+
* - `--cpu-shares=round(cpus x 1024)`: a WEIGHT, not a cap. Under contention a larger size gets more CPU than a
|
|
18
|
+
* smaller one, in order; an idle host lets any job use the idle cores. This is the fair share decision 2 of the
|
|
19
|
+
* issue asked for. EXACTLY proportional only on a runtime with the linear mapping (below): measured, sizes 3:1
|
|
20
|
+
* split CPU about 3:1 on runc 1.1.13 and crun 1.14, and about 2.4:1 on runc 1.5.1 and crun 1.27, whose curved
|
|
21
|
+
* mapping compresses the weights. The order holds on both, which is what a share promises here.
|
|
22
|
+
* - `--shm-size=min(1g, memory/2)`: Chromium needs a large `/dev/shm` (the reason `1g` was there), and `/dev/shm`
|
|
23
|
+
* is charged to the container's memory, so it may not be most of a small size.
|
|
24
|
+
* - `--cpus=<host ceiling>`: the host's CPU budget, else its CPU count minus a reserve (`cpuCeilingCenti`), the same for every job. It
|
|
25
|
+
* bounds ONE job: no single job can use more than the ceiling. It does not keep a core free across jobs, because
|
|
26
|
+
* each container's quota is its own and they do not sum (measured: two busy jobs on a 4-core cpuset with
|
|
27
|
+
* `--cpus=3` used 4.06 cores between them). The reserve across every job is the quota of the one parent cgroup
|
|
28
|
+
* they all run under (`CGROUP_PARENT`, `cpu-reserve.mjs`). It is not the job's size either, which would be the hard cap decision 2
|
|
29
|
+
* rejected; the host budget bounds the jobs' SIZES together, not their use.
|
|
30
|
+
* Absent when the runtime did not say how many CPUs it has (see `hostCpuCeiling`).
|
|
31
|
+
*
|
|
32
|
+
* THE cpu.weight A SHARE BECOMES DEPENDS ON THE OCI RUNTIME'S VERSION (measured, issue #596): runc 1.1.13 and crun
|
|
33
|
+
* 1.14 map shares linearly (1024 to 39, 2048 to 79), runc 1.5.1 and crun 1.27 map 1024 to 100 and 2048 to 174, and a
|
|
34
|
+
* container started without `--cpu-shares` gets 100 on all four. Both mappings keep the ORDER of shares, and every job
|
|
35
|
+
* container carries a share, so the order between jobs, which is what a fair share is about, holds everywhere. What an
|
|
36
|
+
* old runtime changes is a job's weight against a container started WITHOUT a share (the egress proxy, a Valkey): a
|
|
37
|
+
* default job's 79 is below their 100. That is the right direction (the proxy serves every job), and no such container
|
|
38
|
+
* competes with jobs for long, so the flag is passed as is rather than scaled per runtime version.
|
|
39
|
+
*
|
|
40
|
+
* THE OTHER DIRECTION is closed by the parent cgroup, not by a weight scale (issue #596, phase 2). At `cpus` 256 the
|
|
41
|
+
* share is 262144, which a current runtime maps to `cpu.weight` 10000 against the 100 of the egress proxy, Valkey and the
|
|
42
|
+
* host's own services; with no parent such jobs left a proxy-like container 0.01 of a core (measured). Every job runs
|
|
43
|
+
* under `CGROUP_PARENT`, where a job's weight competes only with its sibling jobs and the parent competes as ONE group of
|
|
44
|
+
* weight 100, so the shares are kept as they are and still order the jobs (on Fedora's kernel 6.19 the ratio is
|
|
45
|
+
* compressed under the parent's throttling, the order holds). Only a job run WITHOUT the parent has its shares capped,
|
|
46
|
+
* at `PARENTLESS_SHARES_MAX`.
|
|
47
|
+
*/
|
|
48
|
+
|
|
49
|
+
/** The smallest memory a job may be given: 512 MiB. The runner, pi and a shell need about 200 MB before any tool runs. */
|
|
50
|
+
export const JOB_MEMORY_FLOOR_MIB = 512;
|
|
51
|
+
/** The largest: 1024 GiB. A bound against a typo (`40000g`), and it keeps every byte count a safe integer. */
|
|
52
|
+
export const JOB_MEMORY_CEILING_MIB = 1024 * 1024;
|
|
53
|
+
/** The smallest CPU size: 0.25, as hundredths. Its share, 256, is well inside the runtime's range. */
|
|
54
|
+
export const JOB_CPUS_FLOOR_CENTI = 25;
|
|
55
|
+
/** The largest CPU size: 256, as hundredths, because round(256 x 1024) is 262144, the top of `--cpu-shares`. */
|
|
56
|
+
export const JOB_CPUS_CEILING_CENTI = 256 * 100;
|
|
57
|
+
/** The valid range of `--cpu-shares` on cgroup v2 (runc and crun clamp to it; 2 is the kernel's cgroup v1 minimum). */
|
|
58
|
+
export const CPU_SHARES_MIN = 2;
|
|
59
|
+
export const CPU_SHARES_MAX = 262144;
|
|
60
|
+
/**
|
|
61
|
+
* The one parent cgroup every job container is started under (issue #596, phase 2): a slice name with NO dash, because
|
|
62
|
+
* a dash nests (`pd-jobs.slice` lands in `/pd.slice/pd-jobs.slice/`, measured on docker and podman). Its quota is the
|
|
63
|
+
* host's CPU budget; `cpu-reserve.mjs` says who sets it on which venue and why it is always on the argv.
|
|
64
|
+
*/
|
|
65
|
+
export const CGROUP_PARENT = "pidispatch.slice";
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* The `--cpu-shares` cap for a job that runs WITHOUT the parent (`cpu-reserve.mjs` `cgroupParentFor`): 1024 is weight
|
|
69
|
+
* 100 on current runtimes (39 on old ones), the egress proxy's and Valkey's default, so no job outweighs them. A FAIR
|
|
70
|
+
* share, not a reserve: six busy jobs at 1024 left a proxy-like container 0.57 of a core, one seventh (measured).
|
|
71
|
+
*/
|
|
72
|
+
export const PARENTLESS_SHARES_MAX = 1024;
|
|
73
|
+
/** `/dev/shm`'s ceiling, the `--shm-size=1g` every job had before sizes. */
|
|
74
|
+
export const SHM_CEILING_MIB = 1024;
|
|
75
|
+
|
|
76
|
+
/** The size every job had before issue #596, and the one a deployment that sets nothing still gets. */
|
|
77
|
+
export const DEFAULT_JOB_MEMORY = "4g";
|
|
78
|
+
export const DEFAULT_JOB_CPUS = "2";
|
|
79
|
+
|
|
80
|
+
/** Where a resolved size came from: the job's project row, the deployment's `PI_JOB_*` settings, or the built-in default. */
|
|
81
|
+
export const JOB_SIZE_SOURCES = Object.freeze(["project", "env", "default"]);
|
|
82
|
+
|
|
83
|
+
/** The scope prefix of a project row, as `scoped-limits.mjs` spells it (`PROJECT_SCOPE_PREFIX`; a test holds them equal). */
|
|
84
|
+
const PROJECT_ROW_PREFIX = "project:";
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* A memory size, written as an operator writes it (`512m`, `1536m`, `4g`), in MiB. Throws an Error naming the rule for
|
|
88
|
+
* anything else. Lower-case `m` or `g` only, no zero and no leading zero, no fraction, no bytes or kilobytes: one
|
|
89
|
+
* spelling per size, so a row reads the same in the file, the panel and the record. Strings only: a bare number would
|
|
90
|
+
* have to mean bytes to agree with the runtime, and nobody writes a job's memory in bytes.
|
|
91
|
+
*/
|
|
92
|
+
export function parseMemory(value) {
|
|
93
|
+
if (typeof value !== "string" || !/^[1-9]\d*[mg]$/.test(value)) {
|
|
94
|
+
throw new Error(`a memory size is a whole number of megabytes or gigabytes such as "512m", "1536m" or "4g" (got ${JSON.stringify(value)})`);
|
|
95
|
+
}
|
|
96
|
+
const digits = value.slice(0, -1);
|
|
97
|
+
// Compared as digits first, so a value of any length is refused without ever becoming an unsafe number.
|
|
98
|
+
const ceilingDigits = String(JOB_MEMORY_CEILING_MIB).length;
|
|
99
|
+
const mib = digits.length > ceilingDigits + 1 ? Infinity : Number(digits) * (value.endsWith("g") ? 1024 : 1);
|
|
100
|
+
if (mib < JOB_MEMORY_FLOOR_MIB) throw new Error(`a memory size must be at least ${JOB_MEMORY_FLOOR_MIB}m (got ${JSON.stringify(value)})`);
|
|
101
|
+
if (mib > JOB_MEMORY_CEILING_MIB) throw new Error(`a memory size must be at most ${JOB_MEMORY_CEILING_MIB / 1024}g (got ${JSON.stringify(value)})`);
|
|
102
|
+
return mib;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* A CPU size (`0.5`, `2`, `"1.25"`), in hundredths. A number or a decimal string: above 0, at most two decimals, at
|
|
107
|
+
* least 0.25 and at most 256. Throws an Error naming the rule for anything else. A number is judged by its own decimal
|
|
108
|
+
* spelling (`String(n)`), so `1e-7`, `NaN`, `Infinity` and `0.125` are refused rather than rounded into a size.
|
|
109
|
+
*/
|
|
110
|
+
export function parseCpus(value) {
|
|
111
|
+
const text = typeof value === "number" ? String(value) : value;
|
|
112
|
+
if (typeof text !== "string" || !/^(?:0|[1-9]\d{0,5})(?:\.\d{1,2})?$/.test(text)) {
|
|
113
|
+
throw new Error(`a CPU size is a number above 0 with at most two decimals, such as 0.5, 2 or 1.25 (got ${JSON.stringify(value)})`);
|
|
114
|
+
}
|
|
115
|
+
const [whole, fraction = ""] = text.split(".");
|
|
116
|
+
const centi = Number(whole) * 100 + Number(fraction.padEnd(2, "0"));
|
|
117
|
+
if (centi < JOB_CPUS_FLOOR_CENTI) throw new Error(`a CPU size must be at least ${JOB_CPUS_FLOOR_CENTI / 100} (got ${JSON.stringify(value)})`);
|
|
118
|
+
if (centi > JOB_CPUS_CEILING_CENTI) throw new Error(`a CPU size must be at most ${JOB_CPUS_CEILING_CENTI / 100} (got ${JSON.stringify(value)})`);
|
|
119
|
+
return centi;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* The one spelling of a memory size: `<n>g` when it is whole gigabytes, else `<n>m`. Every `--memory`, `--memory-swap`
|
|
124
|
+
* and stored row is written by this function, and `container-spec.mjs`'s `memoryBytes` reads every value it can write
|
|
125
|
+
* (a test walks the whole accepted range), so the OOM rule's 90%-of-the-limit comparison works at every size.
|
|
126
|
+
*/
|
|
127
|
+
export function formatMemory(memMiB) {
|
|
128
|
+
return memMiB % 1024 === 0 ? `${memMiB / 1024}g` : `${memMiB}m`;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** The one spelling of a CPU size: `2`, `0.5`, `1.25`. Built from the integer, so no float ever prints as `0.30000000000000004`. */
|
|
132
|
+
export function formatCpus(cpuCenti) {
|
|
133
|
+
const whole = Math.floor(cpuCenti / 100);
|
|
134
|
+
const fraction = String(cpuCenti % 100).padStart(2, "0").replace(/0+$/, "");
|
|
135
|
+
return fraction === "" ? String(whole) : `${whole}.${fraction}`;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** `--cpu-shares` for a size: round(cpus x 1024), held inside the runtime's range. */
|
|
139
|
+
export function cpuSharesOf(cpuCenti) {
|
|
140
|
+
return Math.min(CPU_SHARES_MAX, Math.max(CPU_SHARES_MIN, Math.round((cpuCenti * 1024) / 100)));
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** `/dev/shm` for a size, in MiB: half the memory, at most 1g. */
|
|
144
|
+
export function shmMiBOf(memMiB) {
|
|
145
|
+
return Math.min(SHM_CEILING_MIB, Math.floor(memMiB / 2));
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* The CPUs any ONE job may use at most, `--cpus`: the host's count minus the reserve (one CPU when the host has four
|
|
150
|
+
* or more, else none), or null when the count is unknown. Per container: several busy jobs together can still use
|
|
151
|
+
* every core, the reserved one included, unless the parent cgroup they share holds its quota (`cpu-reserve.mjs`).
|
|
152
|
+
*
|
|
153
|
+
* `hostCpus` is the RUNTIME's own count (`docker info`'s `NCPU`, `podman info`'s `host.cpus`), never the worker's
|
|
154
|
+
* `os.availableParallelism()`: on Docker Desktop the daemon runs in a VM with its own count, and Docker refuses a
|
|
155
|
+
* `--cpus` above the CPUs it has. Under a CPU budget the ceiling is the budget instead (`cpuCeilingCenti`), which
|
|
156
|
+
* carries an operator's reserve and a rootless service's own `cpu.max`; this is the ceiling without one.
|
|
157
|
+
*
|
|
158
|
+
* NULL FAILS OPEN, and says so: the argv then carries no `--cpus`, so a job may use every core, reserve included, as a
|
|
159
|
+
* container with no bound does. The shares still apply. The worker logs `cpu_ceiling_unknown` when it happens; it
|
|
160
|
+
* needs a daemon whose `info` did not answer while the job still ran, which only a desktop platform's job user allows.
|
|
161
|
+
*/
|
|
162
|
+
export function hostCpuCeiling(hostCpus) {
|
|
163
|
+
if (!Number.isSafeInteger(hostCpus) || hostCpus < 1) return null;
|
|
164
|
+
return hostCpus - (hostCpus >= 4 ? 1 : 0);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* The `--cpus` every job gets, in hundredths (issue #596, phase 2): the host's CPU BUDGET when one is in force (an
|
|
169
|
+
* integer `cpuBudgetCenti`, `host-budget.mjs`), capped at the runtime's own count where that is known (Docker refuses a
|
|
170
|
+
* `--cpus` above it), else the phase 1 ceiling (`hostCpuCeiling`: the count minus the default reserve), else null. With
|
|
171
|
+
* the budget off or unknown the phase 1 ceiling stands, so a deployment that sets nothing gets the same flag as before
|
|
172
|
+
* whenever `auto` reads the same count. It bounds ONE job; the budget's ledger is what bounds them together.
|
|
173
|
+
*/
|
|
174
|
+
export function cpuCeilingCenti(hostCpus, cpuBudgetCenti = null) {
|
|
175
|
+
const runtime = Number.isSafeInteger(hostCpus) && hostCpus >= 1 ? hostCpus * 100 : null;
|
|
176
|
+
if (Number.isSafeInteger(cpuBudgetCenti) && cpuBudgetCenti > 0) return runtime === null ? cpuBudgetCenti : Math.min(cpuBudgetCenti, runtime);
|
|
177
|
+
const ceiling = hostCpuCeiling(hostCpus);
|
|
178
|
+
return ceiling === null ? null : ceiling * 100;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* The highest CPU count a runtime's own refusal of a `--cpus` names, or null when `text` holds no such refusal. Docker
|
|
183
|
+
* refuses a `--cpus` above the CPUs its daemon has, at create, exit 125: `Range of CPUs is from 0.01 to 4.00, as there
|
|
184
|
+
* are only 4 CPUs available` (measured, Docker 27.4.0). That is what a ceiling computed from a stale count meets after
|
|
185
|
+
* Docker Desktop's VM was given fewer CPUs, so the worker reads its facts again (`cpu_ceiling_stale`). Podman 4.9.3 and
|
|
186
|
+
* 5.8.1 accept a `--cpus` above the host's count (measured, rootless and rootful), so no Podman text matches: a stale
|
|
187
|
+
* count there is a ceiling that bounds less, never a job that cannot start. Only the leading digits are read.
|
|
188
|
+
*/
|
|
189
|
+
export function cpuRangeRefusal(text) {
|
|
190
|
+
const match = /range of cpus is from [0-9.]+ to ([0-9]{1,5})(?:\.[0-9]+)?/i.exec(String(text ?? ""));
|
|
191
|
+
return match ? Number(match[1]) : null;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* The size flags a runtime said it will not enforce, from its parsed `docker info` facts (`parseDaemonFacts`): an
|
|
196
|
+
* empty array, or `--memory-swap` where Docker reports `SwapLimit` false and `--cpu-shares` where it reports
|
|
197
|
+
* `CPUShares` false. Docker then drops the flag with a client warning only and runs the container anyway (no swap
|
|
198
|
+
* accounting, no cpu controller), so a job would run past the bound its size promised and nothing would say so. The
|
|
199
|
+
* worker logs `size_bound_unenforced` per job and doctor warns. A fact that is null (Podman, or a key absent) is no
|
|
200
|
+
* evidence and adds nothing.
|
|
201
|
+
*/
|
|
202
|
+
export function unenforcedSizeFlags(facts) {
|
|
203
|
+
const flags = [];
|
|
204
|
+
if (facts?.swapLimit === false) flags.push("--memory-swap");
|
|
205
|
+
if (facts?.cpuShares === false) flags.push("--cpu-shares");
|
|
206
|
+
return flags;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/**
|
|
210
|
+
* What a size becomes on a container's argv, as `containerSpec` fields: `{ memory, memorySwap, cpus, cpuShares,
|
|
211
|
+
* shmSize }`. `memory` and `memorySwap` are the SAME string by construction; `cpus` is the host ceiling
|
|
212
|
+
* (`cpuCeilingCenti`, which may be fractional, `3.5`, under a CPU budget) or null.
|
|
213
|
+
*/
|
|
214
|
+
export function containerSizing(size, hostCpus = null, cpuBudgetCenti = null) {
|
|
215
|
+
assertJobSize(size);
|
|
216
|
+
const memory = formatMemory(size.memMiB);
|
|
217
|
+
const ceiling = cpuCeilingCenti(hostCpus, cpuBudgetCenti);
|
|
218
|
+
return { memory, memorySwap: memory, cpus: ceiling === null ? null : formatCpus(ceiling), cpuShares: cpuSharesOf(size.cpuCenti), shmSize: formatMemory(shmMiBOf(size.memMiB)) };
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/** Throws unless `size` is `{ memMiB, cpuCenti }` within the floors and ceilings above. */
|
|
222
|
+
export function assertJobSize(size) {
|
|
223
|
+
const mem = size?.memMiB;
|
|
224
|
+
const cpu = size?.cpuCenti;
|
|
225
|
+
if (!Number.isSafeInteger(mem) || mem < JOB_MEMORY_FLOOR_MIB || mem > JOB_MEMORY_CEILING_MIB || !Number.isSafeInteger(cpu) || cpu < JOB_CPUS_FLOOR_CENTI || cpu > JOB_CPUS_CEILING_CENTI) {
|
|
226
|
+
throw new Error(`container spec: refusing a job size that is not { memMiB, cpuCenti } within ${JOB_MEMORY_FLOOR_MIB}..${JOB_MEMORY_CEILING_MIB} MiB and ${JOB_CPUS_FLOOR_CENTI}..${JOB_CPUS_CEILING_CENTI} hundredths: ${JSON.stringify(size ?? null)}`);
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** The built-in size, `{ memMiB: 4096, cpuCenti: 200, source: "default" }`, derived from the two defaults above. */
|
|
231
|
+
export const DEFAULT_JOB_SIZE = Object.freeze({ memMiB: parseMemory(DEFAULT_JOB_MEMORY), cpuCenti: parseCpus(DEFAULT_JOB_CPUS), source: "default" });
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* The deployment's default size from `PI_JOB_MEMORY` and `PI_JOB_CPUS` (unset or empty is the built-in value), as `{
|
|
235
|
+
* memMiB, cpuCenti, memSet, cpuSet }`. Throws an Error naming the key for a value either parser refuses; the worker's
|
|
236
|
+
* config runs this at boot, so a bad value stops the worker there (exit 2) rather than at a job.
|
|
237
|
+
*/
|
|
238
|
+
export function jobSizeDefaults(env = {}) {
|
|
239
|
+
const read = (key, parse, fallback) => {
|
|
240
|
+
const raw = env?.[key];
|
|
241
|
+
if (raw === undefined || raw === null || raw === "") return { value: parse(fallback), set: false };
|
|
242
|
+
try {
|
|
243
|
+
return { value: parse(raw), set: true };
|
|
244
|
+
} catch (error) {
|
|
245
|
+
throw new Error(`${key}: ${error.message}`);
|
|
246
|
+
}
|
|
247
|
+
};
|
|
248
|
+
const mem = read("PI_JOB_MEMORY", parseMemory, DEFAULT_JOB_MEMORY);
|
|
249
|
+
const cpu = read("PI_JOB_CPUS", parseCpus, DEFAULT_JOB_CPUS);
|
|
250
|
+
return { memMiB: mem.value, cpuCenti: cpu.value, memSet: mem.set, cpuSet: cpu.set };
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* The size a job runs at, `{ memMiB, cpuCenti, source }`, resolved ONCE at pickup from the same limits snapshot every
|
|
255
|
+
* other gate reads (issue #596). `project` is the pickup project's id (or null), `limits` the parsed scoped-limits
|
|
256
|
+
* rows, `env` the deployment's settings (`PI_JOB_MEMORY`, `PI_JOB_CPUS`).
|
|
257
|
+
*
|
|
258
|
+
* Field by field: the project row's `memory` and `cpus` win, then the deployment's settings, then the built-in 4g and
|
|
259
|
+
* 2. `source` names the most specific place that supplied ANY field: `project` when the row set memory or CPUs (a row
|
|
260
|
+
* that sets only one takes the other from the deployment), `env` when a setting did and the row none, else `default`.
|
|
261
|
+
*
|
|
262
|
+
* Pure. Never throws on a parsed limits list (its sizes were validated at load); throws on a refused `env` value, which
|
|
263
|
+
* a booted worker cannot have.
|
|
264
|
+
*/
|
|
265
|
+
export function resolveJobSize({ project = null, limits = [], env = {} } = {}) {
|
|
266
|
+
const defaults = jobSizeDefaults(env);
|
|
267
|
+
const row = typeof project === "string" && Array.isArray(limits) ? (limits.find((l) => l?.scope === `${PROJECT_ROW_PREFIX}${project}`) ?? null) : null;
|
|
268
|
+
const rowMem = typeof row?.memory === "string" ? parseMemory(row.memory) : null;
|
|
269
|
+
const rowCpu = row?.cpus !== null && row?.cpus !== undefined ? parseCpus(row.cpus) : null;
|
|
270
|
+
const source = rowMem !== null || rowCpu !== null ? "project" : defaults.memSet || defaults.cpuSet ? "env" : "default";
|
|
271
|
+
return { memMiB: rowMem ?? defaults.memMiB, cpuCenti: rowCpu ?? defaults.cpuCenti, source };
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* A size as a record carries it (INT-RUN-HISTORY-FILE-CONTRACT): `{ memMiB, cpuCenti, source }` REBUILT as an explicit
|
|
276
|
+
* literal, or null for anything that is not one. Integers and a fixed word, so the record stays PII-free.
|
|
277
|
+
*/
|
|
278
|
+
export function recordedJobSize(size) {
|
|
279
|
+
if (size === null || typeof size !== "object" || Array.isArray(size)) return null;
|
|
280
|
+
try {
|
|
281
|
+
assertJobSize(size);
|
|
282
|
+
} catch {
|
|
283
|
+
return null;
|
|
284
|
+
}
|
|
285
|
+
return JOB_SIZE_SOURCES.includes(size.source) ? { memMiB: size.memMiB, cpuCenti: size.cpuCenti, source: size.source } : null;
|
|
286
|
+
}
|
package/src/job-user.mjs
CHANGED
|
@@ -207,9 +207,36 @@ export function jobUserRefusal(causeOrDecision) {
|
|
|
207
207
|
* NOT cached: `unknown`, `unmappable` `runtime-unreadable`, and any decision made without an answered read (an `image`
|
|
208
208
|
* decided on macOS while the daemon was still starting). Each describes an answer rather than a daemon, and a cached one
|
|
209
209
|
* would retry or refuse every later job on that endpoint, or leave the observations unread, until the worker restarted.
|
|
210
|
-
* A cached decision is otherwise kept until the endpoint state changes
|
|
211
|
-
*
|
|
210
|
+
* A cached decision is otherwise kept until the endpoint state changes or for `maxAgeMs` (issue #596), whichever comes
|
|
211
|
+
* first, and `resolveJobUser.invalidate()` drops it at once. The age exists for the facts that move behind an unchanged
|
|
212
|
+
* endpoint: Docker Desktop's VM resized to another CPU count changes every job's `--cpus` ceiling, and a daemon
|
|
213
|
+
* reconfigured on one socket path (rootful to rootless) changes the job user. Ten minutes bounds how long either goes
|
|
214
|
+
* unseen; a lowered CPU count is caught sooner, because Docker refuses the stale `--cpus` and the job path invalidates
|
|
215
|
+
* (`cpu_ceiling_stale`, `run-container.mjs`).
|
|
216
|
+
*
|
|
217
|
+
* STALE WHILE ERROR (issue #596, gate round 2). A read past the age is a fresh read, and one that does not answer (a
|
|
218
|
+
* timeout, a daemon restarting, an unreadable reply) or throws keeps serving the last answered value for the same key
|
|
219
|
+
* until a read answers, logging `job_user_facts_stale` with the failed read's reason once per run of failures. Deciding
|
|
220
|
+
* such a read as a first read would make that pickup unavailable (an infrastructure retry) where, before the age
|
|
221
|
+
* existed, the cached answer was simply reused; the age is there to SEE a change, never to manufacture an outage.
|
|
222
|
+
* `invalidate()` is different on purpose: it is called because the job path just proved the cached CPU count wrong
|
|
223
|
+
* (Docker refused its `--cpus`, exit 125 "Range of CPUs"), so it drops the value entirely. Serving that count again
|
|
224
|
+
* would refuse every job the same way, so after an invalidate the next pickup reads fresh and, if that read fails, is
|
|
225
|
+
* unavailable exactly as a first read on a new endpoint is.
|
|
226
|
+
*/
|
|
227
|
+
export const JOB_USER_FACTS_MAX_AGE_MS = 10 * 60_000;
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* How long a kept answer may be served while every re-read fails (issue #596, phase 1 gate round 3, carried to phase 2).
|
|
231
|
+
* Stale-while-error exists so one slow read is not an outage; it must not let a runtime that stopped answering a day ago
|
|
232
|
+
* keep deciding jobs from what it said then. Past this the kept answer is no longer served and a pickup is decided as a
|
|
233
|
+
* first read is (unavailable, retried as infrastructure, said per job), which is loud where it belongs. Chosen over
|
|
234
|
+
* repeating the stale line every max-age plus a doctor field: one comparison in each cache, no new surface, and the
|
|
235
|
+
* existing per-job unavailability is the honest signal for a runtime that has not answered in a day. Shared by the
|
|
236
|
+
* podman venue's info cache (`cachedPodmanInfo`).
|
|
212
237
|
*/
|
|
238
|
+
export const STALE_FACTS_CEILING_MS = 24 * 60 * 60_000;
|
|
239
|
+
|
|
213
240
|
export function makeJobUserResolver({
|
|
214
241
|
readFacts,
|
|
215
242
|
platform = process.platform,
|
|
@@ -217,18 +244,46 @@ export function makeJobUserResolver({
|
|
|
217
244
|
euid = process.geteuid?.(),
|
|
218
245
|
egid = process.getegid?.(),
|
|
219
246
|
stat = statSync,
|
|
247
|
+
now = Date.now,
|
|
248
|
+
maxAgeMs = JOB_USER_FACTS_MAX_AGE_MS,
|
|
249
|
+
log = () => {},
|
|
220
250
|
} = {}) {
|
|
221
251
|
let cached = null;
|
|
252
|
+
// True while a run of failed re-reads is being answered from `cached`, so the line is written once per run.
|
|
253
|
+
let staleSaid = false;
|
|
222
254
|
const inFlight = new Map();
|
|
255
|
+
resolveJobUser.invalidate = () => {
|
|
256
|
+
cached = null;
|
|
257
|
+
staleSaid = false;
|
|
258
|
+
};
|
|
223
259
|
return resolveJobUser;
|
|
224
260
|
async function resolveJobUser({ endpoint, key }) {
|
|
225
|
-
if (cached && cached.key === key) return cached.value;
|
|
261
|
+
if (cached && cached.key === key && now() - cached.at < maxAgeMs) return cached.value;
|
|
226
262
|
if (inFlight.has(key)) return inFlight.get(key);
|
|
227
263
|
const work = (async () => {
|
|
228
264
|
// Asked on EVERY platform and endpoint, although a VM-backed platform or an endpoint on another machine decides
|
|
229
265
|
// `image` before any daemon fact is read: the same read is where the runtime observations come from (issue
|
|
230
266
|
// #345), and a Docker Desktop host skipped here would never be credited with the bounds its daemon applies.
|
|
231
|
-
|
|
267
|
+
let daemon;
|
|
268
|
+
let threw = null;
|
|
269
|
+
try {
|
|
270
|
+
daemon = await readFacts();
|
|
271
|
+
} catch (error) {
|
|
272
|
+
threw = error;
|
|
273
|
+
daemon = { answered: false, reason: "read-threw", transient: true };
|
|
274
|
+
}
|
|
275
|
+
// The value an expired entry may still serve when this read failed (see STALE WHILE ERROR above). Taken AFTER
|
|
276
|
+
// the read, so an `invalidate()` that landed while it was out is honoured by this very pickup.
|
|
277
|
+
// Never past `STALE_FACTS_CEILING_MS`: an answer that old is not served, whatever the read says now.
|
|
278
|
+
const stale = cached && cached.key === key && now() - cached.at < STALE_FACTS_CEILING_MS ? cached : null;
|
|
279
|
+
if (threw && !stale) throw threw;
|
|
280
|
+
if (stale && daemon?.answered !== true) {
|
|
281
|
+
if (!staleSaid) {
|
|
282
|
+
staleSaid = true;
|
|
283
|
+
log("job_user_facts_stale", { reason: daemon?.reason ?? "no-daemon-facts", ageMs: now() - stale.at });
|
|
284
|
+
}
|
|
285
|
+
return stale.value;
|
|
286
|
+
}
|
|
232
287
|
// A local docker endpoint's display form IS its unix path (credentials never ride a unix URL); with no endpoint,
|
|
233
288
|
// Podman's own shape names the service socket.
|
|
234
289
|
const socketPath = endpoint?.local === true && typeof endpoint.endpoint === "string" && endpoint.endpoint.startsWith("unix://")
|
|
@@ -237,9 +292,15 @@ export function makeJobUserResolver({
|
|
|
237
292
|
const socket = socketFacts(socketPath, { stat });
|
|
238
293
|
const decision = decideJobUser({ platform, release, euid, egid, endpoint, daemon, socket });
|
|
239
294
|
const value = { decision, facts: daemon?.answered ? daemon.facts : null, daemon, socket };
|
|
295
|
+
// An ANSWERED read ends a run of failures even when it decides nothing cacheable (an `unknown`, a
|
|
296
|
+
// `runtime-unreadable`), so the next failure after it is said again (gate round 3 of phase 1).
|
|
297
|
+
if (daemon?.answered === true) staleSaid = false;
|
|
240
298
|
// Cached only for an ANSWERED read: an `image` decided on a VM-backed platform while its daemon was still starting
|
|
241
299
|
// must not pin "not read" for the runtime observations until the endpoint changes.
|
|
242
|
-
if (decision.mode !== "unknown" && decision.cause !== "runtime-unreadable" && daemon?.answered === true)
|
|
300
|
+
if (decision.mode !== "unknown" && decision.cause !== "runtime-unreadable" && daemon?.answered === true) {
|
|
301
|
+
cached = { key, value, at: now() };
|
|
302
|
+
staleSaid = false;
|
|
303
|
+
}
|
|
243
304
|
return value;
|
|
244
305
|
})();
|
|
245
306
|
inFlight.set(key, work);
|