@edgehero/pi-dispatch 3.1.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +38 -0
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +599 -16
- package/src/host-budget.mjs +736 -0
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +2 -1
package/src/live-probes.mjs
CHANGED
|
@@ -28,7 +28,9 @@
|
|
|
28
28
|
*/
|
|
29
29
|
|
|
30
30
|
import { READ_BACK_BY_A_LIVE_PROBE } from "./backend-conformance.mjs";
|
|
31
|
-
import { containerSpec } from "./container-spec.mjs";
|
|
31
|
+
import { containerSpec, memoryBytes } from "./container-spec.mjs";
|
|
32
|
+
import { CGROUP_PARENT, DEFAULT_JOB_SIZE } from "./job-size.mjs";
|
|
33
|
+
import { cpuMaxLine, parseCpuMaxRead } from "./cpu-reserve.mjs";
|
|
32
34
|
import { ISOLATION_FLAGS, buildDockerRunArgs } from "./docker-run.mjs";
|
|
33
35
|
import { DEFAULT_EGRESS_PROXY, EGRESS_PROXY_PORT, createJobNetworkWith, networkEndpoints, networkNameFor, removeNetworkOrSay } from "./egress.mjs";
|
|
34
36
|
import { detachBlockedSentence, makeDetachGate } from "./netns-keeper.mjs";
|
|
@@ -112,8 +114,14 @@ export function liveFixture(root) {
|
|
|
112
114
|
// `relabel` (issue #355) is a job's too: where a job's own mounts carry `:Z`, so do the probe's, because a probe mounted
|
|
113
115
|
// the way no job is would read back a container no job gets. The fixture is doctor's own directory, so its workspace
|
|
114
116
|
// is relabelled like a forge job's clone (`workspaceOwned`); its global overlay directory never is, exactly as a job's.
|
|
115
|
-
|
|
116
|
-
|
|
117
|
+
//
|
|
118
|
+
// `size` and `hostCpus` (issue #596) are a job's too: the deployment's default size and the `--cpus` ceiling of this
|
|
119
|
+
// runtime's own CPU count, so the probe is bounded exactly as a job without a project size is, and reads that back.
|
|
120
|
+
//
|
|
121
|
+
// `cgroupParent` (issue #596, phase 2) is a job's too: the jobs' parent cgroup, or null where this venue runs jobs without
|
|
122
|
+
// it (`cgroupParentFor`), so the probe sits where a job sits and `cgroupParentVerdict` reads that back.
|
|
123
|
+
function probeOptions({ image, name, fixture, user = null, relabel = false, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
|
|
124
|
+
return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, size, hostCpus, cgroupParent, ...fixture };
|
|
117
125
|
}
|
|
118
126
|
|
|
119
127
|
// `buildArgs` (issue #354) is the RUNTIME's job builder, never a second one: each probe below is built by whichever
|
|
@@ -121,32 +129,32 @@ function probeOptions({ image, name, fixture, user = null, relabel = false }) {
|
|
|
121
129
|
// The default is docker's, so every argv here is byte-for-byte what it was.
|
|
122
130
|
|
|
123
131
|
/** The probe container's argv: the job builder's, detached, with `sleep <derived seconds>` as its whole program. */
|
|
124
|
-
export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
|
|
125
|
-
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
|
|
132
|
+
export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
|
|
133
|
+
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
|
|
126
134
|
}
|
|
127
135
|
|
|
128
136
|
/**
|
|
129
137
|
* The pinning probe's argv: the job builder's, detached, against an image this host does not have. Detached so a
|
|
130
138
|
* container that WAS created prints the ID it is removed by; with `--pull=never` in the builder none should be.
|
|
131
139
|
*/
|
|
132
|
-
export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
|
|
133
|
-
return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel }), extraFlags: ["-d"] });
|
|
140
|
+
export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
|
|
141
|
+
return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d"] });
|
|
134
142
|
}
|
|
135
143
|
|
|
136
144
|
/**
|
|
137
145
|
* One ephemeral run's argv (issue #344): the job builder's, detached, running EPHEMERAL_SCRIPT with the nonce and the
|
|
138
146
|
* run's number. The same NAME both times, because "a job id run twice" is the question.
|
|
139
147
|
*/
|
|
140
|
-
export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
|
|
141
|
-
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
|
|
148
|
+
export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
|
|
149
|
+
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
|
|
142
150
|
}
|
|
143
151
|
|
|
144
152
|
/**
|
|
145
153
|
* One peer's argv (issue #344): the job builder's, detached, on its OWN job network, running PEER_SCRIPT, which answers
|
|
146
154
|
* every connection with the nonce for `seconds`. Built with `network` set exactly as a job with egress armed is.
|
|
147
155
|
*/
|
|
148
|
-
export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
|
|
149
|
-
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
|
|
156
|
+
export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
|
|
157
|
+
return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
|
|
150
158
|
}
|
|
151
159
|
|
|
152
160
|
/**
|
|
@@ -156,6 +164,9 @@ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = li
|
|
|
156
164
|
export const STATUS_SCRIPT = [
|
|
157
165
|
"cat /proc/1/status",
|
|
158
166
|
'if [ -f /sys/fs/cgroup/cgroup.controllers ]; then echo "cgroup:v2"; echo "pids.max:$(cat /sys/fs/cgroup/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory.max 2>/dev/null)";',
|
|
167
|
+
// Issue #596: the swap bound (0 when --memory-swap equals --memory), the CPU weight --cpu-shares became, and the
|
|
168
|
+
// --cpus ceiling. cgroup v2 only: v1 spells all three differently, and every venue measured is v2.
|
|
169
|
+
'echo "memory.swap.max:$(cat /sys/fs/cgroup/memory.swap.max 2>/dev/null)"; echo "cpu.weight:$(cat /sys/fs/cgroup/cpu.weight 2>/dev/null)"; echo "cpu.max:$(cat /sys/fs/cgroup/cpu.max 2>/dev/null)";',
|
|
159
170
|
'else echo "cgroup:v1"; echo "pids.max:$(cat /sys/fs/cgroup/pids/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null)"; fi',
|
|
160
171
|
].join("\n");
|
|
161
172
|
|
|
@@ -246,6 +257,9 @@ export function parseStatus(output) {
|
|
|
246
257
|
cgroup: field("cgroup"),
|
|
247
258
|
pidsMax: field("pids.max"),
|
|
248
259
|
memoryMax: field("memory.max"),
|
|
260
|
+
swapMax: field("memory.swap.max"),
|
|
261
|
+
cpuWeight: field("cpu.weight"),
|
|
262
|
+
cpuMax: field("cpu.max"),
|
|
249
263
|
};
|
|
250
264
|
}
|
|
251
265
|
|
|
@@ -257,9 +271,34 @@ export function expectedPidsLimit(flags = ISOLATION_FLAGS) {
|
|
|
257
271
|
|
|
258
272
|
/** The memory bound in bytes, from the spec's own default (`4g`), never a literal. */
|
|
259
273
|
export function expectedMemoryBytes(memory = containerSpec({ image: "i", name: "n", workspace: "/w" }).memory) {
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
274
|
+
return memoryBytes(memory);
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* What a container of this size reads back (issue #596), from the SAME spec builder the probe's argv came from:
|
|
279
|
+
* `{ pidsLimit, memoryBytes, cpuShares, cpuMax }`, `cpuMax` being the `cpu.max` line `--cpus` writes (`<quota> 100000`,
|
|
280
|
+
* or `max 100000` with no ceiling).
|
|
281
|
+
*/
|
|
282
|
+
export function expectedBounds(size = DEFAULT_JOB_SIZE, hostCpus = null, cpuBudgetCenti = null, cgroupParent = CGROUP_PARENT) {
|
|
283
|
+
const spec = containerSpec({ image: "i", name: "n", workspace: "/w", size, hostCpus, cpuBudgetCenti, cgroupParent });
|
|
284
|
+
// ROUNDED (issue #596, gate round 3 of phase 1): under a CPU budget the ceiling may be fractional (`3.3`), and
|
|
285
|
+
// `3.3 * 100000` is 329999.99999999994 in floating point, while the runtime writes the integer quota 330000.
|
|
286
|
+
return { pidsLimit: expectedPidsLimit(), memoryBytes: memoryBytes(spec.memory), cpuShares: spec.cpuShares, cpuMax: spec.cpus === null ? "max 100000" : `${Math.round(Number(spec.cpus) * 100000)} 100000` };
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/**
|
|
290
|
+
* The `cpu.weight` a `--cpu-shares` value becomes, under the two mappings measured (issue #596): runc 1.1.13 and crun
|
|
291
|
+
* 1.14 map linearly (`1 + ((shares - 2) * 9999) / 262142`, integer division: 1024 to 39, 2048 to 79), runc 1.5.1 and
|
|
292
|
+
* crun 1.27 map so 1024 is 100 (`10^((l^2 + 125l) / 612 - 7/34)` rounded up, l = log2(shares): 2048 to 174). Both
|
|
293
|
+
* reproduce every value of the lab's sweep. A weight that is neither is a runtime nobody measured, said as such.
|
|
294
|
+
*/
|
|
295
|
+
export function cpuWeightsFor(shares) {
|
|
296
|
+
if (!Number.isSafeInteger(shares) || shares < 2) return [];
|
|
297
|
+
if (shares >= 262144) return [10000];
|
|
298
|
+
const linear = 1 + Math.floor(((shares - 2) * 9999) / 262142);
|
|
299
|
+
const l = Math.log2(shares);
|
|
300
|
+
const curved = Math.ceil(10 ** ((l * l + 125 * l) / 612 - 7 / 34));
|
|
301
|
+
return [...new Set([linear, curved])];
|
|
263
302
|
}
|
|
264
303
|
|
|
265
304
|
const verdict = (property, ok, detail, extra = {}) => ({ property, ok, detail, ...extra });
|
|
@@ -273,7 +312,7 @@ const notReadBack = (property, why) => verdict(property, false, `not read back:
|
|
|
273
312
|
* FAILURE (the bound was not applied, as on rootless docker without delegation); a bound that could not be read at
|
|
274
313
|
* all is "not read back", never a pass.
|
|
275
314
|
*/
|
|
276
|
-
export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes() } = {}) {
|
|
315
|
+
export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes(), cpuShares = null, cpuMax = null } = {}) {
|
|
277
316
|
if (!status || status.capBnd === null || status.noNewPrivs === null) return notReadBack("isolation", "PID 1's status did not show CapBnd and NoNewPrivs");
|
|
278
317
|
const failures = [];
|
|
279
318
|
if (!/^0+$/.test(status.capBnd)) failures.push(`CapBnd ${status.capBnd} (the capability bounding set is not empty)`);
|
|
@@ -287,10 +326,30 @@ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memo
|
|
|
287
326
|
if (Number(got) !== want) failures.push(`${name} ${got}, expected ${want}`);
|
|
288
327
|
return null;
|
|
289
328
|
};
|
|
290
|
-
const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)]
|
|
329
|
+
const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)];
|
|
330
|
+
// Issue #596, read only by a caller that passes the size's bounds (`expectedBounds`). The swap bound is part of the
|
|
331
|
+
// memory bound: `--memory-swap` equal to `--memory` makes it 0, and anything else lets a job swap past its size.
|
|
332
|
+
// `cpu.max` is the host ceiling (`--cpus`), which keeps the host's reserve, so a different one is a failure too.
|
|
333
|
+
// `cpu.weight` is the job's share, whose number depends on the runtime's version: one neither measured mapping
|
|
334
|
+
// gives is said, not failed, because the ORDER between jobs is what it carries and an unknown mapping may keep it.
|
|
335
|
+
let cpuNote = "";
|
|
336
|
+
if (cpuMax !== null) {
|
|
337
|
+
const swap = status.swapMax;
|
|
338
|
+
if (swap === null || swap === undefined || swap === "") unread.push("memory.swap.max not readable");
|
|
339
|
+
else if (swap !== "0") failures.push(`memory.swap.max ${swap}, expected 0 (a job may swap beyond its memory)`);
|
|
340
|
+
const max = status.cpuMax;
|
|
341
|
+
if (max === null || max === undefined || max === "") unread.push("cpu.max not readable");
|
|
342
|
+
else if (max !== cpuMax) failures.push(`cpu.max ${max}, expected ${cpuMax}`);
|
|
343
|
+
const weight = status.cpuWeight;
|
|
344
|
+
const known = cpuWeightsFor(cpuShares);
|
|
345
|
+
if (weight === null || weight === undefined || weight === "") unread.push("cpu.weight not readable");
|
|
346
|
+
else if (!known.includes(Number(weight))) unread.push(`cpu.weight ${weight} is neither mapping measured for --cpu-shares=${cpuShares} (${known.join(" or ")})`);
|
|
347
|
+
else cpuNote = `, memory.swap.max 0, cpu.max ${cpuMax}, cpu.weight ${weight} (--cpu-shares=${cpuShares})`;
|
|
348
|
+
}
|
|
349
|
+
const missing = unread.filter(Boolean);
|
|
291
350
|
if (failures.length > 0) return verdict("isolation", false, failures.join("; "));
|
|
292
|
-
if (
|
|
293
|
-
return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}`);
|
|
351
|
+
if (missing.length > 0) return notReadBack("isolation", `${missing.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
|
|
352
|
+
return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}${cpuNote}`);
|
|
294
353
|
}
|
|
295
354
|
|
|
296
355
|
/** NON-ROOT: all four Uid fields (real, effective, saved, filesystem) nonzero, on the process the job would be. */
|
|
@@ -612,6 +671,60 @@ export function peerTargetsOf(inspectOutput, { network, name }) {
|
|
|
612
671
|
return { targets: [...[...hosts].map((h) => `${h}:${PEER_PORT}`), ...addresses], addresses };
|
|
613
672
|
}
|
|
614
673
|
|
|
674
|
+
/**
|
|
675
|
+
* WHERE THE RUNTIME PUT A JOB-BUILT CONTAINER (issue #596, phase 2, INT-LIVE-PROBE-CONTRACT): `{ ok, warn?, detail,
|
|
676
|
+
* placed, quota }`, from `<bin> inspect --format={{.HostConfig.CgroupParent}}|{{.State.Pid}}` and this host's own files.
|
|
677
|
+
*
|
|
678
|
+
* Three readings, each as far as this host can see. The runtime's RECORD of the parent must be the one the argv named
|
|
679
|
+
* (`expected`; none where the venue runs jobs without it). The process's own cgroup (`/proc/<pid>/cgroup`, cgroup v2's
|
|
680
|
+
* `0::<path>`) must lie under `/<parent>/`, which proves the placement rather than the flag: readable for a local Linux
|
|
681
|
+
* daemon and rootless Podman, not for a daemon in a VM (Docker Desktop), where the record alone is said as such. The
|
|
682
|
+
* container cannot see any of this itself (a private cgroup namespace shows it `0::/`, measured). Then the parent's
|
|
683
|
+
* `cpu.max`, where readable, against the CPU budget (`cpuBudgetCenti`: an integer, `Infinity` for off, null not
|
|
684
|
+
* checked): a quota that is not the budget is a WARNING, "no host CPU reserve across jobs", never a failure of the
|
|
685
|
+
* placement, because on a systemd host only the operator can set it.
|
|
686
|
+
*/
|
|
687
|
+
export function cgroupParentVerdict({ inspected, expected = CGROUP_PARENT, cpuBudgetCenti = null, readFile = () => {
|
|
688
|
+
throw new Error("no reader");
|
|
689
|
+
}, bin = "docker" }) {
|
|
690
|
+
const out = (ok, detail, extra = {}) => ({ ok, detail, placed: null, quota: undefined, ...extra });
|
|
691
|
+
if (inspected?.code !== 0) return out(false, `not read back: ${bin} inspect did not answer`, { warn: true });
|
|
692
|
+
const [recorded = "", pidText = ""] = String(inspected.stdout ?? "").trim().split("|");
|
|
693
|
+
if (expected === null) {
|
|
694
|
+
if (recorded.includes(CGROUP_PARENT)) return out(false, `the runtime recorded the parent ${recorded} for a venue that runs jobs without one`);
|
|
695
|
+
return out(true, `this venue runs jobs without the ${CGROUP_PARENT} parent (Podman's cgroup manager is not systemd), so their CPU weight is capped at 1024: the egress proxy and Valkey get a fair share of the CPU, not a reserve`, { warn: true });
|
|
696
|
+
}
|
|
697
|
+
// The bare name, or a path ending in it (a runtime may record where the slice resolved); the process's own cgroup
|
|
698
|
+
// below is the proof either way.
|
|
699
|
+
if (recorded.replace(/^\//, "") !== expected && !recorded.endsWith(`/${expected}`)) return out(false, `the runtime recorded the parent ${JSON.stringify(recorded)}, not ${expected}, so this container is outside the jobs' CPU reserve`);
|
|
700
|
+
let cgroupText = null;
|
|
701
|
+
if (/^[1-9][0-9]{0,9}$/.test(pidText)) {
|
|
702
|
+
try {
|
|
703
|
+
cgroupText = readFile(`/proc/${pidText}/cgroup`);
|
|
704
|
+
} catch {
|
|
705
|
+
cgroupText = null;
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
const path = cgroupText === null ? null : (/^0::(\/\S*)$/m.exec(String(cgroupText))?.[1] ?? null);
|
|
709
|
+
if (path === null) return out(true, `the runtime recorded the parent ${expected}; the container's own cgroup is not readable from this host (its process runs in the runtime's VM or another namespace), so the placement and the parent's quota were not read here`, { warn: true });
|
|
710
|
+
const at = path.indexOf(`/${expected}/`);
|
|
711
|
+
if (at === -1) return out(false, `the runtime recorded the parent ${expected}, but the container's cgroup is ${path}, not under it`);
|
|
712
|
+
const parentDir = `/sys/fs/cgroup${path.slice(0, at + expected.length + 1)}`;
|
|
713
|
+
let quota;
|
|
714
|
+
try {
|
|
715
|
+
quota = parseCpuMaxRead(readFile(`${parentDir}/cpu.max`));
|
|
716
|
+
} catch {
|
|
717
|
+
quota = null;
|
|
718
|
+
}
|
|
719
|
+
const placed = path.slice(0, at + expected.length + 1);
|
|
720
|
+
if (quota === null) return out(true, `the container's cgroup is under ${placed}; the parent's cpu.max is not readable here`, { warn: true, placed });
|
|
721
|
+
const shown = quota.cpuCenti === null ? "no quota" : `a quota of ${quota.cpuCenti / 100} CPUs (${cpuMaxLine(quota.cpuCenti)})`;
|
|
722
|
+
if (cpuBudgetCenti === null || cpuBudgetCenti === undefined) return out(true, `the container's cgroup is under ${placed}, which has ${shown}`, { placed, quota: quota.cpuCenti });
|
|
723
|
+
const want = cpuBudgetCenti === Infinity ? null : cpuBudgetCenti;
|
|
724
|
+
if (quota.cpuCenti === want) return out(true, `the container's cgroup is under ${placed}, which has ${want === null ? "no quota, as the CPU budget is off" : `${shown}, the host's CPU budget, so all jobs together keep the reserve`}`, { placed, quota: quota.cpuCenti });
|
|
725
|
+
return out(true, `the container's cgroup is under ${placed}, which has ${shown}, not ${want === null ? "none (the CPU budget is off)" : `the CPU budget of ${want / 100}`}: no host CPU reserve across jobs`, { warn: true, placed, quota: quota.cpuCenti });
|
|
726
|
+
}
|
|
727
|
+
|
|
615
728
|
/**
|
|
616
729
|
* The whole sequence. Returns `{ ran, reason, verdicts, notes, swept, ranAs }`: `ran` false with a `reason` when nothing
|
|
617
730
|
* was read back; the verdicts in READ_BACK_BY_A_LIVE_PROBE's order otherwise; `notes` for a teardown that failed, on
|
|
@@ -660,6 +773,14 @@ export async function runLiveProbes({
|
|
|
660
773
|
// Issue #452, gate round 3: the runtime as this pass already read it (`{ podman, rootless, version } | null`), for the
|
|
661
774
|
// detach gate, so doctor asks the daemon once per run. Absent, the gate reads it itself.
|
|
662
775
|
readRuntime = null,
|
|
776
|
+
// Issue #596: the size every probe container is built at (a job's default) and the runtime's CPU count for the
|
|
777
|
+
// `--cpus` ceiling; the isolation verdict reads both back.
|
|
778
|
+
size = DEFAULT_JOB_SIZE,
|
|
779
|
+
hostCpus = null,
|
|
780
|
+
// Issue #596, phase 2: the parent cgroup a job on this venue runs under (null: none, `cgroupParentFor`) and the CPU
|
|
781
|
+
// budget its quota should be (an integer in hundredths, `Infinity` for off, null or absent: not checked).
|
|
782
|
+
cgroupParent = CGROUP_PARENT,
|
|
783
|
+
cpuBudgetCenti = null,
|
|
663
784
|
}) {
|
|
664
785
|
const notRun = (reason) => ({ ran: false, reason, verdicts: [], notes: [], swept: [] });
|
|
665
786
|
const notLocal = notRun(`this shell's ${bin} CLI is not observed to point at this host, so bind paths and .Mounts would describe another machine; nothing was run`);
|
|
@@ -764,7 +885,7 @@ export async function runLiveProbes({
|
|
|
764
885
|
}
|
|
765
886
|
|
|
766
887
|
// --- the reading container: mounts, status, writes ---
|
|
767
|
-
const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
|
|
888
|
+
const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
|
|
768
889
|
const probeId = reading.entry.id;
|
|
769
890
|
if (reading.result?.code !== 0 || probeId === null) {
|
|
770
891
|
// The runtime's own words, when it printed any (issue #453, gate round 1): "did not start" alone left the operator
|
|
@@ -773,7 +894,7 @@ export async function runLiveProbes({
|
|
|
773
894
|
return { ran: false, reason: `the probe container did not start${said ? ` (${bin} said: ${said})` : ""}, so nothing was read back`, verdicts: [], notes, swept };
|
|
774
895
|
}
|
|
775
896
|
|
|
776
|
-
const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel })).mounts;
|
|
897
|
+
const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel, size, hostCpus, cgroupParent })).mounts;
|
|
777
898
|
const inspected = await step(["inspect", "--format={{json .Mounts}}", probeId]);
|
|
778
899
|
// Issue #345: the mount table as the container itself sees it, by a constant `cat`, for what `.Mounts` does not list.
|
|
779
900
|
const mountinfo = await step(["exec", probeId, "cat", "/proc/self/mountinfo"]);
|
|
@@ -781,8 +902,12 @@ export async function runLiveProbes({
|
|
|
781
902
|
|
|
782
903
|
const statusRun = await step(["exec", probeId, "sh", "-c", STATUS_SCRIPT]);
|
|
783
904
|
const status = statusRun?.code === 0 ? parseStatus(statusRun.stdout) : null;
|
|
784
|
-
const isolation = status ? isolationVerdict(status) : notReadBack("isolation", "the status probe did not run in the container");
|
|
905
|
+
const isolation = status ? isolationVerdict(status, expectedBounds(size, hostCpus, null, cgroupParent)) : notReadBack("isolation", "the status probe did not run in the container");
|
|
785
906
|
const nonRoot = status ? nonRootVerdict(status) : notReadBack("nonRoot", "the status probe did not run in the container");
|
|
907
|
+
// Issue #596, phase 2: where the runtime put the container (its record, and the process's own cgroup where this host
|
|
908
|
+
// can read it) and, where readable, the parent's quota. Its own line, not one of the declared properties.
|
|
909
|
+
const placed = await step(["inspect", "--format={{.HostConfig.CgroupParent}}|{{.State.Pid}}", probeId]);
|
|
910
|
+
const parentRead = cgroupParentVerdict({ inspected: placed, expected: cgroupParent, cpuBudgetCenti, readFile: (path) => fs.readFileSync(path, "utf8"), bin });
|
|
786
911
|
|
|
787
912
|
const written = await step(["exec", probeId, "sh", "-c", WRITE_SCRIPT, "sh", nonce]);
|
|
788
913
|
const readBack = (dir) => {
|
|
@@ -807,7 +932,7 @@ export async function runLiveProbes({
|
|
|
807
932
|
await release(reading.entry);
|
|
808
933
|
|
|
809
934
|
// --- the pinning container: an image this host does not have ---
|
|
810
|
-
const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs }));
|
|
935
|
+
const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
|
|
811
936
|
const after = await step(["image", "inspect", absentImageRef(nonce)]);
|
|
812
937
|
const stillAbsent = after?.code === 0 ? false : typeof after?.code === "number" ? true : null;
|
|
813
938
|
const imagePinning = imagePinningVerdict({ code: pinning.result?.code, output: `${pinning.result?.stdout ?? ""}${pinning.result?.stderr ?? ""}`, stillAbsent, bin });
|
|
@@ -815,7 +940,7 @@ export async function runLiveProbes({
|
|
|
815
940
|
|
|
816
941
|
// --- the ephemeral pair (issue #344): one name, two runs, each waited on until it is gone ---
|
|
817
942
|
const runEphemeral = async (n) => {
|
|
818
|
-
const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs }));
|
|
943
|
+
const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
|
|
819
944
|
const started = result?.code === 0 && entry.id !== null;
|
|
820
945
|
// A HELD NAME is the daemon refusing the create for the name, in its own words (measured: Docker "Conflict. ...
|
|
821
946
|
// is already in use", Podman "that name is already in use"). Not a listed container: after the first run was
|
|
@@ -864,7 +989,7 @@ export async function runLiveProbes({
|
|
|
864
989
|
} catch {
|
|
865
990
|
break;
|
|
866
991
|
}
|
|
867
|
-
const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
|
|
992
|
+
const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
|
|
868
993
|
peers.push(entry);
|
|
869
994
|
if (result?.code !== 0 || entry.id === null) break;
|
|
870
995
|
ids[key] = entry.id;
|
|
@@ -901,7 +1026,7 @@ export async function runLiveProbes({
|
|
|
901
1026
|
const byProperty = { isolation, ephemeral, mountSet, egress: egressVerdict(egress ?? {}), jobToJobIsolation, imagePinning, nonRoot, localFolders };
|
|
902
1027
|
// The uid PID 1 actually ran as, for doctor's job-user line: the decision it was given, read back.
|
|
903
1028
|
const ranAs = Array.isArray(status?.uids) && /^\d+$/.test(status.uids[1] ?? "") ? Number(status.uids[1]) : null;
|
|
904
|
-
return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs };
|
|
1029
|
+
return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs, cgroupParent: parentRead };
|
|
905
1030
|
} finally {
|
|
906
1031
|
for (const entry of owned) await release(entry);
|
|
907
1032
|
for (const entry of networks) await dropNetwork(entry);
|
package/src/prepare.mjs
CHANGED
|
@@ -14,6 +14,7 @@ import { buildForgejoPrompt } from "./forgejo-prompt.mjs";
|
|
|
14
14
|
import { buildAzurePrompt } from "./azure-prompt.mjs";
|
|
15
15
|
import { prepareLocalWorkspace } from "./prepare-local.mjs";
|
|
16
16
|
import { copySkillTree } from "./copy-tree.mjs";
|
|
17
|
+
import { recordedJobSize } from "./job-size.mjs";
|
|
17
18
|
|
|
18
19
|
/**
|
|
19
20
|
* The subdirectory of the per-job dir a trigger's injected skills are copied into, so they reach the
|
|
@@ -82,7 +83,7 @@ export function makePrepareWorkspace({
|
|
|
82
83
|
removeDir = (dir) => rmSync(dir, { recursive: true, force: true }),
|
|
83
84
|
}) {
|
|
84
85
|
ensureDir(jobsDir);
|
|
85
|
-
return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false } = {}) {
|
|
86
|
+
return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false, size = null } = {}) {
|
|
86
87
|
ensureDir(jobsDir);
|
|
87
88
|
const jobDir = mkdtempSync(join(jobsDir, "job-"));
|
|
88
89
|
// Issue #524: a THROW out of anything below leaves `jobDir` to nobody. The processor tears down only what
|
|
@@ -92,7 +93,7 @@ export function makePrepareWorkspace({
|
|
|
92
93
|
// covers the thrown ones, in one place for every kind, rather than in each preparer that can throw.
|
|
93
94
|
// A retry makes a fresh directory, so nothing is lost by removing this one.
|
|
94
95
|
try {
|
|
95
|
-
return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio });
|
|
96
|
+
return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size });
|
|
96
97
|
} catch (error) {
|
|
97
98
|
// GUARDED: a removal that fails (a busy mount, a permission flipped mid-job) must never replace the error
|
|
98
99
|
// that is the job's actual outcome. The directory is then left, which is what happened before this catch.
|
|
@@ -103,7 +104,7 @@ export function makePrepareWorkspace({
|
|
|
103
104
|
}
|
|
104
105
|
};
|
|
105
106
|
|
|
106
|
-
async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio }) {
|
|
107
|
+
async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size }) {
|
|
107
108
|
// The trigger's injected skills (REQ-PER-TRIGGER-SKILLS, issue #60), COPIED here rather than
|
|
108
109
|
// mounted, and copied ONCE for every job kind because this is where local and forge converge.
|
|
109
110
|
//
|
|
@@ -137,6 +138,9 @@ export function makePrepareWorkspace({
|
|
|
137
138
|
...(jobUser ? { jobUser: { user: jobUser.user ?? null, home: jobUser.home ?? null } } : {}),
|
|
138
139
|
// Issue #429: the podman store the run's container lived in, for a podman run only.
|
|
139
140
|
...(typeof podmanStore === "string" && podmanStore !== "" ? { podmanStore } : {}),
|
|
141
|
+
// Issue #596: the size the run's container had, so a re-opened sandbox gets the same memory and CPU weight.
|
|
142
|
+
// Rebuilt (`recordedJobSize`), and only when the processor resolved one; a direct call stamps nothing.
|
|
143
|
+
...(recordedJobSize(size) ? { size: recordedJobSize(size) } : {}),
|
|
140
144
|
};
|
|
141
145
|
if (job.kind === "local") {
|
|
142
146
|
// Harness text above, operator DATA below: the fixed pointer line names /job/event.json so a
|
package/src/processor.mjs
CHANGED
|
@@ -8,7 +8,7 @@ import { PROJECT_CAP_REASON } from "./scoped-limits.mjs";
|
|
|
8
8
|
import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
|
|
9
9
|
import { RESERVED_ENV_NAMES } from "./triggers.mjs";
|
|
10
10
|
import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
|
|
11
|
-
import { COST_CAP_WHYS, RUNNER_POLICY_REASONS } from "./run-history.mjs";
|
|
11
|
+
import { COST_CAP_WHYS, EXIT_OOM_KILLED, RUNNER_POLICY_REASONS } from "./run-history.mjs";
|
|
12
12
|
import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
|
|
13
13
|
import { CAPABILITY_GATES, EXIT_AUTH_CAPABILITY } from "./image-preflight.mjs";
|
|
14
14
|
import { modelListProblem, modelOnList, splitModelEntry } from "./model-ref.mjs";
|
|
@@ -83,6 +83,29 @@ export const OBSERVATION_COMMENT_UNNAMED = "the venue this job runs on did not c
|
|
|
83
83
|
* into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
|
|
84
84
|
*/
|
|
85
85
|
|
|
86
|
+
/** The exit code of a container whose main process was SIGKILLed: a worker's stop or the kernel's OOM killer. */
|
|
87
|
+
const EXIT_SIGKILL = 137;
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Issue #596: whether a run's memory peak reached the container's own limit closely enough for its OOM kill to be the
|
|
91
|
+
* job's own: `memPeak` (bytes, off the decisive exit line) at 90% of `limit` (bytes, the `--memory=` the worker passed)
|
|
92
|
+
* or more. Both must be known; anything else is false, so the 137 stays infrastructure and retries.
|
|
93
|
+
*
|
|
94
|
+
* Why the check exists: `memory.events` `oom_kill` counts a kill by ANY OOM killer, the HOST's included, and the runner
|
|
95
|
+
* tree's `oom_score_adj` of 1000 makes a job the host's first victim when the machine itself runs short. A kill of
|
|
96
|
+
* that kind says nothing about the job's size, and a retry may well pass. A cgroup OOM happens only at the limit, so its
|
|
97
|
+
* peak sits there: measured in the issue #596 lab, `memory.peak` read exactly the bound (64m, 80m) or just under it on
|
|
98
|
+
* every venue. 90% and not 100% is the margin for that "just under"; nothing in between is read as a measurement.
|
|
99
|
+
*
|
|
100
|
+
* What it cannot see (the residual): a host OOM that strikes a job which is ALREADY at 90% of its own limit, page
|
|
101
|
+
* cache included (a job that read large files counts that cache), reads as the job's own OOM. And the peak is a
|
|
102
|
+
* high-water mark over the whole run, so a job that touched 90% early and was killed by the host later reads the same.
|
|
103
|
+
*/
|
|
104
|
+
export function peakReachedLimit(memPeak, limit) {
|
|
105
|
+
if (!Number.isSafeInteger(memPeak) || !Number.isSafeInteger(limit) || memPeak < 0 || limit <= 0) return false;
|
|
106
|
+
return memPeak >= Math.ceil((limit * 9) / 10);
|
|
107
|
+
}
|
|
108
|
+
|
|
86
109
|
// The post-spend terminal comments (issue #288). Every FREE refusal above the container already comments;
|
|
87
110
|
// these are the paths where money was spent and the run still ended without the agent's own status step,
|
|
88
111
|
// which used to tell the issue nothing (REQ-JOB-STATUS-COMMENTS' acceptance -- "exactly one completion or
|
|
@@ -106,6 +129,9 @@ export const TERMINAL_COMMENTS = {
|
|
|
106
129
|
"model-not-allowed": "Stopped: the run tried to call an AI model this trigger does not allow, or to change an AI request in a way it does not allow, so the call was not made. Partial work may exist. Not retried.",
|
|
107
130
|
"cost-cap-unenforceable": "Stopped: this run has a cost limit, and the job image could not enforce it before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
|
|
108
131
|
"model-policy-unenforceable": "Stopped: this run is limited to certain AI models, and the job image could not enforce that before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
|
|
132
|
+
// Issue #596. Never the size or the project: both are operator configuration, and the reader may be an issue author
|
|
133
|
+
// who can act on neither. The worker log and the run record carry them.
|
|
134
|
+
[EXIT_OOM_KILLED]: "Stopped: the job's container ran out of memory and was stopped. Partial work may exist. Not retried, because the same size would stop the same way. The operator can raise this job's memory size.",
|
|
109
135
|
};
|
|
110
136
|
|
|
111
137
|
// Issue #502: the `model-unknown` refusal's comment. Names no model: the reader may be an issue author.
|
|
@@ -238,7 +264,8 @@ export async function runJob(job, deps) {
|
|
|
238
264
|
// Issue #341: which uid this job's container runs as. `(job, { capabilities, observed }) =>` `{ user, home }`
|
|
239
265
|
// (`user` null = the image's own USER), `{ refused, cause }` or `{ unavailable, reason }`. The default runs
|
|
240
266
|
// every job as the image's user, exactly as before, so a wiring that omits it changes nothing. A non-refused answer
|
|
241
|
-
// may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with
|
|
267
|
+
// may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with, and
|
|
268
|
+
// `hostCpus` (issue #596), the runtime's CPU count from the same facts read, which sets the `--cpus` ceiling.
|
|
242
269
|
jobUserPreflight = async () => ({ user: null, home: null }),
|
|
243
270
|
// (session, { piVersion, context }) => { promoted, reason, bytes }. Promotes this job's transcript back into
|
|
244
271
|
// the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
|
|
@@ -300,8 +327,11 @@ export async function runJob(job, deps) {
|
|
|
300
327
|
mintToken,
|
|
301
328
|
isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
|
|
302
329
|
prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
|
|
303
|
-
// runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
|
|
330
|
+
// runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints?, size?, hostCpus?, unenforced? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
|
|
331
|
+
// `size` (issue #596) is `jobSize` below; `hostCpus` the runtime's CPU count off the job-user gate's own facts read.
|
|
332
|
+
// `unenforced` (issue #596) is the size flags that same read says the runtime drops, passed only when there are some.
|
|
304
333
|
// `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
|
|
334
|
+
// `resources` and `exitOomKilled` (issue #596) are the exit line's cgroup block and the supervisor's OOM report.
|
|
305
335
|
// `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
|
|
306
336
|
// `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
|
|
307
337
|
// `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
|
|
@@ -356,6 +386,14 @@ export async function runJob(job, deps) {
|
|
|
356
386
|
// after prepare, before any reserve. Absent on a bare wiring, which checks nothing.
|
|
357
387
|
pickupProject = null,
|
|
358
388
|
folderProject = null,
|
|
389
|
+
// Issue #596: the job's size `{ memMiB, cpuCenti, source }`, resolved at pickup from the same limits snapshot as
|
|
390
|
+
// every other gate (index.mjs). Handed to `prepareWorkspace` (a retained run's manifest records it, so a sandbox
|
|
391
|
+
// reopens the run at its size) and to `runContainer`, never through `job.data`. null on a bare wiring: neither call
|
|
392
|
+
// then carries it, and the container gets the built-in 4g and 2.
|
|
393
|
+
jobSize = null,
|
|
394
|
+
// Issue #596, phase 2: the host's CPU budget at pickup in hundredths (`host-budget.mjs`), or null when it is off or
|
|
395
|
+
// unknown. It becomes every job's `--cpus` (capped at the runtime's count), so no job can use the reserve.
|
|
396
|
+
cpuBudgetCenti = null,
|
|
359
397
|
// Issue #503 part 7: the builtin catalog's model object for (provider, id), or null (model-catalog.mjs
|
|
360
398
|
// `builtinModel`). Read only by the zero-rated check; the default knows no builtin model, so an unwired
|
|
361
399
|
// processor judges overlay models alone and reserves for every other.
|
|
@@ -412,6 +450,8 @@ export async function runJob(job, deps) {
|
|
|
412
450
|
// the record says about it (INT-RUN-HISTORY-FILE-CONTRACT), null until the reservation step ran.
|
|
413
451
|
let dollarHold = null;
|
|
414
452
|
let dollars = null;
|
|
453
|
+
// Issue #596: what the container used (`resources` off its exit line), null until a container ran and reported it.
|
|
454
|
+
let resources = null;
|
|
415
455
|
|
|
416
456
|
try {
|
|
417
457
|
// The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
|
|
@@ -1067,7 +1107,7 @@ export async function runJob(job, deps) {
|
|
|
1067
1107
|
// `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
|
|
1068
1108
|
// `portfolio` (issue #505) only for a job the gate above confirmed: prepare then writes /job/portfolio.json, after
|
|
1069
1109
|
// asking the live file once more. Absent otherwise, so every other job's prepare call is unchanged.
|
|
1070
|
-
prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
|
|
1110
|
+
prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}), ...(jobSize ? { size: jobSize } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
|
|
1071
1111
|
|
|
1072
1112
|
// A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
|
|
1073
1113
|
// or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
|
|
@@ -1270,12 +1310,29 @@ export async function runJob(job, deps) {
|
|
|
1270
1310
|
// then only a signed line is read (run-container.mjs, run-history.mjs `authenticExitLines`). An image that does not
|
|
1271
1311
|
// declare it is read as before, under the #542 trust rule below alone.
|
|
1272
1312
|
const exitAuth = (img.capabilities ?? []).includes(EXIT_AUTH_CAPABILITY);
|
|
1273
|
-
const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}) });
|
|
1313
|
+
const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null, exitOomKilled = false, memoryLimit = null, resources: ranResources = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}), ...(jobSize ? { size: jobSize } : {}), ...(Number.isSafeInteger(jobUser?.hostCpus) ? { hostCpus: jobUser.hostCpus } : {}), ...(Number.isSafeInteger(cpuBudgetCenti) ? { cpuBudgetCenti } : {}), ...(Array.isArray(jobUser?.unenforced) && jobUser.unenforced.length > 0 ? { unenforced: jobUser.unenforced } : {}) });
|
|
1274
1314
|
containerRan = true;
|
|
1315
|
+
// Issue #596: what the container used, off its exit line, rebuilt by the sink (null from a runContainer that predates
|
|
1316
|
+
// the field). Every result and every throw below carries it, so a retried attempt's record says what it used too.
|
|
1317
|
+
resources = ranResources ?? null;
|
|
1318
|
+
// Issue #596: CONFIRMED killed for memory. The image's supervisor (image/runner/src/supervise.mjs) outlives the runner,
|
|
1319
|
+
// whose process tree it gives the highest OOM score, and when the runner dies of SIGKILL with the cgroup's
|
|
1320
|
+
// `oom_kill` above 0 it writes the signed line `code: 137, reason: "oom-killed"` (parseExitOomKilled). All three
|
|
1321
|
+
// facts must agree: that line, verified under this run's key (an unsigned line is a tool's), and the container's
|
|
1322
|
+
// own exit 137. Docker's `oom` event is not used: it fires also when only a child was killed and the job went on
|
|
1323
|
+
// to exit 0, and Podman has no such event at all (both measured in the issue #596 lab). And a fourth: the line's
|
|
1324
|
+
// `memPeak` at 90% of the `--memory` this container got (`memoryLimit`, from runContainer) or more, because
|
|
1325
|
+
// `oom_kill` counts the HOST's OOM killer too (`peakReachedLimit`). A report below it stays infrastructure.
|
|
1326
|
+
const oomReported = exitOomKilled === true && exitAuthResult === "verified" && code === EXIT_SIGKILL;
|
|
1327
|
+
const oomKilled = oomReported && peakReachedLimit(resources?.memPeak ?? null, memoryLimit);
|
|
1328
|
+
// Numbers only (bytes), never a path or a project: the operator's trace of a kill read as the host's. Never on a run
|
|
1329
|
+
// the worker stopped itself (a timeout, a cancel, a shutdown): that 137 is the worker's own, the abort decides it
|
|
1330
|
+
// below, and a line naming the host's OOM killer would send the operator after a kill that never happened.
|
|
1331
|
+
if (oomReported && !oomKilled && !aborted) log("oom_report_below_limit", { jobId: job.id ?? null, memPeak: resources?.memPeak ?? null, memoryLimit });
|
|
1275
1332
|
// `exitAuth: "unverified"` is a run whose image signs its exit line and no signed line was found: the runner died
|
|
1276
1333
|
// before writing one, or a line was forged or taken off the pipe. Its tokens read as unknown and its dollars settle
|
|
1277
1334
|
// at the floor, the same as a container that wrote no exit line at all.
|
|
1278
|
-
log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}) });
|
|
1335
|
+
log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}), ...(oomKilled ? { oomKilled: true } : {}) });
|
|
1279
1336
|
|
|
1280
1337
|
// Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
|
|
1281
1338
|
// so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
|
|
@@ -1334,14 +1391,15 @@ export async function runJob(job, deps) {
|
|
|
1334
1391
|
// mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
|
|
1335
1392
|
// BEFORE the exit-code switch, where the same code would read as a free never-started exit.
|
|
1336
1393
|
if (detached === true) {
|
|
1337
|
-
throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
1394
|
+
throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
|
|
1338
1395
|
}
|
|
1339
1396
|
|
|
1340
1397
|
// A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
|
|
1341
1398
|
// operator's cancel, issue #287) kills the container -> exit 143/137. That is our decision, not an
|
|
1342
1399
|
// infra fault: it is POLICY and must NOT retry, or a wedged job re-runs into a second PR / double
|
|
1343
1400
|
// spend. Keyed on the abort FLAG, not the code -- an unbidden 137 (kernel OOM) carries
|
|
1344
|
-
// `aborted: false`, falls to the switch,
|
|
1401
|
+
// `aborted: false`, and falls to the OOM branch below when the runtime confirmed it, else to the switch, where it
|
|
1402
|
+
// stays infra-retryable (issue #596).
|
|
1345
1403
|
// WHO aborted is an exact-match on `abortReason` (the wiring maps it off `signal.reason`), and the
|
|
1346
1404
|
// match is deliberately closed: "job-timeout-30m", "shutdown", undefined and any future garbage all
|
|
1347
1405
|
// classify as worker-abort, so a pin bump that changes what rides the signal can widen nothing.
|
|
@@ -1352,7 +1410,23 @@ export async function runJob(job, deps) {
|
|
|
1352
1410
|
// Awaited bare like every determinate refusal above: the adapter never throws by contract, and
|
|
1353
1411
|
// the one swallowed comment in this file (the catch's) justifies itself by its position.
|
|
1354
1412
|
await comment(job, TERMINAL_COMMENTS[reason]);
|
|
1355
|
-
return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}) };
|
|
1413
|
+
return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
|
|
1414
|
+
}
|
|
1415
|
+
|
|
1416
|
+
// Issue #596: a runner the kernel killed for memory, CONFIRMED (`oomKilled` above). After the abort branch, so a
|
|
1417
|
+
// worker's own stop is never relabelled, and before the switch, where the same 137 is an unknown exit and retries.
|
|
1418
|
+
// The same size would be killed the same way on every retry, so it is POLICY: returned, never retried, its slot
|
|
1419
|
+
// kept and its dollars settled above like any other paid stop. An UNCONFIRMED 137 (an image without the supervisor,
|
|
1420
|
+
// an unsigned line, a SIGKILL with no OOM kill in the cgroup, a peak short of the limit, which is how a kill by the
|
|
1421
|
+
// host's OOM killer reads) falls through and retries, exactly as before.
|
|
1422
|
+
//
|
|
1423
|
+
// When only a CHILD was killed, the runner survives and ends on its own code, so this branch is not taken: the
|
|
1424
|
+
// outcome is the runner's, and `resources.oomKills` in the record says a process was killed for memory.
|
|
1425
|
+
if (oomKilled) {
|
|
1426
|
+
// The project and the size go to the log and the record, never to the comment (the TERMINAL_COMMENTS rule).
|
|
1427
|
+
log("oom_killed", { jobId: job.id ?? null });
|
|
1428
|
+
await comment(job, TERMINAL_COMMENTS[EXIT_OOM_KILLED]);
|
|
1429
|
+
return { outcome: "policy", reason: EXIT_OOM_KILLED, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
|
|
1356
1430
|
}
|
|
1357
1431
|
|
|
1358
1432
|
switch (code) {
|
|
@@ -1394,6 +1468,7 @@ export async function runJob(job, deps) {
|
|
|
1394
1468
|
// Only when the collector returned a plan (a file, or a confirmed portfolio job with none, `plan-absent`): the
|
|
1395
1469
|
// record's `plan` is null otherwise, and every other result is unchanged.
|
|
1396
1470
|
...(plan ? { plan } : {}),
|
|
1471
|
+
...(resources ? { resources } : {}),
|
|
1397
1472
|
};
|
|
1398
1473
|
}
|
|
1399
1474
|
case EXIT_POLICY: {
|
|
@@ -1420,7 +1495,7 @@ export async function runJob(job, deps) {
|
|
|
1420
1495
|
// `why`. parseExitWhy keeps only a member of the closed COST_CAP_WHYS off a `cost-cap` line that said code 2,
|
|
1421
1496
|
// and it rides only beside a `cost-cap` reason, so a forged line can at worst name the wrong rule of three.
|
|
1422
1497
|
const why = reason === "cost-cap" && COST_CAP_WHYS.includes(exitWhy) ? { why: exitWhy } : {};
|
|
1423
|
-
return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why };
|
|
1498
|
+
return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why, ...(resources ? { resources } : {}) };
|
|
1424
1499
|
}
|
|
1425
1500
|
case EXIT_INFRA:
|
|
1426
1501
|
// NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
|
|
@@ -1428,7 +1503,7 @@ export async function runJob(job, deps) {
|
|
|
1428
1503
|
// the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
|
|
1429
1504
|
// BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
|
|
1430
1505
|
// function never sees (issue #288).
|
|
1431
|
-
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
1506
|
+
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
|
|
1432
1507
|
default:
|
|
1433
1508
|
// THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
|
|
1434
1509
|
// (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
|
|
@@ -1444,9 +1519,9 @@ export async function runJob(job, deps) {
|
|
|
1444
1519
|
// and normalises to this outcome itself.
|
|
1445
1520
|
if (neverStarted) {
|
|
1446
1521
|
// No `dollars` here: the hold is still standing, and the catch refunds it whole with the job-count slots.
|
|
1447
|
-
throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
1522
|
+
throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), resources });
|
|
1448
1523
|
}
|
|
1449
|
-
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
1524
|
+
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
|
|
1450
1525
|
}
|
|
1451
1526
|
} catch (e) {
|
|
1452
1527
|
// A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
|
|
@@ -1625,7 +1700,7 @@ function isNeverStartedRetry(e) {
|
|
|
1625
1700
|
}
|
|
1626
1701
|
|
|
1627
1702
|
export class InfraRetry extends Error {
|
|
1628
|
-
constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars } = {}) {
|
|
1703
|
+
constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars, resources } = {}) {
|
|
1629
1704
|
super(message, cause ? { cause } : undefined);
|
|
1630
1705
|
this.name = "InfraRetry";
|
|
1631
1706
|
this.piDispatchRetry = true;
|
|
@@ -1647,6 +1722,8 @@ export class InfraRetry extends Error {
|
|
|
1647
1722
|
// Issue #501: the dollar reservation's outcome (`dollarsRecord`), or null when no dollar window applied. Set by
|
|
1648
1723
|
// the processor on a throw after the reservation, so a retried attempt's record says what its window was charged.
|
|
1649
1724
|
this.dollars = dollars ?? null;
|
|
1725
|
+
// Issue #596: what the container used, off its exit line, or null; set on a throw after a container ran.
|
|
1726
|
+
this.resources = resources ?? null;
|
|
1650
1727
|
}
|
|
1651
1728
|
}
|
|
1652
1729
|
|