@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,7 +28,9 @@
28
28
  */
29
29
 
30
30
  import { READ_BACK_BY_A_LIVE_PROBE } from "./backend-conformance.mjs";
31
- import { containerSpec } from "./container-spec.mjs";
31
+ import { containerSpec, memoryBytes } from "./container-spec.mjs";
32
+ import { CGROUP_PARENT, DEFAULT_JOB_SIZE } from "./job-size.mjs";
33
+ import { cpuMaxLine, parseCpuMaxRead } from "./cpu-reserve.mjs";
32
34
  import { ISOLATION_FLAGS, buildDockerRunArgs } from "./docker-run.mjs";
33
35
  import { DEFAULT_EGRESS_PROXY, EGRESS_PROXY_PORT, createJobNetworkWith, networkEndpoints, networkNameFor, removeNetworkOrSay } from "./egress.mjs";
34
36
  import { detachBlockedSentence, makeDetachGate } from "./netns-keeper.mjs";
@@ -112,8 +114,14 @@ export function liveFixture(root) {
112
114
  // `relabel` (issue #355) is a job's too: where a job's own mounts carry `:Z`, so do the probe's, because a probe mounted
113
115
  // the way no job is would read back a container no job gets. The fixture is doctor's own directory, so its workspace
114
116
  // is relabelled like a forge job's clone (`workspaceOwned`); its global overlay directory never is, exactly as a job's.
115
- function probeOptions({ image, name, fixture, user = null, relabel = false }) {
116
- return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, ...fixture };
117
+ //
118
+ // `size` and `hostCpus` (issue #596) are a job's too: the deployment's default size and the `--cpus` ceiling of this
119
+ // runtime's own CPU count, so the probe is bounded exactly as a job without a project size is, and reads that back.
120
+ //
121
+ // `cgroupParent` (issue #596, phase 2) is a job's too: the jobs' parent cgroup, or null where this venue runs jobs without
122
+ // it (`cgroupParentFor`), so the probe sits where a job sits and `cgroupParentVerdict` reads that back.
123
+ function probeOptions({ image, name, fixture, user = null, relabel = false, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
124
+ return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, size, hostCpus, cgroupParent, ...fixture };
117
125
  }
118
126
 
119
127
  // `buildArgs` (issue #354) is the RUNTIME's job builder, never a second one: each probe below is built by whichever
@@ -121,32 +129,32 @@ function probeOptions({ image, name, fixture, user = null, relabel = false }) {
121
129
  // The default is docker's, so every argv here is byte-for-byte what it was.
122
130
 
123
131
  /** The probe container's argv: the job builder's, detached, with `sleep <derived seconds>` as its whole program. */
124
- export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
125
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
132
+ export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
133
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
126
134
  }
127
135
 
128
136
  /**
129
137
  * The pinning probe's argv: the job builder's, detached, against an image this host does not have. Detached so a
130
138
  * container that WAS created prints the ID it is removed by; with `--pull=never` in the builder none should be.
131
139
  */
132
- export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
133
- return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel }), extraFlags: ["-d"] });
140
+ export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
141
+ return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d"] });
134
142
  }
135
143
 
136
144
  /**
137
145
  * One ephemeral run's argv (issue #344): the job builder's, detached, running EPHEMERAL_SCRIPT with the nonce and the
138
146
  * run's number. The same NAME both times, because "a job id run twice" is the question.
139
147
  */
140
- export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
141
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
148
+ export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
149
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
142
150
  }
143
151
 
144
152
  /**
145
153
  * One peer's argv (issue #344): the job builder's, detached, on its OWN job network, running PEER_SCRIPT, which answers
146
154
  * every connection with the nonce for `seconds`. Built with `network` set exactly as a job with egress armed is.
147
155
  */
148
- export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
149
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
156
+ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
157
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
150
158
  }
151
159
 
152
160
  /**
@@ -156,6 +164,9 @@ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = li
156
164
  export const STATUS_SCRIPT = [
157
165
  "cat /proc/1/status",
158
166
  'if [ -f /sys/fs/cgroup/cgroup.controllers ]; then echo "cgroup:v2"; echo "pids.max:$(cat /sys/fs/cgroup/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory.max 2>/dev/null)";',
167
+ // Issue #596: the swap bound (0 when --memory-swap equals --memory), the CPU weight --cpu-shares became, and the
168
+ // --cpus ceiling. cgroup v2 only: v1 spells all three differently, and every venue measured is v2.
169
+ 'echo "memory.swap.max:$(cat /sys/fs/cgroup/memory.swap.max 2>/dev/null)"; echo "cpu.weight:$(cat /sys/fs/cgroup/cpu.weight 2>/dev/null)"; echo "cpu.max:$(cat /sys/fs/cgroup/cpu.max 2>/dev/null)";',
159
170
  'else echo "cgroup:v1"; echo "pids.max:$(cat /sys/fs/cgroup/pids/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null)"; fi',
160
171
  ].join("\n");
161
172
 
@@ -246,6 +257,9 @@ export function parseStatus(output) {
246
257
  cgroup: field("cgroup"),
247
258
  pidsMax: field("pids.max"),
248
259
  memoryMax: field("memory.max"),
260
+ swapMax: field("memory.swap.max"),
261
+ cpuWeight: field("cpu.weight"),
262
+ cpuMax: field("cpu.max"),
249
263
  };
250
264
  }
251
265
 
@@ -257,9 +271,34 @@ export function expectedPidsLimit(flags = ISOLATION_FLAGS) {
257
271
 
258
272
  /** The memory bound in bytes, from the spec's own default (`4g`), never a literal. */
259
273
  export function expectedMemoryBytes(memory = containerSpec({ image: "i", name: "n", workspace: "/w" }).memory) {
260
- const m = /^(\d+)([kmg]?)$/i.exec(String(memory));
261
- if (!m) return null;
262
- return Number(m[1]) * { "": 1, k: 1024, m: 1024 ** 2, g: 1024 ** 3 }[m[2].toLowerCase()];
274
+ return memoryBytes(memory);
275
+ }
276
+
277
+ /**
278
+ * What a container of this size reads back (issue #596), from the SAME spec builder the probe's argv came from:
279
+ * `{ pidsLimit, memoryBytes, cpuShares, cpuMax }`, `cpuMax` being the `cpu.max` line `--cpus` writes (`<quota> 100000`,
280
+ * or `max 100000` with no ceiling).
281
+ */
282
+ export function expectedBounds(size = DEFAULT_JOB_SIZE, hostCpus = null, cpuBudgetCenti = null, cgroupParent = CGROUP_PARENT) {
283
+ const spec = containerSpec({ image: "i", name: "n", workspace: "/w", size, hostCpus, cpuBudgetCenti, cgroupParent });
284
+ // ROUNDED (issue #596, gate round 3 of phase 1): under a CPU budget the ceiling may be fractional (`3.3`), and
285
+ // `3.3 * 100000` is 329999.99999999994 in floating point, while the runtime writes the integer quota 330000.
286
+ return { pidsLimit: expectedPidsLimit(), memoryBytes: memoryBytes(spec.memory), cpuShares: spec.cpuShares, cpuMax: spec.cpus === null ? "max 100000" : `${Math.round(Number(spec.cpus) * 100000)} 100000` };
287
+ }
288
+
289
+ /**
290
+ * The `cpu.weight` a `--cpu-shares` value becomes, under the two mappings measured (issue #596): runc 1.1.13 and crun
291
+ * 1.14 map linearly (`1 + ((shares - 2) * 9999) / 262142`, integer division: 1024 to 39, 2048 to 79), runc 1.5.1 and
292
+ * crun 1.27 map so 1024 is 100 (`10^((l^2 + 125l) / 612 - 7/34)` rounded up, l = log2(shares): 2048 to 174). Both
293
+ * reproduce every value of the lab's sweep. A weight that is neither is a runtime nobody measured, said as such.
294
+ */
295
+ export function cpuWeightsFor(shares) {
296
+ if (!Number.isSafeInteger(shares) || shares < 2) return [];
297
+ if (shares >= 262144) return [10000];
298
+ const linear = 1 + Math.floor(((shares - 2) * 9999) / 262142);
299
+ const l = Math.log2(shares);
300
+ const curved = Math.ceil(10 ** ((l * l + 125 * l) / 612 - 7 / 34));
301
+ return [...new Set([linear, curved])];
263
302
  }
264
303
 
265
304
  const verdict = (property, ok, detail, extra = {}) => ({ property, ok, detail, ...extra });
@@ -273,7 +312,7 @@ const notReadBack = (property, why) => verdict(property, false, `not read back:
273
312
  * FAILURE (the bound was not applied, as on rootless docker without delegation); a bound that could not be read at
274
313
  * all is "not read back", never a pass.
275
314
  */
276
- export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes() } = {}) {
315
+ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes(), cpuShares = null, cpuMax = null } = {}) {
277
316
  if (!status || status.capBnd === null || status.noNewPrivs === null) return notReadBack("isolation", "PID 1's status did not show CapBnd and NoNewPrivs");
278
317
  const failures = [];
279
318
  if (!/^0+$/.test(status.capBnd)) failures.push(`CapBnd ${status.capBnd} (the capability bounding set is not empty)`);
@@ -287,10 +326,30 @@ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memo
287
326
  if (Number(got) !== want) failures.push(`${name} ${got}, expected ${want}`);
288
327
  return null;
289
328
  };
290
- const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)].filter(Boolean);
329
+ const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)];
330
+ // Issue #596, read only by a caller that passes the size's bounds (`expectedBounds`). The swap bound is part of the
331
+ // memory bound: `--memory-swap` equal to `--memory` makes it 0, and anything else lets a job swap past its size.
332
+ // `cpu.max` is the host ceiling (`--cpus`), which keeps the host's reserve, so a different one is a failure too.
333
+ // `cpu.weight` is the job's share, whose number depends on the runtime's version: one neither measured mapping
334
+ // gives is said, not failed, because the ORDER between jobs is what it carries and an unknown mapping may keep it.
335
+ let cpuNote = "";
336
+ if (cpuMax !== null) {
337
+ const swap = status.swapMax;
338
+ if (swap === null || swap === undefined || swap === "") unread.push("memory.swap.max not readable");
339
+ else if (swap !== "0") failures.push(`memory.swap.max ${swap}, expected 0 (a job may swap beyond its memory)`);
340
+ const max = status.cpuMax;
341
+ if (max === null || max === undefined || max === "") unread.push("cpu.max not readable");
342
+ else if (max !== cpuMax) failures.push(`cpu.max ${max}, expected ${cpuMax}`);
343
+ const weight = status.cpuWeight;
344
+ const known = cpuWeightsFor(cpuShares);
345
+ if (weight === null || weight === undefined || weight === "") unread.push("cpu.weight not readable");
346
+ else if (!known.includes(Number(weight))) unread.push(`cpu.weight ${weight} is neither mapping measured for --cpu-shares=${cpuShares} (${known.join(" or ")})`);
347
+ else cpuNote = `, memory.swap.max 0, cpu.max ${cpuMax}, cpu.weight ${weight} (--cpu-shares=${cpuShares})`;
348
+ }
349
+ const missing = unread.filter(Boolean);
291
350
  if (failures.length > 0) return verdict("isolation", false, failures.join("; "));
292
- if (unread.length > 0) return notReadBack("isolation", `${unread.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
293
- return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}`);
351
+ if (missing.length > 0) return notReadBack("isolation", `${missing.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
352
+ return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}${cpuNote}`);
294
353
  }
295
354
 
296
355
  /** NON-ROOT: all four Uid fields (real, effective, saved, filesystem) nonzero, on the process the job would be. */
@@ -612,6 +671,60 @@ export function peerTargetsOf(inspectOutput, { network, name }) {
612
671
  return { targets: [...[...hosts].map((h) => `${h}:${PEER_PORT}`), ...addresses], addresses };
613
672
  }
614
673
 
674
+ /**
675
+ * WHERE THE RUNTIME PUT A JOB-BUILT CONTAINER (issue #596, phase 2, INT-LIVE-PROBE-CONTRACT): `{ ok, warn?, detail,
676
+ * placed, quota }`, from `<bin> inspect --format={{.HostConfig.CgroupParent}}|{{.State.Pid}}` and this host's own files.
677
+ *
678
+ * Three readings, each as far as this host can see. The runtime's RECORD of the parent must be the one the argv named
679
+ * (`expected`; none where the venue runs jobs without it). The process's own cgroup (`/proc/<pid>/cgroup`, cgroup v2's
680
+ * `0::<path>`) must lie under `/<parent>/`, which proves the placement rather than the flag: readable for a local Linux
681
+ * daemon and rootless Podman, not for a daemon in a VM (Docker Desktop), where the record alone is said as such. The
682
+ * container cannot see any of this itself (a private cgroup namespace shows it `0::/`, measured). Then the parent's
683
+ * `cpu.max`, where readable, against the CPU budget (`cpuBudgetCenti`: an integer, `Infinity` for off, null not
684
+ * checked): a quota that is not the budget is a WARNING, "no host CPU reserve across jobs", never a failure of the
685
+ * placement, because on a systemd host only the operator can set it.
686
+ */
687
+ export function cgroupParentVerdict({ inspected, expected = CGROUP_PARENT, cpuBudgetCenti = null, readFile = () => {
688
+ throw new Error("no reader");
689
+ }, bin = "docker" }) {
690
+ const out = (ok, detail, extra = {}) => ({ ok, detail, placed: null, quota: undefined, ...extra });
691
+ if (inspected?.code !== 0) return out(false, `not read back: ${bin} inspect did not answer`, { warn: true });
692
+ const [recorded = "", pidText = ""] = String(inspected.stdout ?? "").trim().split("|");
693
+ if (expected === null) {
694
+ if (recorded.includes(CGROUP_PARENT)) return out(false, `the runtime recorded the parent ${recorded} for a venue that runs jobs without one`);
695
+ return out(true, `this venue runs jobs without the ${CGROUP_PARENT} parent (Podman's cgroup manager is not systemd), so their CPU weight is capped at 1024: the egress proxy and Valkey get a fair share of the CPU, not a reserve`, { warn: true });
696
+ }
697
+ // The bare name, or a path ending in it (a runtime may record where the slice resolved); the process's own cgroup
698
+ // below is the proof either way.
699
+ if (recorded.replace(/^\//, "") !== expected && !recorded.endsWith(`/${expected}`)) return out(false, `the runtime recorded the parent ${JSON.stringify(recorded)}, not ${expected}, so this container is outside the jobs' CPU reserve`);
700
+ let cgroupText = null;
701
+ if (/^[1-9][0-9]{0,9}$/.test(pidText)) {
702
+ try {
703
+ cgroupText = readFile(`/proc/${pidText}/cgroup`);
704
+ } catch {
705
+ cgroupText = null;
706
+ }
707
+ }
708
+ const path = cgroupText === null ? null : (/^0::(\/\S*)$/m.exec(String(cgroupText))?.[1] ?? null);
709
+ if (path === null) return out(true, `the runtime recorded the parent ${expected}; the container's own cgroup is not readable from this host (its process runs in the runtime's VM or another namespace), so the placement and the parent's quota were not read here`, { warn: true });
710
+ const at = path.indexOf(`/${expected}/`);
711
+ if (at === -1) return out(false, `the runtime recorded the parent ${expected}, but the container's cgroup is ${path}, not under it`);
712
+ const parentDir = `/sys/fs/cgroup${path.slice(0, at + expected.length + 1)}`;
713
+ let quota;
714
+ try {
715
+ quota = parseCpuMaxRead(readFile(`${parentDir}/cpu.max`));
716
+ } catch {
717
+ quota = null;
718
+ }
719
+ const placed = path.slice(0, at + expected.length + 1);
720
+ if (quota === null) return out(true, `the container's cgroup is under ${placed}; the parent's cpu.max is not readable here`, { warn: true, placed });
721
+ const shown = quota.cpuCenti === null ? "no quota" : `a quota of ${quota.cpuCenti / 100} CPUs (${cpuMaxLine(quota.cpuCenti)})`;
722
+ if (cpuBudgetCenti === null || cpuBudgetCenti === undefined) return out(true, `the container's cgroup is under ${placed}, which has ${shown}`, { placed, quota: quota.cpuCenti });
723
+ const want = cpuBudgetCenti === Infinity ? null : cpuBudgetCenti;
724
+ if (quota.cpuCenti === want) return out(true, `the container's cgroup is under ${placed}, which has ${want === null ? "no quota, as the CPU budget is off" : `${shown}, the host's CPU budget, so all jobs together keep the reserve`}`, { placed, quota: quota.cpuCenti });
725
+ return out(true, `the container's cgroup is under ${placed}, which has ${shown}, not ${want === null ? "none (the CPU budget is off)" : `the CPU budget of ${want / 100}`}: no host CPU reserve across jobs`, { warn: true, placed, quota: quota.cpuCenti });
726
+ }
727
+
615
728
  /**
616
729
  * The whole sequence. Returns `{ ran, reason, verdicts, notes, swept, ranAs }`: `ran` false with a `reason` when nothing
617
730
  * was read back; the verdicts in READ_BACK_BY_A_LIVE_PROBE's order otherwise; `notes` for a teardown that failed, on
@@ -660,6 +773,14 @@ export async function runLiveProbes({
660
773
  // Issue #452, gate round 3: the runtime as this pass already read it (`{ podman, rootless, version } | null`), for the
661
774
  // detach gate, so doctor asks the daemon once per run. Absent, the gate reads it itself.
662
775
  readRuntime = null,
776
+ // Issue #596: the size every probe container is built at (a job's default) and the runtime's CPU count for the
777
+ // `--cpus` ceiling; the isolation verdict reads both back.
778
+ size = DEFAULT_JOB_SIZE,
779
+ hostCpus = null,
780
+ // Issue #596, phase 2: the parent cgroup a job on this venue runs under (null: none, `cgroupParentFor`) and the CPU
781
+ // budget its quota should be (an integer in hundredths, `Infinity` for off, null or absent: not checked).
782
+ cgroupParent = CGROUP_PARENT,
783
+ cpuBudgetCenti = null,
663
784
  }) {
664
785
  const notRun = (reason) => ({ ran: false, reason, verdicts: [], notes: [], swept: [] });
665
786
  const notLocal = notRun(`this shell's ${bin} CLI is not observed to point at this host, so bind paths and .Mounts would describe another machine; nothing was run`);
@@ -764,7 +885,7 @@ export async function runLiveProbes({
764
885
  }
765
886
 
766
887
  // --- the reading container: mounts, status, writes ---
767
- const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
888
+ const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
768
889
  const probeId = reading.entry.id;
769
890
  if (reading.result?.code !== 0 || probeId === null) {
770
891
  // The runtime's own words, when it printed any (issue #453, gate round 1): "did not start" alone left the operator
@@ -773,7 +894,7 @@ export async function runLiveProbes({
773
894
  return { ran: false, reason: `the probe container did not start${said ? ` (${bin} said: ${said})` : ""}, so nothing was read back`, verdicts: [], notes, swept };
774
895
  }
775
896
 
776
- const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel })).mounts;
897
+ const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel, size, hostCpus, cgroupParent })).mounts;
777
898
  const inspected = await step(["inspect", "--format={{json .Mounts}}", probeId]);
778
899
  // Issue #345: the mount table as the container itself sees it, by a constant `cat`, for what `.Mounts` does not list.
779
900
  const mountinfo = await step(["exec", probeId, "cat", "/proc/self/mountinfo"]);
@@ -781,8 +902,12 @@ export async function runLiveProbes({
781
902
 
782
903
  const statusRun = await step(["exec", probeId, "sh", "-c", STATUS_SCRIPT]);
783
904
  const status = statusRun?.code === 0 ? parseStatus(statusRun.stdout) : null;
784
- const isolation = status ? isolationVerdict(status) : notReadBack("isolation", "the status probe did not run in the container");
905
+ const isolation = status ? isolationVerdict(status, expectedBounds(size, hostCpus, null, cgroupParent)) : notReadBack("isolation", "the status probe did not run in the container");
785
906
  const nonRoot = status ? nonRootVerdict(status) : notReadBack("nonRoot", "the status probe did not run in the container");
907
+ // Issue #596, phase 2: where the runtime put the container (its record, and the process's own cgroup where this host
908
+ // can read it) and, where readable, the parent's quota. Its own line, not one of the declared properties.
909
+ const placed = await step(["inspect", "--format={{.HostConfig.CgroupParent}}|{{.State.Pid}}", probeId]);
910
+ const parentRead = cgroupParentVerdict({ inspected: placed, expected: cgroupParent, cpuBudgetCenti, readFile: (path) => fs.readFileSync(path, "utf8"), bin });
786
911
 
787
912
  const written = await step(["exec", probeId, "sh", "-c", WRITE_SCRIPT, "sh", nonce]);
788
913
  const readBack = (dir) => {
@@ -807,7 +932,7 @@ export async function runLiveProbes({
807
932
  await release(reading.entry);
808
933
 
809
934
  // --- the pinning container: an image this host does not have ---
810
- const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs }));
935
+ const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
811
936
  const after = await step(["image", "inspect", absentImageRef(nonce)]);
812
937
  const stillAbsent = after?.code === 0 ? false : typeof after?.code === "number" ? true : null;
813
938
  const imagePinning = imagePinningVerdict({ code: pinning.result?.code, output: `${pinning.result?.stdout ?? ""}${pinning.result?.stderr ?? ""}`, stillAbsent, bin });
@@ -815,7 +940,7 @@ export async function runLiveProbes({
815
940
 
816
941
  // --- the ephemeral pair (issue #344): one name, two runs, each waited on until it is gone ---
817
942
  const runEphemeral = async (n) => {
818
- const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs }));
943
+ const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
819
944
  const started = result?.code === 0 && entry.id !== null;
820
945
  // A HELD NAME is the daemon refusing the create for the name, in its own words (measured: Docker "Conflict. ...
821
946
  // is already in use", Podman "that name is already in use"). Not a listed container: after the first run was
@@ -864,7 +989,7 @@ export async function runLiveProbes({
864
989
  } catch {
865
990
  break;
866
991
  }
867
- const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
992
+ const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
868
993
  peers.push(entry);
869
994
  if (result?.code !== 0 || entry.id === null) break;
870
995
  ids[key] = entry.id;
@@ -901,7 +1026,7 @@ export async function runLiveProbes({
901
1026
  const byProperty = { isolation, ephemeral, mountSet, egress: egressVerdict(egress ?? {}), jobToJobIsolation, imagePinning, nonRoot, localFolders };
902
1027
  // The uid PID 1 actually ran as, for doctor's job-user line: the decision it was given, read back.
903
1028
  const ranAs = Array.isArray(status?.uids) && /^\d+$/.test(status.uids[1] ?? "") ? Number(status.uids[1]) : null;
904
- return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs };
1029
+ return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs, cgroupParent: parentRead };
905
1030
  } finally {
906
1031
  for (const entry of owned) await release(entry);
907
1032
  for (const entry of networks) await dropNetwork(entry);
package/src/prepare.mjs CHANGED
@@ -14,6 +14,7 @@ import { buildForgejoPrompt } from "./forgejo-prompt.mjs";
14
14
  import { buildAzurePrompt } from "./azure-prompt.mjs";
15
15
  import { prepareLocalWorkspace } from "./prepare-local.mjs";
16
16
  import { copySkillTree } from "./copy-tree.mjs";
17
+ import { recordedJobSize } from "./job-size.mjs";
17
18
 
18
19
  /**
19
20
  * The subdirectory of the per-job dir a trigger's injected skills are copied into, so they reach the
@@ -82,7 +83,7 @@ export function makePrepareWorkspace({
82
83
  removeDir = (dir) => rmSync(dir, { recursive: true, force: true }),
83
84
  }) {
84
85
  ensureDir(jobsDir);
85
- return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false } = {}) {
86
+ return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false, size = null } = {}) {
86
87
  ensureDir(jobsDir);
87
88
  const jobDir = mkdtempSync(join(jobsDir, "job-"));
88
89
  // Issue #524: a THROW out of anything below leaves `jobDir` to nobody. The processor tears down only what
@@ -92,7 +93,7 @@ export function makePrepareWorkspace({
92
93
  // covers the thrown ones, in one place for every kind, rather than in each preparer that can throw.
93
94
  // A retry makes a fresh directory, so nothing is lost by removing this one.
94
95
  try {
95
- return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio });
96
+ return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size });
96
97
  } catch (error) {
97
98
  // GUARDED: a removal that fails (a busy mount, a permission flipped mid-job) must never replace the error
98
99
  // that is the job's actual outcome. The directory is then left, which is what happened before this catch.
@@ -103,7 +104,7 @@ export function makePrepareWorkspace({
103
104
  }
104
105
  };
105
106
 
106
- async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio }) {
107
+ async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size }) {
107
108
  // The trigger's injected skills (REQ-PER-TRIGGER-SKILLS, issue #60), COPIED here rather than
108
109
  // mounted, and copied ONCE for every job kind because this is where local and forge converge.
109
110
  //
@@ -137,6 +138,9 @@ export function makePrepareWorkspace({
137
138
  ...(jobUser ? { jobUser: { user: jobUser.user ?? null, home: jobUser.home ?? null } } : {}),
138
139
  // Issue #429: the podman store the run's container lived in, for a podman run only.
139
140
  ...(typeof podmanStore === "string" && podmanStore !== "" ? { podmanStore } : {}),
141
+ // Issue #596: the size the run's container had, so a re-opened sandbox gets the same memory and CPU weight.
142
+ // Rebuilt (`recordedJobSize`), and only when the processor resolved one; a direct call stamps nothing.
143
+ ...(recordedJobSize(size) ? { size: recordedJobSize(size) } : {}),
140
144
  };
141
145
  if (job.kind === "local") {
142
146
  // Harness text above, operator DATA below: the fixed pointer line names /job/event.json so a
package/src/processor.mjs CHANGED
@@ -8,7 +8,7 @@ import { PROJECT_CAP_REASON } from "./scoped-limits.mjs";
8
8
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
9
9
  import { RESERVED_ENV_NAMES } from "./triggers.mjs";
10
10
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
11
- import { COST_CAP_WHYS, RUNNER_POLICY_REASONS } from "./run-history.mjs";
11
+ import { COST_CAP_WHYS, EXIT_OOM_KILLED, RUNNER_POLICY_REASONS } from "./run-history.mjs";
12
12
  import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
13
13
  import { CAPABILITY_GATES, EXIT_AUTH_CAPABILITY } from "./image-preflight.mjs";
14
14
  import { modelListProblem, modelOnList, splitModelEntry } from "./model-ref.mjs";
@@ -83,6 +83,29 @@ export const OBSERVATION_COMMENT_UNNAMED = "the venue this job runs on did not c
83
83
  * into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
84
84
  */
85
85
 
86
+ /** The exit code of a container whose main process was SIGKILLed: a worker's stop or the kernel's OOM killer. */
87
+ const EXIT_SIGKILL = 137;
88
+
89
+ /**
90
+ * Issue #596: whether a run's memory peak reached the container's own limit closely enough for its OOM kill to be the
91
+ * job's own: `memPeak` (bytes, off the decisive exit line) at 90% of `limit` (bytes, the `--memory=` the worker passed)
92
+ * or more. Both must be known; anything else is false, so the 137 stays infrastructure and retries.
93
+ *
94
+ * Why the check exists: `memory.events` `oom_kill` counts a kill by ANY OOM killer, the HOST's included, and the runner
95
+ * tree's `oom_score_adj` of 1000 makes a job the host's first victim when the machine itself runs short. A kill of
96
+ * that kind says nothing about the job's size, and a retry may well pass. A cgroup OOM happens only at the limit, so its
97
+ * peak sits there: measured in the issue #596 lab, `memory.peak` read exactly the bound (64m, 80m) or just under it on
98
+ * every venue. 90% and not 100% is the margin for that "just under"; nothing in between is read as a measurement.
99
+ *
100
+ * What it cannot see (the residual): a host OOM that strikes a job which is ALREADY at 90% of its own limit, page
101
+ * cache included (a job that read large files counts that cache), reads as the job's own OOM. And the peak is a
102
+ * high-water mark over the whole run, so a job that touched 90% early and was killed by the host later reads the same.
103
+ */
104
+ export function peakReachedLimit(memPeak, limit) {
105
+ if (!Number.isSafeInteger(memPeak) || !Number.isSafeInteger(limit) || memPeak < 0 || limit <= 0) return false;
106
+ return memPeak >= Math.ceil((limit * 9) / 10);
107
+ }
108
+
86
109
  // The post-spend terminal comments (issue #288). Every FREE refusal above the container already comments;
87
110
  // these are the paths where money was spent and the run still ended without the agent's own status step,
88
111
  // which used to tell the issue nothing (REQ-JOB-STATUS-COMMENTS' acceptance -- "exactly one completion or
@@ -106,6 +129,9 @@ export const TERMINAL_COMMENTS = {
106
129
  "model-not-allowed": "Stopped: the run tried to call an AI model this trigger does not allow, or to change an AI request in a way it does not allow, so the call was not made. Partial work may exist. Not retried.",
107
130
  "cost-cap-unenforceable": "Stopped: this run has a cost limit, and the job image could not enforce it before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
108
131
  "model-policy-unenforceable": "Stopped: this run is limited to certain AI models, and the job image could not enforce that before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
132
+ // Issue #596. Never the size or the project: both are operator configuration, and the reader may be an issue author
133
+ // who can act on neither. The worker log and the run record carry them.
134
+ [EXIT_OOM_KILLED]: "Stopped: the job's container ran out of memory and was stopped. Partial work may exist. Not retried, because the same size would stop the same way. The operator can raise this job's memory size.",
109
135
  };
110
136
 
111
137
  // Issue #502: the `model-unknown` refusal's comment. Names no model: the reader may be an issue author.
@@ -238,7 +264,8 @@ export async function runJob(job, deps) {
238
264
  // Issue #341: which uid this job's container runs as. `(job, { capabilities, observed }) =>` `{ user, home }`
239
265
  // (`user` null = the image's own USER), `{ refused, cause }` or `{ unavailable, reason }`. The default runs
240
266
  // every job as the image's user, exactly as before, so a wiring that omits it changes nothing. A non-refused answer
241
- // may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with.
267
+ // may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with, and
268
+ // `hostCpus` (issue #596), the runtime's CPU count from the same facts read, which sets the `--cpus` ceiling.
242
269
  jobUserPreflight = async () => ({ user: null, home: null }),
243
270
  // (session, { piVersion, context }) => { promoted, reason, bytes }. Promotes this job's transcript back into
244
271
  // the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
@@ -300,8 +327,11 @@ export async function runJob(job, deps) {
300
327
  mintToken,
301
328
  isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
302
329
  prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
303
- // runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
330
+ // runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints?, size?, hostCpus?, unenforced? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
331
+ // `size` (issue #596) is `jobSize` below; `hostCpus` the runtime's CPU count off the job-user gate's own facts read.
332
+ // `unenforced` (issue #596) is the size flags that same read says the runtime drops, passed only when there are some.
304
333
  // `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
334
+ // `resources` and `exitOomKilled` (issue #596) are the exit line's cgroup block and the supervisor's OOM report.
305
335
  // `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
306
336
  // `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
307
337
  // `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
@@ -356,6 +386,14 @@ export async function runJob(job, deps) {
356
386
  // after prepare, before any reserve. Absent on a bare wiring, which checks nothing.
357
387
  pickupProject = null,
358
388
  folderProject = null,
389
+ // Issue #596: the job's size `{ memMiB, cpuCenti, source }`, resolved at pickup from the same limits snapshot as
390
+ // every other gate (index.mjs). Handed to `prepareWorkspace` (a retained run's manifest records it, so a sandbox
391
+ // reopens the run at its size) and to `runContainer`, never through `job.data`. null on a bare wiring: neither call
392
+ // then carries it, and the container gets the built-in 4g and 2.
393
+ jobSize = null,
394
+ // Issue #596, phase 2: the host's CPU budget at pickup in hundredths (`host-budget.mjs`), or null when it is off or
395
+ // unknown. It becomes every job's `--cpus` (capped at the runtime's count), so no job can use the reserve.
396
+ cpuBudgetCenti = null,
359
397
  // Issue #503 part 7: the builtin catalog's model object for (provider, id), or null (model-catalog.mjs
360
398
  // `builtinModel`). Read only by the zero-rated check; the default knows no builtin model, so an unwired
361
399
  // processor judges overlay models alone and reserves for every other.
@@ -412,6 +450,8 @@ export async function runJob(job, deps) {
412
450
  // the record says about it (INT-RUN-HISTORY-FILE-CONTRACT), null until the reservation step ran.
413
451
  let dollarHold = null;
414
452
  let dollars = null;
453
+ // Issue #596: what the container used (`resources` off its exit line), null until a container ran and reported it.
454
+ let resources = null;
415
455
 
416
456
  try {
417
457
  // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
@@ -1067,7 +1107,7 @@ export async function runJob(job, deps) {
1067
1107
  // `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
1068
1108
  // `portfolio` (issue #505) only for a job the gate above confirmed: prepare then writes /job/portfolio.json, after
1069
1109
  // asking the live file once more. Absent otherwise, so every other job's prepare call is unchanged.
1070
- prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
1110
+ prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}), ...(jobSize ? { size: jobSize } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
1071
1111
 
1072
1112
  // A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
1073
1113
  // or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
@@ -1270,12 +1310,29 @@ export async function runJob(job, deps) {
1270
1310
  // then only a signed line is read (run-container.mjs, run-history.mjs `authenticExitLines`). An image that does not
1271
1311
  // declare it is read as before, under the #542 trust rule below alone.
1272
1312
  const exitAuth = (img.capabilities ?? []).includes(EXIT_AUTH_CAPABILITY);
1273
- const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}) });
1313
+ const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null, exitOomKilled = false, memoryLimit = null, resources: ranResources = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}), ...(jobSize ? { size: jobSize } : {}), ...(Number.isSafeInteger(jobUser?.hostCpus) ? { hostCpus: jobUser.hostCpus } : {}), ...(Number.isSafeInteger(cpuBudgetCenti) ? { cpuBudgetCenti } : {}), ...(Array.isArray(jobUser?.unenforced) && jobUser.unenforced.length > 0 ? { unenforced: jobUser.unenforced } : {}) });
1274
1314
  containerRan = true;
1315
+ // Issue #596: what the container used, off its exit line, rebuilt by the sink (null from a runContainer that predates
1316
+ // the field). Every result and every throw below carries it, so a retried attempt's record says what it used too.
1317
+ resources = ranResources ?? null;
1318
+ // Issue #596: CONFIRMED killed for memory. The image's supervisor (image/runner/src/supervise.mjs) outlives the runner,
1319
+ // whose process tree it gives the highest OOM score, and when the runner dies of SIGKILL with the cgroup's
1320
+ // `oom_kill` above 0 it writes the signed line `code: 137, reason: "oom-killed"` (parseExitOomKilled). All three
1321
+ // facts must agree: that line, verified under this run's key (an unsigned line is a tool's), and the container's
1322
+ // own exit 137. Docker's `oom` event is not used: it fires also when only a child was killed and the job went on
1323
+ // to exit 0, and Podman has no such event at all (both measured in the issue #596 lab). And a fourth: the line's
1324
+ // `memPeak` at 90% of the `--memory` this container got (`memoryLimit`, from runContainer) or more, because
1325
+ // `oom_kill` counts the HOST's OOM killer too (`peakReachedLimit`). A report below it stays infrastructure.
1326
+ const oomReported = exitOomKilled === true && exitAuthResult === "verified" && code === EXIT_SIGKILL;
1327
+ const oomKilled = oomReported && peakReachedLimit(resources?.memPeak ?? null, memoryLimit);
1328
+ // Numbers only (bytes), never a path or a project: the operator's trace of a kill read as the host's. Never on a run
1329
+ // the worker stopped itself (a timeout, a cancel, a shutdown): that 137 is the worker's own, the abort decides it
1330
+ // below, and a line naming the host's OOM killer would send the operator after a kill that never happened.
1331
+ if (oomReported && !oomKilled && !aborted) log("oom_report_below_limit", { jobId: job.id ?? null, memPeak: resources?.memPeak ?? null, memoryLimit });
1275
1332
  // `exitAuth: "unverified"` is a run whose image signs its exit line and no signed line was found: the runner died
1276
1333
  // before writing one, or a line was forged or taken off the pipe. Its tokens read as unknown and its dollars settle
1277
1334
  // at the floor, the same as a container that wrote no exit line at all.
1278
- log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}) });
1335
+ log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}), ...(oomKilled ? { oomKilled: true } : {}) });
1279
1336
 
1280
1337
  // Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
1281
1338
  // so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
@@ -1334,14 +1391,15 @@ export async function runJob(job, deps) {
1334
1391
  // mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
1335
1392
  // BEFORE the exit-code switch, where the same code would read as a free never-started exit.
1336
1393
  if (detached === true) {
1337
- throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1394
+ throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1338
1395
  }
1339
1396
 
1340
1397
  // A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
1341
1398
  // operator's cancel, issue #287) kills the container -> exit 143/137. That is our decision, not an
1342
1399
  // infra fault: it is POLICY and must NOT retry, or a wedged job re-runs into a second PR / double
1343
1400
  // spend. Keyed on the abort FLAG, not the code -- an unbidden 137 (kernel OOM) carries
1344
- // `aborted: false`, falls to the switch, and stays infra-retryable.
1401
+ // `aborted: false`, and falls to the OOM branch below when the runtime confirmed it, else to the switch, where it
1402
+ // stays infra-retryable (issue #596).
1345
1403
  // WHO aborted is an exact-match on `abortReason` (the wiring maps it off `signal.reason`), and the
1346
1404
  // match is deliberately closed: "job-timeout-30m", "shutdown", undefined and any future garbage all
1347
1405
  // classify as worker-abort, so a pin bump that changes what rides the signal can widen nothing.
@@ -1352,7 +1410,23 @@ export async function runJob(job, deps) {
1352
1410
  // Awaited bare like every determinate refusal above: the adapter never throws by contract, and
1353
1411
  // the one swallowed comment in this file (the catch's) justifies itself by its position.
1354
1412
  await comment(job, TERMINAL_COMMENTS[reason]);
1355
- return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}) };
1413
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
1414
+ }
1415
+
1416
+ // Issue #596: a runner the kernel killed for memory, CONFIRMED (`oomKilled` above). After the abort branch, so a
1417
+ // worker's own stop is never relabelled, and before the switch, where the same 137 is an unknown exit and retries.
1418
+ // The same size would be killed the same way on every retry, so it is POLICY: returned, never retried, its slot
1419
+ // kept and its dollars settled above like any other paid stop. An UNCONFIRMED 137 (an image without the supervisor,
1420
+ // an unsigned line, a SIGKILL with no OOM kill in the cgroup, a peak short of the limit, which is how a kill by the
1421
+ // host's OOM killer reads) falls through and retries, exactly as before.
1422
+ //
1423
+ // When only a CHILD was killed, the runner survives and ends on its own code, so this branch is not taken: the
1424
+ // outcome is the runner's, and `resources.oomKills` in the record says a process was killed for memory.
1425
+ if (oomKilled) {
1426
+ // The project and the size go to the log and the record, never to the comment (the TERMINAL_COMMENTS rule).
1427
+ log("oom_killed", { jobId: job.id ?? null });
1428
+ await comment(job, TERMINAL_COMMENTS[EXIT_OOM_KILLED]);
1429
+ return { outcome: "policy", reason: EXIT_OOM_KILLED, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
1356
1430
  }
1357
1431
 
1358
1432
  switch (code) {
@@ -1394,6 +1468,7 @@ export async function runJob(job, deps) {
1394
1468
  // Only when the collector returned a plan (a file, or a confirmed portfolio job with none, `plan-absent`): the
1395
1469
  // record's `plan` is null otherwise, and every other result is unchanged.
1396
1470
  ...(plan ? { plan } : {}),
1471
+ ...(resources ? { resources } : {}),
1397
1472
  };
1398
1473
  }
1399
1474
  case EXIT_POLICY: {
@@ -1420,7 +1495,7 @@ export async function runJob(job, deps) {
1420
1495
  // `why`. parseExitWhy keeps only a member of the closed COST_CAP_WHYS off a `cost-cap` line that said code 2,
1421
1496
  // and it rides only beside a `cost-cap` reason, so a forged line can at worst name the wrong rule of three.
1422
1497
  const why = reason === "cost-cap" && COST_CAP_WHYS.includes(exitWhy) ? { why: exitWhy } : {};
1423
- return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why };
1498
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why, ...(resources ? { resources } : {}) };
1424
1499
  }
1425
1500
  case EXIT_INFRA:
1426
1501
  // NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
@@ -1428,7 +1503,7 @@ export async function runJob(job, deps) {
1428
1503
  // the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
1429
1504
  // BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
1430
1505
  // function never sees (issue #288).
1431
- throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1506
+ throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1432
1507
  default:
1433
1508
  // THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
1434
1509
  // (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
@@ -1444,9 +1519,9 @@ export async function runJob(job, deps) {
1444
1519
  // and normalises to this outcome itself.
1445
1520
  if (neverStarted) {
1446
1521
  // No `dollars` here: the hold is still standing, and the catch refunds it whole with the job-count slots.
1447
- throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1522
+ throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), resources });
1448
1523
  }
1449
- throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1524
+ throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1450
1525
  }
1451
1526
  } catch (e) {
1452
1527
  // A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
@@ -1625,7 +1700,7 @@ function isNeverStartedRetry(e) {
1625
1700
  }
1626
1701
 
1627
1702
  export class InfraRetry extends Error {
1628
- constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars } = {}) {
1703
+ constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars, resources } = {}) {
1629
1704
  super(message, cause ? { cause } : undefined);
1630
1705
  this.name = "InfraRetry";
1631
1706
  this.piDispatchRetry = true;
@@ -1647,6 +1722,8 @@ export class InfraRetry extends Error {
1647
1722
  // Issue #501: the dollar reservation's outcome (`dollarsRecord`), or null when no dollar window applied. Set by
1648
1723
  // the processor on a throw after the reservation, so a retried attempt's record says what its window was charged.
1649
1724
  this.dollars = dollars ?? null;
1725
+ // Issue #596: what the container used, off its exit line, or null; set on a throw after a container ran.
1726
+ this.resources = resources ?? null;
1650
1727
  }
1651
1728
  }
1652
1729