@edgehero/pi-dispatch 3.0.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -28,7 +28,9 @@
28
28
  */
29
29
 
30
30
  import { READ_BACK_BY_A_LIVE_PROBE } from "./backend-conformance.mjs";
31
- import { containerSpec } from "./container-spec.mjs";
31
+ import { containerSpec, memoryBytes } from "./container-spec.mjs";
32
+ import { CGROUP_PARENT, DEFAULT_JOB_SIZE } from "./job-size.mjs";
33
+ import { cpuMaxLine, parseCpuMaxRead } from "./cpu-reserve.mjs";
32
34
  import { ISOLATION_FLAGS, buildDockerRunArgs } from "./docker-run.mjs";
33
35
  import { DEFAULT_EGRESS_PROXY, EGRESS_PROXY_PORT, createJobNetworkWith, networkEndpoints, networkNameFor, removeNetworkOrSay } from "./egress.mjs";
34
36
  import { detachBlockedSentence, makeDetachGate } from "./netns-keeper.mjs";
@@ -112,8 +114,14 @@ export function liveFixture(root) {
112
114
  // `relabel` (issue #355) is a job's too: where a job's own mounts carry `:Z`, so do the probe's, because a probe mounted
113
115
  // the way no job is would read back a container no job gets. The fixture is doctor's own directory, so its workspace
114
116
  // is relabelled like a forge job's clone (`workspaceOwned`); its global overlay directory never is, exactly as a job's.
115
- function probeOptions({ image, name, fixture, user = null, relabel = false }) {
116
- return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, ...fixture };
117
+ //
118
+ // `size` and `hostCpus` (issue #596) are a job's too: the deployment's default size and the `--cpus` ceiling of this
119
+ // runtime's own CPU count, so the probe is bounded exactly as a job without a project size is, and reads that back.
120
+ //
121
+ // `cgroupParent` (issue #596, phase 2) is a job's too: the jobs' parent cgroup, or null where this venue runs jobs without
122
+ // it (`cgroupParentFor`), so the probe sits where a job sits and `cgroupParentVerdict` reads that back.
123
+ function probeOptions({ image, name, fixture, user = null, relabel = false, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
124
+ return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, size, hostCpus, cgroupParent, ...fixture };
117
125
  }
118
126
 
119
127
  // `buildArgs` (issue #354) is the RUNTIME's job builder, never a second one: each probe below is built by whichever
@@ -121,32 +129,32 @@ function probeOptions({ image, name, fixture, user = null, relabel = false }) {
121
129
  // The default is docker's, so every argv here is byte-for-byte what it was.
122
130
 
123
131
  /** The probe container's argv: the job builder's, detached, with `sleep <derived seconds>` as its whole program. */
124
- export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
125
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
132
+ export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
133
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
126
134
  }
127
135
 
128
136
  /**
129
137
  * The pinning probe's argv: the job builder's, detached, against an image this host does not have. Detached so a
130
138
  * container that WAS created prints the ID it is removed by; with `--pull=never` in the builder none should be.
131
139
  */
132
- export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
133
- return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel }), extraFlags: ["-d"] });
140
+ export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
141
+ return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d"] });
134
142
  }
135
143
 
136
144
  /**
137
145
  * One ephemeral run's argv (issue #344): the job builder's, detached, running EPHEMERAL_SCRIPT with the nonce and the
138
146
  * run's number. The same NAME both times, because "a job id run twice" is the question.
139
147
  */
140
- export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
141
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
148
+ export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
149
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
142
150
  }
143
151
 
144
152
  /**
145
153
  * One peer's argv (issue #344): the job builder's, detached, on its OWN job network, running PEER_SCRIPT, which answers
146
154
  * every connection with the nonce for `seconds`. Built with `network` set exactly as a job with egress armed is.
147
155
  */
148
- export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
149
- return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
156
+ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs, size = DEFAULT_JOB_SIZE, hostCpus = null, cgroupParent = CGROUP_PARENT }) {
157
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel, size, hostCpus, cgroupParent }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
150
158
  }
151
159
 
152
160
  /**
@@ -156,6 +164,9 @@ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = li
156
164
  export const STATUS_SCRIPT = [
157
165
  "cat /proc/1/status",
158
166
  'if [ -f /sys/fs/cgroup/cgroup.controllers ]; then echo "cgroup:v2"; echo "pids.max:$(cat /sys/fs/cgroup/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory.max 2>/dev/null)";',
167
+ // Issue #596: the swap bound (0 when --memory-swap equals --memory), the CPU weight --cpu-shares became, and the
168
+ // --cpus ceiling. cgroup v2 only: v1 spells all three differently, and every venue measured is v2.
169
+ 'echo "memory.swap.max:$(cat /sys/fs/cgroup/memory.swap.max 2>/dev/null)"; echo "cpu.weight:$(cat /sys/fs/cgroup/cpu.weight 2>/dev/null)"; echo "cpu.max:$(cat /sys/fs/cgroup/cpu.max 2>/dev/null)";',
159
170
  'else echo "cgroup:v1"; echo "pids.max:$(cat /sys/fs/cgroup/pids/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null)"; fi',
160
171
  ].join("\n");
161
172
 
@@ -246,6 +257,9 @@ export function parseStatus(output) {
246
257
  cgroup: field("cgroup"),
247
258
  pidsMax: field("pids.max"),
248
259
  memoryMax: field("memory.max"),
260
+ swapMax: field("memory.swap.max"),
261
+ cpuWeight: field("cpu.weight"),
262
+ cpuMax: field("cpu.max"),
249
263
  };
250
264
  }
251
265
 
@@ -257,9 +271,34 @@ export function expectedPidsLimit(flags = ISOLATION_FLAGS) {
257
271
 
258
272
  /** The memory bound in bytes, from the spec's own default (`4g`), never a literal. */
259
273
  export function expectedMemoryBytes(memory = containerSpec({ image: "i", name: "n", workspace: "/w" }).memory) {
260
- const m = /^(\d+)([kmg]?)$/i.exec(String(memory));
261
- if (!m) return null;
262
- return Number(m[1]) * { "": 1, k: 1024, m: 1024 ** 2, g: 1024 ** 3 }[m[2].toLowerCase()];
274
+ return memoryBytes(memory);
275
+ }
276
+
277
+ /**
278
+ * What a container of this size reads back (issue #596), from the SAME spec builder the probe's argv came from:
279
+ * `{ pidsLimit, memoryBytes, cpuShares, cpuMax }`, `cpuMax` being the `cpu.max` line `--cpus` writes (`<quota> 100000`,
280
+ * or `max 100000` with no ceiling).
281
+ */
282
+ export function expectedBounds(size = DEFAULT_JOB_SIZE, hostCpus = null, cpuBudgetCenti = null, cgroupParent = CGROUP_PARENT) {
283
+ const spec = containerSpec({ image: "i", name: "n", workspace: "/w", size, hostCpus, cpuBudgetCenti, cgroupParent });
284
+ // ROUNDED (issue #596, gate round 3 of phase 1): under a CPU budget the ceiling may be fractional (`3.3`), and
285
+ // `3.3 * 100000` is 329999.99999999994 in floating point, while the runtime writes the integer quota 330000.
286
+ return { pidsLimit: expectedPidsLimit(), memoryBytes: memoryBytes(spec.memory), cpuShares: spec.cpuShares, cpuMax: spec.cpus === null ? "max 100000" : `${Math.round(Number(spec.cpus) * 100000)} 100000` };
287
+ }
288
+
289
+ /**
290
+ * The `cpu.weight` a `--cpu-shares` value becomes, under the two mappings measured (issue #596): runc 1.1.13 and crun
291
+ * 1.14 map linearly (`1 + ((shares - 2) * 9999) / 262142`, integer division: 1024 to 39, 2048 to 79), runc 1.5.1 and
292
+ * crun 1.27 map so 1024 is 100 (`10^((l^2 + 125l) / 612 - 7/34)` rounded up, l = log2(shares): 2048 to 174). Both
293
+ * reproduce every value of the lab's sweep. A weight that is neither is a runtime nobody measured, said as such.
294
+ */
295
+ export function cpuWeightsFor(shares) {
296
+ if (!Number.isSafeInteger(shares) || shares < 2) return [];
297
+ if (shares >= 262144) return [10000];
298
+ const linear = 1 + Math.floor(((shares - 2) * 9999) / 262142);
299
+ const l = Math.log2(shares);
300
+ const curved = Math.ceil(10 ** ((l * l + 125 * l) / 612 - 7 / 34));
301
+ return [...new Set([linear, curved])];
263
302
  }
264
303
 
265
304
  const verdict = (property, ok, detail, extra = {}) => ({ property, ok, detail, ...extra });
@@ -273,7 +312,7 @@ const notReadBack = (property, why) => verdict(property, false, `not read back:
273
312
  * FAILURE (the bound was not applied, as on rootless docker without delegation); a bound that could not be read at
274
313
  * all is "not read back", never a pass.
275
314
  */
276
- export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes() } = {}) {
315
+ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes(), cpuShares = null, cpuMax = null } = {}) {
277
316
  if (!status || status.capBnd === null || status.noNewPrivs === null) return notReadBack("isolation", "PID 1's status did not show CapBnd and NoNewPrivs");
278
317
  const failures = [];
279
318
  if (!/^0+$/.test(status.capBnd)) failures.push(`CapBnd ${status.capBnd} (the capability bounding set is not empty)`);
@@ -287,10 +326,30 @@ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memo
287
326
  if (Number(got) !== want) failures.push(`${name} ${got}, expected ${want}`);
288
327
  return null;
289
328
  };
290
- const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)].filter(Boolean);
329
+ const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)];
330
+ // Issue #596, read only by a caller that passes the size's bounds (`expectedBounds`). The swap bound is part of the
331
+ // memory bound: `--memory-swap` equal to `--memory` makes it 0, and anything else lets a job swap past its size.
332
+ // `cpu.max` is the host ceiling (`--cpus`), which keeps the host's reserve, so a different one is a failure too.
333
+ // `cpu.weight` is the job's share, whose number depends on the runtime's version: one neither measured mapping
334
+ // gives is said, not failed, because the ORDER between jobs is what it carries and an unknown mapping may keep it.
335
+ let cpuNote = "";
336
+ if (cpuMax !== null) {
337
+ const swap = status.swapMax;
338
+ if (swap === null || swap === undefined || swap === "") unread.push("memory.swap.max not readable");
339
+ else if (swap !== "0") failures.push(`memory.swap.max ${swap}, expected 0 (a job may swap beyond its memory)`);
340
+ const max = status.cpuMax;
341
+ if (max === null || max === undefined || max === "") unread.push("cpu.max not readable");
342
+ else if (max !== cpuMax) failures.push(`cpu.max ${max}, expected ${cpuMax}`);
343
+ const weight = status.cpuWeight;
344
+ const known = cpuWeightsFor(cpuShares);
345
+ if (weight === null || weight === undefined || weight === "") unread.push("cpu.weight not readable");
346
+ else if (!known.includes(Number(weight))) unread.push(`cpu.weight ${weight} is neither mapping measured for --cpu-shares=${cpuShares} (${known.join(" or ")})`);
347
+ else cpuNote = `, memory.swap.max 0, cpu.max ${cpuMax}, cpu.weight ${weight} (--cpu-shares=${cpuShares})`;
348
+ }
349
+ const missing = unread.filter(Boolean);
291
350
  if (failures.length > 0) return verdict("isolation", false, failures.join("; "));
292
- if (unread.length > 0) return notReadBack("isolation", `${unread.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
293
- return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}`);
351
+ if (missing.length > 0) return notReadBack("isolation", `${missing.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
352
+ return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}${cpuNote}`);
294
353
  }
295
354
 
296
355
  /** NON-ROOT: all four Uid fields (real, effective, saved, filesystem) nonzero, on the process the job would be. */
@@ -612,6 +671,60 @@ export function peerTargetsOf(inspectOutput, { network, name }) {
612
671
  return { targets: [...[...hosts].map((h) => `${h}:${PEER_PORT}`), ...addresses], addresses };
613
672
  }
614
673
 
674
+ /**
675
+ * WHERE THE RUNTIME PUT A JOB-BUILT CONTAINER (issue #596, phase 2, INT-LIVE-PROBE-CONTRACT): `{ ok, warn?, detail,
676
+ * placed, quota }`, from `<bin> inspect --format={{.HostConfig.CgroupParent}}|{{.State.Pid}}` and this host's own files.
677
+ *
678
+ * Three readings, each as far as this host can see. The runtime's RECORD of the parent must be the one the argv named
679
+ * (`expected`; none where the venue runs jobs without it). The process's own cgroup (`/proc/<pid>/cgroup`, cgroup v2's
680
+ * `0::<path>`) must lie under `/<parent>/`, which proves the placement rather than the flag: readable for a local Linux
681
+ * daemon and rootless Podman, not for a daemon in a VM (Docker Desktop), where the record alone is said as such. The
682
+ * container cannot see any of this itself (a private cgroup namespace shows it `0::/`, measured). Then the parent's
683
+ * `cpu.max`, where readable, against the CPU budget (`cpuBudgetCenti`: an integer, `Infinity` for off, null not
684
+ * checked): a quota that is not the budget is a WARNING, "no host CPU reserve across jobs", never a failure of the
685
+ * placement, because on a systemd host only the operator can set it.
686
+ */
687
+ export function cgroupParentVerdict({ inspected, expected = CGROUP_PARENT, cpuBudgetCenti = null, readFile = () => {
688
+ throw new Error("no reader");
689
+ }, bin = "docker" }) {
690
+ const out = (ok, detail, extra = {}) => ({ ok, detail, placed: null, quota: undefined, ...extra });
691
+ if (inspected?.code !== 0) return out(false, `not read back: ${bin} inspect did not answer`, { warn: true });
692
+ const [recorded = "", pidText = ""] = String(inspected.stdout ?? "").trim().split("|");
693
+ if (expected === null) {
694
+ if (recorded.includes(CGROUP_PARENT)) return out(false, `the runtime recorded the parent ${recorded} for a venue that runs jobs without one`);
695
+ return out(true, `this venue runs jobs without the ${CGROUP_PARENT} parent (Podman's cgroup manager is not systemd), so their CPU weight is capped at 1024: the egress proxy and Valkey get a fair share of the CPU, not a reserve`, { warn: true });
696
+ }
697
+ // The bare name, or a path ending in it (a runtime may record where the slice resolved); the process's own cgroup
698
+ // below is the proof either way.
699
+ if (recorded.replace(/^\//, "") !== expected && !recorded.endsWith(`/${expected}`)) return out(false, `the runtime recorded the parent ${JSON.stringify(recorded)}, not ${expected}, so this container is outside the jobs' CPU reserve`);
700
+ let cgroupText = null;
701
+ if (/^[1-9][0-9]{0,9}$/.test(pidText)) {
702
+ try {
703
+ cgroupText = readFile(`/proc/${pidText}/cgroup`);
704
+ } catch {
705
+ cgroupText = null;
706
+ }
707
+ }
708
+ const path = cgroupText === null ? null : (/^0::(\/\S*)$/m.exec(String(cgroupText))?.[1] ?? null);
709
+ if (path === null) return out(true, `the runtime recorded the parent ${expected}; the container's own cgroup is not readable from this host (its process runs in the runtime's VM or another namespace), so the placement and the parent's quota were not read here`, { warn: true });
710
+ const at = path.indexOf(`/${expected}/`);
711
+ if (at === -1) return out(false, `the runtime recorded the parent ${expected}, but the container's cgroup is ${path}, not under it`);
712
+ const parentDir = `/sys/fs/cgroup${path.slice(0, at + expected.length + 1)}`;
713
+ let quota;
714
+ try {
715
+ quota = parseCpuMaxRead(readFile(`${parentDir}/cpu.max`));
716
+ } catch {
717
+ quota = null;
718
+ }
719
+ const placed = path.slice(0, at + expected.length + 1);
720
+ if (quota === null) return out(true, `the container's cgroup is under ${placed}; the parent's cpu.max is not readable here`, { warn: true, placed });
721
+ const shown = quota.cpuCenti === null ? "no quota" : `a quota of ${quota.cpuCenti / 100} CPUs (${cpuMaxLine(quota.cpuCenti)})`;
722
+ if (cpuBudgetCenti === null || cpuBudgetCenti === undefined) return out(true, `the container's cgroup is under ${placed}, which has ${shown}`, { placed, quota: quota.cpuCenti });
723
+ const want = cpuBudgetCenti === Infinity ? null : cpuBudgetCenti;
724
+ if (quota.cpuCenti === want) return out(true, `the container's cgroup is under ${placed}, which has ${want === null ? "no quota, as the CPU budget is off" : `${shown}, the host's CPU budget, so all jobs together keep the reserve`}`, { placed, quota: quota.cpuCenti });
725
+ return out(true, `the container's cgroup is under ${placed}, which has ${shown}, not ${want === null ? "none (the CPU budget is off)" : `the CPU budget of ${want / 100}`}: no host CPU reserve across jobs`, { warn: true, placed, quota: quota.cpuCenti });
726
+ }
727
+
615
728
  /**
616
729
  * The whole sequence. Returns `{ ran, reason, verdicts, notes, swept, ranAs }`: `ran` false with a `reason` when nothing
617
730
  * was read back; the verdicts in READ_BACK_BY_A_LIVE_PROBE's order otherwise; `notes` for a teardown that failed, on
@@ -660,6 +773,14 @@ export async function runLiveProbes({
660
773
  // Issue #452, gate round 3: the runtime as this pass already read it (`{ podman, rootless, version } | null`), for the
661
774
  // detach gate, so doctor asks the daemon once per run. Absent, the gate reads it itself.
662
775
  readRuntime = null,
776
+ // Issue #596: the size every probe container is built at (a job's default) and the runtime's CPU count for the
777
+ // `--cpus` ceiling; the isolation verdict reads both back.
778
+ size = DEFAULT_JOB_SIZE,
779
+ hostCpus = null,
780
+ // Issue #596, phase 2: the parent cgroup a job on this venue runs under (null: none, `cgroupParentFor`) and the CPU
781
+ // budget its quota should be (an integer in hundredths, `Infinity` for off, null or absent: not checked).
782
+ cgroupParent = CGROUP_PARENT,
783
+ cpuBudgetCenti = null,
663
784
  }) {
664
785
  const notRun = (reason) => ({ ran: false, reason, verdicts: [], notes: [], swept: [] });
665
786
  const notLocal = notRun(`this shell's ${bin} CLI is not observed to point at this host, so bind paths and .Mounts would describe another machine; nothing was run`);
@@ -764,7 +885,7 @@ export async function runLiveProbes({
764
885
  }
765
886
 
766
887
  // --- the reading container: mounts, status, writes ---
767
- const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
888
+ const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
768
889
  const probeId = reading.entry.id;
769
890
  if (reading.result?.code !== 0 || probeId === null) {
770
891
  // The runtime's own words, when it printed any (issue #453, gate round 1): "did not start" alone left the operator
@@ -773,7 +894,7 @@ export async function runLiveProbes({
773
894
  return { ran: false, reason: `the probe container did not start${said ? ` (${bin} said: ${said})` : ""}, so nothing was read back`, verdicts: [], notes, swept };
774
895
  }
775
896
 
776
- const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel })).mounts;
897
+ const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel, size, hostCpus, cgroupParent })).mounts;
777
898
  const inspected = await step(["inspect", "--format={{json .Mounts}}", probeId]);
778
899
  // Issue #345: the mount table as the container itself sees it, by a constant `cat`, for what `.Mounts` does not list.
779
900
  const mountinfo = await step(["exec", probeId, "cat", "/proc/self/mountinfo"]);
@@ -781,8 +902,12 @@ export async function runLiveProbes({
781
902
 
782
903
  const statusRun = await step(["exec", probeId, "sh", "-c", STATUS_SCRIPT]);
783
904
  const status = statusRun?.code === 0 ? parseStatus(statusRun.stdout) : null;
784
- const isolation = status ? isolationVerdict(status) : notReadBack("isolation", "the status probe did not run in the container");
905
+ const isolation = status ? isolationVerdict(status, expectedBounds(size, hostCpus, null, cgroupParent)) : notReadBack("isolation", "the status probe did not run in the container");
785
906
  const nonRoot = status ? nonRootVerdict(status) : notReadBack("nonRoot", "the status probe did not run in the container");
907
+ // Issue #596, phase 2: where the runtime put the container (its record, and the process's own cgroup where this host
908
+ // can read it) and, where readable, the parent's quota. Its own line, not one of the declared properties.
909
+ const placed = await step(["inspect", "--format={{.HostConfig.CgroupParent}}|{{.State.Pid}}", probeId]);
910
+ const parentRead = cgroupParentVerdict({ inspected: placed, expected: cgroupParent, cpuBudgetCenti, readFile: (path) => fs.readFileSync(path, "utf8"), bin });
786
911
 
787
912
  const written = await step(["exec", probeId, "sh", "-c", WRITE_SCRIPT, "sh", nonce]);
788
913
  const readBack = (dir) => {
@@ -807,7 +932,7 @@ export async function runLiveProbes({
807
932
  await release(reading.entry);
808
933
 
809
934
  // --- the pinning container: an image this host does not have ---
810
- const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs }));
935
+ const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
811
936
  const after = await step(["image", "inspect", absentImageRef(nonce)]);
812
937
  const stillAbsent = after?.code === 0 ? false : typeof after?.code === "number" ? true : null;
813
938
  const imagePinning = imagePinningVerdict({ code: pinning.result?.code, output: `${pinning.result?.stdout ?? ""}${pinning.result?.stderr ?? ""}`, stillAbsent, bin });
@@ -815,7 +940,7 @@ export async function runLiveProbes({
815
940
 
816
941
  // --- the ephemeral pair (issue #344): one name, two runs, each waited on until it is gone ---
817
942
  const runEphemeral = async (n) => {
818
- const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs }));
943
+ const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs, size, hostCpus, cgroupParent }));
819
944
  const started = result?.code === 0 && entry.id !== null;
820
945
  // A HELD NAME is the daemon refusing the create for the name, in its own words (measured: Docker "Conflict. ...
821
946
  // is already in use", Podman "that name is already in use"). Not a listed container: after the first run was
@@ -864,7 +989,7 @@ export async function runLiveProbes({
864
989
  } catch {
865
990
  break;
866
991
  }
867
- const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
992
+ const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs, size, hostCpus, cgroupParent }));
868
993
  peers.push(entry);
869
994
  if (result?.code !== 0 || entry.id === null) break;
870
995
  ids[key] = entry.id;
@@ -901,7 +1026,7 @@ export async function runLiveProbes({
901
1026
  const byProperty = { isolation, ephemeral, mountSet, egress: egressVerdict(egress ?? {}), jobToJobIsolation, imagePinning, nonRoot, localFolders };
902
1027
  // The uid PID 1 actually ran as, for doctor's job-user line: the decision it was given, read back.
903
1028
  const ranAs = Array.isArray(status?.uids) && /^\d+$/.test(status.uids[1] ?? "") ? Number(status.uids[1]) : null;
904
- return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs };
1029
+ return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs, cgroupParent: parentRead };
905
1030
  } finally {
906
1031
  for (const entry of owned) await release(entry);
907
1032
  for (const entry of networks) await dropNetwork(entry);
@@ -91,7 +91,7 @@ function overlayDeclares(overlay, provider, id) {
91
91
 
92
92
  /**
93
93
  * Would pi COMPOSE this overlay provider (PR #536's review)? A file that passes pi's schema can still lose a provider
94
- * at the next step: pi 0.99.1's `applyModelsJson` and `modelFromJson` (`pi-coding-agent/dist/core/provider-composer.js`)
94
+ * at the next step: the pinned pi's `applyModelsJson` and `modelFromJson` (`pi-coding-agent/dist/core/provider-composer.js`)
95
95
  * throw for `oauth` with no `baseUrl`, and per model no resolvable `api` or
96
96
  * `baseUrl`, or a `contextWindow` or `maxTokens` at or below zero. pi then keeps the provider's BUILTIN models
97
97
  * (`ModelRuntime.composeProvider` falls back to the base) and drops every model the overlay added, so such a model
@@ -518,6 +518,28 @@ export function unreportedUsageModels(models, { builtinModel = () => null, built
518
518
  /** The exact `apiKey` a keyless provider's models.json entry carries: pi resolves `$NAME` from the job's environment. */
519
519
  export const KEYLESS_API_KEY = `$${KEYLESS_ENV_NAME}`;
520
520
 
521
+ /**
522
+ * Provider ids pi renamed, old to new (issue #587: pi 1.0.3 renamed `azure-openai-responses` to `azure`, and its catalog
523
+ * file with it; the api id `azure-openai-responses` is unchanged). Pinned against the catalog by
524
+ * worker/test/env-allowlist.test.mjs: every old id is gone from pi, every new id is there. A rename is not a refusal of
525
+ * its own: the old id is already refused before spend as a provider pi does not have, and this only lets that refusal
526
+ * and doctor say which id was meant.
527
+ */
528
+ export const RENAMED_PROVIDERS = Object.freeze({ "azure-openai-responses": "azure" });
529
+
530
+ /**
531
+ * The "did you mean" for a provider pi renamed, or "" (issue #587). Only for an old id pi no longer has, whose new id pi
532
+ * does have, and that the overlay's models.json (`models`, parsed, or null when there is none) does not declare: a
533
+ * provider the file declares under the old id is the operator's own custom provider, and the hint would be wrong there.
534
+ * `overlayUnread`: the overlay exists but could not be read, so whether it declares the id is not known: no guess.
535
+ */
536
+ export function providerRenameHint(provider, { models = null, piProviders = [], overlayUnread = false } = {}) {
537
+ if (typeof provider !== "string" || !Object.hasOwn(RENAMED_PROVIDERS, provider)) return "";
538
+ const renamed = RENAMED_PROVIDERS[provider];
539
+ if (overlayUnread || piProviders.includes(provider) || !piProviders.includes(renamed) || providerOf(models, provider) !== null) return "";
540
+ return ` pi renamed the provider "${provider}" to "${renamed}" in pi 1.0.3: did you mean "${renamed}"?`;
541
+ }
542
+
521
543
  /**
522
544
  * The second way into the credential gate, said once for every refusal and doctor line that names it (issue #503).
523
545
  */
@@ -1,5 +1,5 @@
1
1
  /**
2
- * The overlay `models.json`, read the way pi reads it (issue #502, PR #536's review): pi 0.99.1's
2
+ * The overlay `models.json`, read the way pi reads it (issue #502, PR #536's review): pi's
3
3
  * `ModelConfig.load` (`pi-coding-agent/dist/core/model-config.js`) strips a leading BOM, strips `//` comments and
4
4
  * trailing commas outside strings, parses, and then checks the whole document against its TypeBox schema. ANY
5
5
  * schema error drops the WHOLE file: pi then knows none of its providers or models.
@@ -121,7 +121,8 @@ const T = {
121
121
  const opt = (check) => [check, true];
122
122
  const req = (check) => [check, false];
123
123
 
124
- // The schema, transcribed from model-config.js at the 0.99.1 pin, in its order.
124
+ // The schema, transcribed from model-config.js at the pin (1.0.3; worker/test/models-json.test.mjs holds the file's
125
+ // content hash), in its order.
125
126
  const PercentileCutoffs = T.obj({ p50: opt(T.num()), p75: opt(T.num()), p90: opt(T.num()), p99: opt(T.num()) });
126
127
  const numOrStr = T.union(T.num(), T.str());
127
128
  const OpenRouterRouting = T.obj({
@@ -142,6 +143,9 @@ const OpenRouterRouting = T.obj({
142
143
  const VercelGatewayRouting = T.obj({ only: opt(T.arr(T.str())), order: opt(T.arr(T.str())) });
143
144
  const ThinkingValue = T.union(T.str(), T.nul());
144
145
  const ThinkingLevelMap = T.obj({ off: opt(ThinkingValue), minimal: opt(ThinkingValue), low: opt(ThinkingValue), medium: opt(ThinkingValue), high: opt(ThinkingValue), xhigh: opt(ThinkingValue), max: opt(ThinkingValue) });
146
+ // pi 1.0.2 (issue #587): a model's sampling parameters, and the same per thinking level, merged per call over them.
147
+ const SamplingParams = T.rec(T.unknown());
148
+ const SamplingParamsByThinkingLevel = T.obj({ off: opt(SamplingParams), minimal: opt(SamplingParams), low: opt(SamplingParams), medium: opt(SamplingParams), high: opt(SamplingParams), xhigh: opt(SamplingParams), max: opt(SamplingParams) });
145
149
  const KwargScalar = T.union(T.str(), T.num(), T.bool(), T.nul());
146
150
  const KwargVariable = T.obj({ $var: req(T.union(T.lit("thinking.enabled"), T.lit("thinking.effort"))), omitWhenOff: opt(T.bool()) });
147
151
  const Kwarg = T.union(KwargScalar, KwargVariable);
@@ -214,7 +218,8 @@ const ModelDefinition = T.obj({
214
218
  promptCache: opt(ModelPromptCache),
215
219
  contextWindow: opt(T.num()),
216
220
  maxTokens: opt(T.num()),
217
- samplingParams: opt(T.rec(T.unknown())),
221
+ samplingParams: opt(SamplingParams),
222
+ samplingParamsByThinkingLevel: opt(SamplingParamsByThinkingLevel),
218
223
  headers: opt(headers),
219
224
  compat: opt(ProviderCompat),
220
225
  });
@@ -228,7 +233,8 @@ const ModelOverride = T.obj({
228
233
  promptCache: opt(ModelPromptCache),
229
234
  contextWindow: opt(T.num()),
230
235
  maxTokens: opt(T.num()),
231
- samplingParams: opt(T.rec(T.unknown())),
236
+ samplingParams: opt(SamplingParams),
237
+ samplingParamsByThinkingLevel: opt(SamplingParamsByThinkingLevel),
232
238
  headers: opt(headers),
233
239
  compat: opt(ProviderCompat),
234
240
  });
@@ -13,7 +13,7 @@
13
13
 
14
14
  import { composedCost, endpointsForModel, isZeroCost, modelEntryOf } from "./model-endpoints.mjs";
15
15
 
16
- /** The hosts pi's catalog serves openai-completions on: the runner's `COMPLETIONS_CATALOG_HOSTS`, copied. */
16
+ /** The hosts pi's catalog serves openai-completions on: the runner's `COMPLETIONS_CATALOG_HOSTS`, generated the same way. */
17
17
  export const COMPLETIONS_CATALOG_HOSTS = Object.freeze([
18
18
  "api.ant-ling.com",
19
19
  "api.cerebras.ai",
@@ -56,7 +56,7 @@ export function completionsOwnServer(model) {
56
56
  }
57
57
 
58
58
  /**
59
- * The parts of a model the rule reads, composed as pi 0.99.1's provider-composer.js composes them from the overlay
59
+ * The parts of a model the rule reads, composed as the pinned pi's provider-composer.js composes them from the overlay
60
60
  * `models.json` (`models`, parsed, or null) and the builtin catalog (`builtinModel(provider, id)`, injected):
61
61
  * - a model the overlay DEFINES: its own `api` and `baseUrl`, else its provider's, else those of pi's DEFAULTS
62
62
  * model (`findModelDefaults`, mirrored in `definedDefaults`); its compat field from its `modelOverrides` entry,
@@ -94,7 +94,7 @@ export function outputCapView({ models, provider, modelId, builtinModel = () =>
94
94
  const str = (v) => (typeof v === "string" ? v : undefined);
95
95
 
96
96
  /**
97
- * The `{ api, baseUrl }` pi 0.99.1 composes for the overlay-defined model `modelId` (provider-composer.js
97
+ * The `{ api, baseUrl }` the pinned pi composes for the overlay-defined model `modelId` (provider-composer.js
98
98
  * `applyModelsJson`): the provider's chat models start as the catalog's (each on the provider's `baseUrl` when it sets
99
99
  * one), and each definition in the file's order is composed and then replaces the model of its id or joins the list.
100
100
  * A definition's `api` is its own, else the provider's, else its DEFAULTS model's, and its `baseUrl` likewise, where the
package/src/prepare.mjs CHANGED
@@ -14,6 +14,7 @@ import { buildForgejoPrompt } from "./forgejo-prompt.mjs";
14
14
  import { buildAzurePrompt } from "./azure-prompt.mjs";
15
15
  import { prepareLocalWorkspace } from "./prepare-local.mjs";
16
16
  import { copySkillTree } from "./copy-tree.mjs";
17
+ import { recordedJobSize } from "./job-size.mjs";
17
18
 
18
19
  /**
19
20
  * The subdirectory of the per-job dir a trigger's injected skills are copied into, so they reach the
@@ -82,7 +83,7 @@ export function makePrepareWorkspace({
82
83
  removeDir = (dir) => rmSync(dir, { recursive: true, force: true }),
83
84
  }) {
84
85
  ensureDir(jobsDir);
85
- return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false } = {}) {
86
+ return async function prepareWorkspace(job, token, { queueJobId, piVersion = null, jobUser = null, podmanStore = null, portfolio = false, size = null } = {}) {
86
87
  ensureDir(jobsDir);
87
88
  const jobDir = mkdtempSync(join(jobsDir, "job-"));
88
89
  // Issue #524: a THROW out of anything below leaves `jobDir` to nobody. The processor tears down only what
@@ -92,7 +93,7 @@ export function makePrepareWorkspace({
92
93
  // covers the thrown ones, in one place for every kind, rather than in each preparer that can throw.
93
94
  // A retry makes a fresh directory, so nothing is lost by removing this one.
94
95
  try {
95
- return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio });
96
+ return await prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size });
96
97
  } catch (error) {
97
98
  // GUARDED: a removal that fails (a busy mount, a permission flipped mid-job) must never replace the error
98
99
  // that is the job's actual outcome. The directory is then left, which is what happened before this catch.
@@ -103,7 +104,7 @@ export function makePrepareWorkspace({
103
104
  }
104
105
  };
105
106
 
106
- async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio }) {
107
+ async function prepareInto(job, token, jobDir, { queueJobId, piVersion, jobUser, podmanStore, portfolio, size }) {
107
108
  // The trigger's injected skills (REQ-PER-TRIGGER-SKILLS, issue #60), COPIED here rather than
108
109
  // mounted, and copied ONCE for every job kind because this is where local and forge converge.
109
110
  //
@@ -137,6 +138,9 @@ export function makePrepareWorkspace({
137
138
  ...(jobUser ? { jobUser: { user: jobUser.user ?? null, home: jobUser.home ?? null } } : {}),
138
139
  // Issue #429: the podman store the run's container lived in, for a podman run only.
139
140
  ...(typeof podmanStore === "string" && podmanStore !== "" ? { podmanStore } : {}),
141
+ // Issue #596: the size the run's container had, so a re-opened sandbox gets the same memory and CPU weight.
142
+ // Rebuilt (`recordedJobSize`), and only when the processor resolved one; a direct call stamps nothing.
143
+ ...(recordedJobSize(size) ? { size: recordedJobSize(size) } : {}),
140
144
  };
141
145
  if (job.kind === "local") {
142
146
  // Harness text above, operator DATA below: the fixed pointer line names /job/event.json so a