@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -113,6 +113,40 @@ VALKEY_URL=redis://127.0.0.1:6379
113
113
  # docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
114
114
  # On rootful Podman, run these and every docker command through a docker context pointed at podman.sock, with the real docker CLI (docs/podman.md)
115
115
  PI_JOB_IMAGE=pi-job:latest
116
+ # How much memory and CPU each job container gets, unless its project's row in the scoped-limits file sets its own.
117
+ # See docs/scoped-limits.md, and docs/sizing.md for how to choose sizes.
118
+ # Memory is a whole number of megabytes or gigabytes (512m, 1536m, 4g), at least 512m.
119
+ # CPUs is a number with at most two decimals (0.5, 2, 1.25), at least 0.25.
120
+ # A bad value stops the worker at boot.
121
+ # A job gets no swap beyond its memory where the runtime enforces swap limits (doctor warns where it does not).
122
+ # CPUs are a weight, not a cap: under contention a job with more CPUs gets more CPU than one with fewer.
123
+ # On an idle host one job may use every core up to the CPU budget below, whenever one is in force (with the
124
+ # default auto budget, every core but one on a host with 4 or more; with the CPU budget off, the same).
125
+ # Default 4g and 2.
126
+ # PI_JOB_MEMORY=4g
127
+ # PI_JOB_CPUS=2
128
+ # How much memory and CPU this host's jobs may hold together: the host budget. A job starts only when its size fits
129
+ # beside what already runs here, and a waiting job keeps its place, so a big job is not starved by a stream of small ones.
130
+ # PI_CONCURRENCY still caps the number of jobs; whichever is reached first applies. See docs/multi-host.md,
131
+ # and docs/sizing.md for a worked example.
132
+ # Each budget is auto, a value, or off (no limit on that resource).
133
+ # auto reads the container runtime: its memory and CPU count (and, on rootless Podman, the user service's own limits),
134
+ # minus the reserve below, and never less than one job of the default size above.
135
+ # A value is a memory amount (64g, 49152m) or a number of CPUs (12, 3.5). It is the budget itself; the reserve is not
136
+ # taken from it. A value below one job of the default size stops the worker at boot.
137
+ # A job's cpus count as CPU reserved for it (the runtime uses them as a weight), so they must fit the CPU budget.
138
+ # A job too big for this host's budget, or for its project's hostShare of it, is refused before anything is spent
139
+ # when it can run nowhere else: on this host's own queue, or on any queue without PI_WORKER_NAME (no fleet is
140
+ # declared, so it never waits for another host). With PI_WORKER_NAME set, a job on the shared queue waits for a
141
+ # host it fits on.
142
+ # Default auto.
143
+ # PI_HOST_MEMORY_BUDGET=auto
144
+ # PI_HOST_CPU_BUDGET=auto
145
+ # What auto leaves for the host itself (the system, the egress proxy, Valkey, the worker).
146
+ # auto: memory 10% of the host's memory, at least 1g and at most 4g; CPUs 1 when the host has 4 or more, else 0.
147
+ # Or a value (2g, 0.5, or 0 for none). A bad value stops the worker at boot.
148
+ # PI_HOST_RESERVE_MEMORY=auto
149
+ # PI_HOST_RESERVE_CPUS=auto
116
150
  # where per-job /job inputs live (default: <OS temp dir>/pi-dispatch-<your uid>/jobs, one per account, created 0700)
117
151
  # Another account's directory there, or one you set that another account owns, stops the worker at boot (exit 2), and doctor names it
118
152
  # Pin it (and PI_SANDBOX_DIR) yourself if the worker runs as a different account than your /dispatch panel or `pi-dispatch sandbox`: the default is per account
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "3.1.0",
3
+ "version": "4.0.0",
4
4
  "type": "module",
5
5
  "description": "The pi-dispatch worker and CLI: runs the pi coding agent as a self-hosted service, one locked down Docker or Podman container per job, with spend caps checked before anything is spent, plus init, up, doctor and service install.",
6
6
  "keywords": [
@@ -63,6 +63,10 @@
63
63
  "./backends": "./src/backends.mjs",
64
64
  "./backend-registry": "./src/backend-registry.mjs",
65
65
  "./container-spec": "./src/container-spec.mjs",
66
+ "./job-size": "./src/job-size.mjs",
67
+ "./size-suggest": "./src/size-suggest.mjs",
68
+ "./size-records": "./src/size-records.mjs",
69
+ "./host-budget": "./src/host-budget.mjs",
66
70
  "./backend-conformance": "./src/backend-conformance.mjs",
67
71
  "./live-probes": "./src/live-probes.mjs",
68
72
  "./triggers": "./src/triggers.mjs",
@@ -105,7 +109,7 @@
105
109
  "start": "node src/cli.mjs worker"
106
110
  },
107
111
  "dependencies": {
108
- "@earendil-works/pi-ai": "1.0.3",
112
+ "@earendil-works/pi-ai": "1.0.4",
109
113
  "@octokit/auth-app": "8.2.0",
110
114
  "@octokit/rest": "22.0.1",
111
115
  "bullmq": "5.80.4",
@@ -28,6 +28,7 @@ import { BACKENDS, DEFAULT_BACKEND, DOCKER_NEVER_STARTED_EXITS } from "./backend
28
28
  import { DEFAULT_EGRESS_PROXY, ENDPOINT_LISTED_STATES, networkEndpoints, removeNetworkOrSay } from "./egress.mjs";
29
29
  import { makeDetachGate } from "./netns-keeper.mjs";
30
30
  import { isDeterminateFsCode } from "./transient.mjs";
31
+ import { SIZE_LABEL_CPU, SIZE_LABEL_MEM } from "./container-spec.mjs";
31
32
 
32
33
  const execDocker = promisify(execFile);
33
34
 
@@ -401,6 +402,74 @@ export function isJobNamespace(name) {
401
402
  return typeof name === "string" && name.startsWith(JOB_NAME_PREFIX);
402
403
  }
403
404
 
405
+ /**
406
+ * The states in which a job container runs nothing and never will again (gate round 1 of phase 2): Docker's
407
+ * `exited` and `dead`, Podman's `exited` and `stopped`. Such a container uses no memory and no CPU, yet `ps -a` lists
408
+ * it (an `--rm` whose removal failed), so counting it as running kept an orphan's hold until a worker restart.
409
+ */
410
+ export const CONTAINER_GONE_STATES = Object.freeze(new Set(["exited", "dead", "stopped"]));
411
+ /**
412
+ * The states in which a job container has not started yet: Docker's `created`, Podman's `created` and `configured`. NOT
413
+ * gone by itself: a `docker run` client still alive after its stop timed out may yet start it. So it is REMOVED
414
+ * (`rm -f`, exact name, our namespace only), and gone once the removal is answered.
415
+ */
416
+ export const CONTAINER_UNSTARTED_STATES = Object.freeze(new Set(["created", "configured"]));
417
+
418
+ /** `ps -a` lines of `{{.Names}}\t{{.State}}...`: `[{ name, state, rest }]`, only names in the job namespace, `rest` the later fields. */
419
+ function psRows(stdout) {
420
+ return String(stdout ?? "")
421
+ .split("\n")
422
+ .map((line) => line.split("\t").map((f) => f.trim()))
423
+ .filter(([name, state]) => isJobNamespace(name) && typeof state === "string")
424
+ .map(([name, state, ...rest]) => ({ name, state: state.toLowerCase(), rest }));
425
+ }
426
+
427
+ /**
428
+ * Whether a job container is GONE (issue #596, phase 2): `true` when the runtime lists no container of exactly that name,
429
+ * or lists it in a state that runs nothing (`CONTAINER_GONE_STATES`), or lists it not yet started and removes it
430
+ * (`CONTAINER_UNSTARTED_STATES`); `false` when it lists one in any other state (running, paused, restarting, removing,
431
+ * one nothing here has measured); `null` when it could not be asked or the removal was not answered. The host budget
432
+ * keeps the hold of a job whose stop did not take until this says `true` (`host-budget.mjs` `sweep`), so `null` keeps
433
+ * it: an unanswered listing must never free room a running container may still use. `-a` with the STATE, because a
434
+ * stopped container is listed by `ps -a` and is gone in every sense the budget cares about, and the anchored name test,
435
+ * because `--filter name=` is a SUBSTRING match (the reaper's measured reason).
436
+ */
437
+ export function makeContainerGone({ exec = execReaperBounded, binOf = () => "docker" } = {}) {
438
+ return async (name, venue) => {
439
+ if (typeof name !== "string" || !isJobNamespace(name)) return null;
440
+ const bin = binOf(venue);
441
+ try {
442
+ const { stdout } = await exec(bin, ["ps", "-a", "--filter", `name=${name}`, "--format", "{{.Names}}\t{{.State}}"]);
443
+ const row = psRows(stdout).find((r) => r.name === name);
444
+ if (!row) return true;
445
+ if (CONTAINER_GONE_STATES.has(row.state)) return true;
446
+ if (!CONTAINER_UNSTARTED_STATES.has(row.state)) return false;
447
+ await exec(bin, ["rm", "-f", name]);
448
+ return true;
449
+ } catch {
450
+ return null;
451
+ }
452
+ };
453
+ }
454
+
455
+ /**
456
+ * The job containers a venue still lists (gate round 1 of phase 2), for the host budget's boot seed: `ps -a`
457
+ * by the job namespace, each with its state and its two size labels, as `[{ name, memMiB, cpuCenti }]` (a label absent
458
+ * or not a positive integer is null). A container in a state that runs nothing (`CONTAINER_GONE_STATES`) is left out:
459
+ * it holds nothing. One not yet started is kept, so the sweep removes it before its room is given back. THROWS when
460
+ * the runtime does not answer: the budget then admits nothing until a listing is read, because an empty ledger beside
461
+ * containers nobody counted is an overcommit.
462
+ */
463
+ export function makeJobContainerLister({ exec = execReaperBounded, bin = "docker" } = {}) {
464
+ const int = (v) => (/^[1-9][0-9]{0,8}$/.test(v ?? "") ? Number(v) : null);
465
+ return async () => {
466
+ const { stdout } = await exec(bin, ["ps", "-a", "--filter", `name=${JOB_NAME_PREFIX}`, "--format", `{{.Names}}\t{{.State}}\t{{.Label "${SIZE_LABEL_MEM}"}}\t{{.Label "${SIZE_LABEL_CPU}"}}`]);
467
+ return psRows(stdout)
468
+ .filter((r) => !CONTAINER_GONE_STATES.has(r.state))
469
+ .map((r) => ({ name: r.name, memMiB: int(r.rest[0]), cpuCenti: int(r.rest[1]) }));
470
+ };
471
+ }
472
+
404
473
  /**
405
474
  * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind.
406
475
  *
@@ -34,10 +34,11 @@ import { JOB_NAME_PREFIX, execDockerBounded, jobContainerName, makeReaper, makeS
34
34
  import { BACKENDS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_CONF_WIDENS_JOB, PODMAN_NETWORK_HELPER_KEYS, PODMAN_WIDENING_KEYS, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
35
35
  import { CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
36
36
  import { buildPodmanRunArgs } from "./docker-run.mjs";
37
+ import { cgroupParentFor } from "./cpu-reserve.mjs";
37
38
  import { DEFAULT_EGRESS_PROXY, makeEgressPreflight } from "./egress.mjs";
38
39
  import { NETNS_KEEPER, NETNS_KEEPER_FORMAT, QUADLET_FILES, STARTED_AT_FORMAT, judgeNetnsKeeper, netnsKeeperRemedy, podmanNeedsNetnsKeeper } from "./podman-stack.mjs";
39
40
  import { makeImagePreflight } from "./image-preflight.mjs";
40
- import { DAEMON_FACTS_TIMEOUT_MS } from "./job-user.mjs";
41
+ import { DAEMON_FACTS_TIMEOUT_MS, JOB_USER_FACTS_MAX_AGE_MS, STALE_FACTS_CEILING_MS } from "./job-user.mjs";
41
42
  import { PODMAN_INFO_ARGS, parsePodmanInfo } from "./daemon-facts.mjs";
42
43
 
43
44
  // Moved to the leaf `daemon-facts.mjs` (issue #452, gate round 3) and re-exported, so every importer keeps its path.
@@ -106,25 +107,50 @@ export function makePodmanInfoReader({ run = (args) => execDockerBounded(args, {
106
107
  const CACHED = Symbol("podman-info-cached");
107
108
 
108
109
  /**
109
- * `readInfo` with its first ANSWERED result kept, and concurrent callers sharing one read. An unanswered read is never
110
- * kept: a Podman that timed out once must be asked again, or every later job would retry on a stale failure. Idempotent,
111
- * so the boot wiring can wrap a reader once and hand the same one to the bundle and to its own boot read.
110
+ * `readInfo` with its last ANSWERED result kept for `maxAgeMs`, and concurrent callers sharing one read. An unanswered
111
+ * read is never kept: a Podman that timed out once must be asked again, or every later job would retry on a stale
112
+ * failure. Idempotent, so the boot wiring can wrap a reader once and hand the same one to the bundle and to its own boot
113
+ * read (a second wrap keeps the first wrap's clock and age).
114
+ *
115
+ * The age (issue #596, gate round 2) is the job-user resolver's (`JOB_USER_FACTS_MAX_AGE_MS`), for the same reason:
116
+ * this read carries the host's CPU count that every job's `--cpus` ceiling is built from, and a raised or lowered count
117
+ * must be seen without a restart. STALE WHILE ERROR, also as there: a read past the age that does not answer keeps
118
+ * serving the kept answer until a read does, logging `podman_info_stale` with the failed read's reason once per run of
119
+ * failures, because the age exists to see a change and must not turn one slow `podman info` into a failed pickup.
120
+ * Podman has no `invalidate`: it accepts a `--cpus` above the host's count (4.9.3 and 5.8.1, measured), so no job
121
+ * refusal proves the kept count wrong.
112
122
  */
113
- export function cachedPodmanInfo(readInfo) {
123
+ export function cachedPodmanInfo(readInfo, { now = Date.now, maxAgeMs = JOB_USER_FACTS_MAX_AGE_MS, log = () => {} } = {}) {
114
124
  if (readInfo?.[CACHED]) return readInfo;
115
125
  let kept = null;
126
+ let keptAt = 0;
127
+ let staleSaid = false;
116
128
  let inFlight = null;
117
129
  const cached = async () => {
118
- if (kept) return kept;
130
+ if (kept && now() - keptAt < maxAgeMs) return kept;
119
131
  if (inFlight) return inFlight;
120
132
  inFlight = (async () => {
133
+ let read;
121
134
  try {
122
- const read = await readInfo();
123
- if (read?.answered === true && read.info) kept = read;
124
- return read;
135
+ read = await readInfo();
125
136
  } catch {
126
- return { answered: false, reason: "spawn-failed", transient: true };
137
+ read = { answered: false, reason: "spawn-failed", transient: true };
138
+ }
139
+ if (read?.answered === true && read.info) {
140
+ kept = read;
141
+ keptAt = now();
142
+ staleSaid = false;
143
+ return read;
144
+ }
145
+ // Not past `STALE_FACTS_CEILING_MS` (job-user.mjs says why): then the failed read is the answer, as a first one is.
146
+ if (kept && now() - keptAt < STALE_FACTS_CEILING_MS) {
147
+ if (!staleSaid) {
148
+ staleSaid = true;
149
+ log("podman_info_stale", { reason: read?.reason ?? "unanswered", ageMs: now() - keptAt });
150
+ }
151
+ return kept;
127
152
  }
153
+ return read;
128
154
  })();
129
155
  try {
130
156
  return await inFlight;
@@ -1081,7 +1107,7 @@ export function makePodmanBackend(opts = {}) {
1081
1107
  if (reap !== undefined && typeof reap !== "function") throw new Error(`backend "${PODMAN_BACKEND}": reap must be a function (makePodmanReaper)`);
1082
1108
  const spawnSeam = spawnFn ? { spawnFn } : {};
1083
1109
  const execSeam = exec ? { exec } : spawnFn ? { exec: execViaSpawn(spawnFn) } : {};
1084
- const info = cachedPodmanInfo(readInfo);
1110
+ const info = cachedPodmanInfo(readInfo, { log });
1085
1111
 
1086
1112
  const runContainer = makeRunContainerFn({
1087
1113
  image,
@@ -1098,7 +1124,9 @@ export function makePodmanBackend(opts = {}) {
1098
1124
  forgeHosts,
1099
1125
  neverStartedExits: PODMAN_NEVER_STARTED_EXITS,
1100
1126
  bin: "podman",
1101
- buildArgs: buildPodmanRunArgs,
1127
+ // Issue #596, phase 2: every job under the one parent cgroup (`CGROUP_PARENT`), except where Podman reports a cgroup
1128
+ // manager other than systemd (`cgroupParentFor` says why), from the SAME `podman info` this job was admitted on.
1129
+ buildArgs: (opts) => buildPodmanRunArgs({ ...opts, cgroupParent: cgroupParentFor({ podman: true, cgroupManager: info.peek?.()?.info?.cgroupManager ?? null }) }),
1102
1130
  // Issue #452, gate round 4: the teardown's detach gate uses the `podman info` this venue admitted jobs on, never a
1103
1131
  // read of its own; before any answered read it falls back to one. A refused teardown is logged with its token.
1104
1132
  teardownRuntime: () => {
@@ -1145,7 +1173,10 @@ export function makePodmanBackend(opts = {}) {
1145
1173
  // Issue #429: the store this job's container lives in rides beside its user, so a retained run records it and a
1146
1174
  // sandbox or the retention sweep can tell another store's empty answer from "not open".
1147
1175
  const store = read?.answered === true ? read.info?.graphRoot : null;
1148
- return chosen.user && typeof store === "string" ? { ...chosen, store } : chosen;
1176
+ // Issue #596: the host's CPU count from the same read, for the job's `--cpus` ceiling. Absent when Podman did not say.
1177
+ const hostCpus = read?.answered === true ? read.info?.hostCpus : null;
1178
+ const stored = chosen.user && typeof store === "string" ? { ...chosen, store } : chosen;
1179
+ return chosen.user && Number.isSafeInteger(hostCpus) ? { ...stored, hostCpus } : stored;
1149
1180
  };
1150
1181
 
1151
1182
  return {
package/src/config.mjs CHANGED
@@ -15,6 +15,8 @@ import { SWEEP_INTERVAL_HOURS, SWEEP_INTERVAL_MAX_HOURS } from "./retention-swee
15
15
  import { parseSecretProfiles } from "./secret-profiles.mjs";
16
16
  import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, parseWaitProfiles } from "./wait-for.mjs";
17
17
  import { imageRefProblem } from "./image-ref.mjs";
18
+ import { jobSizeDefaults } from "./job-size.mjs";
19
+ import { hostBudgetSettings } from "./host-budget.mjs";
18
20
  import { CONTAINER_ENV_NAMES, KEYLESS_ENV_NAME, RUNNER_ENV_NAMES } from "./reserved-env.mjs";
19
21
  import { modelListProblem } from "./model-ref.mjs";
20
22
  import { DOLLAR_ENV_NAMES, DOLLAR_WINDOW_KEYS, checkDollarInvariant, optionalUsdMicros } from "./money.mjs";
@@ -291,6 +293,34 @@ function refuseBackendShortfall(config) {
291
293
  if (first) throw configError(first);
292
294
  }
293
295
 
296
+ /**
297
+ * The deployment's default job size (issue #596): `{ memMiB, cpuCenti, memSet, cpuSet }` from `PI_JOB_MEMORY` and
298
+ * `PI_JOB_CPUS`, unset or empty meaning the built-in 4g and 2. A value the size parsers refuse (`job-size.mjs`) is a
299
+ * config error naming the key and the rule, so a typo stops the worker at boot rather than every job at its start.
300
+ */
301
+ export function jobSizeFrom(env) {
302
+ try {
303
+ return jobSizeDefaults(env);
304
+ } catch (error) {
305
+ throw configError(error.message);
306
+ }
307
+ }
308
+
309
+ /**
310
+ * The host budget's four settings (issue #596, phase 2): `PI_HOST_MEMORY_BUDGET`, `PI_HOST_CPU_BUDGET` (each `auto`, a
311
+ * value or `off`) and `PI_HOST_RESERVE_MEMORY`, `PI_HOST_RESERVE_CPUS` (each `auto` or a value), judged against the
312
+ * deployment's default job size so a budget below one default job is refused here. A bad value is a config error naming
313
+ * the key and the rule, at boot (exit 2), never a refusal per job. ENV ONLY in this release, never the settings overlay:
314
+ * a budget that moved under running jobs would strand the holds taken against the old one.
315
+ */
316
+ export function hostBudgetFrom(env, jobSize = jobSizeFrom(env)) {
317
+ try {
318
+ return hostBudgetSettings(env, jobSize);
319
+ } catch (error) {
320
+ throw configError(error.message);
321
+ }
322
+ }
323
+
294
324
  /**
295
325
  * PI_JOB_IMAGE as the worker runs it (issue #471): unset or empty is pi-job:latest (`||`, so "" falls back), and any
296
326
  * other value is judged by the one image rule `run.image` is (`image-ref.mjs`). A refused value is a config error at
@@ -396,6 +426,14 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
396
426
  dailyCostUsd: usdSetting(env, "PI_DAILY_COST_USD"),
397
427
  weeklyCostUsd: usdSetting(env, "PI_WEEKLY_COST_USD"),
398
428
  monthlyCostUsd: usdSetting(env, "PI_MONTHLY_COST_USD"),
429
+ // Issue #596: the deployment's default job size (`PI_JOB_MEMORY`, `PI_JOB_CPUS`; 4g and 2 when unset), refused here at
430
+ // boot (exit 2) naming the key, never at a job. A project row's size overrides it (INT-SCOPED-LIMITS-FILE-CONTRACT).
431
+ // The VALIDATION is this field's job; nothing reads its value. The pickup resolves each job's size from the same two
432
+ // settings (`jobSizeEnv`, start.mjs) with the same parser, so a value that got past here cannot be read otherwise.
433
+ jobSize: jobSizeFrom(env),
434
+ // Issue #596, phase 2: the host budget's settings (`hostBudgetFrom`), refused here at boot naming the key. The worker
435
+ // builds its one budget from them (start.mjs), and doctor reads them with the same function.
436
+ hostBudget: hostBudgetFrom(env),
399
437
  jobImage: jobImageFrom(env), // || (not ??) so an empty string falls back; "" is falsy and would throw inside buildDockerRunArgs AFTER a budget slot was reserved
400
438
  globalPiDir: resolveGlobalPiDir(env, fileExists), // REQ-GLOBAL-PI-OVERLAY: operator's ~/.pi/agent subset, :ro-mounted; null = off
401
439
  allowGlobalExtensions: globalExtensionsEnabled(env), // REQ-GLOBAL-PI-OVERLAY: ON unless PI_GLOBAL_ALLOW_EXTENSIONS=0
@@ -20,10 +20,13 @@
20
20
  * without reaching the argv builder. Nothing about the value changed in the move -- same fields, same
21
21
  * order, same guards -- because a byte-identical local deployment is #227's first constraint.
22
22
  *
23
- * It imports NOTHING, which is the property `packages.mjs` depends on when it derives the staged-packages
24
- * root from `CONTAINER_GLOBAL_PI_DIR` rather than re-typing the path.
23
+ * It imports nothing but `job-size.mjs` (issue #596), itself a leaf that imports nothing, which is the property
24
+ * `packages.mjs` depends on when it derives the staged-packages root from `CONTAINER_GLOBAL_PI_DIR` rather than
25
+ * re-typing the path.
25
26
  */
26
27
 
28
+ import { CGROUP_PARENT, DEFAULT_JOB_SIZE, PARENTLESS_SHARES_MAX, containerSizing } from "./job-size.mjs";
29
+
27
30
  /**
28
31
  * Where the operator's global pi overlay lands INSIDE the container (REQ-GLOBAL-PI-OVERLAY). Exported
29
32
  * because packages.mjs derives the staged-packages root from it: the mount and that root are ONE fact on
@@ -90,6 +93,13 @@ export function assertJobUser(user) {
90
93
  */
91
94
  export const USERNS_MODES = Object.freeze(["keep-id"]);
92
95
 
96
+ /**
97
+ * The two labels every job container carries (issue #596, phase 2): its memory size in MiB and its CPU size in
98
+ * hundredths, so `doctor` can sum what runs on this host and hold it against the host budget's ledger.
99
+ */
100
+ export const SIZE_LABEL_MEM = "pi.dispatch.mem";
101
+ export const SIZE_LABEL_CPU = "pi.dispatch.cpu";
102
+
93
103
  /** Throws unless `userns` is null/undefined or a member of `USERNS_MODES` (issue #354). */
94
104
  export function assertUserns(userns) {
95
105
  if (userns === null || userns === undefined) return;
@@ -123,7 +133,19 @@ export function assertUserns(userns) {
123
133
  * mounted /session:rw. Per-job, like jobDir -- never the shared store.
124
134
  * @param globalPiDir host path to the operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY); mounted /opt/pi-global:ro
125
135
  * @param name container name (for `docker stop` at the timeout)
126
- * @param memory e.g. "4g"; cpus e.g. "2"
136
+ * @param size the job's size, `{ memMiB, cpuCenti }` (issue #596, `job-size.mjs`); the built-in 4g and 2 when
137
+ * absent. It becomes `memory`, `memorySwap` (equal: no swap beyond memory), `cpuShares` and `shmSize`
138
+ * through ONE function, `containerSizing`, so no caller can pair a memory with another swap bound.
139
+ * @param hostCpus the runtime's own CPU count (`docker info` `NCPU`, `podman info` `host.cpus`), or null. It sets
140
+ * `cpus`, the host ceiling every job shares (`hostCpuCeiling`), never the job's own size.
141
+ * @param cpuBudgetCenti the host's CPU budget in hundredths (`host-budget.mjs`), or null when it is off or unknown. With
142
+ * one, `cpus` is the budget capped at the runtime's count (`cpuCeilingCenti`), so no job can use the
143
+ * CPUs `auto` reserved for the host; without, the phase 1 ceiling stands.
144
+ * @param cgroupParent the parent cgroup the container is started under (issue #596, phase 2): `CGROUP_PARENT` (the
145
+ * default, for every container this builder makes) or null where the runtime cannot take one
146
+ * (`cgroupParentFor`). Every job shares the parent, whose quota is the host's CPU budget, so the
147
+ * reserve holds across ALL jobs; with null the job's `cpuShares` are capped at `PARENTLESS_SHARES_MAX`
148
+ * so no job outweighs the egress proxy and Valkey. Anything else is refused.
127
149
  * @param network the per-job egress network this container joins (REQ-EGRESS-ALLOWLIST); null = the
128
150
  * docker default bridge, which is what every job did before that requirement existed
129
151
  * @param user "<uid>:<gid>" the job runs as (issue #341), or null for the image's own USER. Portable: it says WHO
@@ -148,8 +170,10 @@ export function containerSpec({
148
170
  sessionDir,
149
171
  globalPiDir,
150
172
  name,
151
- memory = "4g",
152
- cpus = "2",
173
+ size = DEFAULT_JOB_SIZE,
174
+ hostCpus = null,
175
+ cpuBudgetCenti = null,
176
+ cgroupParent = CGROUP_PARENT,
153
177
  network = null,
154
178
  user = null,
155
179
  userns = null,
@@ -164,6 +188,9 @@ export function containerSpec({
164
188
  assertCidFile(cidFile);
165
189
  if (!name) throw new Error("docker run: container name is required");
166
190
  if (!workspace) throw new Error("docker run: workspace mount is required");
191
+ // Throws on a size outside the floors and ceilings, before any field is built (issue #596).
192
+ const sizing = containerSizing(size, hostCpus, cpuBudgetCenti);
193
+ if (cgroupParent !== CGROUP_PARENT && cgroupParent !== null) throw new Error(`docker run: refusing a cgroup parent other than ${CGROUP_PARENT} or none: ${JSON.stringify(cgroupParent)}`);
167
194
  // Booleans, strictly: a truthy string from a caller that forwarded an option bag must not re-own host directories.
168
195
  if (typeof relabel !== "boolean") throw new Error(`docker run: relabel must be a boolean; got ${typeof relabel}`);
169
196
  if (typeof workspaceOwned !== "boolean") throw new Error(`docker run: workspaceOwned must be a boolean; got ${typeof workspaceOwned}`);
@@ -207,8 +234,20 @@ export function containerSpec({
207
234
  return {
208
235
  image,
209
236
  name,
210
- memory,
211
- cpus,
237
+ // Issue #596: the job's size. `memorySwap` equals `memory` (no swap beyond it), `cpuShares` is the job's weight,
238
+ // `shmSize` is min(1g, memory/2), and `cpus` is the host ceiling or null (then `--cpus` is absent).
239
+ memory: sizing.memory,
240
+ memorySwap: sizing.memorySwap,
241
+ cpus: sizing.cpus,
242
+ // Without the parent (issue #596, phase 2) a job's weight would compete with the proxy's and Valkey's directly, so it
243
+ // is capped at their default weight; under the parent it competes only with sibling jobs and orders them.
244
+ cpuShares: cgroupParent === null ? Math.min(sizing.cpuShares, PARENTLESS_SHARES_MAX) : sizing.cpuShares,
245
+ shmSize: sizing.shmSize,
246
+ cgroupParent,
247
+ // Issue #596, phase 2: the size the job was started at, as two labels on the container, so doctor can hold the host
248
+ // budget's ledger against what actually runs (`pi.dispatch.mem` in MiB, `pi.dispatch.cpu` in hundredths). Integers
249
+ // from the validated size, never anything a job or an operator writes as text.
250
+ labels: { [SIZE_LABEL_MEM]: String(size.memMiB), [SIZE_LABEL_CPU]: String(size.cpuCenti) },
212
251
  network,
213
252
  user,
214
253
  // Issue #354. Always present and `null` by default (the parameter default turns an `undefined` into it, so a spec
@@ -319,3 +358,27 @@ export function copyDowngrades(spec) {
319
358
  .map((b, i) => ({ container: b.container, was: b.readOnlyEnforcedBy, becomes: copied[i].readOnlyEnforcedBy }))
320
359
  .filter((d) => d.was !== null && d.was !== d.becomes);
321
360
  }
361
+
362
+ /**
363
+ * A container memory bound (`containerSpec`'s `memory`, the `--memory=` value: `4g`, `512m`, `1024k`, a plain byte
364
+ * count) in bytes, or null for anything else. Binary units, as Docker and Podman read them. Null is the safe answer for
365
+ * every caller: the live probe then reports the bound as not checked, and the OOM classification (issue #596) does not
366
+ * confirm a kill it cannot compare with the limit.
367
+ *
368
+ * Stricter than Docker's own parser (no fractions, no `t`), and that is safe only because every bound the worker
369
+ * passes is written by `formatMemory` from a size `parseMemory` accepted: `job-size.test.mjs` walks every such size
370
+ * through the argv and back through this function.
371
+ */
372
+ export function memoryBytes(memory) {
373
+ const m = /^(\d{1,15})([bkmg]?)$/i.exec(String(memory ?? ""));
374
+ if (!m) return null;
375
+ const bytes = Number(m[1]) * { "": 1, b: 1, k: 1024, m: 1024 ** 2, g: 1024 ** 3 }[m[2].toLowerCase()];
376
+ return Number.isSafeInteger(bytes) && bytes > 0 ? bytes : null;
377
+ }
378
+
379
+ /** The `--memory=` bound in a runtime argv, in bytes (`memoryBytes`), or null when the argv carries none. */
380
+ export function memoryBytesOfArgs(args) {
381
+ if (!Array.isArray(args)) return null;
382
+ const flag = args.find((a) => typeof a === "string" && a.startsWith("--memory="));
383
+ return flag === undefined ? null : memoryBytes(flag.slice("--memory=".length));
384
+ }