@edgehero/pi-dispatch 3.1.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +38 -0
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +599 -16
- package/src/host-budget.mjs +736 -0
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +2 -1
package/.env.example
CHANGED
|
@@ -113,6 +113,40 @@ VALKEY_URL=redis://127.0.0.1:6379
|
|
|
113
113
|
# docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)
|
|
114
114
|
# On rootful Podman, run these and every docker command through a docker context pointed at podman.sock, with the real docker CLI (docs/podman.md)
|
|
115
115
|
PI_JOB_IMAGE=pi-job:latest
|
|
116
|
+
# How much memory and CPU each job container gets, unless its project's row in the scoped-limits file sets its own.
|
|
117
|
+
# See docs/scoped-limits.md, and docs/sizing.md for how to choose sizes.
|
|
118
|
+
# Memory is a whole number of megabytes or gigabytes (512m, 1536m, 4g), at least 512m.
|
|
119
|
+
# CPUs is a number with at most two decimals (0.5, 2, 1.25), at least 0.25.
|
|
120
|
+
# A bad value stops the worker at boot.
|
|
121
|
+
# A job gets no swap beyond its memory where the runtime enforces swap limits (doctor warns where it does not).
|
|
122
|
+
# CPUs are a weight, not a cap: under contention a job with more CPUs gets more CPU than one with fewer.
|
|
123
|
+
# On an idle host one job may use every core up to the CPU budget below, whenever one is in force (with the
|
|
124
|
+
# default auto budget, every core but one on a host with 4 or more; with the CPU budget off, the same).
|
|
125
|
+
# Default 4g and 2.
|
|
126
|
+
# PI_JOB_MEMORY=4g
|
|
127
|
+
# PI_JOB_CPUS=2
|
|
128
|
+
# How much memory and CPU this host's jobs may hold together: the host budget. A job starts only when its size fits
|
|
129
|
+
# beside what already runs here, and a waiting job keeps its place, so a big job is not starved by a stream of small ones.
|
|
130
|
+
# PI_CONCURRENCY still caps the number of jobs; whichever is reached first applies. See docs/multi-host.md,
|
|
131
|
+
# and docs/sizing.md for a worked example.
|
|
132
|
+
# Each budget is auto, a value, or off (no limit on that resource).
|
|
133
|
+
# auto reads the container runtime: its memory and CPU count (and, on rootless Podman, the user service's own limits),
|
|
134
|
+
# minus the reserve below, and never less than one job of the default size above.
|
|
135
|
+
# A value is a memory amount (64g, 49152m) or a number of CPUs (12, 3.5). It is the budget itself; the reserve is not
|
|
136
|
+
# taken from it. A value below one job of the default size stops the worker at boot.
|
|
137
|
+
# A job's cpus count as CPU reserved for it (the runtime uses them as a weight), so they must fit the CPU budget.
|
|
138
|
+
# A job too big for this host's budget, or for its project's hostShare of it, is refused before anything is spent
|
|
139
|
+
# when it can run nowhere else: on this host's own queue, or on any queue without PI_WORKER_NAME (no fleet is
|
|
140
|
+
# declared, so it never waits for another host). With PI_WORKER_NAME set, a job on the shared queue waits for a
|
|
141
|
+
# host it fits on.
|
|
142
|
+
# Default auto.
|
|
143
|
+
# PI_HOST_MEMORY_BUDGET=auto
|
|
144
|
+
# PI_HOST_CPU_BUDGET=auto
|
|
145
|
+
# What auto leaves for the host itself (the system, the egress proxy, Valkey, the worker).
|
|
146
|
+
# auto: memory 10% of the host's memory, at least 1g and at most 4g; CPUs 1 when the host has 4 or more, else 0.
|
|
147
|
+
# Or a value (2g, 0.5, or 0 for none). A bad value stops the worker at boot.
|
|
148
|
+
# PI_HOST_RESERVE_MEMORY=auto
|
|
149
|
+
# PI_HOST_RESERVE_CPUS=auto
|
|
116
150
|
# where per-job /job inputs live (default: <OS temp dir>/pi-dispatch-<your uid>/jobs, one per account, created 0700)
|
|
117
151
|
# Another account's directory there, or one you set that another account owns, stops the worker at boot (exit 2), and doctor names it
|
|
118
152
|
# Pin it (and PI_SANDBOX_DIR) yourself if the worker runs as a different account than your /dispatch panel or `pi-dispatch sandbox`: the default is per account
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "4.0.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "The pi-dispatch worker and CLI: runs the pi coding agent as a self-hosted service, one locked down Docker or Podman container per job, with spend caps checked before anything is spent, plus init, up, doctor and service install.",
|
|
6
6
|
"keywords": [
|
|
@@ -63,6 +63,10 @@
|
|
|
63
63
|
"./backends": "./src/backends.mjs",
|
|
64
64
|
"./backend-registry": "./src/backend-registry.mjs",
|
|
65
65
|
"./container-spec": "./src/container-spec.mjs",
|
|
66
|
+
"./job-size": "./src/job-size.mjs",
|
|
67
|
+
"./size-suggest": "./src/size-suggest.mjs",
|
|
68
|
+
"./size-records": "./src/size-records.mjs",
|
|
69
|
+
"./host-budget": "./src/host-budget.mjs",
|
|
66
70
|
"./backend-conformance": "./src/backend-conformance.mjs",
|
|
67
71
|
"./live-probes": "./src/live-probes.mjs",
|
|
68
72
|
"./triggers": "./src/triggers.mjs",
|
|
@@ -105,7 +109,7 @@
|
|
|
105
109
|
"start": "node src/cli.mjs worker"
|
|
106
110
|
},
|
|
107
111
|
"dependencies": {
|
|
108
|
-
"@earendil-works/pi-ai": "1.0.
|
|
112
|
+
"@earendil-works/pi-ai": "1.0.4",
|
|
109
113
|
"@octokit/auth-app": "8.2.0",
|
|
110
114
|
"@octokit/rest": "22.0.1",
|
|
111
115
|
"bullmq": "5.80.4",
|
package/src/backend-local.mjs
CHANGED
|
@@ -28,6 +28,7 @@ import { BACKENDS, DEFAULT_BACKEND, DOCKER_NEVER_STARTED_EXITS } from "./backend
|
|
|
28
28
|
import { DEFAULT_EGRESS_PROXY, ENDPOINT_LISTED_STATES, networkEndpoints, removeNetworkOrSay } from "./egress.mjs";
|
|
29
29
|
import { makeDetachGate } from "./netns-keeper.mjs";
|
|
30
30
|
import { isDeterminateFsCode } from "./transient.mjs";
|
|
31
|
+
import { SIZE_LABEL_CPU, SIZE_LABEL_MEM } from "./container-spec.mjs";
|
|
31
32
|
|
|
32
33
|
const execDocker = promisify(execFile);
|
|
33
34
|
|
|
@@ -401,6 +402,74 @@ export function isJobNamespace(name) {
|
|
|
401
402
|
return typeof name === "string" && name.startsWith(JOB_NAME_PREFIX);
|
|
402
403
|
}
|
|
403
404
|
|
|
405
|
+
/**
|
|
406
|
+
* The states in which a job container runs nothing and never will again (gate round 1 of phase 2): Docker's
|
|
407
|
+
* `exited` and `dead`, Podman's `exited` and `stopped`. Such a container uses no memory and no CPU, yet `ps -a` lists
|
|
408
|
+
* it (an `--rm` whose removal failed), so counting it as running kept an orphan's hold until a worker restart.
|
|
409
|
+
*/
|
|
410
|
+
export const CONTAINER_GONE_STATES = Object.freeze(new Set(["exited", "dead", "stopped"]));
|
|
411
|
+
/**
|
|
412
|
+
* The states in which a job container has not started yet: Docker's `created`, Podman's `created` and `configured`. NOT
|
|
413
|
+
* gone by itself: a `docker run` client still alive after its stop timed out may yet start it. So it is REMOVED
|
|
414
|
+
* (`rm -f`, exact name, our namespace only), and gone once the removal is answered.
|
|
415
|
+
*/
|
|
416
|
+
export const CONTAINER_UNSTARTED_STATES = Object.freeze(new Set(["created", "configured"]));
|
|
417
|
+
|
|
418
|
+
/** `ps -a` lines of `{{.Names}}\t{{.State}}...`: `[{ name, state, rest }]`, only names in the job namespace, `rest` the later fields. */
|
|
419
|
+
function psRows(stdout) {
|
|
420
|
+
return String(stdout ?? "")
|
|
421
|
+
.split("\n")
|
|
422
|
+
.map((line) => line.split("\t").map((f) => f.trim()))
|
|
423
|
+
.filter(([name, state]) => isJobNamespace(name) && typeof state === "string")
|
|
424
|
+
.map(([name, state, ...rest]) => ({ name, state: state.toLowerCase(), rest }));
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* Whether a job container is GONE (issue #596, phase 2): `true` when the runtime lists no container of exactly that name,
|
|
429
|
+
* or lists it in a state that runs nothing (`CONTAINER_GONE_STATES`), or lists it not yet started and removes it
|
|
430
|
+
* (`CONTAINER_UNSTARTED_STATES`); `false` when it lists one in any other state (running, paused, restarting, removing,
|
|
431
|
+
* one nothing here has measured); `null` when it could not be asked or the removal was not answered. The host budget
|
|
432
|
+
* keeps the hold of a job whose stop did not take until this says `true` (`host-budget.mjs` `sweep`), so `null` keeps
|
|
433
|
+
* it: an unanswered listing must never free room a running container may still use. `-a` with the STATE, because a
|
|
434
|
+
* stopped container is listed by `ps -a` and is gone in every sense the budget cares about, and the anchored name test,
|
|
435
|
+
* because `--filter name=` is a SUBSTRING match (the reaper's measured reason).
|
|
436
|
+
*/
|
|
437
|
+
export function makeContainerGone({ exec = execReaperBounded, binOf = () => "docker" } = {}) {
|
|
438
|
+
return async (name, venue) => {
|
|
439
|
+
if (typeof name !== "string" || !isJobNamespace(name)) return null;
|
|
440
|
+
const bin = binOf(venue);
|
|
441
|
+
try {
|
|
442
|
+
const { stdout } = await exec(bin, ["ps", "-a", "--filter", `name=${name}`, "--format", "{{.Names}}\t{{.State}}"]);
|
|
443
|
+
const row = psRows(stdout).find((r) => r.name === name);
|
|
444
|
+
if (!row) return true;
|
|
445
|
+
if (CONTAINER_GONE_STATES.has(row.state)) return true;
|
|
446
|
+
if (!CONTAINER_UNSTARTED_STATES.has(row.state)) return false;
|
|
447
|
+
await exec(bin, ["rm", "-f", name]);
|
|
448
|
+
return true;
|
|
449
|
+
} catch {
|
|
450
|
+
return null;
|
|
451
|
+
}
|
|
452
|
+
};
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
/**
|
|
456
|
+
* The job containers a venue still lists (gate round 1 of phase 2), for the host budget's boot seed: `ps -a`
|
|
457
|
+
* by the job namespace, each with its state and its two size labels, as `[{ name, memMiB, cpuCenti }]` (a label absent
|
|
458
|
+
* or not a positive integer is null). A container in a state that runs nothing (`CONTAINER_GONE_STATES`) is left out:
|
|
459
|
+
* it holds nothing. One not yet started is kept, so the sweep removes it before its room is given back. THROWS when
|
|
460
|
+
* the runtime does not answer: the budget then admits nothing until a listing is read, because an empty ledger beside
|
|
461
|
+
* containers nobody counted is an overcommit.
|
|
462
|
+
*/
|
|
463
|
+
export function makeJobContainerLister({ exec = execReaperBounded, bin = "docker" } = {}) {
|
|
464
|
+
const int = (v) => (/^[1-9][0-9]{0,8}$/.test(v ?? "") ? Number(v) : null);
|
|
465
|
+
return async () => {
|
|
466
|
+
const { stdout } = await exec(bin, ["ps", "-a", "--filter", `name=${JOB_NAME_PREFIX}`, "--format", `{{.Names}}\t{{.State}}\t{{.Label "${SIZE_LABEL_MEM}"}}\t{{.Label "${SIZE_LABEL_CPU}"}}`]);
|
|
467
|
+
return psRows(stdout)
|
|
468
|
+
.filter((r) => !CONTAINER_GONE_STATES.has(r.state))
|
|
469
|
+
.map((r) => ({ name: r.name, memMiB: int(r.rest[0]), cpuCenti: int(r.rest[1]) }));
|
|
470
|
+
};
|
|
471
|
+
}
|
|
472
|
+
|
|
404
473
|
/**
|
|
405
474
|
* Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind.
|
|
406
475
|
*
|
package/src/backend-podman.mjs
CHANGED
|
@@ -34,10 +34,11 @@ import { JOB_NAME_PREFIX, execDockerBounded, jobContainerName, makeReaper, makeS
|
|
|
34
34
|
import { BACKENDS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_CONF_WIDENS_JOB, PODMAN_NETWORK_HELPER_KEYS, PODMAN_WIDENING_KEYS, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
|
|
35
35
|
import { CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
|
|
36
36
|
import { buildPodmanRunArgs } from "./docker-run.mjs";
|
|
37
|
+
import { cgroupParentFor } from "./cpu-reserve.mjs";
|
|
37
38
|
import { DEFAULT_EGRESS_PROXY, makeEgressPreflight } from "./egress.mjs";
|
|
38
39
|
import { NETNS_KEEPER, NETNS_KEEPER_FORMAT, QUADLET_FILES, STARTED_AT_FORMAT, judgeNetnsKeeper, netnsKeeperRemedy, podmanNeedsNetnsKeeper } from "./podman-stack.mjs";
|
|
39
40
|
import { makeImagePreflight } from "./image-preflight.mjs";
|
|
40
|
-
import { DAEMON_FACTS_TIMEOUT_MS } from "./job-user.mjs";
|
|
41
|
+
import { DAEMON_FACTS_TIMEOUT_MS, JOB_USER_FACTS_MAX_AGE_MS, STALE_FACTS_CEILING_MS } from "./job-user.mjs";
|
|
41
42
|
import { PODMAN_INFO_ARGS, parsePodmanInfo } from "./daemon-facts.mjs";
|
|
42
43
|
|
|
43
44
|
// Moved to the leaf `daemon-facts.mjs` (issue #452, gate round 3) and re-exported, so every importer keeps its path.
|
|
@@ -106,25 +107,50 @@ export function makePodmanInfoReader({ run = (args) => execDockerBounded(args, {
|
|
|
106
107
|
const CACHED = Symbol("podman-info-cached");
|
|
107
108
|
|
|
108
109
|
/**
|
|
109
|
-
* `readInfo` with its
|
|
110
|
-
* kept: a Podman that timed out once must be asked again, or every later job would retry on a stale
|
|
111
|
-
* so the boot wiring can wrap a reader once and hand the same one to the bundle and to its own boot
|
|
110
|
+
* `readInfo` with its last ANSWERED result kept for `maxAgeMs`, and concurrent callers sharing one read. An unanswered
|
|
111
|
+
* read is never kept: a Podman that timed out once must be asked again, or every later job would retry on a stale
|
|
112
|
+
* failure. Idempotent, so the boot wiring can wrap a reader once and hand the same one to the bundle and to its own boot
|
|
113
|
+
* read (a second wrap keeps the first wrap's clock and age).
|
|
114
|
+
*
|
|
115
|
+
* The age (issue #596, gate round 2) is the job-user resolver's (`JOB_USER_FACTS_MAX_AGE_MS`), for the same reason:
|
|
116
|
+
* this read carries the host's CPU count that every job's `--cpus` ceiling is built from, and a raised or lowered count
|
|
117
|
+
* must be seen without a restart. STALE WHILE ERROR, also as there: a read past the age that does not answer keeps
|
|
118
|
+
* serving the kept answer until a read does, logging `podman_info_stale` with the failed read's reason once per run of
|
|
119
|
+
* failures, because the age exists to see a change and must not turn one slow `podman info` into a failed pickup.
|
|
120
|
+
* Podman has no `invalidate`: it accepts a `--cpus` above the host's count (4.9.3 and 5.8.1, measured), so no job
|
|
121
|
+
* refusal proves the kept count wrong.
|
|
112
122
|
*/
|
|
113
|
-
export function cachedPodmanInfo(readInfo) {
|
|
123
|
+
export function cachedPodmanInfo(readInfo, { now = Date.now, maxAgeMs = JOB_USER_FACTS_MAX_AGE_MS, log = () => {} } = {}) {
|
|
114
124
|
if (readInfo?.[CACHED]) return readInfo;
|
|
115
125
|
let kept = null;
|
|
126
|
+
let keptAt = 0;
|
|
127
|
+
let staleSaid = false;
|
|
116
128
|
let inFlight = null;
|
|
117
129
|
const cached = async () => {
|
|
118
|
-
if (kept) return kept;
|
|
130
|
+
if (kept && now() - keptAt < maxAgeMs) return kept;
|
|
119
131
|
if (inFlight) return inFlight;
|
|
120
132
|
inFlight = (async () => {
|
|
133
|
+
let read;
|
|
121
134
|
try {
|
|
122
|
-
|
|
123
|
-
if (read?.answered === true && read.info) kept = read;
|
|
124
|
-
return read;
|
|
135
|
+
read = await readInfo();
|
|
125
136
|
} catch {
|
|
126
|
-
|
|
137
|
+
read = { answered: false, reason: "spawn-failed", transient: true };
|
|
138
|
+
}
|
|
139
|
+
if (read?.answered === true && read.info) {
|
|
140
|
+
kept = read;
|
|
141
|
+
keptAt = now();
|
|
142
|
+
staleSaid = false;
|
|
143
|
+
return read;
|
|
144
|
+
}
|
|
145
|
+
// Not past `STALE_FACTS_CEILING_MS` (job-user.mjs says why): then the failed read is the answer, as a first one is.
|
|
146
|
+
if (kept && now() - keptAt < STALE_FACTS_CEILING_MS) {
|
|
147
|
+
if (!staleSaid) {
|
|
148
|
+
staleSaid = true;
|
|
149
|
+
log("podman_info_stale", { reason: read?.reason ?? "unanswered", ageMs: now() - keptAt });
|
|
150
|
+
}
|
|
151
|
+
return kept;
|
|
127
152
|
}
|
|
153
|
+
return read;
|
|
128
154
|
})();
|
|
129
155
|
try {
|
|
130
156
|
return await inFlight;
|
|
@@ -1081,7 +1107,7 @@ export function makePodmanBackend(opts = {}) {
|
|
|
1081
1107
|
if (reap !== undefined && typeof reap !== "function") throw new Error(`backend "${PODMAN_BACKEND}": reap must be a function (makePodmanReaper)`);
|
|
1082
1108
|
const spawnSeam = spawnFn ? { spawnFn } : {};
|
|
1083
1109
|
const execSeam = exec ? { exec } : spawnFn ? { exec: execViaSpawn(spawnFn) } : {};
|
|
1084
|
-
const info = cachedPodmanInfo(readInfo);
|
|
1110
|
+
const info = cachedPodmanInfo(readInfo, { log });
|
|
1085
1111
|
|
|
1086
1112
|
const runContainer = makeRunContainerFn({
|
|
1087
1113
|
image,
|
|
@@ -1098,7 +1124,9 @@ export function makePodmanBackend(opts = {}) {
|
|
|
1098
1124
|
forgeHosts,
|
|
1099
1125
|
neverStartedExits: PODMAN_NEVER_STARTED_EXITS,
|
|
1100
1126
|
bin: "podman",
|
|
1101
|
-
|
|
1127
|
+
// Issue #596, phase 2: every job under the one parent cgroup (`CGROUP_PARENT`), except where Podman reports a cgroup
|
|
1128
|
+
// manager other than systemd (`cgroupParentFor` says why), from the SAME `podman info` this job was admitted on.
|
|
1129
|
+
buildArgs: (opts) => buildPodmanRunArgs({ ...opts, cgroupParent: cgroupParentFor({ podman: true, cgroupManager: info.peek?.()?.info?.cgroupManager ?? null }) }),
|
|
1102
1130
|
// Issue #452, gate round 4: the teardown's detach gate uses the `podman info` this venue admitted jobs on, never a
|
|
1103
1131
|
// read of its own; before any answered read it falls back to one. A refused teardown is logged with its token.
|
|
1104
1132
|
teardownRuntime: () => {
|
|
@@ -1145,7 +1173,10 @@ export function makePodmanBackend(opts = {}) {
|
|
|
1145
1173
|
// Issue #429: the store this job's container lives in rides beside its user, so a retained run records it and a
|
|
1146
1174
|
// sandbox or the retention sweep can tell another store's empty answer from "not open".
|
|
1147
1175
|
const store = read?.answered === true ? read.info?.graphRoot : null;
|
|
1148
|
-
|
|
1176
|
+
// Issue #596: the host's CPU count from the same read, for the job's `--cpus` ceiling. Absent when Podman did not say.
|
|
1177
|
+
const hostCpus = read?.answered === true ? read.info?.hostCpus : null;
|
|
1178
|
+
const stored = chosen.user && typeof store === "string" ? { ...chosen, store } : chosen;
|
|
1179
|
+
return chosen.user && Number.isSafeInteger(hostCpus) ? { ...stored, hostCpus } : stored;
|
|
1149
1180
|
};
|
|
1150
1181
|
|
|
1151
1182
|
return {
|
package/src/config.mjs
CHANGED
|
@@ -15,6 +15,8 @@ import { SWEEP_INTERVAL_HOURS, SWEEP_INTERVAL_MAX_HOURS } from "./retention-swee
|
|
|
15
15
|
import { parseSecretProfiles } from "./secret-profiles.mjs";
|
|
16
16
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, parseWaitProfiles } from "./wait-for.mjs";
|
|
17
17
|
import { imageRefProblem } from "./image-ref.mjs";
|
|
18
|
+
import { jobSizeDefaults } from "./job-size.mjs";
|
|
19
|
+
import { hostBudgetSettings } from "./host-budget.mjs";
|
|
18
20
|
import { CONTAINER_ENV_NAMES, KEYLESS_ENV_NAME, RUNNER_ENV_NAMES } from "./reserved-env.mjs";
|
|
19
21
|
import { modelListProblem } from "./model-ref.mjs";
|
|
20
22
|
import { DOLLAR_ENV_NAMES, DOLLAR_WINDOW_KEYS, checkDollarInvariant, optionalUsdMicros } from "./money.mjs";
|
|
@@ -291,6 +293,34 @@ function refuseBackendShortfall(config) {
|
|
|
291
293
|
if (first) throw configError(first);
|
|
292
294
|
}
|
|
293
295
|
|
|
296
|
+
/**
|
|
297
|
+
* The deployment's default job size (issue #596): `{ memMiB, cpuCenti, memSet, cpuSet }` from `PI_JOB_MEMORY` and
|
|
298
|
+
* `PI_JOB_CPUS`, unset or empty meaning the built-in 4g and 2. A value the size parsers refuse (`job-size.mjs`) is a
|
|
299
|
+
* config error naming the key and the rule, so a typo stops the worker at boot rather than every job at its start.
|
|
300
|
+
*/
|
|
301
|
+
export function jobSizeFrom(env) {
|
|
302
|
+
try {
|
|
303
|
+
return jobSizeDefaults(env);
|
|
304
|
+
} catch (error) {
|
|
305
|
+
throw configError(error.message);
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* The host budget's four settings (issue #596, phase 2): `PI_HOST_MEMORY_BUDGET`, `PI_HOST_CPU_BUDGET` (each `auto`, a
|
|
311
|
+
* value or `off`) and `PI_HOST_RESERVE_MEMORY`, `PI_HOST_RESERVE_CPUS` (each `auto` or a value), judged against the
|
|
312
|
+
* deployment's default job size so a budget below one default job is refused here. A bad value is a config error naming
|
|
313
|
+
* the key and the rule, at boot (exit 2), never a refusal per job. ENV ONLY in this release, never the settings overlay:
|
|
314
|
+
* a budget that moved under running jobs would strand the holds taken against the old one.
|
|
315
|
+
*/
|
|
316
|
+
export function hostBudgetFrom(env, jobSize = jobSizeFrom(env)) {
|
|
317
|
+
try {
|
|
318
|
+
return hostBudgetSettings(env, jobSize);
|
|
319
|
+
} catch (error) {
|
|
320
|
+
throw configError(error.message);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
|
|
294
324
|
/**
|
|
295
325
|
* PI_JOB_IMAGE as the worker runs it (issue #471): unset or empty is pi-job:latest (`||`, so "" falls back), and any
|
|
296
326
|
* other value is judged by the one image rule `run.image` is (`image-ref.mjs`). A refused value is a config error at
|
|
@@ -396,6 +426,14 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
|
|
|
396
426
|
dailyCostUsd: usdSetting(env, "PI_DAILY_COST_USD"),
|
|
397
427
|
weeklyCostUsd: usdSetting(env, "PI_WEEKLY_COST_USD"),
|
|
398
428
|
monthlyCostUsd: usdSetting(env, "PI_MONTHLY_COST_USD"),
|
|
429
|
+
// Issue #596: the deployment's default job size (`PI_JOB_MEMORY`, `PI_JOB_CPUS`; 4g and 2 when unset), refused here at
|
|
430
|
+
// boot (exit 2) naming the key, never at a job. A project row's size overrides it (INT-SCOPED-LIMITS-FILE-CONTRACT).
|
|
431
|
+
// The VALIDATION is this field's job; nothing reads its value. The pickup resolves each job's size from the same two
|
|
432
|
+
// settings (`jobSizeEnv`, start.mjs) with the same parser, so a value that got past here cannot be read otherwise.
|
|
433
|
+
jobSize: jobSizeFrom(env),
|
|
434
|
+
// Issue #596, phase 2: the host budget's settings (`hostBudgetFrom`), refused here at boot naming the key. The worker
|
|
435
|
+
// builds its one budget from them (start.mjs), and doctor reads them with the same function.
|
|
436
|
+
hostBudget: hostBudgetFrom(env),
|
|
399
437
|
jobImage: jobImageFrom(env), // || (not ??) so an empty string falls back; "" is falsy and would throw inside buildDockerRunArgs AFTER a budget slot was reserved
|
|
400
438
|
globalPiDir: resolveGlobalPiDir(env, fileExists), // REQ-GLOBAL-PI-OVERLAY: operator's ~/.pi/agent subset, :ro-mounted; null = off
|
|
401
439
|
allowGlobalExtensions: globalExtensionsEnabled(env), // REQ-GLOBAL-PI-OVERLAY: ON unless PI_GLOBAL_ALLOW_EXTENSIONS=0
|
package/src/container-spec.mjs
CHANGED
|
@@ -20,10 +20,13 @@
|
|
|
20
20
|
* without reaching the argv builder. Nothing about the value changed in the move -- same fields, same
|
|
21
21
|
* order, same guards -- because a byte-identical local deployment is #227's first constraint.
|
|
22
22
|
*
|
|
23
|
-
* It imports
|
|
24
|
-
* root from `CONTAINER_GLOBAL_PI_DIR` rather than
|
|
23
|
+
* It imports nothing but `job-size.mjs` (issue #596), itself a leaf that imports nothing, which is the property
|
|
24
|
+
* `packages.mjs` depends on when it derives the staged-packages root from `CONTAINER_GLOBAL_PI_DIR` rather than
|
|
25
|
+
* re-typing the path.
|
|
25
26
|
*/
|
|
26
27
|
|
|
28
|
+
import { CGROUP_PARENT, DEFAULT_JOB_SIZE, PARENTLESS_SHARES_MAX, containerSizing } from "./job-size.mjs";
|
|
29
|
+
|
|
27
30
|
/**
|
|
28
31
|
* Where the operator's global pi overlay lands INSIDE the container (REQ-GLOBAL-PI-OVERLAY). Exported
|
|
29
32
|
* because packages.mjs derives the staged-packages root from it: the mount and that root are ONE fact on
|
|
@@ -90,6 +93,13 @@ export function assertJobUser(user) {
|
|
|
90
93
|
*/
|
|
91
94
|
export const USERNS_MODES = Object.freeze(["keep-id"]);
|
|
92
95
|
|
|
96
|
+
/**
|
|
97
|
+
* The two labels every job container carries (issue #596, phase 2): its memory size in MiB and its CPU size in
|
|
98
|
+
* hundredths, so `doctor` can sum what runs on this host and hold it against the host budget's ledger.
|
|
99
|
+
*/
|
|
100
|
+
export const SIZE_LABEL_MEM = "pi.dispatch.mem";
|
|
101
|
+
export const SIZE_LABEL_CPU = "pi.dispatch.cpu";
|
|
102
|
+
|
|
93
103
|
/** Throws unless `userns` is null/undefined or a member of `USERNS_MODES` (issue #354). */
|
|
94
104
|
export function assertUserns(userns) {
|
|
95
105
|
if (userns === null || userns === undefined) return;
|
|
@@ -123,7 +133,19 @@ export function assertUserns(userns) {
|
|
|
123
133
|
* mounted /session:rw. Per-job, like jobDir -- never the shared store.
|
|
124
134
|
* @param globalPiDir host path to the operator's global pi overlay (REQ-GLOBAL-PI-OVERLAY); mounted /opt/pi-global:ro
|
|
125
135
|
* @param name container name (for `docker stop` at the timeout)
|
|
126
|
-
* @param
|
|
136
|
+
* @param size the job's size, `{ memMiB, cpuCenti }` (issue #596, `job-size.mjs`); the built-in 4g and 2 when
|
|
137
|
+
* absent. It becomes `memory`, `memorySwap` (equal: no swap beyond memory), `cpuShares` and `shmSize`
|
|
138
|
+
* through ONE function, `containerSizing`, so no caller can pair a memory with another swap bound.
|
|
139
|
+
* @param hostCpus the runtime's own CPU count (`docker info` `NCPU`, `podman info` `host.cpus`), or null. It sets
|
|
140
|
+
* `cpus`, the host ceiling every job shares (`hostCpuCeiling`), never the job's own size.
|
|
141
|
+
* @param cpuBudgetCenti the host's CPU budget in hundredths (`host-budget.mjs`), or null when it is off or unknown. With
|
|
142
|
+
* one, `cpus` is the budget capped at the runtime's count (`cpuCeilingCenti`), so no job can use the
|
|
143
|
+
* CPUs `auto` reserved for the host; without, the phase 1 ceiling stands.
|
|
144
|
+
* @param cgroupParent the parent cgroup the container is started under (issue #596, phase 2): `CGROUP_PARENT` (the
|
|
145
|
+
* default, for every container this builder makes) or null where the runtime cannot take one
|
|
146
|
+
* (`cgroupParentFor`). Every job shares the parent, whose quota is the host's CPU budget, so the
|
|
147
|
+
* reserve holds across ALL jobs; with null the job's `cpuShares` are capped at `PARENTLESS_SHARES_MAX`
|
|
148
|
+
* so no job outweighs the egress proxy and Valkey. Anything else is refused.
|
|
127
149
|
* @param network the per-job egress network this container joins (REQ-EGRESS-ALLOWLIST); null = the
|
|
128
150
|
* docker default bridge, which is what every job did before that requirement existed
|
|
129
151
|
* @param user "<uid>:<gid>" the job runs as (issue #341), or null for the image's own USER. Portable: it says WHO
|
|
@@ -148,8 +170,10 @@ export function containerSpec({
|
|
|
148
170
|
sessionDir,
|
|
149
171
|
globalPiDir,
|
|
150
172
|
name,
|
|
151
|
-
|
|
152
|
-
|
|
173
|
+
size = DEFAULT_JOB_SIZE,
|
|
174
|
+
hostCpus = null,
|
|
175
|
+
cpuBudgetCenti = null,
|
|
176
|
+
cgroupParent = CGROUP_PARENT,
|
|
153
177
|
network = null,
|
|
154
178
|
user = null,
|
|
155
179
|
userns = null,
|
|
@@ -164,6 +188,9 @@ export function containerSpec({
|
|
|
164
188
|
assertCidFile(cidFile);
|
|
165
189
|
if (!name) throw new Error("docker run: container name is required");
|
|
166
190
|
if (!workspace) throw new Error("docker run: workspace mount is required");
|
|
191
|
+
// Throws on a size outside the floors and ceilings, before any field is built (issue #596).
|
|
192
|
+
const sizing = containerSizing(size, hostCpus, cpuBudgetCenti);
|
|
193
|
+
if (cgroupParent !== CGROUP_PARENT && cgroupParent !== null) throw new Error(`docker run: refusing a cgroup parent other than ${CGROUP_PARENT} or none: ${JSON.stringify(cgroupParent)}`);
|
|
167
194
|
// Booleans, strictly: a truthy string from a caller that forwarded an option bag must not re-own host directories.
|
|
168
195
|
if (typeof relabel !== "boolean") throw new Error(`docker run: relabel must be a boolean; got ${typeof relabel}`);
|
|
169
196
|
if (typeof workspaceOwned !== "boolean") throw new Error(`docker run: workspaceOwned must be a boolean; got ${typeof workspaceOwned}`);
|
|
@@ -207,8 +234,20 @@ export function containerSpec({
|
|
|
207
234
|
return {
|
|
208
235
|
image,
|
|
209
236
|
name,
|
|
210
|
-
memory,
|
|
211
|
-
cpus
|
|
237
|
+
// Issue #596: the job's size. `memorySwap` equals `memory` (no swap beyond it), `cpuShares` is the job's weight,
|
|
238
|
+
// `shmSize` is min(1g, memory/2), and `cpus` is the host ceiling or null (then `--cpus` is absent).
|
|
239
|
+
memory: sizing.memory,
|
|
240
|
+
memorySwap: sizing.memorySwap,
|
|
241
|
+
cpus: sizing.cpus,
|
|
242
|
+
// Without the parent (issue #596, phase 2) a job's weight would compete with the proxy's and Valkey's directly, so it
|
|
243
|
+
// is capped at their default weight; under the parent it competes only with sibling jobs and orders them.
|
|
244
|
+
cpuShares: cgroupParent === null ? Math.min(sizing.cpuShares, PARENTLESS_SHARES_MAX) : sizing.cpuShares,
|
|
245
|
+
shmSize: sizing.shmSize,
|
|
246
|
+
cgroupParent,
|
|
247
|
+
// Issue #596, phase 2: the size the job was started at, as two labels on the container, so doctor can hold the host
|
|
248
|
+
// budget's ledger against what actually runs (`pi.dispatch.mem` in MiB, `pi.dispatch.cpu` in hundredths). Integers
|
|
249
|
+
// from the validated size, never anything a job or an operator writes as text.
|
|
250
|
+
labels: { [SIZE_LABEL_MEM]: String(size.memMiB), [SIZE_LABEL_CPU]: String(size.cpuCenti) },
|
|
212
251
|
network,
|
|
213
252
|
user,
|
|
214
253
|
// Issue #354. Always present and `null` by default (the parameter default turns an `undefined` into it, so a spec
|
|
@@ -319,3 +358,27 @@ export function copyDowngrades(spec) {
|
|
|
319
358
|
.map((b, i) => ({ container: b.container, was: b.readOnlyEnforcedBy, becomes: copied[i].readOnlyEnforcedBy }))
|
|
320
359
|
.filter((d) => d.was !== null && d.was !== d.becomes);
|
|
321
360
|
}
|
|
361
|
+
|
|
362
|
+
/**
|
|
363
|
+
* A container memory bound (`containerSpec`'s `memory`, the `--memory=` value: `4g`, `512m`, `1024k`, a plain byte
|
|
364
|
+
* count) in bytes, or null for anything else. Binary units, as Docker and Podman read them. Null is the safe answer for
|
|
365
|
+
* every caller: the live probe then reports the bound as not checked, and the OOM classification (issue #596) does not
|
|
366
|
+
* confirm a kill it cannot compare with the limit.
|
|
367
|
+
*
|
|
368
|
+
* Stricter than Docker's own parser (no fractions, no `t`), and that is safe only because every bound the worker
|
|
369
|
+
* passes is written by `formatMemory` from a size `parseMemory` accepted: `job-size.test.mjs` walks every such size
|
|
370
|
+
* through the argv and back through this function.
|
|
371
|
+
*/
|
|
372
|
+
export function memoryBytes(memory) {
|
|
373
|
+
const m = /^(\d{1,15})([bkmg]?)$/i.exec(String(memory ?? ""));
|
|
374
|
+
if (!m) return null;
|
|
375
|
+
const bytes = Number(m[1]) * { "": 1, b: 1, k: 1024, m: 1024 ** 2, g: 1024 ** 3 }[m[2].toLowerCase()];
|
|
376
|
+
return Number.isSafeInteger(bytes) && bytes > 0 ? bytes : null;
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
/** The `--memory=` bound in a runtime argv, in bytes (`memoryBytes`), or null when the argv carries none. */
|
|
380
|
+
export function memoryBytesOfArgs(args) {
|
|
381
|
+
if (!Array.isArray(args)) return null;
|
|
382
|
+
const flag = args.find((a) => typeof a === "string" && a.startsWith("--memory="));
|
|
383
|
+
return flag === undefined ? null : memoryBytes(flag.slice("--memory=".length));
|
|
384
|
+
}
|