@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,8 @@ import { spawn } from "node:child_process";
2
2
  import { randomBytes } from "node:crypto";
3
3
  import { readFileSync, rmSync } from "node:fs";
4
4
  import { DOCKER_NEVER_STARTED_EXITS } from "./backends.mjs";
5
- import { CONTAINER_HOME } from "./container-spec.mjs";
5
+ import { CONTAINER_HOME, memoryBytesOfArgs } from "./container-spec.mjs";
6
+ import { cpuCeilingCenti, cpuRangeRefusal, DEFAULT_JOB_SIZE } from "./job-size.mjs";
6
7
  import { buildDockerRunArgs, CONTAINER_SESSION_FILE, insideDir } from "./docker-run.mjs";
7
8
  import { createJobNetwork, networkNameFor, removeJobNetwork } from "./egress.mjs";
8
9
  import { buildContainerEnv } from "./env-allowlist.mjs";
@@ -11,7 +12,7 @@ import { InfraRetry } from "./processor.mjs";
11
12
 
12
13
  /**
13
14
  * The real `runContainer` the processor injects. Launches one job container and returns
14
- * `{ code, aborted, turns, tokens, session, usage, context, exitReason }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
15
+ * `{ code, aborted, turns, tokens, session, usage, context, exitReason, resources?, exitOomKilled?, memoryLimit? }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
15
16
  * the 30-min timeout or graceful shutdown), which the processor classifies as POLICY (no retry) per
16
17
  * INT-RUNNER-EXIT-CODE-PROTOCOL. The numeric `code` alone cannot say this: a worker SIGKILL and a
17
18
  * kernel OOM both surface as 137, so the abort FLAG -- not the code -- is the discriminator.
@@ -64,6 +65,10 @@ export function makeRunContainer({
64
65
  log = () => {},
65
66
  // Issue #545: the per-job exit-line key, 32 random bytes as hex. A seam so a test can name the key it verifies with.
66
67
  mintExitKey = () => randomBytes(32).toString("hex"),
68
+ // Issue #596: called when the runtime refused this job's `--cpus` as above its CPU count (`cpu_ceiling_stale`), so
69
+ // the wiring drops the cached runtime facts the ceiling came from and the next pickup reads them again. The local
70
+ // venue's is the job-user resolver's `invalidate`; a no-op elsewhere (Podman accepts any `--cpus`, measured).
71
+ onCpuCeilingStale = () => {},
67
72
  }) {
68
73
  // async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
69
74
  // a rejection, uniformly awaitable by the processor and by tests.
@@ -75,7 +80,15 @@ export function makeRunContainer({
75
80
  // `exitAuth` (issue #545) is the processor's, off the image preflight: true when the job image declares `exitAuth`, so
76
81
  // its runner reads a key from stdin and signs its exit line with it. Defaults off, so a caller that predates it, and
77
82
  // every image that does not declare it, runs exactly as before and is read exactly as before.
78
- return async function runContainer({ job, token, prepared, secrets = {}, name, signal, user = null, home = null, relabel = false, modelEndpoints = null, exitAuth = false }) {
83
+ // `size` (issue #596) is the job's `{ memMiB, cpuCenti, source }`, resolved at pickup from the limits snapshot and
84
+ // handed here as an argument, never through `job.data`; absent, the built-in 4g and 2. `hostCpus` is the runtime's own
85
+ // CPU count from the job user's facts read, which sets the `--cpus` ceiling; null leaves `--cpus` off (fails open, and
86
+ // says so in `cpu_ceiling_unknown`). `cpuBudgetCenti` (phase 2) is the host's CPU budget at pickup, which replaces that
87
+ // ceiling while it is in force (`cpuCeilingCenti`).
88
+ // `unenforced` (issue #596) is the size flags this runtime said it drops (`unenforcedSizeFlags`: `--memory-swap` where
89
+ // Docker reports SwapLimit false, `--cpu-shares` where it reports CPUShares false), logged per job as
90
+ // `size_bound_unenforced` so a run past its size's bound is named where it happens. The flags stay on the argv.
91
+ return async function runContainer({ job, token, prepared, secrets = {}, name, signal, user = null, home = null, relabel = false, modelEndpoints = null, exitAuth = false, size = DEFAULT_JOB_SIZE, hostCpus = null, cpuBudgetCenti = null, unenforced = [] }) {
79
92
  if (signal?.aborted) return { code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null }; // killed before it could start
80
93
  // Minted per attempt, held in this closure and the sink's, and handed to the container on its stdin only: never in
81
94
  // the argv (a host `ps` shows it), never in the env (the container's /proc/1/environ shows it), never logged.
@@ -175,6 +188,9 @@ export function makeRunContainer({
175
188
  sessionDir: prepared.session?.hostDir,
176
189
  globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
177
190
  name,
191
+ size, // issue #596: memory, swap equal to it, the CPU weight and /dev/shm, through containerSpec's one function
192
+ hostCpus, // issue #596: the `--cpus` ceiling every job shares; null leaves the flag off
193
+ cpuBudgetCenti, // issue #596, phase 2: the host's CPU budget, the ceiling while it is in force
178
194
  network, // REQ-EGRESS-ALLOWLIST: null when no policy is armed, and the flag is then absent
179
195
  user, // issue #341: the worker's own "<uid>:<gid>" on a daemon that enforces bind-mount ownership, else null
180
196
  cidFile, // issue #345: where the CLI writes this attempt's container ID, read below when the run exits "never started"
@@ -190,6 +206,13 @@ export function makeRunContainer({
190
206
  ...(exitKey !== null ? { extraFlags: ["-i"] } : {}),
191
207
  });
192
208
 
209
+ // Issue #596: the memory bound this container actually got, in bytes, read off the argv the runtime is handed (both
210
+ // builders spell it `--memory=`, from the job's size), never re-derived from the environment or the size. The
211
+ // processor confirms an OOM only when the run's peak reached 90% of it. null when the argv names none.
212
+ const memoryLimit = memoryBytesOfArgs(args);
213
+ if (!args.some((a) => typeof a === "string" && a.startsWith("--cpus="))) log("cpu_ceiling_unknown", { container: name });
214
+ if (Array.isArray(unenforced) && unenforced.length > 0) log("size_bound_unenforced", { container: name, flags: unenforced.filter((f) => typeof f === "string" && /^--[a-z-]{1,32}$/.test(f)) });
215
+
193
216
  // REQ-EGRESS-ALLOWLIST. This job's own --internal network, created here rather than at boot because
194
217
  // it holds exactly two endpoints -- this container and the proxy -- and that is what makes job-to-job
195
218
  // traffic structurally impossible rather than merely discouraged. A shared network could not do it:
@@ -205,6 +228,10 @@ export function makeRunContainer({
205
228
  // Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
206
229
  // sanitizes internally. No container mount, no env var -- the sink lives on this side only.
207
230
  const sink = openJobLog(name, { exitKey });
231
+ // Issue #596: the head of the CLI's stderr, kept only to recognise the runtime's refusal of a stale `--cpus`
232
+ // (`cpuRangeRefusal`). Bounded, and never logged: only the count it names leaves this function.
233
+ let stderrHead = "";
234
+ const STDERR_HEAD_MAX = 4096;
208
235
 
209
236
  const run = new Promise((resolve, reject) => {
210
237
  const child = spawnFn(bin, args, { stdio: [exitKey !== null ? "pipe" : "ignore", "pipe", "pipe"] });
@@ -223,7 +250,10 @@ export function makeRunContainer({
223
250
  } catch {}
224
251
  };
225
252
  child.stdout?.on("data", tee);
226
- child.stderr?.on("data", tee);
253
+ child.stderr?.on("data", (chunk) => {
254
+ if (stderrHead.length < STDERR_HEAD_MAX) stderrHead = (stderrHead + String(chunk)).slice(0, STDERR_HEAD_MAX);
255
+ tee(chunk);
256
+ });
227
257
  // docker not found / daemon down -- a transient infra fault, so tag it retryable
228
258
  // (CONST-RETRY-INFRA-ONLY). `reason` also cues the processor to release the budget slot,
229
259
  // since a container that never started spent nothing.
@@ -243,19 +273,23 @@ export function makeRunContainer({
243
273
  // RUNNER_POLICY_REASONS set and to a line that itself said code 2. The processor still decides
244
274
  // the retry class from `code` alone; this only picks the label inside exit 2.
245
275
  let exitReason = null;
246
- // Issue #501 (PR #542's review, round 3): the LAST exit line's own `code`, which the dollar settlement
247
- // compares with the container's real exit code before it trusts that line's cost.
276
+ // Issue #501 (PR #542's review, round 3): the decisive exit line's own `code` (decisiveExitLine), which the
277
+ // dollar settlement compares with the container's real exit code before it trusts that line's cost.
248
278
  let exitLineCode = null;
249
279
  // Issue #507: the cost guard's rule on a `cost-cap` line, already filtered by parseExitWhy to COST_CAP_WHYS.
250
280
  let exitWhy = null;
251
281
  // Issue #545: null (no key issued), "verified" or "unverified" (a key, and no exit line carried it).
252
282
  let exitAuthResult = null;
283
+ // Issue #596: what the container used, off its exit line (parseExitResources), or null.
284
+ let resources = null;
285
+ // Issue #596: the image's supervisor reported the runner killed for memory (parseExitOomKilled).
286
+ let exitOomKilled = false;
253
287
  try {
254
288
  // `context = null` is a DEFAULT rather than a plain destructure: an injected sink that
255
289
  // predates the field returns no such key, and `undefined` would then reach the record's
256
290
  // shape where every other absence is spelled `null`.
257
291
  // `exitReason` defaults the same way, for the same reason.
258
- ({ turns, tokens, session, usage, context = null, exitReason = null, exitLineCode = null, exitWhy = null, exitAuth: exitAuthResult = null } = await sink.close());
292
+ ({ turns, tokens, session, usage, context = null, exitReason = null, exitLineCode = null, exitWhy = null, exitAuth: exitAuthResult = null, resources = null, exitOomKilled = false } = await sink.close());
259
293
  } catch {
260
294
  turns = null;
261
295
  tokens = null;
@@ -265,6 +299,8 @@ export function makeRunContainer({
265
299
  exitReason = null;
266
300
  exitLineCode = null;
267
301
  exitWhy = null;
302
+ resources = null;
303
+ exitOomKilled = false;
268
304
  // The sink could not say, and a key was issued: nothing it returned was verified.
269
305
  exitAuthResult = exitKey !== null ? "unverified" : null;
270
306
  }
@@ -279,12 +315,17 @@ export function makeRunContainer({
279
315
  exitReason = null;
280
316
  exitLineCode = null;
281
317
  exitWhy = null;
318
+ resources = null;
319
+ exitOomKilled = false;
282
320
  }
283
321
  // Spread only when a key was issued, so a run without one resolves the very object it always did.
284
322
  const auth = exitKey !== null ? { exitAuth: exitAuthResult } : {};
285
323
  // `exitWhy` only when the line named one, so every other run resolves the object it always did.
286
324
  const why = exitWhy !== null ? { exitWhy } : {};
287
- resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...auth } : { code: code ?? 1, aborted: false, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...auth });
325
+ // `resources` (issue #596) only when the exit line carried a block, so every other run resolves the object it always did.
326
+ // `memoryLimit` only beside the OOM report, the one place it is read, for the same reason.
327
+ const used = { ...(resources !== null && resources !== undefined ? { resources } : {}), ...(exitOomKilled === true ? { exitOomKilled: true, memoryLimit } : {}) };
328
+ resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...used, ...auth } : { code: code ?? 1, aborted: false, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...used, ...auth });
288
329
  });
289
330
  });
290
331
 
@@ -293,6 +334,19 @@ export function makeRunContainer({
293
334
  // that answer. What a failure leaves behind is a memberless network, which the boot reaper sweeps.
294
335
  try {
295
336
  const result = await run;
337
+ // Issue #596: the runtime refused the ceiling as above its CPU count. The count it was computed from is stale
338
+ // (Docker Desktop's VM given fewer CPUs since the facts were read), and every later job would fail the same
339
+ // way until a restart, so the facts are dropped and the next pickup reads them again. This attempt stays a
340
+ // never-started one (refunded and retried as infrastructure), and the line names both counts.
341
+ const runtimeCpus = !result.aborted && (neverStartedExits ?? []).includes(result.code) && Number.isSafeInteger(hostCpus) ? cpuRangeRefusal(stderrHead) : null;
342
+ if (runtimeCpus !== null) {
343
+ log("cpu_ceiling_stale", { container: name, cpus: (cpuCeilingCenti(hostCpus, cpuBudgetCenti) ?? 0) / 100, hostCpus, runtimeCpus });
344
+ try {
345
+ onCpuCeilingStale();
346
+ } catch {
347
+ // a wiring fault must not rewrite this attempt's answer; the facts age out on their own (`maxAgeMs`)
348
+ }
349
+ }
296
350
  // Issue #345: an exit that says "never started" is checked against the cidfile BEFORE the network goes, so a
297
351
  // container found running is stopped while its network still exists. Only when the worker did not abort it.
298
352
  if (!result.aborted && (neverStartedExits ?? []).includes(result.code) && (await stopDetached({ spawnFn, cidFile, fs, bin, ...detachedCheck }))) {