@edgehero/pi-dispatch 3.0.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/processor.mjs CHANGED
@@ -8,7 +8,7 @@ import { PROJECT_CAP_REASON } from "./scoped-limits.mjs";
8
8
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
9
9
  import { RESERVED_ENV_NAMES } from "./triggers.mjs";
10
10
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
11
- import { COST_CAP_WHYS, RUNNER_POLICY_REASONS } from "./run-history.mjs";
11
+ import { COST_CAP_WHYS, EXIT_OOM_KILLED, RUNNER_POLICY_REASONS } from "./run-history.mjs";
12
12
  import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
13
13
  import { CAPABILITY_GATES, EXIT_AUTH_CAPABILITY } from "./image-preflight.mjs";
14
14
  import { modelListProblem, modelOnList, splitModelEntry } from "./model-ref.mjs";
@@ -83,6 +83,29 @@ export const OBSERVATION_COMMENT_UNNAMED = "the venue this job runs on did not c
83
83
  * into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
84
84
  */
85
85
 
86
+ /** The exit code of a container whose main process was SIGKILLed: a worker's stop or the kernel's OOM killer. */
87
+ const EXIT_SIGKILL = 137;
88
+
89
+ /**
90
+ * Issue #596: whether a run's memory peak reached the container's own limit closely enough for its OOM kill to be the
91
+ * job's own: `memPeak` (bytes, off the decisive exit line) at 90% of `limit` (bytes, the `--memory=` the worker passed)
92
+ * or more. Both must be known; anything else is false, so the 137 stays infrastructure and retries.
93
+ *
94
+ * Why the check exists: `memory.events` `oom_kill` counts a kill by ANY OOM killer, the HOST's included, and the runner
95
+ * tree's `oom_score_adj` of 1000 makes a job the host's first victim when the machine itself runs short. A kill of
96
+ * that kind says nothing about the job's size, and a retry may well pass. A cgroup OOM happens only at the limit, so its
97
+ * peak sits there: measured in the issue #596 lab, `memory.peak` read exactly the bound (64m, 80m) or just under it on
98
+ * every venue. 90% and not 100% is the margin for that "just under"; nothing in between is read as a measurement.
99
+ *
100
+ * What it cannot see (the residual): a host OOM that strikes a job which is ALREADY at 90% of its own limit, page
101
+ * cache included (a job that read large files counts that cache), reads as the job's own OOM. And the peak is a
102
+ * high-water mark over the whole run, so a job that touched 90% early and was killed by the host later reads the same.
103
+ */
104
+ export function peakReachedLimit(memPeak, limit) {
105
+ if (!Number.isSafeInteger(memPeak) || !Number.isSafeInteger(limit) || memPeak < 0 || limit <= 0) return false;
106
+ return memPeak >= Math.ceil((limit * 9) / 10);
107
+ }
108
+
86
109
  // The post-spend terminal comments (issue #288). Every FREE refusal above the container already comments;
87
110
  // these are the paths where money was spent and the run still ended without the agent's own status step,
88
111
  // which used to tell the issue nothing (REQ-JOB-STATUS-COMMENTS' acceptance -- "exactly one completion or
@@ -106,6 +129,9 @@ export const TERMINAL_COMMENTS = {
106
129
  "model-not-allowed": "Stopped: the run tried to call an AI model this trigger does not allow, or to change an AI request in a way it does not allow, so the call was not made. Partial work may exist. Not retried.",
107
130
  "cost-cap-unenforceable": "Stopped: this run has a cost limit, and the job image could not enforce it before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
108
131
  "model-policy-unenforceable": "Stopped: this run is limited to certain AI models, and the job image could not enforce that before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
132
+ // Issue #596. Never the size or the project: both are operator configuration, and the reader may be an issue author
133
+ // who can act on neither. The worker log and the run record carry them.
134
+ [EXIT_OOM_KILLED]: "Stopped: the job's container ran out of memory and was stopped. Partial work may exist. Not retried, because the same size would stop the same way. The operator can raise this job's memory size.",
109
135
  };
110
136
 
111
137
  // Issue #502: the `model-unknown` refusal's comment. Names no model: the reader may be an issue author.
@@ -238,7 +264,8 @@ export async function runJob(job, deps) {
238
264
  // Issue #341: which uid this job's container runs as. `(job, { capabilities, observed }) =>` `{ user, home }`
239
265
  // (`user` null = the image's own USER), `{ refused, cause }` or `{ unavailable, reason }`. The default runs
240
266
  // every job as the image's user, exactly as before, so a wiring that omits it changes nothing. A non-refused answer
241
- // may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with.
267
+ // may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with, and
268
+ // `hostCpus` (issue #596), the runtime's CPU count from the same facts read, which sets the `--cpus` ceiling.
242
269
  jobUserPreflight = async () => ({ user: null, home: null }),
243
270
  // (session, { piVersion, context }) => { promoted, reason, bytes }. Promotes this job's transcript back into
244
271
  // the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
@@ -300,8 +327,11 @@ export async function runJob(job, deps) {
300
327
  mintToken,
301
328
  isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
302
329
  prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
303
- // runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
330
+ // runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints?, size?, hostCpus?, unenforced? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
331
+ // `size` (issue #596) is `jobSize` below; `hostCpus` the runtime's CPU count off the job-user gate's own facts read.
332
+ // `unenforced` (issue #596) is the size flags that same read says the runtime drops, passed only when there are some.
304
333
  // `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
334
+ // `resources` and `exitOomKilled` (issue #596) are the exit line's cgroup block and the supervisor's OOM report.
305
335
  // `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
306
336
  // `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
307
337
  // `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
@@ -356,6 +386,14 @@ export async function runJob(job, deps) {
356
386
  // after prepare, before any reserve. Absent on a bare wiring, which checks nothing.
357
387
  pickupProject = null,
358
388
  folderProject = null,
389
+ // Issue #596: the job's size `{ memMiB, cpuCenti, source }`, resolved at pickup from the same limits snapshot as
390
+ // every other gate (index.mjs). Handed to `prepareWorkspace` (a retained run's manifest records it, so a sandbox
391
+ // reopens the run at its size) and to `runContainer`, never through `job.data`. null on a bare wiring: neither call
392
+ // then carries it, and the container gets the built-in 4g and 2.
393
+ jobSize = null,
394
+ // Issue #596, phase 2: the host's CPU budget at pickup in hundredths (`host-budget.mjs`), or null when it is off or
395
+ // unknown. It becomes every job's `--cpus` (capped at the runtime's count), so no job can use the reserve.
396
+ cpuBudgetCenti = null,
359
397
  // Issue #503 part 7: the builtin catalog's model object for (provider, id), or null (model-catalog.mjs
360
398
  // `builtinModel`). Read only by the zero-rated check; the default knows no builtin model, so an unwired
361
399
  // processor judges overlay models alone and reserves for every other.
@@ -412,6 +450,8 @@ export async function runJob(job, deps) {
412
450
  // the record says about it (INT-RUN-HISTORY-FILE-CONTRACT), null until the reservation step ran.
413
451
  let dollarHold = null;
414
452
  let dollars = null;
453
+ // Issue #596: what the container used (`resources` off its exit line), null until a container ran and reported it.
454
+ let resources = null;
415
455
 
416
456
  try {
417
457
  // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
@@ -1067,7 +1107,7 @@ export async function runJob(job, deps) {
1067
1107
  // `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
1068
1108
  // `portfolio` (issue #505) only for a job the gate above confirmed: prepare then writes /job/portfolio.json, after
1069
1109
  // asking the live file once more. Absent otherwise, so every other job's prepare call is unchanged.
1070
- prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
1110
+ prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}), ...(jobSize ? { size: jobSize } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
1071
1111
 
1072
1112
  // A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
1073
1113
  // or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
@@ -1270,12 +1310,29 @@ export async function runJob(job, deps) {
1270
1310
  // then only a signed line is read (run-container.mjs, run-history.mjs `authenticExitLines`). An image that does not
1271
1311
  // declare it is read as before, under the #542 trust rule below alone.
1272
1312
  const exitAuth = (img.capabilities ?? []).includes(EXIT_AUTH_CAPABILITY);
1273
- const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}) });
1313
+ const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null, exitOomKilled = false, memoryLimit = null, resources: ranResources = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}), ...(jobSize ? { size: jobSize } : {}), ...(Number.isSafeInteger(jobUser?.hostCpus) ? { hostCpus: jobUser.hostCpus } : {}), ...(Number.isSafeInteger(cpuBudgetCenti) ? { cpuBudgetCenti } : {}), ...(Array.isArray(jobUser?.unenforced) && jobUser.unenforced.length > 0 ? { unenforced: jobUser.unenforced } : {}) });
1274
1314
  containerRan = true;
1315
+ // Issue #596: what the container used, off its exit line, rebuilt by the sink (null from a runContainer that predates
1316
+ // the field). Every result and every throw below carries it, so a retried attempt's record says what it used too.
1317
+ resources = ranResources ?? null;
1318
+ // Issue #596: CONFIRMED killed for memory. The image's supervisor (image/runner/src/supervise.mjs) outlives the runner,
1319
+ // whose process tree it gives the highest OOM score, and when the runner dies of SIGKILL with the cgroup's
1320
+ // `oom_kill` above 0 it writes the signed line `code: 137, reason: "oom-killed"` (parseExitOomKilled). All three
1321
+ // facts must agree: that line, verified under this run's key (an unsigned line is a tool's), and the container's
1322
+ // own exit 137. Docker's `oom` event is not used: it fires also when only a child was killed and the job went on
1323
+ // to exit 0, and Podman has no such event at all (both measured in the issue #596 lab). And a fourth: the line's
1324
+ // `memPeak` at 90% of the `--memory` this container got (`memoryLimit`, from runContainer) or more, because
1325
+ // `oom_kill` counts the HOST's OOM killer too (`peakReachedLimit`). A report below it stays infrastructure.
1326
+ const oomReported = exitOomKilled === true && exitAuthResult === "verified" && code === EXIT_SIGKILL;
1327
+ const oomKilled = oomReported && peakReachedLimit(resources?.memPeak ?? null, memoryLimit);
1328
+ // Numbers only (bytes), never a path or a project: the operator's trace of a kill read as the host's. Never on a run
1329
+ // the worker stopped itself (a timeout, a cancel, a shutdown): that 137 is the worker's own, the abort decides it
1330
+ // below, and a line naming the host's OOM killer would send the operator after a kill that never happened.
1331
+ if (oomReported && !oomKilled && !aborted) log("oom_report_below_limit", { jobId: job.id ?? null, memPeak: resources?.memPeak ?? null, memoryLimit });
1275
1332
  // `exitAuth: "unverified"` is a run whose image signs its exit line and no signed line was found: the runner died
1276
1333
  // before writing one, or a line was forged or taken off the pipe. Its tokens read as unknown and its dollars settle
1277
1334
  // at the floor, the same as a container that wrote no exit line at all.
1278
- log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}) });
1335
+ log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}), ...(oomKilled ? { oomKilled: true } : {}) });
1279
1336
 
1280
1337
  // Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
1281
1338
  // so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
@@ -1334,14 +1391,15 @@ export async function runJob(job, deps) {
1334
1391
  // mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
1335
1392
  // BEFORE the exit-code switch, where the same code would read as a free never-started exit.
1336
1393
  if (detached === true) {
1337
- throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1394
+ throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1338
1395
  }
1339
1396
 
1340
1397
  // A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
1341
1398
  // operator's cancel, issue #287) kills the container -> exit 143/137. That is our decision, not an
1342
1399
  // infra fault: it is POLICY and must NOT retry, or a wedged job re-runs into a second PR / double
1343
1400
  // spend. Keyed on the abort FLAG, not the code -- an unbidden 137 (kernel OOM) carries
1344
- // `aborted: false`, falls to the switch, and stays infra-retryable.
1401
+ // `aborted: false`, and falls to the OOM branch below when the runtime confirmed it, else to the switch, where it
1402
+ // stays infra-retryable (issue #596).
1345
1403
  // WHO aborted is an exact-match on `abortReason` (the wiring maps it off `signal.reason`), and the
1346
1404
  // match is deliberately closed: "job-timeout-30m", "shutdown", undefined and any future garbage all
1347
1405
  // classify as worker-abort, so a pin bump that changes what rides the signal can widen nothing.
@@ -1352,7 +1410,23 @@ export async function runJob(job, deps) {
1352
1410
  // Awaited bare like every determinate refusal above: the adapter never throws by contract, and
1353
1411
  // the one swallowed comment in this file (the catch's) justifies itself by its position.
1354
1412
  await comment(job, TERMINAL_COMMENTS[reason]);
1355
- return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}) };
1413
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
1414
+ }
1415
+
1416
+ // Issue #596: a runner the kernel killed for memory, CONFIRMED (`oomKilled` above). After the abort branch, so a
1417
+ // worker's own stop is never relabelled, and before the switch, where the same 137 is an unknown exit and retries.
1418
+ // The same size would be killed the same way on every retry, so it is POLICY: returned, never retried, its slot
1419
+ // kept and its dollars settled above like any other paid stop. An UNCONFIRMED 137 (an image without the supervisor,
1420
+ // an unsigned line, a SIGKILL with no OOM kill in the cgroup, a peak short of the limit, which is how a kill by the
1421
+ // host's OOM killer reads) falls through and retries, exactly as before.
1422
+ //
1423
+ // When only a CHILD was killed, the runner survives and ends on its own code, so this branch is not taken: the
1424
+ // outcome is the runner's, and `resources.oomKills` in the record says a process was killed for memory.
1425
+ if (oomKilled) {
1426
+ // The project and the size go to the log and the record, never to the comment (the TERMINAL_COMMENTS rule).
1427
+ log("oom_killed", { jobId: job.id ?? null });
1428
+ await comment(job, TERMINAL_COMMENTS[EXIT_OOM_KILLED]);
1429
+ return { outcome: "policy", reason: EXIT_OOM_KILLED, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...(resources ? { resources } : {}) };
1356
1430
  }
1357
1431
 
1358
1432
  switch (code) {
@@ -1394,6 +1468,7 @@ export async function runJob(job, deps) {
1394
1468
  // Only when the collector returned a plan (a file, or a confirmed portfolio job with none, `plan-absent`): the
1395
1469
  // record's `plan` is null otherwise, and every other result is unchanged.
1396
1470
  ...(plan ? { plan } : {}),
1471
+ ...(resources ? { resources } : {}),
1397
1472
  };
1398
1473
  }
1399
1474
  case EXIT_POLICY: {
@@ -1420,7 +1495,7 @@ export async function runJob(job, deps) {
1420
1495
  // `why`. parseExitWhy keeps only a member of the closed COST_CAP_WHYS off a `cost-cap` line that said code 2,
1421
1496
  // and it rides only beside a `cost-cap` reason, so a forged line can at worst name the wrong rule of three.
1422
1497
  const why = reason === "cost-cap" && COST_CAP_WHYS.includes(exitWhy) ? { why: exitWhy } : {};
1423
- return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why };
1498
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why, ...(resources ? { resources } : {}) };
1424
1499
  }
1425
1500
  case EXIT_INFRA:
1426
1501
  // NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
@@ -1428,7 +1503,7 @@ export async function runJob(job, deps) {
1428
1503
  // the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
1429
1504
  // BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
1430
1505
  // function never sees (issue #288).
1431
- throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1506
+ throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1432
1507
  default:
1433
1508
  // THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
1434
1509
  // (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
@@ -1444,9 +1519,9 @@ export async function runJob(job, deps) {
1444
1519
  // and normalises to this outcome itself.
1445
1520
  if (neverStarted) {
1446
1521
  // No `dollars` here: the hold is still standing, and the catch refunds it whole with the job-count slots.
1447
- throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1522
+ throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), resources });
1448
1523
  }
1449
- throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1524
+ throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars, resources });
1450
1525
  }
1451
1526
  } catch (e) {
1452
1527
  // A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
@@ -1625,7 +1700,7 @@ function isNeverStartedRetry(e) {
1625
1700
  }
1626
1701
 
1627
1702
  export class InfraRetry extends Error {
1628
- constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars } = {}) {
1703
+ constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars, resources } = {}) {
1629
1704
  super(message, cause ? { cause } : undefined);
1630
1705
  this.name = "InfraRetry";
1631
1706
  this.piDispatchRetry = true;
@@ -1647,6 +1722,8 @@ export class InfraRetry extends Error {
1647
1722
  // Issue #501: the dollar reservation's outcome (`dollarsRecord`), or null when no dollar window applied. Set by
1648
1723
  // the processor on a throw after the reservation, so a retried attempt's record says what its window was charged.
1649
1724
  this.dollars = dollars ?? null;
1725
+ // Issue #596: what the container used, off its exit line, or null; set on a throw after a container ran.
1726
+ this.resources = resources ?? null;
1650
1727
  }
1651
1728
  }
1652
1729
 
@@ -64,11 +64,12 @@ export const CONTAINER_ENV_NAMES = new Set([
64
64
 
65
65
  /**
66
66
  * Issue #500: the names the RUNNER sets in its own environment inside the container, for its descendants: the child
67
- * ledger directory and the runner's pid (image/runner/src/usage-meter.mjs, openChildLedger). The worker never writes
67
+ * ledger directory and the runner's pid (image/runner/src/usage-meter.mjs, openChildLedger), and the hash of the price
68
+ * table the runner hands its children (publishPriceTable, issue #587's review). The worker never writes
68
69
  * them, so they are not in the set above (which a test pins to what buildContainerEnv emits). They are reserved all the
69
70
  * same: a trigger binding one through `run.secrets`, or a host value forwarded through `PI_FORWARD_ENV`, would arrive in
70
71
  * the runner's environment before the runner sets its own, and a value the runner then failed to replace would point
71
72
  * every pi child's ledger at a directory the agent chose. Refused at load in both lists, and deleted by
72
73
  * buildContainerEnv after both loops as the backstop.
73
74
  */
74
- export const RUNNER_ENV_NAMES = new Set(["PI_DISPATCH_CHILD_LEDGER", "PI_DISPATCH_RUNNER_PID"]);
75
+ export const RUNNER_ENV_NAMES = new Set(["PI_DISPATCH_CHILD_LEDGER", "PI_DISPATCH_PRICE_TABLE", "PI_DISPATCH_RUNNER_PID"]);
@@ -2,7 +2,8 @@ import { spawn } from "node:child_process";
2
2
  import { randomBytes } from "node:crypto";
3
3
  import { readFileSync, rmSync } from "node:fs";
4
4
  import { DOCKER_NEVER_STARTED_EXITS } from "./backends.mjs";
5
- import { CONTAINER_HOME } from "./container-spec.mjs";
5
+ import { CONTAINER_HOME, memoryBytesOfArgs } from "./container-spec.mjs";
6
+ import { cpuCeilingCenti, cpuRangeRefusal, DEFAULT_JOB_SIZE } from "./job-size.mjs";
6
7
  import { buildDockerRunArgs, CONTAINER_SESSION_FILE, insideDir } from "./docker-run.mjs";
7
8
  import { createJobNetwork, networkNameFor, removeJobNetwork } from "./egress.mjs";
8
9
  import { buildContainerEnv } from "./env-allowlist.mjs";
@@ -11,7 +12,7 @@ import { InfraRetry } from "./processor.mjs";
11
12
 
12
13
  /**
13
14
  * The real `runContainer` the processor injects. Launches one job container and returns
14
- * `{ code, aborted, turns, tokens, session, usage, context, exitReason }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
15
+ * `{ code, aborted, turns, tokens, session, usage, context, exitReason, resources?, exitOomKilled?, memoryLimit? }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
15
16
  * the 30-min timeout or graceful shutdown), which the processor classifies as POLICY (no retry) per
16
17
  * INT-RUNNER-EXIT-CODE-PROTOCOL. The numeric `code` alone cannot say this: a worker SIGKILL and a
17
18
  * kernel OOM both surface as 137, so the abort FLAG -- not the code -- is the discriminator.
@@ -64,6 +65,10 @@ export function makeRunContainer({
64
65
  log = () => {},
65
66
  // Issue #545: the per-job exit-line key, 32 random bytes as hex. A seam so a test can name the key it verifies with.
66
67
  mintExitKey = () => randomBytes(32).toString("hex"),
68
+ // Issue #596: called when the runtime refused this job's `--cpus` as above its CPU count (`cpu_ceiling_stale`), so
69
+ // the wiring drops the cached runtime facts the ceiling came from and the next pickup reads them again. The local
70
+ // venue's is the job-user resolver's `invalidate`; a no-op elsewhere (Podman accepts any `--cpus`, measured).
71
+ onCpuCeilingStale = () => {},
67
72
  }) {
68
73
  // async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
69
74
  // a rejection, uniformly awaitable by the processor and by tests.
@@ -75,7 +80,15 @@ export function makeRunContainer({
75
80
  // `exitAuth` (issue #545) is the processor's, off the image preflight: true when the job image declares `exitAuth`, so
76
81
  // its runner reads a key from stdin and signs its exit line with it. Defaults off, so a caller that predates it, and
77
82
  // every image that does not declare it, runs exactly as before and is read exactly as before.
78
- return async function runContainer({ job, token, prepared, secrets = {}, name, signal, user = null, home = null, relabel = false, modelEndpoints = null, exitAuth = false }) {
83
+ // `size` (issue #596) is the job's `{ memMiB, cpuCenti, source }`, resolved at pickup from the limits snapshot and
84
+ // handed here as an argument, never through `job.data`; absent, the built-in 4g and 2. `hostCpus` is the runtime's own
85
+ // CPU count from the job user's facts read, which sets the `--cpus` ceiling; null leaves `--cpus` off (fails open, and
86
+ // says so in `cpu_ceiling_unknown`). `cpuBudgetCenti` (phase 2) is the host's CPU budget at pickup, which replaces that
87
+ // ceiling while it is in force (`cpuCeilingCenti`).
88
+ // `unenforced` (issue #596) is the size flags this runtime said it drops (`unenforcedSizeFlags`: `--memory-swap` where
89
+ // Docker reports SwapLimit false, `--cpu-shares` where it reports CPUShares false), logged per job as
90
+ // `size_bound_unenforced` so a run past its size's bound is named where it happens. The flags stay on the argv.
91
+ return async function runContainer({ job, token, prepared, secrets = {}, name, signal, user = null, home = null, relabel = false, modelEndpoints = null, exitAuth = false, size = DEFAULT_JOB_SIZE, hostCpus = null, cpuBudgetCenti = null, unenforced = [] }) {
79
92
  if (signal?.aborted) return { code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null }; // killed before it could start
80
93
  // Minted per attempt, held in this closure and the sink's, and handed to the container on its stdin only: never in
81
94
  // the argv (a host `ps` shows it), never in the env (the container's /proc/1/environ shows it), never logged.
@@ -175,6 +188,9 @@ export function makeRunContainer({
175
188
  sessionDir: prepared.session?.hostDir,
176
189
  globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
177
190
  name,
191
+ size, // issue #596: memory, swap equal to it, the CPU weight and /dev/shm, through containerSpec's one function
192
+ hostCpus, // issue #596: the `--cpus` ceiling every job shares; null leaves the flag off
193
+ cpuBudgetCenti, // issue #596, phase 2: the host's CPU budget, the ceiling while it is in force
178
194
  network, // REQ-EGRESS-ALLOWLIST: null when no policy is armed, and the flag is then absent
179
195
  user, // issue #341: the worker's own "<uid>:<gid>" on a daemon that enforces bind-mount ownership, else null
180
196
  cidFile, // issue #345: where the CLI writes this attempt's container ID, read below when the run exits "never started"
@@ -190,6 +206,13 @@ export function makeRunContainer({
190
206
  ...(exitKey !== null ? { extraFlags: ["-i"] } : {}),
191
207
  });
192
208
 
209
+ // Issue #596: the memory bound this container actually got, in bytes, read off the argv the runtime is handed (both
210
+ // builders spell it `--memory=`, from the job's size), never re-derived from the environment or the size. The
211
+ // processor confirms an OOM only when the run's peak reached 90% of it. null when the argv names none.
212
+ const memoryLimit = memoryBytesOfArgs(args);
213
+ if (!args.some((a) => typeof a === "string" && a.startsWith("--cpus="))) log("cpu_ceiling_unknown", { container: name });
214
+ if (Array.isArray(unenforced) && unenforced.length > 0) log("size_bound_unenforced", { container: name, flags: unenforced.filter((f) => typeof f === "string" && /^--[a-z-]{1,32}$/.test(f)) });
215
+
193
216
  // REQ-EGRESS-ALLOWLIST. This job's own --internal network, created here rather than at boot because
194
217
  // it holds exactly two endpoints -- this container and the proxy -- and that is what makes job-to-job
195
218
  // traffic structurally impossible rather than merely discouraged. A shared network could not do it:
@@ -205,6 +228,10 @@ export function makeRunContainer({
205
228
  // Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
206
229
  // sanitizes internally. No container mount, no env var -- the sink lives on this side only.
207
230
  const sink = openJobLog(name, { exitKey });
231
+ // Issue #596: the head of the CLI's stderr, kept only to recognise the runtime's refusal of a stale `--cpus`
232
+ // (`cpuRangeRefusal`). Bounded, and never logged: only the count it names leaves this function.
233
+ let stderrHead = "";
234
+ const STDERR_HEAD_MAX = 4096;
208
235
 
209
236
  const run = new Promise((resolve, reject) => {
210
237
  const child = spawnFn(bin, args, { stdio: [exitKey !== null ? "pipe" : "ignore", "pipe", "pipe"] });
@@ -223,7 +250,10 @@ export function makeRunContainer({
223
250
  } catch {}
224
251
  };
225
252
  child.stdout?.on("data", tee);
226
- child.stderr?.on("data", tee);
253
+ child.stderr?.on("data", (chunk) => {
254
+ if (stderrHead.length < STDERR_HEAD_MAX) stderrHead = (stderrHead + String(chunk)).slice(0, STDERR_HEAD_MAX);
255
+ tee(chunk);
256
+ });
227
257
  // docker not found / daemon down -- a transient infra fault, so tag it retryable
228
258
  // (CONST-RETRY-INFRA-ONLY). `reason` also cues the processor to release the budget slot,
229
259
  // since a container that never started spent nothing.
@@ -243,19 +273,23 @@ export function makeRunContainer({
243
273
  // RUNNER_POLICY_REASONS set and to a line that itself said code 2. The processor still decides
244
274
  // the retry class from `code` alone; this only picks the label inside exit 2.
245
275
  let exitReason = null;
246
- // Issue #501 (PR #542's review, round 3): the LAST exit line's own `code`, which the dollar settlement
247
- // compares with the container's real exit code before it trusts that line's cost.
276
+ // Issue #501 (PR #542's review, round 3): the decisive exit line's own `code` (decisiveExitLine), which the
277
+ // dollar settlement compares with the container's real exit code before it trusts that line's cost.
248
278
  let exitLineCode = null;
249
279
  // Issue #507: the cost guard's rule on a `cost-cap` line, already filtered by parseExitWhy to COST_CAP_WHYS.
250
280
  let exitWhy = null;
251
281
  // Issue #545: null (no key issued), "verified" or "unverified" (a key, and no exit line carried it).
252
282
  let exitAuthResult = null;
283
+ // Issue #596: what the container used, off its exit line (parseExitResources), or null.
284
+ let resources = null;
285
+ // Issue #596: the image's supervisor reported the runner killed for memory (parseExitOomKilled).
286
+ let exitOomKilled = false;
253
287
  try {
254
288
  // `context = null` is a DEFAULT rather than a plain destructure: an injected sink that
255
289
  // predates the field returns no such key, and `undefined` would then reach the record's
256
290
  // shape where every other absence is spelled `null`.
257
291
  // `exitReason` defaults the same way, for the same reason.
258
- ({ turns, tokens, session, usage, context = null, exitReason = null, exitLineCode = null, exitWhy = null, exitAuth: exitAuthResult = null } = await sink.close());
292
+ ({ turns, tokens, session, usage, context = null, exitReason = null, exitLineCode = null, exitWhy = null, exitAuth: exitAuthResult = null, resources = null, exitOomKilled = false } = await sink.close());
259
293
  } catch {
260
294
  turns = null;
261
295
  tokens = null;
@@ -265,6 +299,8 @@ export function makeRunContainer({
265
299
  exitReason = null;
266
300
  exitLineCode = null;
267
301
  exitWhy = null;
302
+ resources = null;
303
+ exitOomKilled = false;
268
304
  // The sink could not say, and a key was issued: nothing it returned was verified.
269
305
  exitAuthResult = exitKey !== null ? "unverified" : null;
270
306
  }
@@ -279,12 +315,17 @@ export function makeRunContainer({
279
315
  exitReason = null;
280
316
  exitLineCode = null;
281
317
  exitWhy = null;
318
+ resources = null;
319
+ exitOomKilled = false;
282
320
  }
283
321
  // Spread only when a key was issued, so a run without one resolves the very object it always did.
284
322
  const auth = exitKey !== null ? { exitAuth: exitAuthResult } : {};
285
323
  // `exitWhy` only when the line named one, so every other run resolves the object it always did.
286
324
  const why = exitWhy !== null ? { exitWhy } : {};
287
- resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...auth } : { code: code ?? 1, aborted: false, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...auth });
325
+ // `resources` (issue #596) only when the exit line carried a block, so every other run resolves the object it always did.
326
+ // `memoryLimit` only beside the OOM report, the one place it is read, for the same reason.
327
+ const used = { ...(resources !== null && resources !== undefined ? { resources } : {}), ...(exitOomKilled === true ? { exitOomKilled: true, memoryLimit } : {}) };
328
+ resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...used, ...auth } : { code: code ?? 1, aborted: false, turns, tokens, session, usage, context, exitReason, exitLineCode, ...why, ...used, ...auth });
288
329
  });
289
330
  });
290
331
 
@@ -293,6 +334,19 @@ export function makeRunContainer({
293
334
  // that answer. What a failure leaves behind is a memberless network, which the boot reaper sweeps.
294
335
  try {
295
336
  const result = await run;
337
+ // Issue #596: the runtime refused the ceiling as above its CPU count. The count it was computed from is stale
338
+ // (Docker Desktop's VM given fewer CPUs since the facts were read), and every later job would fail the same
339
+ // way until a restart, so the facts are dropped and the next pickup reads them again. This attempt stays a
340
+ // never-started one (refunded and retried as infrastructure), and the line names both counts.
341
+ const runtimeCpus = !result.aborted && (neverStartedExits ?? []).includes(result.code) && Number.isSafeInteger(hostCpus) ? cpuRangeRefusal(stderrHead) : null;
342
+ if (runtimeCpus !== null) {
343
+ log("cpu_ceiling_stale", { container: name, cpus: (cpuCeilingCenti(hostCpus, cpuBudgetCenti) ?? 0) / 100, hostCpus, runtimeCpus });
344
+ try {
345
+ onCpuCeilingStale();
346
+ } catch {
347
+ // a wiring fault must not rewrite this attempt's answer; the facts age out on their own (`maxAgeMs`)
348
+ }
349
+ }
296
350
  // Issue #345: an exit that says "never started" is checked against the cidfile BEFORE the network goes, so a
297
351
  // container found running is stopped while its network still exists. Only when the worker did not abort it.
298
352
  if (!result.aborted && (neverStartedExits ?? []).includes(result.code) && (await stopDetached({ spawnFn, cidFile, fs, bin, ...detachedCheck }))) {