@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/start.mjs CHANGED
@@ -25,9 +25,12 @@ import { builtinModel, checkModelsKnown } from "./model-catalog.mjs";
25
25
  import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
26
26
  import { cronFingerprint } from "./fingerprint.mjs";
27
27
  import { makeHostRegistry } from "./host-registry.mjs";
28
+ import { budgetField, readUserServiceLimits } from "./host-budget.mjs";
29
+ import { makeCpuReserve, reservePlan } from "./cpu-reserve.mjs";
28
30
  import { makeImagePreflight } from "./image-preflight.mjs";
29
31
  import { createWorker, JOB_TIMEOUT_MS, STALLED_FAILED_REASON } from "./index.mjs";
30
32
  import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, jobUserRefusal, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
33
+ import { unenforcedSizeFlags } from "./job-size.mjs";
31
34
  import { makeCollectChain } from "./outbox.mjs";
32
35
  import { makeCollectPlan } from "./outbox-plan.mjs";
33
36
  import { makePortfolioSnapshot } from "./portfolio-snapshot.mjs";
@@ -41,7 +44,7 @@ import { scrubCredentials } from "./redact.mjs";
41
44
  import { makeCheckOnceSpent, makeCheckPortfolioFlag, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
42
45
  import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "./watch-closer.mjs";
43
46
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
44
- import { checkProjectRows, danglingProjectRows, dollarRowsWithoutCap, loadScopedLimits, scopeClaimRows } from "./scoped-limits.mjs";
47
+ import { checkProjectRows, danglingProjectRows, dollarRowsWithoutCap, loadScopedLimits, SCOPED_LIMITS_VERSION, scopeClaimRows } from "./scoped-limits.mjs";
45
48
  import { escapeControls, loadProjects, projectOf, projectsFingerprint } from "./projects.mjs";
46
49
  import { envelopeDigest, envelopeInsideJobPaths, loadEnvelopeChecked } from "./envelope.mjs";
47
50
  import { NO_ENVELOPE_FINGERPRINT, makeAllocationAudit, makeAllocationLogReaper, makeAllocationState } from "./allocation.mjs";
@@ -50,7 +53,7 @@ import { makeOnFailure } from "./on-failure.mjs";
50
53
  import { makeWaitChecker } from "./wait-check.mjs";
51
54
  import { makeWaitState } from "./wait-state.mjs";
52
55
  import { hostQueueName, makeQueue } from "./queue.mjs";
53
- import { endpointShown, makeDockerEndpointResolver, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
56
+ import { endpointShown, execDockerBounded, makeContainerGone, makeDockerEndpointResolver, makeJobContainerLister, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
54
57
  import { NETNS_KEEPER_MIN_AGE_MS, NETNS_KEEPER_YOUNG_MARGIN_MS, runtimeFromFacts } from "./netns-keeper.mjs";
55
58
  import { makeBackendRegistry, reapAll, resolveBackendName } from "./backend-registry.mjs";
56
59
  import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, isPerMachineHost, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
@@ -60,7 +63,7 @@ import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootf
60
63
  import { makeRunContainer } from "./run-container.mjs";
61
64
  import { resolveProviderCredential } from "./env-allowlist.mjs";
62
65
  import { makeSecretsResolver } from "./secrets.mjs";
63
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
66
+ import { buildRecord, EXIT_OOM_KILLED, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
64
67
  import { makeRunMirror, readMirroredRecord } from "./run-mirror.mjs";
65
68
  import { readOverlay, resolveSettings } from "./runtime-settings.mjs";
66
69
  import { usdFingerprint } from "./dollar-fingerprint.mjs";
@@ -575,6 +578,7 @@ export async function startWorker(
575
578
  makeAuth = makeGitHubAuth,
576
579
  makeHost = makeGitHubHost,
577
580
  createWorkerFn = createWorker,
581
+ makeCpuReserve: makeCpuReserveFn = makeCpuReserve,
578
582
  makeReaper: makeReaperFn = makeReaper,
579
583
  makeBackendRegistry: makeBackendRegistryFn = makeBackendRegistry,
580
584
  // Additional backend bundles, in registration order after `local` (which is built only while blessed,
@@ -606,6 +610,13 @@ export async function startWorker(
606
610
  // Issue #341: the one `docker info` the job-user decision reads, and the process facts it reads beside it.
607
611
  // Seams for the same reason as the endpoint: a wiring test decides what the daemon and the process say.
608
612
  readDaemonFacts: readDaemonFactsFn = makeDaemonFactsReader(),
613
+ // Issue #596, phase 2: a cgroup file's text (a rootless account's `memory.max` and `cpu.max`), for the host budget.
614
+ readCgroupFile: readCgroupFileFn = (path) => readFileSync(path, "utf8"),
615
+ // Issue #596, phase 2: whether an orphaned job container is gone, per venue's CLI.
616
+ containerGone: containerGoneFn = null,
617
+ // Issue #596, gate round 1 of phase 2: the job containers a venue's CLI still lists after the boot reaper,
618
+ // with their size labels, `(bin) => async () => [{ name, memMiB, cpuCenti }]`, throwing when the CLI does not answer.
619
+ listJobContainers: listJobContainersFn = (bin) => makeJobContainerLister({ bin }),
609
620
  // `home` (issue #354) is the account whose rootless Podman runs the podman venue's jobs: its own mounts.conf and
610
621
  // containers.conf are read from there. Absent in a test's identity, it falls to the observation's own default.
611
622
  jobUserIdentity = { platform: process.platform, release: osRelease(), euid: process.geteuid?.(), egid: process.getegid?.(), home: homedir() },
@@ -822,6 +833,9 @@ export async function startWorker(
822
833
  // The per-job read (below, `observationPreflight`) logs only when the answer CHANGES from the last one, so a
823
834
  // deliberate, standing redirect writes one line at boot rather than one per job.
824
835
  let endpointSeen = bootEndpoint ? dockerEndpointState(bootEndpoint) : null;
836
+ // Issue #596, phase 2: the endpoint the last job-user read asked about, so the host budget's tick reads the SAME cached
837
+ // facts a job was decided from (and re-reads them when they age), never a second daemon of its own choosing.
838
+ let budgetEndpoint = bootEndpoint ? { endpoint: bootEndpoint, key: endpointSeen } : null;
825
839
 
826
840
  // Issue #341: WHO job containers run as on this daemon (`DES-JOB-USER-INFERRED-READ-BACK-ON-REQUEST`). Decided
827
841
  // from facts, never a probe container, cached per endpoint state. Bounded like the boot image read, because a
@@ -829,7 +843,7 @@ export async function startWorker(
829
843
  // default venue: rootless, userns-remap, a root worker and Docker Desktop on Linux cannot run any local job here.
830
844
  // An unknown answer (a daemon still starting) boots, so a unit with RestartPreventExitStatus=2 is never stranded
831
845
  // by one; so does `runtime-unreadable`, which a later job re-reads.
832
- const resolveJobUser = makeJobUserResolver({ readFacts: readDaemonFactsFn, ...jobUserIdentity });
846
+ const resolveJobUser = makeJobUserResolver({ readFacts: readDaemonFactsFn, ...jobUserIdentity, now, log });
833
847
  // Issue #345: a floor that needs a DAEMON observation waits the facts read's own bound (plus a margin), not the image
834
848
  // read's 5 s: a busy host's `docker info` is the slow read, and a floor this boot met from the table before would
835
849
  // otherwise exit 1 on every restart of a healthy daemon.
@@ -897,7 +911,42 @@ export async function startWorker(
897
911
  // Wrapped once in `cachedPodmanInfo` and the SAME wrapper is handed to the bundle below, so an answer read here is the
898
912
  // one the first job is decided from rather than a second spawn. Bounded twice: the reader's own timeout (docker info's
899
913
  // 15 s, reused) and this fuse two seconds past it, because a spawn that never settles has no timeout to fire.
900
- const podmanInfo = podmanBlessed ? cachedPodmanInfo(readPodmanInfoFn) : null;
914
+ const podmanInfo = podmanBlessed ? cachedPodmanInfo(readPodmanInfoFn, { now, log }) : null;
915
+ // Issue #596, phase 2: what the host budget's `auto` is computed from, read on its tick (off every job path) from the
916
+ // two cached readers above: each blessed venue's memory and CPU count, the SMALLER where both answered (two venues on
917
+ // one host share it; a desktop VM is the smaller), and on rootless Podman the user service's own `memory.max` and
918
+ // `cpu.max` beside them. A venue that did not answer adds nothing, so the budget is unknown only when none did.
919
+ // The same reads also say how each venue manages cgroups (`reserveVenues`), from which the CPU reserve keeps the jobs'
920
+ // parent cgroup's quota at the budget (`cpu-reserve.mjs`, on every refresh, off every job path).
921
+ const readHostFacts = async () => {
922
+ const views = [];
923
+ const reserveVenues = [];
924
+ let user = {};
925
+ if (localBlessed && budgetEndpoint) {
926
+ const read = await resolveJobUser(budgetEndpoint).catch(() => null);
927
+ if (read?.daemon?.answered === true) {
928
+ views.push(read.daemon.facts);
929
+ reserveVenues.push({ venue: DEFAULT_BACKEND, facts: read.daemon.facts, endpointLocal: budgetEndpoint.endpoint?.local === true });
930
+ }
931
+ }
932
+ if (podmanInfo) {
933
+ const read = await podmanInfo().catch(() => null);
934
+ if (read?.answered === true) {
935
+ views.push(read.info);
936
+ reserveVenues.push({ venue: PODMAN_BACKEND, facts: read.info, endpointLocal: read.info?.serviceIsRemote === false });
937
+ if (read.info?.rootless === true) user = readUserServiceLimits({ uid: jobUserIdentity.euid, readFile: readCgroupFileFn });
938
+ }
939
+ }
940
+ const least = (key) => {
941
+ const known = views.map((v) => v?.[key]).filter((v) => Number.isSafeInteger(v));
942
+ return known.length > 0 ? Math.min(...known) : null;
943
+ };
944
+ return { memTotalMiB: least("memTotalMiB"), hostCpus: least("hostCpus"), ...user, reserveVenues };
945
+ };
946
+ // Issue #596, phase 2: the aggregate CPU reserve. One per worker; the helper container (Docker's cgroupfs driver) runs
947
+ // the job image this worker already pins, with `--pull=never`, never an image fetched for it.
948
+ const cpuReserve = makeCpuReserveFn({ run: (bin, args, { timeoutMs }) => execDockerBounded(args, { bin, timeoutMs }), image: config.jobImage, now, log });
949
+ const syncCpuReserve = (budget, facts) => cpuReserve.sync({ cpuCenti: budget.cpuCenti, plans: (facts?.reserveVenues ?? []).map((v) => reservePlan({ ...v, platform: jobUserIdentity.platform })) });
901
950
  const bootPodmanRead = podmanInfo
902
951
  ? await settleWithin(
903
952
  Promise.resolve()
@@ -1319,7 +1368,7 @@ export async function startWorker(
1319
1368
  // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
1320
1369
  // no job then issues a single extra Valkey command.
1321
1370
  const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
1322
- const recordRun = ({ job, result, error, startedAt, endedAt, project }) => {
1371
+ const recordRun = ({ job, result, error, startedAt, endedAt, project, size = null }) => {
1323
1372
  // The project (issue #499) was resolved at the pickup gate and rides here as `project` (an id or null), so a live
1324
1373
  // edit of projects.json mid-run cannot make the record disagree with what the job was counted against. A record
1325
1374
  // path that ends BEFORE the pickup gate (the wait gate's refusals) passes none, and resolves from the live ref
@@ -1329,7 +1378,7 @@ export async function startWorker(
1329
1378
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
1330
1379
  // The default venue rides the same way and for the same reason (#277): it is the value the registry
1331
1380
  // below is built with, so the record resolves a job's venue exactly as dispatch does.
1332
- const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId });
1381
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId, size });
1333
1382
  writeRecord(record);
1334
1383
  // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
1335
1384
  // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
@@ -1564,6 +1613,11 @@ export async function startWorker(
1564
1613
  // Resolved once: `Intl` is not free, and this value cannot change without a restart.
1565
1614
  const hostTz = Intl.DateTimeFormat().resolvedOptions().timeZone ?? "";
1566
1615
  const registry = makeHostRegistryFn({ redis, name: config.workerName, log });
1616
+ // Issue #596, phase 2: one integer of the host budget's snapshot for the beat, or "" before the worker exists.
1617
+ const snapshotField = (key) => {
1618
+ const snap = worker?.hostBudget?.snapshot?.();
1619
+ return Number.isSafeInteger(snap?.[key]) ? String(snap[key]) : "";
1620
+ };
1567
1621
  // NOT awaited, and that is load-bearing rather than an optimisation. `makeRedisClient` sets
1568
1622
  // `maxRetriesPerRequest: null` -- required for BullMQ's blocking connections -- which means a command
1569
1623
  // issued against an unreachable server QUEUES FOREVER instead of rejecting. Awaiting the first beat
@@ -1617,6 +1671,29 @@ export async function startWorker(
1617
1671
  // Issue #504 part B: the digest of this host's live envelope (`envelopeDigest`, 16 hex, never a value), so doctor can
1618
1672
  // name a host whose envelope differs; such a host refuses governed jobs as `envelope-mismatch`. `none` without one.
1619
1673
  fpEnvelope: () => envelope.digest ?? NO_ENVELOPE_FINGERPRINT,
1674
+ // Issue #596: the highest scoped-limits version this build reads, so doctor can name a worker that would keep its
1675
+ // last good file (so no size and no later edit to the file applies on it) once the file is version 3. An integer.
1676
+ limitsVersion: SCOPED_LIMITS_VERSION,
1677
+ // Issue #596, phase 2: this host's budget and what its jobs hold against it, integers (MiB and hundredths of a CPU),
1678
+ // `off` for a budget switched off and "" while unknown, so doctor and a forge job's never-fits check can read which
1679
+ // host a size fits on. Thunks over the worker's one budget, so every beat says what is held now.
1680
+ budgetMemMiB: () => budgetField(worker?.hostBudget?.current().memMiB ?? null),
1681
+ budgetCpuCenti: () => budgetField(worker?.hostBudget?.current().cpuCenti ?? null),
1682
+ usedMemMiB: () => snapshotField("usedMemMiB"),
1683
+ usedCpuCenti: () => snapshotField("usedCpuCenti"),
1684
+ heldMemMiB: () => snapshotField("heldMemMiB"),
1685
+ heldCpuCenti: () => snapshotField("heldCpuCenti"),
1686
+ budgetRunning: () => snapshotField("running"),
1687
+ budgetHolds: () => snapshotField("holds"),
1688
+ budgetOrphans: () => snapshotField("orphans"),
1689
+ // whether the boot listing of the job containers left from before this worker started has been read
1690
+ // (`listed`); until a venue's is, the worker admits no job on it (`unlisted:<venue>[,<venue>]`, gate round 2 of
1691
+ // phase 2), and doctor names the venues. "" before the worker exists.
1692
+ budgetSeed: () => {
1693
+ const snap = worker?.hostBudget?.snapshot?.();
1694
+ if (typeof snap?.seeded !== "boolean") return "";
1695
+ return snap.seeded ? "listed" : `unlisted:${(snap.unseeded ?? []).join(",")}`;
1696
+ },
1620
1697
  });
1621
1698
 
1622
1699
 
@@ -1653,6 +1730,7 @@ export async function startWorker(
1653
1730
  if (endpoint.local === null && endpoint.transient) return { unavailable: true, reason: endpoint.reason };
1654
1731
  return { refused: true, message: endpointRefusal, observations: [DOCKER_ENDPOINT_LOCAL] };
1655
1732
  }
1733
+ budgetEndpoint = { endpoint, key: state };
1656
1734
  const jobUser = await resolveJobUser({ endpoint, key: state });
1657
1735
  const unit = await readRootfulService({ endpoint, daemon: jobUser.daemon, readService: readPodmanServiceFn });
1658
1736
  // One read of each host path for this job's two checks (`onceFs`, gate round 3 of PR #473), fresh per job.
@@ -1702,6 +1780,7 @@ export async function startWorker(
1702
1780
  const venue = resolveBackendName(job, config.defaultBackend);
1703
1781
  if (venue !== DEFAULT_BACKEND) return { user: null, home: null };
1704
1782
  const endpoint = observed?.endpoint ?? (await resolveDockerEndpointFn());
1783
+ if (!observed?.jobUser) budgetEndpoint = { endpoint, key: dockerEndpointState(endpoint) };
1705
1784
  const admittedOn = observed?.jobUser ?? (await resolveJobUser({ endpoint, key: dockerEndpointState(endpoint) }));
1706
1785
  const { decision, socket, facts } = admittedOn;
1707
1786
  // Issue #452, gate round 5: the runtime THIS job is admitted on, recorded per job for its teardown's detach gate. The
@@ -1718,7 +1797,12 @@ export async function startWorker(
1718
1797
  // undecidable daemon runs nothing), and only when true: every host this does not apply to keeps the answer
1719
1798
  // shape, and so the argv, it had before.
1720
1799
  if (chosen.refused || chosen.unavailable) return chosen;
1721
- return relabelsPrivateMounts(facts, endpoint, jobUserIdentity.platform) ? { ...chosen, relabel: true } : chosen;
1800
+ // Issue #596: the daemon's CPU count from that same read, for the job's `--cpus` ceiling. Absent when it did not say.
1801
+ // Beside it, the size flags the same daemon said it drops (SwapLimit or CPUShares false), which the job logs.
1802
+ const unenforced = unenforcedSizeFlags(facts);
1803
+ const counted = Number.isSafeInteger(facts?.hostCpus) ? { ...chosen, hostCpus: facts.hostCpus } : chosen;
1804
+ const sized = unenforced.length > 0 ? { ...counted, unenforced } : counted;
1805
+ return relabelsPrivateMounts(facts, endpoint, jobUserIdentity.platform) ? { ...sized, relabel: true } : sized;
1722
1806
  };
1723
1807
 
1724
1808
  // #227: WHERE this job's container runs. The three functions that decide whether a container may start and
@@ -1781,6 +1865,9 @@ export async function startWorker(
1781
1865
  // reads. A refused teardown is logged with its token.
1782
1866
  teardownRuntime: (job) => takeAdmittedRuntime(job),
1783
1867
  log,
1868
+ // Issue #596: Docker refused a job's `--cpus` as above its CPU count, so the count the resolver cached is
1869
+ // stale (a resized Docker Desktop VM); the next pickup reads the daemon again.
1870
+ onCpuCeilingStale: () => resolveJobUser.invalidate?.(),
1784
1871
  }),
1785
1872
  }),
1786
1873
  observationPreflight: localObservationPreflight,
@@ -1897,7 +1984,8 @@ export async function startWorker(
1897
1984
  // each already comments -- a delivery storm against a spent cap must not page anyone), and
1898
1985
  // `operator-cancel`, because the operator initiated it and a push telling them what they just did is
1899
1986
  // noise with a pager attached.
1900
- const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", ...RUNNER_POLICY_REASONS]);
1987
+ // `oom-killed` (issue #596) is a paid terminal the operator alone can fix (a job's memory size), so it pages too.
1988
+ const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", EXIT_OOM_KILLED, ...RUNNER_POLICY_REASONS]);
1901
1989
  // One predicate for the completed listener and the lost-lock path below, so a record replays exactly the page its
1902
1990
  // result would have sent.
1903
1991
  const pagesAsPolicy = (result) => Boolean(onFailure) && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason) && result.budgetReserved !== false;
@@ -1920,6 +2008,39 @@ export async function startWorker(
1920
2008
 
1921
2009
  const worker = createWorkerFn({
1922
2010
  connection: valkeyConn(),
2011
+ // Issue #596, phase 2 (DES-HOST-BUDGET): the host budget's inputs. `createWorker` builds the ONE budget from them and
2012
+ // shares it between both queues' processors. The settings and the default size were refused at boot if bad.
2013
+ hostBudget: {
2014
+ settings: config.hostBudget,
2015
+ jobDefault: { memMiB: config.jobSize.memMiB, cpuCenti: config.jobSize.cpuCenti },
2016
+ readFacts: readHostFacts,
2017
+ onRefresh: syncCpuReserve,
2018
+ containerGone: containerGoneFn ?? makeContainerGone({ binOf: (venue) => (resolveBackendName(venue ?? {}, config.defaultBackend) === PODMAN_BACKEND ? "podman" : "docker") }),
2019
+ // AFTER the boot reaper (above), every blessed venue's remaining job containers, seeded into the ledger
2020
+ // as orphans from their size labels. PER VENUE (gate round 2 of phase 2): a venue that cannot be listed
2021
+ // stops only its own jobs until a tick reads it, because a container nobody counted is an overcommit, while the
2022
+ // other venue's jobs run. A venue that holds no container by construction counts as none rather than as unread:
2023
+ // its binary is absent (ENOENT, nothing of it can run), or its boot job-user decision is `unmappable` for a
2024
+ // cause that refuses a boot (the venue's boot-refusing set: every job on it is refused before a container, so this
2025
+ // worker starts none there). A cause decided from one unreadable answer is re-decided per job and can clear, so
2026
+ // it leaves the venue unread.
2027
+ survivors: Object.fromEntries(
2028
+ [...(localBlessed ? [[DEFAULT_BACKEND, "docker", bootDecision, BOOT_REFUSING_JOB_USER_CAUSES]] : []), ...(podmanBlessed ? [[PODMAN_BACKEND, "podman", bootPodmanDecision, PODMAN_BOOT_REFUSING_CAUSES]] : [])].map(([backend, bin, decision, stable]) => [
2029
+ backend,
2030
+ async () => {
2031
+ try {
2032
+ return (await listJobContainersFn(bin)()).map((c) => ({ ...c, venue: { backend } }));
2033
+ } catch (error) {
2034
+ if (error?.code === "ENOENT" || (decision?.mode === "unmappable" && stable.has(decision.cause))) return [];
2035
+ throw error;
2036
+ }
2037
+ },
2038
+ ]),
2039
+ ),
2040
+ defaultVenue: config.defaultBackend,
2041
+ now,
2042
+ log,
2043
+ },
1923
2044
  // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
1924
2045
  // not enough to find the runtime holding it once there is more than one venue.
1925
2046
  stopContainer: backends.stopContainer,
@@ -1947,6 +2068,9 @@ export async function startWorker(
1947
2068
  scopedLimits: () => scopedLimits.current,
1948
2069
  // Issue #499: the projects snapshot, read by the pickup gate once, beside the limits snapshot above.
1949
2070
  projects: () => projects.current,
2071
+ // Issue #596: the deployment's default job size, the two settings `loadConfig` already refused at boot if bad. ENV
2072
+ // ONLY, never the settings overlay, so a size cannot change under a running worker.
2073
+ jobSizeEnv: { PI_JOB_MEMORY: env.PI_JOB_MEMORY, PI_JOB_CPUS: env.PI_JOB_CPUS },
1950
2074
  // Issue #504 part B: the live envelope and its digest, and the reconcile the pickup runs before it narrows a job's
1951
2075
  // dollar ledgers by the applied split. Without an envelope `current()` is null, and `fleetGoverned` asks whether an
1952
2076
  // applied split exists: if it does, this host's jobs refuse as envelope-mismatch rather than run ungoverned.
package/src/triggers.mjs CHANGED
@@ -1309,7 +1309,8 @@ function validateMaxCostUsd(run, at, path) {
1309
1309
  * trigger, while naming what to take away cannot widen on a bump.
1310
1310
  *
1311
1311
  * MEMBERSHIP IS VALIDATED against `EXCLUDABLE_TOOL_NAMES` because pi ignores unknown names in
1312
- * `excludeTools` silently (verified at the pin: the set is only ever consulted by a filter), which puts
1312
+ * `excludeTools` silently (verified at the pin: pi matches entries by exact name or `*` pattern, so an
1313
+ * unknown exact name matches nothing, silently), which puts
1313
1314
  * a misspelled exclusion in `run.backend`'s destructive-absence class -- the job runs WITH the tool
1314
1315
  * while the file reads as though it was off. The near-miss sweep covers the KEY for the same reason.
1315
1316
  * No charset check: membership subsumes it, and no known name carries the comma the container env