@edgehero/pi-dispatch 3.1.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/doctor.mjs CHANGED
@@ -59,7 +59,7 @@ import { spawn as nodeSpawn } from "node:child_process";
59
59
  import { randomBytes } from "node:crypto";
60
60
  import { DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_VALKEY_URL, accountTempRoot, allowedModelsFrom, defaultLogsDir, defaultSandboxDir, defaultSettingsFile, defaultWorkerName, globalExtensionsEnabled, jobsDirOwnerFix, jobsDirPath, sandboxDirOwnerFix, legacyTempStateDir, logsDirPath, modelEndpointsFilePath, delimitedList, envelopeFilePath, pauseWindowsFilePath, projectsFilePath, safeHomeDir, scopedLimitsFilePath, settingsFilePath, underOsTempDir } from "./config.mjs";
61
61
  import { SYSTEMD_HAZARD_SHAPES, decodeEnvFile, envFileHazard, envValueShown, quotedRegions, readEnvAssignments, renderEnvValue, envFileWrapperInternal, wrapperInternalSentence } from "./env-file.mjs";
62
- import { canonicalScope, danglingProjectRows, dollarRowsBelowJobCap, dollarRowsWithoutCap, isModelScope, isProjectScope, loadScopedLimits, parseScopedLimits } from "./scoped-limits.mjs";
62
+ import { canonicalScope, danglingProjectRows, dollarRowsBelowJobCap, dollarRowsWithoutCap, isModelScope, isProjectScope, loadScopedLimits, parseScopedLimits, scopedLimitsVersionFor } from "./scoped-limits.mjs";
63
63
  import { EMPTY_PROJECTS_FINGERPRINT, loadProjects, projectsFingerprint } from "./projects.mjs";
64
64
  import { parseModelsJson, stripBom, stripJsonComments } from "./models-json.mjs";
65
65
  import { isTransientOverlayRead, overlayProviderProblem } from "./model-catalog.mjs";
@@ -87,7 +87,7 @@ import { ABSENT, ASSERTED, DAEMON_APPLIES_BOUNDS, DEFAULT_BACKEND, DOCKER_ENDPOI
87
87
  import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_FIRST_START_TIMEOUT_MS, PODMAN_INFO_TIMEOUT_MS, PODMAN_JOB_USER_FIX, decidePodmanJobUser, makePodmanInfoReader, observePodman, observeRootlessNetns, podmanConfFix, podmanConfWidening, resolvePodmanImageUser } from "./backend-podman.mjs";
88
88
  import { PODMAN_PINNED_FLAGS, buildPodmanRunArgs, containerSpec, podmanArgsFromSpec } from "./docker-run.mjs";
89
89
  import { PODMAN_SERVICE_TIMEOUT_MS, PODMAN_SERVICE_UNIT, makePodmanServiceReader, observeHost, observeRootfulConf, readRootfulService, rootfulConfFix, rootfulConfRetries, rootfulConfResidual, rootfulUnreadList } from "./runtime-observations.mjs";
90
- import { endpointShown, makeDockerEndpointResolver, quotedShown } from "./backend-local.mjs";
90
+ import { JOB_NAME_PREFIX, endpointShown, makeDockerEndpointResolver, quotedShown } from "./backend-local.mjs";
91
91
  import { DEFAULT_EGRESS_PROXY, STOPPED_PROXY_STATES, EGRESS_CANARY_NET_PREFIX, EGRESS_CANARY_PROBE_PREFIX, EGRESS_ENDPOINT_PROBE_PREFIX, egressArmed, egressCanaryNetwork, egressCanaryProbe, egressEndpointProbe, egressEnv, egressProxyName, egressProxyUrl, networkEndpoints, removeNetworkOrSay } from "./egress.mjs";
92
92
  import { detachBlockedSentence, makeDetachGate, runtimeFromFacts } from "./netns-keeper.mjs";
93
93
  import { runLiveProbes } from "./live-probes.mjs";
@@ -95,7 +95,12 @@ import { VALKEY_PASSWORD_KEY, VALKEY_PASSWORD_HOWTO, VALKEY_PORT_KEY, isLoopback
95
95
  import { urlShown, valkeyContextFromResolution, valkeyPasswordFor, valkeyUrlProblem } from "./valkey-endpoint.mjs";
96
96
  import { SANDBOX_TOMBSTONE_STUCK_MS, isSandboxTombstone, sandboxTombstoneAge } from "./sandbox-store.mjs";
97
97
  import { installedUnitPaths, readUnitSeam, readUnitUser } from "./service.mjs";
98
- import { CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
98
+ import { CONTAINER_HOME, SHIPPED_IMAGE_UID, SIZE_LABEL_CPU, SIZE_LABEL_MEM } from "./container-spec.mjs";
99
+ import { DEFAULT_JOB_SIZE, cpuCeilingCenti, formatCpus, formatMemory, jobSizeDefaults, resolveJobSize } from "./job-size.mjs";
100
+ import { SUGGEST_WINDOW_DAYS, cpusText, hostCap, refusalWords, sizeRefusal, suggestSize, suggestionCall, suggestionEvidence } from "./size-suggest.mjs";
101
+ import { SIZING_RECORD_MAX_BYTES, readSizingRecords } from "./size-records.mjs";
102
+ import { CGROUP_PARENT, cgroupParentFor, operatorQuotaCommand, readQuota, reservePlan, userQuotaCommand } from "./cpu-reserve.mjs";
103
+ import { HOST_BUDGET_KEYS, computeHostBudget, hostBudgetSettings, largestFit, neverFits, projectBudgetRow, publishedBudget, readUserServiceLimits } from "./host-budget.mjs";
99
104
  import { makeImagePreflight, normalizeImageId } from "./image-preflight.mjs";
100
105
  import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, JOB_USER_FIX, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
101
106
  import { parseSecretProfiles } from "./secret-profiles.mjs";
@@ -274,6 +279,9 @@ export async function runDoctor(shellVars = process.env, deps = {}) {
274
279
  // Issue #458 (PR #463 round 2): the clock the keeper's age is judged on, epoch ms. Its own name, not `now`: `--live`
275
280
  // pairs `now` with `delay`, and a clock that does not advance without its `delay` would never reach a deadline.
276
281
  wallClock = Date.now,
282
+ // Issue #596, phase 3: the run records the size suggestions read, `(nowMs) => records`. A seam so a test decides
283
+ // the runs; absent, the logs directory's records of the window are read (`readSizingRecords`).
284
+ readRunRecords,
277
285
  // Issue #448: `systemctl show podman.service`, read only where the local daemon is rootful Podman on this host. A seam
278
286
  // so a test decides what the unit says; absent, it spawns systemctl through `spawn`, as the docker reads do.
279
287
  readPodmanService,
@@ -375,7 +383,7 @@ export async function runDoctor(shellVars = process.env, deps = {}) {
375
383
  return { ...(await valkeyAuthState(url, { context, withoutPassword })), passwordSet: Boolean(sent.password), from: sent.from };
376
384
  }
377
385
  : null;
378
- const seams = { cwd, out, spawn, probeValkey, valkeyAuth: valkeyAuthSeam, readHosts, ...(modelCatalog ? { modelCatalog } : {}), ...(piModelLoader ? { piModelLoader } : {}), ...(dollarKeysExist ? { dollarKeysExist } : {}), ...(readAppliedSplit ? { readAppliedSplit } : {}), fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home, providerOracle, facts, jobUserIdentity, stat, passwd, readUnit, readEnvFile: readEnvFileShared, observationFs, jobsDirFs, jobsDirUid, valkeyOwner: valkeyOwnerSeam, isAlive, pid, runTimeouts, live: live === true, wallClock, venueChecks, userName, proxyFilesExist, proxyFileIsDirectory, ...(includeNeeds ? { includeNeeds } : {}), ...(declaredEndpoints ? { declaredEndpoints } : {}), ...(readOverlayFile ? { readOverlayFile } : {}), ...(lstatOverlayFile ? { lstatOverlayFile } : {}), ...(hostAddresses ? { hostAddresses } : {}), ...(readProxyConf ? { readProxyConf } : {}), ...(readPackagedConf ? { readPackagedProxyConf: readPackagedConf } : {}), ...(readPodmanService ? { readPodmanService } : {}), serviceEnvFile: envValues === null ? null : serviceEnvFileOf(envValues, envPath, serviceEnvLoader(platform)) };
386
+ const seams = { cwd, out, spawn, probeValkey, valkeyAuth: valkeyAuthSeam, readHosts, ...(modelCatalog ? { modelCatalog } : {}), ...(piModelLoader ? { piModelLoader } : {}), ...(dollarKeysExist ? { dollarKeysExist } : {}), ...(readAppliedSplit ? { readAppliedSplit } : {}), fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home, providerOracle, facts, jobUserIdentity, stat, passwd, readUnit, readEnvFile: readEnvFileShared, observationFs, jobsDirFs, jobsDirUid, valkeyOwner: valkeyOwnerSeam, isAlive, pid, runTimeouts, live: live === true, wallClock, ...(readRunRecords ? { readRunRecords } : {}), venueChecks, userName, proxyFilesExist, proxyFileIsDirectory, ...(includeNeeds ? { includeNeeds } : {}), ...(declaredEndpoints ? { declaredEndpoints } : {}), ...(readOverlayFile ? { readOverlayFile } : {}), ...(lstatOverlayFile ? { lstatOverlayFile } : {}), ...(hostAddresses ? { hostAddresses } : {}), ...(readProxyConf ? { readProxyConf } : {}), ...(readPackagedConf ? { readPackagedProxyConf: readPackagedConf } : {}), ...(readPodmanService ? { readPodmanService } : {}), serviceEnvFile: envValues === null ? null : serviceEnvFileOf(envValues, envPath, serviceEnvLoader(platform)) };
379
387
  // Issue #471: every other service key, resolved ONCE for the whole run (the fix pass's re-collect and `--live` judge the
380
388
  // same resolution). THE RULE (PR #474's round cap, after three rounds of trust patches): no program doctor starts is
381
389
  // handed anything from `.env`. Every child gets this shell's own environment, the one it had before #471; a `.env`
@@ -587,7 +595,7 @@ export const ENV_FILE_READABLE_KEYS = Object.freeze(["PI_PAUSE_WINDOWS_FILE", "P
587
595
  export const GITHUB_SERVICE_KEYS = Object.freeze(["GITHUB_AUTH_SOURCE", "GITHUB_APP_ID", "GITHUB_APP_INSTALLATION_ID", "GITHUB_APP_PRIVATE_KEY_PATH", "GITHUB_APP_PRIVATE_KEY"]);
588
596
  /** Issue #471: the worker's settings doctor judges, which it read from this shell alone while the service read them from
589
597
  * `.env`. TEMP is TMPDIR's twin in the worker's temp root; PI_CODING_AGENT_DIR is where the worker reads auth.json. */
590
- export const WORKER_SERVICE_KEYS = Object.freeze(["PI_JOB_IMAGE", "PI_TRIGGERS_FILE", "PI_LOGS_DIR", "PI_SETTINGS_FILE", "PI_SESSIONS_DIR", "PI_SESSIONS_TTL_DAYS", "PI_SESSION_MAX_AGE_DAYS", "PI_SESSION_MAX_CONTEXT_PCT", "PI_SESSION_MAX_RESUME_CHAIN", "PI_GLOBAL_PI_DIR", "PI_GLOBAL_ALLOW_EXTENSIONS", "PI_FORWARD_ENV", "PI_AUTH_FROM_PI", "PI_CODING_AGENT_DIR", "PI_BACKEND_FLOOR", "PI_SECRET_PROFILES", "PI_SECRET_RESOLVER_ROOTS", "PI_WAIT_PROFILES", "PI_WAIT_AFTER_MAX_MS", "PI_SANDBOX_RETENTION_HOURS", "PI_ALLOWED_MODELS", "PI_DISPATCH_RUN_ROOTS", "GITHUB_PAT_VAR", "TEMP", ...Object.values(DOLLAR_ENV_NAMES)]);
598
+ export const WORKER_SERVICE_KEYS = Object.freeze(["PI_JOB_IMAGE", "PI_JOB_MEMORY", "PI_JOB_CPUS", "PI_CONCURRENCY", ...Object.values(HOST_BUDGET_KEYS), "PI_TRIGGERS_FILE", "PI_LOGS_DIR", "PI_SETTINGS_FILE", "PI_SESSIONS_DIR", "PI_SESSIONS_TTL_DAYS", "PI_SESSION_MAX_AGE_DAYS", "PI_SESSION_MAX_CONTEXT_PCT", "PI_SESSION_MAX_RESUME_CHAIN", "PI_GLOBAL_PI_DIR", "PI_GLOBAL_ALLOW_EXTENSIONS", "PI_FORWARD_ENV", "PI_AUTH_FROM_PI", "PI_CODING_AGENT_DIR", "PI_BACKEND_FLOOR", "PI_SECRET_PROFILES", "PI_SECRET_RESOLVER_ROOTS", "PI_WAIT_PROFILES", "PI_WAIT_AFTER_MAX_MS", "PI_SANDBOX_RETENTION_HOURS", "PI_ALLOWED_MODELS", "PI_DISPATCH_RUN_ROOTS", "GITHUB_PAT_VAR", "TEMP", ...Object.values(DOLLAR_ENV_NAMES)]);
591
599
  /** Issue #471: the receiver's keys doctor judges its boot by (the receiver's unit reads the same `.env`). */
592
600
  export const RECEIVER_SERVICE_KEYS = Object.freeze(["WEBHOOK_SECRET", "RECEIVER_PORT", "GITLAB_TOKEN", "GITLAB_URL", "GITLAB_WEBHOOK_MODE", "GITLAB_WEBHOOK_SECRET", "FORGEJO_URL", "FORGEJO_TOKEN", "FORGEJO_WEBHOOK_SECRET", "AZURE_ORG_URL", "AZURE_TOKEN", "AZURE_WEBHOOK_MODE", "AZURE_WEBHOOK_SECRET", "AZURE_WEBHOOK_HEADER"]);
593
601
  /**
@@ -1827,9 +1835,58 @@ export async function collectChecks(shellVars, seams) {
1827
1835
  ? { answered: false, reason: "docker-not-found", transient: true }
1828
1836
  : null);
1829
1837
  checks.push(...backendChecks(env, { endpoint, daemon, fs: seams.observationFs, unit: jobUser.unit, ...(podman ? { podman: podman.observed } : {}) }));
1838
+ // Issue #596: the default job size, and what this host's runtime says about the bounds a size becomes.
1839
+ // Each venue this deployment runs, from that venue's own read: `undefined` leaves a venue out, null is "not read".
1840
+ // Issue #596, phase 2: the host budget a worker started now would compute, from the same two reads, and what it means
1841
+ // beside PI_CONCURRENCY and the project sizes in the scoped-limits file (a file that does not load adds no project).
1842
+ // Computed first, because its CPU budget is every job's `--cpus` and the size lines say so.
1843
+ const budgetView = doctorHostBudget(env, { daemon: localUsed ? (daemon ?? null) : undefined, podman: podman?.observed?.read ?? undefined, readFile: (path) => (seams.observationFs ?? { readFileSync }).readFileSync(path, "utf8"), euid: seams.jobUserIdentity?.euid ?? null });
1844
+ checks.push(...jobSizeChecks(env, { daemon: localUsed ? (daemon ?? null) : undefined, podman: podman?.observed?.read !== undefined && podman?.observed?.read !== null ? podman.observed.read : undefined, cpuBudgetCenti: Number.isSafeInteger(budgetView.cpuCenti) ? budgetView.cpuCenti : null }));
1845
+ const concurrencyHere = /^[1-9][0-9]{0,5}$/.test(String(env.PI_CONCURRENCY ?? "").trim()) ? Number(String(env.PI_CONCURRENCY).trim()) : 3;
1846
+ // Said again once the registry is read (below), in this place, when a worker that declares no fleet has peers.
1847
+ const budgetChecksArgs = { concurrency: concurrencyHere, limits: scopedLimitFacts.parseError === null ? scopedLimitFacts.limits : [], env };
1848
+ const budgetChecksAt = checks.length;
1849
+ const budgetChecksHere = hostBudgetChecks(budgetView, budgetChecksArgs);
1850
+ checks.push(...budgetChecksHere);
1851
+ // Issue #596, phase 3: one line per project with its size and what its recent runs suggest, capped at what this host
1852
+ // offers. The records are read only when there is a project to suggest for, and not at all when the scoped-limits
1853
+ // file does not load: every size read without it would be the default, not the project's, and a project that has a
1854
+ // row would be told to `dispatch_limit_add` one.
1855
+ const sizingProjects = readProjectFacts(env, fileExists).projects;
1856
+ if (sizingProjects.length > 0 && !budgetView.error) {
1857
+ if (scopedLimitFacts.parseError !== null) {
1858
+ checks.push({ ok: true, label: "size suggestions: off until the scoped-limits file loads (a size read without it would be the default, not the project's)" });
1859
+ } else {
1860
+ const nowMs = (typeof seams.wallClock === "function" ? seams.wallClock : Date.now)();
1861
+ const read = typeof seams.readRunRecords === "function" ? { records: seams.readRunRecords(nowMs), skipped: 0 } : readSizingRecords(logsDirPath(env, home), { nowMs });
1862
+ const f = budgetView.facts ?? {};
1863
+ const least = (...vs) => {
1864
+ const known = vs.filter((v) => Number.isSafeInteger(v) && v > 0);
1865
+ return known.length === 0 ? null : Math.min(...known);
1866
+ };
1867
+ const total = { memMiB: least(f.memTotalMiB, f.userMemMiB), cpuCenti: least(Number.isSafeInteger(f.hostCpus) ? f.hostCpus * 100 : null, f.userCpuCenti) };
1868
+ checks.push(...sizeSuggestionChecks({ projects: sizingProjects, limits: budgetChecksArgs.limits, env, records: read.records, budget: { memMiB: budgetView.memMiB, cpuCenti: budgetView.cpuCenti }, total, nowMs }));
1869
+ if (read.skipped > 0) checks.push({ ok: true, label: `size suggestions: ${read.skipped} run record${read.skipped === 1 ? "" : "s"} over ${SIZING_RECORD_MAX_BYTES / 1024} KiB skipped, not read` });
1870
+ }
1871
+ }
1872
+ // Issue #596, phase 2: the aggregate CPU reserve per venue, read (never written) the way the worker reads it.
1873
+ if (!budgetView.error) {
1874
+ const reserveReads = await doctorCpuReserve({
1875
+ daemon: localUsed ? (daemon ?? null) : undefined,
1876
+ podman: podman?.observed?.read ?? undefined,
1877
+ endpointLocal: endpoint?.local === true,
1878
+ platform: seams.jobUserIdentity?.platform ?? seams.platform ?? process.platform,
1879
+ run: (bin, args, { timeoutMs }) => dockerRunVia(spawn, timeoutMs, { bin })(args),
1880
+ image: jobImage,
1881
+ imagePresent: imageCode === 0,
1882
+ });
1883
+ checks.push(...cpuReserveChecks(reserveReads, budgetView.cpuCenti));
1884
+ }
1830
1885
  checks.push(...jobUser.checks);
1831
1886
  if (podman) checks.push(...podman.checks);
1832
1887
  if (facts) facts.jobUser = jobUser.forLive;
1888
+ // Issue #596, phase 2: the CPU budget the parent's quota is read back against by `--live`.
1889
+ if (facts) facts.cpuBudgetCenti = budgetView.cpuCenti ?? null;
1833
1890
  // Issue #355: the same answer, kept for `--live`, which decides from it whether its probes' own mounts carry `:Z`.
1834
1891
  if (facts) facts.daemon = jobUser.daemon;
1835
1892
  // Issue #354: which read-backs `--live` runs, and the podman venue's facts for its own.
@@ -2399,6 +2456,33 @@ export async function collectChecks(shellVars, seams) {
2399
2456
  // Valkey doctor reads, a host row is another party's text, and a control byte in a name or zone must not reach the
2400
2457
  // terminal. The registry's own charset already refuses them at the source; this is the reader not relying on it.
2401
2458
  const peers = (fleet.hosts ?? []).map((h) => ({ ...h, name: printable(h.name), tz: h.tz ? printable(h.tz) : h.tz })).filter((h) => h.name !== workerNameOf(declaredWorkerName));
2459
+ if (!env.PI_WORKER_NAME && peers.length > 0) checks.splice(budgetChecksAt, budgetChecksHere.length, ...hostBudgetChecks(budgetView, { ...budgetChecksArgs, peers: true }));
2460
+ // Issue #596, phase 2: this host's own row, when its worker runs and publishes one. Its host budget ledger is held
2461
+ // against the size labels of the job containers each venue's runtime lists, read only when the row carries the ledger.
2462
+ const selfRow = (fleet.hosts ?? []).find((h) => printable(h.name) === workerNameOf(declaredWorkerName)) ?? null;
2463
+ // a worker that has not read its boot listing of the job containers left from before it started admits nothing
2464
+ // on that venue. Per venue since gate round 2 of phase 2: `unlisted:<venue>[,<venue>]`, each a backend name;
2465
+ // a bare `unlisted` (a worker of this round's first draft) is the whole host.
2466
+ const seedField = typeof selfRow?.budgetSeed === "string" ? selfRow.budgetSeed : "";
2467
+ if (seedField === "unlisted" || seedField.startsWith("unlisted:")) {
2468
+ const unread = seedField
2469
+ .slice("unlisted:".length)
2470
+ .split(",")
2471
+ .filter((v) => /^[a-z][a-z0-9-]{0,31}$/.test(v));
2472
+ const bins = unread.map((v) => (v === "podman" ? "`podman ps -a`" : "`docker ps -a`"));
2473
+ const where = unread.length === 0 ? "NO job" : `NO job on the ${unread.join(" and ")} venue${unread.length === 1 ? "" : "s"} (the other venues' jobs still run)`;
2474
+ checks.push({ ok: false, warn: true, label: `this host's worker admits ${where}: it could not list the job containers left from before it started there, so it cannot count what they hold (host_budget_seed_unread)`, fix: `make the runtime answer for the worker's account (${bins.length > 0 ? [...new Set(bins)].join(", ") : "`docker ps -a`, `podman ps -a`"}); the worker asks again every few seconds and starts admitting once it reads the listing` });
2475
+ }
2476
+ if (selfRow && typeof selfRow.usedMemMiB === "string" && selfRow.usedMemMiB !== "") {
2477
+ const listed = [];
2478
+ let listedAll = true;
2479
+ for (const bin of [...(localUsed ? ["docker"] : []), ...(podmanUsed ? ["podman"] : [])]) {
2480
+ const ps = await runCmdCapture(spawn, bin, [...SIZE_LABEL_PS_ARGS], { stdoutOnly: true });
2481
+ if (ps.code !== 0) listedAll = false;
2482
+ else listed.push(...parseSizeLabels(ps.output));
2483
+ }
2484
+ if (listedAll) checks.push(...budgetLedgerChecks({ ...selfRow }, listed));
2485
+ }
2402
2486
  // The applied split (issue #504 part B): one GET whenever this command may talk to the Valkey, so a single host with
2403
2487
  // no envelope that refuses every job is told why. `{ digest }`, `{ undecodable: true }` for a key that exists and does
2404
2488
  // not decode (the worker's EXISTS still counts it as governed), or null (no split, or no answer: nothing is said).
@@ -2494,6 +2578,13 @@ export async function collectChecks(shellVars, seams) {
2494
2578
  // "no projects" against healthy peers would send the operator to the wrong host.
2495
2579
  const projectFactsHere = readProjectFacts(env, fileExists);
2496
2580
  if (projectFactsHere.parseError === null) checks.push(...fleetProjectsChecks(projectsFingerprint(projectFactsHere.projects), peers));
2581
+ // Issue #596: a peer that predates version 3 keeps its last good scoped-limits file once this one is version 3, so
2582
+ // no edit to the file applies on it. Judged on the DECLARED version, or the derived one if that is higher (gate
2583
+ // round 2). SKIPPED when this host's file does not load.
2584
+ if (scopedLimitFacts.parseError === null) checks.push(...fleetSizeChecks(Math.max(Number(scopedLimitFacts.declaredVersion) || 0, scopedLimitsVersionFor(scopedLimitFacts.limits)), peers));
2585
+ // Issue #596, phase 2: each host's budget and use, and which hosts each project's size fits on, from every row that
2586
+ // publishes a budget (this host's own included). SKIPPED when this host's scoped-limits file does not load.
2587
+ if (scopedLimitFacts.parseError === null) checks.push(...fleetBudgetChecks((fleet.hosts ?? []).map((h) => ({ ...h, name: printable(h.name) })), { limits: scopedLimitFacts.limits, env }));
2497
2588
  // Issue #504 part B: one applied split, judged on every host against its own envelope. A host whose envelope digest
2498
2589
  // differs refuses every governed job as `envelope-mismatch`. SKIPPED when this host's file does not load, for the
2499
2590
  // projects check's reason above.
@@ -4470,7 +4561,12 @@ function readScopedLimitFacts(env, fileExists) {
4470
4561
  // guarded read was ever reached. `readFileSync` is synchronous, so no test timeout can interrupt it
4471
4562
  // -- the failure mode is a job that never ends rather than one that goes red.
4472
4563
  if (!statSync(path).isFile()) return { limits: [], parseError: `scoped-limits file is not a regular file: ${path}`, path };
4473
- return { limits: parseScopedLimits(readFileSync(path, "utf8"), path), parseError: null, path };
4564
+ const text = readFileSync(path, "utf8");
4565
+ const limits = parseScopedLimits(text, path);
4566
+ // Issue #596, gate round 2: the version the file DECLARES, beside the rows. An older worker refuses by the declared
4567
+ // number, so a hand-written `"version": 3` with no size field is as unreadable to it as one with a size. The parse
4568
+ // above already accepted this text, so this one cannot throw.
4569
+ return { limits, declaredVersion: JSON.parse(text).version, parseError: null, path };
4474
4570
  } catch (e) {
4475
4571
  return { limits: [], parseError: e?.message ?? String(e), path };
4476
4572
  }
@@ -4847,6 +4943,25 @@ export async function fleetDollarChecks(mine, peers, { dollarKeysExist = async (
4847
4943
  * Hosts are named, never a project's members or name: the registry carries a digest, so "different" is all a reader
4848
4944
  * can know.
4849
4945
  */
4946
+ /**
4947
+ * The fleet's job sizes (issue #596): WARNS when this host's scoped-limits file is version 3 (it declares 3, or a row
4948
+ * carries a size) and a peer publishes no `limitsVersion` of 3 or more. Such a worker refuses the file only when it
4949
+ * LOADS it, at boot; a running one keeps its LAST GOOD file on reload (`scoped_limits_reload_invalid`), so the size and
4950
+ * every later edit to the file (job counts, `concurrent`, dollar caps) do not apply on it until it is upgraded and
4951
+ * restarted, with nothing on its jobs saying so. Nothing is said while the file is version 1 or 2.
4952
+ */
4953
+ export function fleetSizeChecks(fileVersion, peers) {
4954
+ if (fileVersion < 3) return [];
4955
+ const old = peers.filter((h) => !(Number(h.limitsVersion) >= 3));
4956
+ if (old.length === 0) return [];
4957
+ return [{
4958
+ ok: false,
4959
+ warn: true,
4960
+ label: `${old.map((h) => h.name).join(", ")} ${old.length === 1 ? "predates" : "predate"} job sizes (scoped-limits version 3), while this host's file is version 3: a running worker from before keeps its last good file, so neither the size nor any later edit to the file (job counts, concurrent, dollar caps) applies on it until it is upgraded and restarted, and it refuses the file at its next start`,
4961
+ fix: "upgrade and restart every worker on this Valkey before scoped-limits.json is written as version 3",
4962
+ }];
4963
+ }
4964
+
4850
4965
  export function fleetProjectsChecks(mine, peers) {
4851
4966
  const checks = [];
4852
4967
  const opinions = peers.filter((h) => typeof h.fpProjects === "string" && h.fpProjects !== "");
@@ -6173,7 +6288,7 @@ async function egressChecks(env, seams, { dockerCode, imageCode, jobImage, endpo
6173
6288
  * then replaced by none; `CANARY_NO_WORKSPACE` is a path nothing creates, so if a later edit ever kept the mount, the
6174
6289
  * run would fail on a missing source rather than bind a real directory.
6175
6290
  */
6176
- export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, proxy, image, url, user = null, script = egressCanaryScript(url), name = egressCanaryProbe(slug, pid), httpProxy = false }) {
6291
+ export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, proxy, image, url, user = null, script = egressCanaryScript(url), name = egressCanaryProbe(slug, pid), httpProxy = false, size = DEFAULT_JOB_SIZE, hostCpus = null }) {
6177
6292
  if (bin !== "podman") {
6178
6293
  return [
6179
6294
  "run",
@@ -6203,7 +6318,9 @@ export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, prox
6203
6318
  // HOME as a podman job gets it (`resolvePodmanImageUser` always answers CONTAINER_HOME): under keep-id the job user's
6204
6319
  // passwd entry otherwise names this host's home path, which does not exist in the image, and the canary loads pi as
6205
6320
  // that user. No credential rides along: the canary proves the route, and a 401 from the provider is its success.
6206
- const { mounts: _placeholder, ...spec } = containerSpec({ image, name, env: { HOME: CONTAINER_HOME, ...egressEnv({ proxy, armed: true }) }, workspace: CANARY_NO_WORKSPACE, network, user, userns: "keep-id", extraFlags: ["--entrypoint", "node"] });
6321
+ // Issue #596: at a job's size (the deployment's default, which doctor reads from the same settings the worker does) and
6322
+ // under the same `--cpus` ceiling, so the canary's container is a job's in its bounds as well as its flags.
6323
+ const { mounts: _placeholder, ...spec } = containerSpec({ image, name, env: { HOME: CONTAINER_HOME, ...egressEnv({ proxy, armed: true }) }, workspace: CANARY_NO_WORKSPACE, network, user, userns: "keep-id", extraFlags: ["--entrypoint", "node"], size, hostCpus });
6207
6324
  return [...podmanArgsFromSpec({ ...spec, mounts: [] }), "-e", script];
6208
6325
  }
6209
6326
 
@@ -6233,7 +6350,7 @@ const CANARY_NO_WORKSPACE = "/nonexistent/pi-dispatch-egress-canary-mounts-nothi
6233
6350
  * `endpoints` (issue #503) are the declared model endpoints to prove on the same network after the three
6234
6351
  * (`runEndpointProbes`), `[]` by default, so the conformance script and a deployment with none run exactly the three.
6235
6352
  */
6236
- export async function runEgressCanary({ run, bin = "docker", proxy, image, pid = process.pid, user = null, probeRun = null, endpoints = [], gate = makeDetachGate((args, opts) => run(args, { timeoutMs: opts?.timeoutMs ?? CANARY_STEP_TIMEOUT_MS }), { bin }) }) {
6353
+ export async function runEgressCanary({ run, bin = "docker", proxy, image, pid = process.pid, user = null, probeRun = null, endpoints = [], size = DEFAULT_JOB_SIZE, hostCpus = null, gate = makeDetachGate((args, opts) => run(args, { timeoutMs: opts?.timeoutMs ?? CANARY_STEP_TIMEOUT_MS }), { bin }) }) {
6237
6354
  const venue = canaryVenueFor(bin);
6238
6355
  const probe = probeRun ?? ((args) => run(args, { timeoutMs: bin === "podman" ? PODMAN_FIRST_START_TIMEOUT_MS : RUN_TIMEOUTS.cmd }));
6239
6356
  const checks = [];
@@ -6312,7 +6429,7 @@ export async function runEgressCanary({ run, bin = "docker", proxy, image, pid =
6312
6429
  [CANARY_PROBE_SLUGS[1], "an unlisted host", "https://example.com/", false],
6313
6430
  [CANARY_PROBE_SLUGS[2], "plain HTTP to a listed host off port 80", "http://api.anthropic.com:443/", false, egressCanaryPlainScript("http://api.anthropic.com:443/", { proxyUrl: egressProxyUrl(proxy) })],
6314
6431
  ]) {
6315
- const answer = await probe(egressCanaryProbeArgs({ bin, slug, pid, network: net, proxy, image, url, user, ...(script ? { script } : {}) }));
6432
+ const answer = await probe(egressCanaryProbeArgs({ bin, slug, pid, network: net, proxy, image, url, user, size, hostCpus, ...(script ? { script } : {}) }));
6316
6433
  // The script exits 0 (reached) or 3 (blocked). Anything else is the container not running it -- a name clash,
6317
6434
  // the image, the daemon -- which is no reading at all, and must not pass for a deny.
6318
6435
  // `code === null` is the ONE case where a container may still be RUNNING under a name we chose: the
@@ -6388,7 +6505,7 @@ export async function runEgressCanary({ run, bin = "docker", proxy, image, pid =
6388
6505
  if (staleRunner) {
6389
6506
  checks.push({ ok: false, warn: true, label: `${venue.prefix}Model endpoints: not probed, because the job image has no runner module (above), and their probes take the runner's route`, fix: "use a job image built after issue #427, then re-run doctor" });
6390
6507
  } else {
6391
- checks.push(...(await runEndpointProbes({ probe, bin, pid, network: net, proxy, image, user, endpoints, venue, unfinished })));
6508
+ checks.push(...(await runEndpointProbes({ probe, bin, pid, network: net, proxy, image, user, endpoints, venue, unfinished, size, hostCpus })));
6392
6509
  }
6393
6510
  }
6394
6511
  } finally {
@@ -6604,7 +6721,7 @@ export function undeclaredPortNear(endpoint, endpoints) {
6604
6721
  * variable as well on docker: the runner's dispatcher sends an `http://` origin to HTTP_PROXY, which docker's canary
6605
6722
  * argv does not otherwise carry (podman's is a job's environment and has it).
6606
6723
  */
6607
- async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user, endpoints, venue, unfinished }) {
6724
+ async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user, endpoints, venue, unfinished, size = DEFAULT_JOB_SIZE, hostCpus = null }) {
6608
6725
  const checks = [];
6609
6726
  for (const endpoint of endpointsById(endpoints)) {
6610
6727
  const next = undeclaredPortNear(endpoint, endpoints);
@@ -6617,7 +6734,7 @@ async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user,
6617
6734
  ];
6618
6735
  for (const [slug, url, script] of runs) {
6619
6736
  const name = egressEndpointProbe(slug, endpoint.id, pid);
6620
- const answer = await probe(egressCanaryProbeArgs({ bin, slug, name, pid, network, proxy, image, url, user, script, httpProxy: true }));
6737
+ const answer = await probe(egressCanaryProbeArgs({ bin, slug, name, pid, network, proxy, image, url, user, script, httpProxy: true, size, hostCpus }));
6621
6738
  if (answer?.code === null && answer.ended !== "error") unfinished.push(name);
6622
6739
  checks.push(endpointProbeCheck({ slug, endpoint, answer, venue, bin, proxy, next }));
6623
6740
  }
@@ -7359,6 +7476,451 @@ async function defaultProbeValkey(url) {
7359
7476
  }
7360
7477
  }
7361
7478
 
7479
+ /**
7480
+ * The deployment's default job size as doctor reads it (issue #596): `PI_JOB_MEMORY` and `PI_JOB_CPUS` through the
7481
+ * worker's own rule (`jobSizeDefaults`), or the built-in 4g and 2 when they do not parse (`jobSizeChecks` reports that as
7482
+ * the boot refusal it is, and the read-backs then probe the size a fixed configuration would get).
7483
+ */
7484
+ export function doctorJobSize(env) {
7485
+ try {
7486
+ const d = jobSizeDefaults(env);
7487
+ return { memMiB: d.memMiB, cpuCenti: d.cpuCenti, source: d.memSet || d.cpuSet ? "env" : "default" };
7488
+ } catch {
7489
+ return DEFAULT_JOB_SIZE;
7490
+ }
7491
+ }
7492
+
7493
+ /**
7494
+ * The job size lines (issue #596): the default size every job without a project size gets, the `--cpus` ceiling each
7495
+ * venue's runtime gives every job, and WARNINGS where a bound will not hold: the ceiling is unknown (the runtime did
7496
+ * not answer, or gave no CPU count, so jobs run with no `--cpus`, the worker's `cpu_ceiling_unknown`), or the Docker
7497
+ * daemon reports `SwapLimit` or `CPUShares` false (it drops `--memory-swap` or `--cpu-shares` with a client warning
7498
+ * only, the worker's `size_bound_unenforced`). A setting that does not parse is a FAILURE: the worker refuses to start
7499
+ * on it (`loadConfig`).
7500
+ * `daemon` is the local venue's one `docker info` answer (null where it was not read), `undefined` where this
7501
+ * deployment does not run `local`; `podman` is the podman venue's one `podman info` read the same way.
7502
+ */
7503
+ export function jobSizeChecks(env, { daemon = undefined, podman = undefined, cpuBudgetCenti = null } = {}) {
7504
+ let d;
7505
+ try {
7506
+ d = jobSizeDefaults(env);
7507
+ } catch (error) {
7508
+ return [{ ok: false, label: `job size does not parse: ${error.message}, so the worker REFUSES TO START`, fix: "set PI_JOB_MEMORY like 512m, 1536m or 4g and PI_JOB_CPUS like 0.5 or 2 (or unset them for 4g and 2), then re-run doctor" }];
7509
+ }
7510
+ const where = d.memSet || d.cpuSet ? "PI_JOB_MEMORY and PI_JOB_CPUS" : "the built-in default";
7511
+ const checks = [{ ok: true, label: `Job size: ${formatMemory(d.memMiB)} of memory with no swap beyond it, and the CPU weight of ${formatCpus(d.cpuCenti)} CPUs, per job (${where}; a project row's memory and cpus override it, docs/scoped-limits.md)` }];
7512
+ const venues = [];
7513
+ // `reason` is the reader's own token (`timeout`, `unparseable`, ...), never the runtime's text.
7514
+ const reasonOf = (read) => (typeof read?.reason === "string" && /^[a-z0-9-]{1,40}$/.test(read.reason) ? read.reason : "not read");
7515
+ if (daemon !== undefined) venues.push({ venue: "local", answered: daemon?.answered === true, reason: reasonOf(daemon), hostCpus: daemon?.answered === true ? daemon.facts?.hostCpus : null, cmd: "docker info" });
7516
+ if (podman !== undefined) venues.push({ venue: "podman", answered: podman?.answered === true, reason: reasonOf(podman), hostCpus: podman?.answered === true ? podman.info?.hostCpus : null, cmd: "podman info" });
7517
+ for (const v of venues) {
7518
+ // Issue #596, phase 2: the host's CPU budget is every job's `--cpus` (capped at the runtime's count) once it is known.
7519
+ const ceilingCenti = Number.isSafeInteger(v.hostCpus) ? cpuCeilingCenti(v.hostCpus, cpuBudgetCenti) : null;
7520
+ const ceiling = ceilingCenti === null ? null : formatCpus(ceilingCenti);
7521
+ if (ceiling !== null) {
7522
+ // "any ONE job": each container's quota is its own and they do not sum, so busy jobs together can still use
7523
+ // every core (measured, issue #596); the host budget's ledger bounds the jobs' CPU SIZES together, and a reserve
7524
+ // that holds across their USE is the parent cgroup the lab is measuring.
7525
+ checks.push({ ok: true, label: `${v.venue}: any one job may use at most ${ceiling} of this runtime's ${v.hostCpus} CPUs (--cpus); under contention a larger size gets more CPU than a smaller one (--cpu-shares)` });
7526
+ } else {
7527
+ checks.push({ ok: false, warn: true, label: `${v.venue}: ${v.answered ? `\`${v.cmd}\` gave no CPU count` : `\`${v.cmd}\` gave no answer that says its CPU count (${v.reason})`}, so the CPU ceiling is unknown and a job that runs gets no --cpus: it may use every core of the host (cpu_ceiling_unknown)`, fix: `make \`${v.cmd}\` answer for the worker's account with its CPU count, then re-run doctor; the worker reads it with every job's user` });
7528
+ }
7529
+ }
7530
+ const facts = daemon?.answered === true ? daemon.facts : null;
7531
+ if (facts?.swapLimit === false) {
7532
+ checks.push({ ok: false, warn: true, label: "local: the Docker daemon reports SwapLimit false, so it drops --memory-swap and a job may swap beyond its memory (size_bound_unenforced)", fix: "enable swap accounting in the kernel (cgroup v2, or swapaccount=1 on cgroup v1), restart Docker, then re-run doctor" });
7533
+ }
7534
+ if (facts?.cpuShares === false) {
7535
+ checks.push({ ok: false, warn: true, label: "local: the Docker daemon reports CPUShares false, so it drops --cpu-shares and jobs get no CPU weight by size (size_bound_unenforced)", fix: "enable the cpu cgroup controller for Docker (cgroup v2 with cpu delegated), restart Docker, then re-run doctor" });
7536
+ }
7537
+ return checks;
7538
+ }
7539
+
7540
+ /**
7541
+ * The host budget as doctor computes it (issue #596, phase 2, DES-HOST-BUDGET), by the worker's own functions from the
7542
+ * same reads the size lines use: `{ settings, jobDefault, facts, memMiB, cpuCenti, detail }`, or `{ error, jobDefault }`
7543
+ * when a setting does not parse (the worker refuses to start on it). `daemon` is the local venue's `docker info` answer
7544
+ * and `podman` the podman venue's `podman info` read, either absent; on rootless Podman the user service's `memory.max`
7545
+ * and `cpu.max` are read beside them (`readFile`, the observation seam), as the worker reads them. Doctor answers for the
7546
+ * CONFIGURATION, so this is the budget a worker started now would compute; the registry row says what a running one has.
7547
+ */
7548
+ export function doctorHostBudget(env, { daemon = undefined, podman = undefined, readFile = null, euid = null } = {}) {
7549
+ let jobDefault;
7550
+ try {
7551
+ const d = jobSizeDefaults(env);
7552
+ jobDefault = { memMiB: d.memMiB, cpuCenti: d.cpuCenti };
7553
+ } catch {
7554
+ jobDefault = { memMiB: DEFAULT_JOB_SIZE.memMiB, cpuCenti: DEFAULT_JOB_SIZE.cpuCenti };
7555
+ }
7556
+ let settings;
7557
+ try {
7558
+ settings = hostBudgetSettings(env, jobDefault);
7559
+ } catch (error) {
7560
+ return { error: error.message, jobDefault };
7561
+ }
7562
+ const views = [];
7563
+ let user = {};
7564
+ if (daemon?.answered === true) views.push(daemon.facts);
7565
+ if (podman?.answered === true) {
7566
+ views.push(podman.info);
7567
+ if (podman.info?.rootless === true && typeof readFile === "function") user = readUserServiceLimits({ uid: euid, readFile });
7568
+ }
7569
+ const least = (key) => {
7570
+ const known = views.map((v) => v?.[key]).filter((v) => Number.isSafeInteger(v));
7571
+ return known.length > 0 ? Math.min(...known) : null;
7572
+ };
7573
+ const facts = { memTotalMiB: least("memTotalMiB"), hostCpus: least("hostCpus"), ...user };
7574
+ return { settings, jobDefault, facts, ...computeHostBudget(settings, facts, jobDefault) };
7575
+ }
7576
+
7577
+ /**
7578
+ * The aggregate CPU reserve's reads for doctor (issue #596, phase 2): per venue this deployment runs, its `reservePlan`
7579
+ * from the same facts the size lines use and, where a method exists, the parent's quota read the way the worker reads it
7580
+ * (`readQuota`: `systemctl show`, or on Docker's cgroupfs driver a read-only one-shot helper of the job image, run only
7581
+ * when that image is present). Doctor never WRITES a quota; the worker does at boot where it may.
7582
+ * `run(bin, args, { timeoutMs })` resolves `{ code, stdout, error }`.
7583
+ */
7584
+ export async function doctorCpuReserve({ daemon = undefined, podman = undefined, endpointLocal = false, platform = process.platform, run, image, imagePresent = false }) {
7585
+ const venues = [];
7586
+ if (daemon !== undefined && daemon?.answered === true) venues.push({ venue: "local", facts: daemon.facts, endpointLocal });
7587
+ if (podman !== undefined && podman?.answered === true) venues.push({ venue: "podman", facts: podman.info, endpointLocal: podman.info?.serviceIsRemote === false });
7588
+ const out = [];
7589
+ for (const v of venues) {
7590
+ const plan = reservePlan({ ...v, platform });
7591
+ let read = null;
7592
+ if (plan.parent && plan.method === "helper" && !imagePresent) read = { ok: false, reason: "job-image-absent" };
7593
+ else if (plan.parent && plan.method) read = await readQuota(plan, { run, image });
7594
+ out.push({ plan, read, cgroupManager: v.facts?.cgroupDriver ?? v.facts?.cgroupManager ?? null });
7595
+ }
7596
+ return out;
7597
+ }
7598
+
7599
+ /** What each method means for an operator, said once per held line. */
7600
+ const RESERVE_METHOD_SAID = Object.freeze({
7601
+ "user-systemd": "set by the worker through this account's systemd user manager when it starts, and kept across reboots",
7602
+ helper: "written by the worker when it starts through a one-shot helper container of the job image; a Docker Desktop restart drops it and the worker writes it again within ten minutes",
7603
+ "system-systemd": "set by the operator with systemctl, and kept across reboots",
7604
+ });
7605
+
7606
+ /** Why a venue keeps no quota, said in the warning, with the fix beside it. `N` is the budget's percentage. */
7607
+ function reserveUnmanaged(why, want) {
7608
+ const pct = want === null ? "CPUQuota=" : `CPUQuota=${want}%`;
7609
+ const table = {
7610
+ "cgroup-v1": ["the host runs cgroup v1, where the worker keeps no quota (every venue measured is cgroup v2)", `move the host to cgroup v2, or set the parent's quota yourself: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
7611
+ "remote-daemon": ["the Docker daemon uses the systemd driver and is not observed on this host, so its slice cannot be read from here", `on the daemon's host, run once as root: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
7612
+ "rootless-docker": ["rootless Docker's slice is its own account's, which the worker does not manage", `as the daemon's account: \`systemctl --user set-property ${CGROUP_PARENT} ${pct}\``],
7613
+ "driver-unknown": ["the runtime did not say which cgroup driver it uses", "make `docker info` report its CgroupDriver, then re-run doctor"],
7614
+ "podman-rootful-remote": ["this Podman is rootful or remote, whose slice the worker does not manage", `on Podman's host, run once as root: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
7615
+ };
7616
+ return table[why] ?? ["the worker keeps no quota on this venue", `set it yourself: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``];
7617
+ }
7618
+
7619
+ /**
7620
+ * The CPU reserve lines (issue #596, phase 2): per venue, whether every job runs under the one parent cgroup and whether
7621
+ * its quota is the host's CPU budget, so all jobs TOGETHER leave the reserve free. WARNINGS, never failures: without the
7622
+ * quota jobs still share the parent (which already keeps a large job from starving the egress proxy and Valkey), and on
7623
+ * a systemd host only root can set it, so the line prints the one command. `reads` is `doctorCpuReserve`'s answer and
7624
+ * `cpuCenti` the budget doctor computed (an integer, `Infinity` for off, null for unknown).
7625
+ */
7626
+ export function cpuReserveChecks(reads, cpuCenti) {
7627
+ const checks = [];
7628
+ for (const { plan, read, cgroupManager } of reads) {
7629
+ const v = plan.venue;
7630
+ if (!plan.parent) {
7631
+ checks.push({ ok: false, warn: true, label: `${v}: jobs run without the ${CGROUP_PARENT} parent cgroup (Podman uses the ${cgroupManager ?? "unknown"} cgroup manager here), so each job's CPU weight is capped at 1024: the egress proxy and Valkey get a fair share of the CPU, not a reserve`, fix: "run the worker as a systemd user service with linger on, so Podman uses the systemd cgroup manager, then re-run doctor" });
7632
+ continue;
7633
+ }
7634
+ if (cpuCenti === null || cpuCenti === undefined) {
7635
+ checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs yet: the CPU budget is unknown, so no quota is kept on ${CGROUP_PARENT} until it is (jobs still share the parent)`, fix: "see the host budget line above" });
7636
+ continue;
7637
+ }
7638
+ const want = cpuCenti === Infinity ? null : cpuCenti;
7639
+ const pct = want === null ? "none" : `${formatCpus(want)} CPUs`;
7640
+ if (!plan.method) {
7641
+ const [why, fix] = reserveUnmanaged(plan.why, want);
7642
+ if (want === null) checks.push({ ok: true, label: `${v}: the CPU budget is off (PI_HOST_CPU_BUDGET=off), so no quota is set on ${CGROUP_PARENT}; ${why}` });
7643
+ else checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs: every job runs under ${CGROUP_PARENT}, but ${why}`, fix });
7644
+ continue;
7645
+ }
7646
+ const fixFor = (target) =>
7647
+ plan.method === "system-systemd"
7648
+ ? `run once, as root (persistent across reboots): \`${operatorQuotaCommand(target)}\``
7649
+ : plan.method === "user-systemd"
7650
+ ? `the worker sets it when it starts; start or restart it, or run as the worker's account: \`${userQuotaCommand(target)}\``
7651
+ : "the worker writes it when it starts and re-checks it every ten minutes; start or restart the worker";
7652
+ if (!read?.ok) {
7653
+ checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs could be confirmed: the quota of ${CGROUP_PARENT} was not readable (${read?.reason ?? "not read"})`, fix: fixFor(want) });
7654
+ continue;
7655
+ }
7656
+ const has = read.cpuCenti === null ? "no CPU quota" : `a quota of ${formatCpus(read.cpuCenti)} CPUs`;
7657
+ if (read.cpuCenti === want) {
7658
+ if (want === null) checks.push({ ok: true, label: `${v}: the CPU budget is off (PI_HOST_CPU_BUDGET=off), so no quota is set on ${CGROUP_PARENT}: jobs share the parent and together may use every core` });
7659
+ else checks.push({ ok: true, label: `${v}: every job runs under ${CGROUP_PARENT}, whose quota is ${pct}, the host's CPU budget, so all jobs together leave the reserve free (${RESERVE_METHOD_SAID[plan.method]})` });
7660
+ continue;
7661
+ }
7662
+ if (want === null) checks.push({ ok: false, warn: true, label: `${v}: the CPU budget is off, but ${CGROUP_PARENT} still has ${has}, so jobs together are held to it`, fix: fixFor(null) });
7663
+ else checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs: ${CGROUP_PARENT} has ${has}, not the CPU budget of ${pct}`, fix: fixFor(want) });
7664
+ }
7665
+ return checks;
7666
+ }
7667
+
7668
+ /** A memory budget or size for a line: `28g`, `7936m`, `off`, or `unknown`. */
7669
+ function budgetMemShown(memMiB) {
7670
+ return memMiB === Infinity ? "off" : Number.isSafeInteger(memMiB) ? formatMemory(memMiB) : "unknown";
7671
+ }
7672
+ /** A CPU budget or size for a line: `7`, `3.5`, `off`, or `unknown`. */
7673
+ function budgetCpuShown(cpuCenti) {
7674
+ return cpuCenti === Infinity ? "off" : Number.isSafeInteger(cpuCenti) ? formatCpus(cpuCenti) : "unknown";
7675
+ }
7676
+
7677
+ /**
7678
+ * Every project row's size, as `[{ id, size, hostShare, minJobs }]` in file order: the project rows of the limits that
7679
+ * set `memory` or `cpus`, each resolved by the worker's own function (the deployment's settings fill a field the row
7680
+ * leaves out). A project whose row sets no size runs at the default and is not listed.
7681
+ */
7682
+ export function projectSizes(limits, env) {
7683
+ const rows = (Array.isArray(limits) ? limits : []).filter((l) => isProjectScope(l?.scope) && (typeof l.memory === "string" || (l.cpus !== null && l.cpus !== undefined)));
7684
+ const sizes = [];
7685
+ for (const row of rows) {
7686
+ const id = row.scope.slice("project:".length);
7687
+ try {
7688
+ const size = resolveJobSize({ project: id, limits, env });
7689
+ sizes.push({ id, size, ...projectBudgetRow(limits, id) });
7690
+ } catch {
7691
+ // A size the worker refuses is reported by the scoped-limits lines; it fits nowhere and is left out here.
7692
+ }
7693
+ }
7694
+ return sizes;
7695
+ }
7696
+
7697
+ /**
7698
+ * The host budget lines (issue #596, phase 2): the budget and where each half comes from, which of it and
7699
+ * `PI_CONCURRENCY` binds first, and WARNINGS for what the budget will refuse or cannot keep: a project size larger than
7700
+ * the budget (`job-size-exceeds-host` on this host's own queue) or than its `hostShare` of it (`job-size-exceeds-share`), a project whose
7701
+ * `minJobs` times its size is more than its `hostShare` of the budget, and all projects' minimums together above the
7702
+ * budget. Warnings, never failures: the budget is per host, and on a fleet a forge job waits for a host it fits on. A
7703
+ * setting that does not parse is a FAILURE, because the worker refuses to start on it.
7704
+ */
7705
+ export function hostBudgetChecks(view, { concurrency = 3, limits = [], env = {}, peers = false } = {}) {
7706
+ if (view.error) return [{ ok: false, label: `host budget does not parse: ${view.error}, so the worker REFUSES TO START`, fix: "set PI_HOST_MEMORY_BUDGET and PI_HOST_CPU_BUDGET to auto, off or a value such as 64g or 12, and PI_HOST_RESERVE_MEMORY and PI_HOST_RESERVE_CPUS to auto or a value (or unset all four for auto), then re-run doctor" }];
7707
+ const { memMiB, cpuCenti, detail, settings } = view;
7708
+ // a worker without PI_WORKER_NAME drains no host queue and declares no fleet, so it refuses every never-fits job.
7709
+ const multiHost = Boolean(env?.PI_WORKER_NAME);
7710
+ // Why such a worker refuses: alone, there is no other host; beside peers in the registry, it simply declares no fleet.
7711
+ const noFleet = (code) => `every job of it is refused before anything is spent (${code}): ${peers ? "this worker declares no fleet (no PI_WORKER_NAME), so it refuses jobs a bigger peer could run" : "with no PI_WORKER_NAME this host declares no fleet, so there is no other host to wait for"}`;
7712
+ const nameFix = peers ? ", or set PI_WORKER_NAME on this worker so a job of it on the shared queue waits for a peer it fits on" : "";
7713
+ const memWhy = settings.memory.mode === "off" ? "off: PI_HOST_MEMORY_BUDGET" : settings.memory.mode === "value" ? "PI_HOST_MEMORY_BUDGET" : detail.memTotalMiB === null ? "auto" : `auto: ${formatMemory(detail.memTotalMiB)} here, ${formatMemory(detail.memReserveMiB)} kept for the host${detail.memFloored ? ", raised to one job of the default size" : ""}`;
7714
+ const cpuWhy = settings.cpus.mode === "off" ? "off: PI_HOST_CPU_BUDGET" : settings.cpus.mode === "value" ? "PI_HOST_CPU_BUDGET" : detail.cpuTotalCenti === null ? "auto" : `auto: ${formatCpus(detail.cpuTotalCenti)} here, ${formatCpus(detail.cpuReserveCenti)} kept for the host${detail.cpuFloored ? ", raised to one job of the default size" : ""}`;
7715
+ const checks = [{ ok: true, label: `Host budget: memory ${budgetMemShown(memMiB)} (${memWhy}), CPUs ${budgetCpuShown(cpuCenti)} (${cpuWhy}); a job starts only when its size fits beside what already runs on this host` }];
7716
+ // the CPU half is a reservation in the budget's arithmetic and a weight at the runtime, and an operator
7717
+ // sizing a project by "it only needs the cores when it is busy" must know the budget does not see it that way.
7718
+ if (cpuCenti !== Infinity) checks.push({ ok: true, label: "The budget counts each job's cpus as CPU reserved for it, although the runtime uses them as a weight (a busy job may use idle cores beyond them): so a job's cpus must fit the CPU budget beside what runs, even on an idle host" });
7719
+ const unknown = [memMiB === null ? "memory" : null, cpuCenti === null ? "CPU count" : null].filter(Boolean);
7720
+ if (unknown.length > 0) {
7721
+ checks.push({ ok: false, warn: true, label: `host budget: the runtime gave no ${unknown.join(" or ")}, so a worker holds no job back on ${unknown.length === 2 ? "either" : "it"} until it does (host_budget_unknown)`, fix: "make the runtime's info answer for the worker's account (`docker info`, `podman info`), or set PI_HOST_MEMORY_BUDGET and PI_HOST_CPU_BUDGET to values, then re-run doctor" });
7722
+ }
7723
+ // How many jobs of the DEFAULT size the budget holds at once, against PI_CONCURRENCY: whichever is smaller binds.
7724
+ const per = (budget, size) => (budget === null || budget === Infinity ? Infinity : Math.floor(budget / size));
7725
+ const fitDefault = Math.min(per(memMiB, view.jobDefault.memMiB), per(cpuCenti, view.jobDefault.cpuCenti));
7726
+ if (fitDefault === Infinity) checks.push({ ok: true, label: `PI_CONCURRENCY (${concurrency}) is the only bound on how many jobs run at once here: the budget is ${unknown.length > 0 ? "not known yet" : "off"}` });
7727
+ else if (concurrency <= fitDefault) checks.push({ ok: true, label: `PI_CONCURRENCY (${concurrency}) binds first: the budget holds ${fitDefault} job${fitDefault === 1 ? "" : "s"} of the default size (${formatMemory(view.jobDefault.memMiB)}, ${formatCpus(view.jobDefault.cpuCenti)} CPUs) at once` });
7728
+ else checks.push({ ok: true, label: `The host budget binds first: it holds ${fitDefault} job${fitDefault === 1 ? "" : "s"} of the default size (${formatMemory(view.jobDefault.memMiB)}, ${formatCpus(view.jobDefault.cpuCenti)} CPUs) at once, fewer than PI_CONCURRENCY (${concurrency}); bigger sizes fit fewer` });
7729
+ const budget = { memMiB, cpuCenti };
7730
+ const sizes = projectSizes(limits, env);
7731
+ let minMem = 0;
7732
+ let minCpu = 0;
7733
+ for (const p of sizes) {
7734
+ const shown = `${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs`;
7735
+ const misfit = neverFits(p.size, budget, p.hostShare);
7736
+ if (misfit === "host") {
7737
+ checks.push({ ok: false, warn: true, label: `project ${p.id}: its job size (${shown}) is larger than this host's budget (${budgetMemShown(memMiB)}, ${budgetCpuShown(cpuCenti)} CPUs), so ${multiHost ? "a job of it on this host's own queue is refused before anything is spent (job-size-exceeds-host), and one on the shared queue waits for a host it fits on" : noFleet("job-size-exceeds-host")}`, fix: `lower project:${p.id}'s memory or cpus in scoped-limits.json, or raise this host's budget${nameFix}` });
7738
+ } else if (misfit === "share") {
7739
+ checks.push({ ok: false, warn: true, label: `project ${p.id}: its job size (${shown}) is larger than its hostShare (${p.hostShare}%) of this host's budget, so ${multiHost ? "a job of it on this host's own queue is refused before anything is spent (job-size-exceeds-share), and one on the shared queue waits for a host it fits on" : noFleet("job-size-exceeds-share")}`, fix: `raise project:${p.id}'s hostShare or lower its size in scoped-limits.json${nameFix}` });
7740
+ }
7741
+ if (p.minJobs > 0) {
7742
+ minMem += p.minJobs * p.size.memMiB;
7743
+ minCpu += p.minJobs * p.size.cpuCenti;
7744
+ const share = p.hostShare ?? 100;
7745
+ const room = largestFit(budget, share);
7746
+ const over = (need, cap) => Number.isSafeInteger(cap) && need > cap;
7747
+ if (misfit === null && (over(p.minJobs * p.size.memMiB, room.memMiB) || over(p.minJobs * p.size.cpuCenti, room.cpuCenti))) {
7748
+ checks.push({ ok: false, warn: true, label: `project ${p.id}: minJobs ${p.minJobs} of its size (${shown}) is more than its hostShare (${share}%) of this host's budget, so this host can never keep room for all of them at once`, fix: `lower project:${p.id}'s minJobs or size, or raise its hostShare or this host's budget` });
7749
+ }
7750
+ }
7751
+ }
7752
+ const overAll = (need, cap) => Number.isSafeInteger(cap) && need > cap;
7753
+ if (overAll(minMem, memMiB) || overAll(minCpu, cpuCenti)) {
7754
+ checks.push({ ok: false, warn: true, label: `the projects' minJobs together (${formatMemory(minMem)}, ${formatCpus(minCpu)} CPUs) are more than this host's budget (${budgetMemShown(memMiB)}, ${budgetCpuShown(cpuCenti)} CPUs), so this host cannot keep every minimum at once: the oldest waiting jobs are served first`, fix: "lower some projects' minJobs in scoped-limits.json, or raise this host's budget" });
7755
+ }
7756
+ return checks;
7757
+ }
7758
+
7759
+ /**
7760
+ * The fleet's budgets (issue #596, phase 2), from the registry rows (this host's own included): one line per host that
7761
+ * publishes a budget (its budget, what its jobs hold, what its holds keep, and the largest project size that fits it),
7762
+ * one line per sized project naming the hosts it fits on (a WARNING when none does: its jobs on the shared queue wait,
7763
+ * never refused, until a host it fits on is live; a host restarting is missing from the registry for that while), and a
7764
+ * WARNING per host whose budget is below the projects' minJobs
7765
+ * together. Nothing when no host publishes a budget (workers from before it).
7766
+ */
7767
+ export function fleetBudgetChecks(rows, { limits = [], env = {} } = {}) {
7768
+ const hosts = (Array.isArray(rows) ? rows : []).map((row) => ({ name: row.name, budget: publishedBudget(row), row })).filter((h) => h.budget.memMiB !== null && h.budget.cpuCenti !== null);
7769
+ if (hosts.length === 0) return [];
7770
+ const sizes = projectSizes(limits, env);
7771
+ const checks = [];
7772
+ const int = (v) => (typeof v === "string" && /^[0-9]{1,15}$/.test(v) ? Number(v) : null);
7773
+ for (const h of hosts) {
7774
+ const fitting = sizes.filter((p) => neverFits(p.size, h.budget, p.hostShare) === null).sort((a, b) => b.size.memMiB - a.size.memMiB || b.size.cpuCenti - a.size.cpuCenti);
7775
+ const largest = fitting.length > 0 ? `largest project size that fits: ${formatMemory(fitting[0].size.memMiB)}, ${formatCpus(fitting[0].size.cpuCenti)} CPUs (${fitting[0].id})` : sizes.length > 0 ? "no project's size fits" : "no project sets a size";
7776
+ const usedMem = int(h.row.usedMemMiB);
7777
+ const usedCpu = int(h.row.usedCpuCenti);
7778
+ const used = usedMem !== null && usedCpu !== null ? `, in use ${usedMem === 0 ? "0" : formatMemory(usedMem)} and ${formatCpus(usedCpu)} CPUs` : "";
7779
+ checks.push({ ok: true, label: `Host ${h.name}: budget ${budgetMemShown(h.budget.memMiB)} and ${budgetCpuShown(h.budget.cpuCenti)} CPUs${used}; ${largest}` });
7780
+ let minMem = 0;
7781
+ let minCpu = 0;
7782
+ for (const p of sizes) {
7783
+ minMem += p.minJobs * p.size.memMiB;
7784
+ minCpu += p.minJobs * p.size.cpuCenti;
7785
+ }
7786
+ const over = (need, cap) => Number.isSafeInteger(cap) && need > cap;
7787
+ if (over(minMem, h.budget.memMiB) || over(minCpu, h.budget.cpuCenti)) {
7788
+ checks.push({ ok: false, warn: true, label: `Host ${h.name}: the projects' minJobs together (${formatMemory(minMem)}, ${formatCpus(minCpu)} CPUs) are more than its budget, so it cannot keep every minimum at once`, fix: "lower some projects' minJobs, or raise that host's budget" });
7789
+ }
7790
+ }
7791
+ for (const p of sizes) {
7792
+ const on = hosts.filter((h) => neverFits(p.size, h.budget, p.hostShare) === null).map((h) => h.name);
7793
+ if (on.length > 0) checks.push({ ok: true, label: `Project ${p.id} (${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs) fits on: ${on.join(", ")}` });
7794
+ else checks.push({ ok: false, warn: true, label: `Project ${p.id} (${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs) fits on no live host's budget, so its jobs on the shared queue wait (they are never refused for it) until a host it fits on is live`, fix: `if a host it fits on is restarting, wait for it; else lower project:${p.id}'s size, or raise a host's budget` });
7795
+ }
7796
+ return checks;
7797
+ }
7798
+
7799
+ /**
7800
+ * One line per project in projects.json with its job size and what its recent runs suggest (issue #596, phase 3,
7801
+ * DES-SIZE-SUGGESTIONS), from the one pure `suggestSize` the panel and the insights page also call: "fits", "not enough
7802
+ * runs", or a suggestion naming the exact `dispatch_limit_edit` call (`dispatch_limit_add` for a project with no row)
7803
+ * that applies it. A RAISE is a warning with the call as its fix; a lowering is a fact line carrying the call.
7804
+ *
7805
+ * Every raise is CAPPED at what this host offers the project (`hostCap`: its budget per dimension, the project's
7806
+ * `hostShare` of it where the row has one, or with the budget off or unknown the runtime's memory and CPU count, `total`): where the cap binds the line says the project's runs need more than
7807
+ * this host offers, and where the size already is the cap it says so and offers no call. Nothing here advises growing
7808
+ * the host's budget: that budget is what the host promised every other project. The call is offered exactly when this
7809
+ * host's admission would accept the suggested pair (`sizeRefusal`, the one rule the panel and the insights page also
7810
+ * use): a pair it would refuse for ever is flagged, naming the dimension that does not fit, and carries NO call
7811
+ * (applying it would trade one refusal for the same refusal); with the budget off or unknown nothing is refused, so a
7812
+ * size above the runtime's own memory or CPU count is only noted beside its call. Facts (memory pressure at the limit, the
7813
+ * host's CPU ceiling) ride along as information, with no call. Nothing here applies a size: the numbers come from
7814
+ * inside the jobs' containers, and an operator confirms the call.
7815
+ */
7816
+ export function sizeSuggestionChecks({ projects = [], limits = [], env = {}, records = [], budget = null, total = null, nowMs }) {
7817
+ const checks = [];
7818
+ for (const project of Array.isArray(projects) ? projects : []) {
7819
+ const id = project?.id;
7820
+ // the project's hostShare of an integer budget: a job above it is refused here (`job-size-exceeds-share`), so a
7821
+ // raise is capped at the share, never at the whole budget, or the line would offer a call to a size that never runs.
7822
+ const share = projectBudgetRow(limits, id).hostShare;
7823
+ const cap = hostCap(budget, total, share);
7824
+ let current;
7825
+ try {
7826
+ current = resolveJobSize({ project: id, limits, env });
7827
+ } catch {
7828
+ continue; // a refused PI_JOB_MEMORY or PI_JOB_CPUS fails the size lines above; the worker does not start
7829
+ }
7830
+ const s = suggestSize({ project: id, records, current, cap, now: nowMs });
7831
+ const words = suggestionEvidence(s);
7832
+ const size = `${formatMemory(current.memMiB)}, ${cpusText(current.cpuCenti)}`;
7833
+ const call = suggestionCall(s, limits);
7834
+ const facts = [words.memoryFact, words.cpuFact].filter(Boolean);
7835
+ const factTail = facts.length > 0 ? `; ${facts.join("; ")}` : "";
7836
+ const held = s.memory.suggested === null && s.memory.held !== null;
7837
+ if (call === null && !held) {
7838
+ const both = s.memory.reason === "not-enough-runs" && s.cpu.reason === "not-enough-runs";
7839
+ const why = both ? `not enough runs to suggest a size yet (${words.memory} in the last ${SUGGEST_WINDOW_DAYS} days)` : `fits its runs (memory: ${s.memory.reason === "fits" ? words.memory : `${s.memory.reason}, ${words.memory}`}; CPUs: ${s.cpu.reason === "fits" ? words.cpu : `${s.cpu.reason}, ${words.cpu}`})`;
7840
+ checks.push({ ok: true, label: `project ${id}: size ${size}${both ? ": " : " "}${why}${factTail}` });
7841
+ continue;
7842
+ }
7843
+ const parts = [];
7844
+ if (s.memory.suggested) parts.push(`memory ${formatMemory(s.memory.suggested)} (${s.memory.reason}: ${words.memory}${s.memory.held === "cap" ? `; ${words.memoryHeld}` : ""})`);
7845
+ if (s.cpu.suggested) parts.push(`${cpusText(s.cpu.suggested)} (${s.cpu.reason}: ${words.cpu})`);
7846
+ const heldText = held ? `${words.memory}; ${words.memoryHeld}` : "";
7847
+ // THE ONE RULE (`sizeRefusal`, shared with the panel and the insights page): the call is withheld exactly when this
7848
+ // host's admission would refuse the suggested pair for ever (job-size-exceeds-host or -share), whichever dimension
7849
+ // does it. With the budget off or unknown admission refuses nothing, so the call is offered, and a size above the
7850
+ // runtime's own memory or CPU count is only flagged.
7851
+ const pair = { memMiB: s.memory.suggested ?? current.memMiB, cpuCenti: s.cpu.suggested ?? current.cpuCenti };
7852
+ const refusal = call === null ? null : sizeRefusal(pair, [budget], share);
7853
+ const totals = refusal !== null ? [] : [s.memory.overBudget ? `memory ${formatMemory(s.memory.suggested)} is above ${formatMemory(cap.memMiB)}, this host's own memory` : null, s.cpu.overBudget ? `${cpusText(s.cpu.suggested)} ${s.cpu.suggested === 100 ? "is" : "are"} above ${cpusText(cap.cpuCenti)}, this host's own CPU count` : null].filter(Boolean);
7854
+ const over = refusal !== null ? [refusalWords(refusal, s, share, budget)] : [];
7855
+ const note = totals.length > 0 ? ` (note: ${totals.join(" and ")}; its budget is off or unknown, so the worker admits it)` : "";
7856
+ const raise = (s.memory.suggested ?? 0) > current.memMiB || (s.cpu.suggested ?? 0) > current.cpuCenti;
7857
+ const suggests = parts.length > 0 ? `its runs in the last ${SUGGEST_WINDOW_DAYS} days suggest ${parts.join(" and ")}` : "";
7858
+ const label = `project ${id}: size ${size}; ${[heldText ? `in the last ${SUGGEST_WINDOW_DAYS} days ${heldText}` : "", suggests].filter(Boolean).join("; ")}${over.length > 0 ? `, but ${over.join(" and ")}, so a job of it would never fit this host` : ""}${note}${factTail}`;
7859
+ // a size that would never fit is never offered as a call: applying it would turn a refusal into the same refusal.
7860
+ const apply = call === null ? "" : over.length > 0 ? "" : `apply it in the admin panel with ${call} (an operator confirms it; nothing applies a size by itself)`;
7861
+ const never = over.length > 0 ? "no call is offered: a job of that size would never fit this host" : "";
7862
+ if (held) {
7863
+ const above = Number.isSafeInteger(s.memory.cap) && current.memMiB > s.memory.cap;
7864
+ const none = s.memory.held === "largest" ? (above ? "no larger memory is offered: the size is already above what this host offers" : "no larger memory is offered: no larger size fits this host") : "no memory call is offered while the largest size this host offers is unknown (its budget is off or unknown and its runtime's memory was not read)";
7865
+ checks.push({ ok: false, warn: true, label, fix: apply ? `${none}; for the rest, ${apply}` : never ? `${none}; ${never}` : none });
7866
+ } else if (raise || over.length > 0 || totals.length > 0) {
7867
+ checks.push({ ok: false, warn: true, label, fix: apply || never });
7868
+ } else {
7869
+ checks.push({ ok: true, label: `${label}; ${apply.replace(/ \(an operator confirms it; nothing applies a size by itself\)$/, "")}` });
7870
+ }
7871
+ }
7872
+ return checks;
7873
+ }
7874
+
7875
+ /** The `ps` that lists this runtime's job containers with their two size labels (issue #596, phase 2). */
7876
+ export const SIZE_LABEL_PS_ARGS = Object.freeze(["ps", "--filter", `name=${JOB_NAME_PREFIX}`, "--format", `{{.Names}}\t{{.Label "${SIZE_LABEL_MEM}"}}\t{{.Label "${SIZE_LABEL_CPU}"}}`]);
7877
+
7878
+ /**
7879
+ * The running job containers' size labels, from `SIZE_LABEL_PS_ARGS`' output: `[{ name, memMiB, cpuCenti }]`, a label
7880
+ * that is absent or not an integer read as null. Only names in the job namespace (the filter is a substring match).
7881
+ */
7882
+ export function parseSizeLabels(stdout) {
7883
+ const int = (v) => (/^[1-9][0-9]{0,8}$/.test(v ?? "") ? Number(v) : null);
7884
+ return String(stdout ?? "")
7885
+ .split("\n")
7886
+ .map((line) => line.split("\t"))
7887
+ .filter(([name]) => typeof name === "string" && name.startsWith(JOB_NAME_PREFIX))
7888
+ .map(([name, mem, cpu]) => ({ name, memMiB: int(mem?.trim()), cpuCenti: int(cpu?.trim()) }));
7889
+ }
7890
+
7891
+ /**
7892
+ * The ledger held against what runs (issue #596, phase 2): this host's registry row says what its worker's budget
7893
+ * counts (`usedMemMiB`, `usedCpuCenti`, orphans included); the job containers' labels say what runs. A WARNING when they
7894
+ * differ (a job starting or ending between the two reads differs for a moment, so the fix says to re-run first), and
7895
+ * when a job container carries no size label (one started by a worker from before the labels). Nothing without a row
7896
+ * that publishes the two fields.
7897
+ */
7898
+ export function budgetLedgerChecks(selfRow, containers) {
7899
+ const int = (v) => (typeof v === "string" && /^[0-9]{1,15}$/.test(v) ? Number(v) : null);
7900
+ const usedMem = int(selfRow?.usedMemMiB);
7901
+ const usedCpu = int(selfRow?.usedCpuCenti);
7902
+ if (usedMem === null || usedCpu === null || !Array.isArray(containers)) return [];
7903
+ const labelled = containers.filter((c) => c.memMiB !== null && c.cpuCenti !== null);
7904
+ const unlabelled = containers.length - labelled.length;
7905
+ const mem = labelled.reduce((sum, c) => sum + c.memMiB, 0);
7906
+ const cpu = labelled.reduce((sum, c) => sum + c.cpuCenti, 0);
7907
+ const checks = [];
7908
+ if (mem === usedMem && cpu === usedCpu) {
7909
+ checks.push({ ok: true, label: `Host budget ledger matches the running job containers (${containers.length} running, ${mem === 0 ? "0" : formatMemory(mem)} and ${formatCpus(cpu)} CPUs)` });
7910
+ } else {
7911
+ checks.push({ ok: false, warn: true, label: `Host budget ledger holds ${usedMem === 0 ? "0" : formatMemory(usedMem)} and ${formatCpus(usedCpu)} CPUs, while the running job containers are labelled ${mem === 0 ? "0" : formatMemory(mem)} and ${formatCpus(cpu)} CPUs`, fix: "re-run doctor: a job starting or ending between the two reads differs for a moment. A difference that stays is a container the worker does not count, or one whose stop failed (it keeps its hold until the runtime says it is gone)" });
7912
+ }
7913
+ // NAMED, so an operator can find each one (`pi-job-<id>`, a job id and never payload text). A worker that
7914
+ // started beside one counts it at the largest size a project row or the default could have started it at, capped
7915
+ // at the budget, so the ledger line above differs while it runs.
7916
+ if (unlabelled > 0) {
7917
+ const names = containers.filter((c) => c.memMiB === null || c.cpuCenti === null).map((c) => c.name);
7918
+ const shown = names.length > 5 ? `${names.slice(0, 5).join(", ")} and ${names.length - 5} more` : names.join(", ");
7919
+ checks.push({ ok: false, warn: true, label: `${unlabelled} running job container${unlabelled === 1 ? " carries" : "s carry"} no size label (${shown}), so the ledger cannot be checked against ${unlabelled === 1 ? "it" : "them"} (started by a worker from before the host budget); a worker that found ${unlabelled === 1 ? "it" : "them"} at its start counts each at the largest size a project may run at, capped at the budget, until ${unlabelled === 1 ? "it is" : "they are"} gone`, fix: "nothing to do: the label is on every container a current worker starts; stop one early with `docker stop <name>` (or `podman stop`) to give its room back sooner" });
7920
+ }
7921
+ return checks;
7922
+ }
7923
+
7362
7924
  /**
7363
7925
  * WHERE this deployment's jobs run, and what that place actually guarantees (issue #227).
7364
7926
  *
@@ -8333,6 +8895,12 @@ export async function liveChecks(env, seams, facts) {
8333
8895
  const relabel = relabelsPrivateMounts(facts.daemon?.answered ? facts.daemon.facts : null, facts.endpoint, ids.platform ?? seams.platform);
8334
8896
  const result = await runLiveProbes({
8335
8897
  image: facts.jobImage ?? jobImageOf(env).image,
8898
+ // Issue #596: the deployment's default size, and the `--cpus` ceiling from the same `docker info` answer, so the
8899
+ // probe is built at the size a job gets and reads back memory, swap, weight and ceiling against it.
8900
+ size: doctorJobSize(env),
8901
+ hostCpus: facts.daemon?.answered ? (facts.daemon.facts?.hostCpus ?? null) : null,
8902
+ // Issue #596, phase 2: the parent's quota is read back against the CPU budget doctor computed above.
8903
+ cpuBudgetCenti: facts.cpuBudgetCenti ?? null,
8336
8904
  endpoint: facts.endpoint,
8337
8905
  // Asked again right before the first probe command, through the same resolver as the collection's read.
8338
8906
  resolveEndpoint: makeDockerEndpointResolver({ run: dockerRunVia(spawn) }),
@@ -8411,6 +8979,14 @@ function readBackChecks({ venue, bin, result, user, ids, relabel, facts, userFix
8411
8979
  return { ok: false, label: `${prefix}: ${v.property} does NOT hold -- declared ${declared}, observed: ${v.detail}`, fix };
8412
8980
  }),
8413
8981
  );
8982
+ // Issue #596, phase 2: where the runtime put the probe (the jobs' parent cgroup) and, where readable, the parent's
8983
+ // quota against the CPU budget. A line of its own: the reserve is not one of the declared properties.
8984
+ const parent = result.cgroupParent;
8985
+ if (parent) {
8986
+ if (!parent.ok && !parent.warn) checks.push({ ok: false, label: `${prefix}: cgroup parent does NOT hold -- ${parent.detail}`, fix: `the worker builds every job under the ${CGROUP_PARENT} parent (--cgroup-parent); a runtime that records another parent or places the container elsewhere keeps no CPU reserve across jobs: check the runtime's cgroup driver (\`${bin} info\`) and re-run \`pi-dispatch doctor --live\`` });
8987
+ else if (parent.warn) checks.push({ ok: false, warn: true, label: `${prefix}: cgroup parent: ${parent.detail}`, fix: "see the CPU reserve lines above for what keeps the host's reserve on this venue" });
8988
+ else checks.push({ ok: true, label: `${prefix}: cgroup parent holds (${parent.detail})` });
8989
+ }
8414
8990
  checks.push(...noteChecks());
8415
8991
  // What a green read-back does NOT mean, on a line of its own so a row of ✓ is never read as more than it is.
8416
8992
  // Each sentence says only what DID happen: a probe that was not read back has its own line above saying why, and
@@ -8470,10 +9046,17 @@ export async function podmanLiveChecks(env, seams, facts) {
8470
9046
  const readInfo = makePodmanInfoReader({ run: dockerRunVia(spawn, PODMAN_INFO_TIMEOUT_MS, { bin: "podman" }) });
8471
9047
  const run = liveRunVia(spawn, { bin: "podman" });
8472
9048
  const image = facts.jobImage ?? jobImageOf(env).image;
8473
- const canary = await podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, announce: (line) => out(`\nread back on podman: ${line}\n`) });
9049
+ const canary = await podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, size: doctorJobSize(env), announce: (line) => out(`\nread back on podman: ${line}\n`) });
8474
9050
  const egress = { armed: podman.egress.armed, results: canary.results, proxy: podman.egress.proxy, proxyRunning: podman.egress.proxyRunning, keeperBlocked: podman.egress.keeperBlocked ?? null };
8475
9051
  const result = await runLiveProbes({
8476
9052
  image,
9053
+ // Issue #596: as the docker read-back, from this account's own `podman info`.
9054
+ size: doctorJobSize(env),
9055
+ hostCpus: podman.info?.hostCpus ?? null,
9056
+ // Issue #596, phase 2: the parent as this venue's jobs get it (none where Podman's cgroup manager is not systemd),
9057
+ // and its quota against the CPU budget doctor computed above.
9058
+ cgroupParent: cgroupParentFor({ podman: true, cgroupManager: podman.info?.cgroupManager ?? null }),
9059
+ cpuBudgetCenti: facts.cpuBudgetCenti ?? null,
8477
9060
  endpoint: podman.info,
8478
9061
  resolveEndpoint: async () => {
8479
9062
  const again = await readInfo();
@@ -8533,7 +9116,7 @@ export async function podmanLiveChecks(env, seams, facts) {
8533
9116
  * judges a pid against THIS process table and the canary's containers must start where the section looked, and both of
8534
9117
  * those are false the moment CONTAINER_HOST points elsewhere. The re-ask costs one spawn and only with egress armed.
8535
9118
  */
8536
- async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, announce = () => {} }) {
9119
+ async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, size = DEFAULT_JOB_SIZE, announce = () => {} }) {
8537
9120
  const none = { checks: [], results: [] };
8538
9121
  if (podman.egress?.armed === false) return none;
8539
9122
  // Issue #458: on Podman 4.x without the keeper, the canary's own teardown (and the stale sweep's detach) is the step
@@ -8559,7 +9142,7 @@ async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive,
8559
9142
  const endpoints = podman.egress.endpoints ?? [];
8560
9143
  const more = endpoints.length > 0 ? `; then three per declared model endpoint (${endpointsById(endpoints).map((e) => e.id).join(", ")}), named ${EGRESS_ENDPOINT_PROBE_PREFIX}<probe>-<id>-${pid}, removed the same way` : "";
8561
9144
  announce(`starting ${probes.slice(0, -1).join(", ")} and ${probes.at(-1)} from ${image} (as the job user ${podman.user}) on the --internal network ${egressCanaryNetwork(pid)}, with ${podman.egress.proxy} attached, to read the egress allowlist back; all three are removed when the canary ends${more}`);
8562
- const canary = await runEgressCanary({ run, bin: "podman", proxy: podman.egress.proxy, image, pid, user: podman.user, gate, endpoints });
9145
+ const canary = await runEgressCanary({ run, bin: "podman", proxy: podman.egress.proxy, image, pid, user: podman.user, gate, endpoints, size, hostCpus: again.info?.hostCpus ?? null });
8563
9146
  return { checks: [...checks, ...canary.checks], results: canary.results };
8564
9147
  }
8565
9148