@edgehero/pi-dispatch 3.1.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +38 -0
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +599 -16
- package/src/host-budget.mjs +736 -0
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +2 -1
package/src/doctor.mjs
CHANGED
|
@@ -59,7 +59,7 @@ import { spawn as nodeSpawn } from "node:child_process";
|
|
|
59
59
|
import { randomBytes } from "node:crypto";
|
|
60
60
|
import { DEFAULT_MODEL, DEFAULT_PROVIDER, DEFAULT_VALKEY_URL, accountTempRoot, allowedModelsFrom, defaultLogsDir, defaultSandboxDir, defaultSettingsFile, defaultWorkerName, globalExtensionsEnabled, jobsDirOwnerFix, jobsDirPath, sandboxDirOwnerFix, legacyTempStateDir, logsDirPath, modelEndpointsFilePath, delimitedList, envelopeFilePath, pauseWindowsFilePath, projectsFilePath, safeHomeDir, scopedLimitsFilePath, settingsFilePath, underOsTempDir } from "./config.mjs";
|
|
61
61
|
import { SYSTEMD_HAZARD_SHAPES, decodeEnvFile, envFileHazard, envValueShown, quotedRegions, readEnvAssignments, renderEnvValue, envFileWrapperInternal, wrapperInternalSentence } from "./env-file.mjs";
|
|
62
|
-
import { canonicalScope, danglingProjectRows, dollarRowsBelowJobCap, dollarRowsWithoutCap, isModelScope, isProjectScope, loadScopedLimits, parseScopedLimits } from "./scoped-limits.mjs";
|
|
62
|
+
import { canonicalScope, danglingProjectRows, dollarRowsBelowJobCap, dollarRowsWithoutCap, isModelScope, isProjectScope, loadScopedLimits, parseScopedLimits, scopedLimitsVersionFor } from "./scoped-limits.mjs";
|
|
63
63
|
import { EMPTY_PROJECTS_FINGERPRINT, loadProjects, projectsFingerprint } from "./projects.mjs";
|
|
64
64
|
import { parseModelsJson, stripBom, stripJsonComments } from "./models-json.mjs";
|
|
65
65
|
import { isTransientOverlayRead, overlayProviderProblem } from "./model-catalog.mjs";
|
|
@@ -87,7 +87,7 @@ import { ABSENT, ASSERTED, DAEMON_APPLIES_BOUNDS, DEFAULT_BACKEND, DOCKER_ENDPOI
|
|
|
87
87
|
import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_FIRST_START_TIMEOUT_MS, PODMAN_INFO_TIMEOUT_MS, PODMAN_JOB_USER_FIX, decidePodmanJobUser, makePodmanInfoReader, observePodman, observeRootlessNetns, podmanConfFix, podmanConfWidening, resolvePodmanImageUser } from "./backend-podman.mjs";
|
|
88
88
|
import { PODMAN_PINNED_FLAGS, buildPodmanRunArgs, containerSpec, podmanArgsFromSpec } from "./docker-run.mjs";
|
|
89
89
|
import { PODMAN_SERVICE_TIMEOUT_MS, PODMAN_SERVICE_UNIT, makePodmanServiceReader, observeHost, observeRootfulConf, readRootfulService, rootfulConfFix, rootfulConfRetries, rootfulConfResidual, rootfulUnreadList } from "./runtime-observations.mjs";
|
|
90
|
-
import { endpointShown, makeDockerEndpointResolver, quotedShown } from "./backend-local.mjs";
|
|
90
|
+
import { JOB_NAME_PREFIX, endpointShown, makeDockerEndpointResolver, quotedShown } from "./backend-local.mjs";
|
|
91
91
|
import { DEFAULT_EGRESS_PROXY, STOPPED_PROXY_STATES, EGRESS_CANARY_NET_PREFIX, EGRESS_CANARY_PROBE_PREFIX, EGRESS_ENDPOINT_PROBE_PREFIX, egressArmed, egressCanaryNetwork, egressCanaryProbe, egressEndpointProbe, egressEnv, egressProxyName, egressProxyUrl, networkEndpoints, removeNetworkOrSay } from "./egress.mjs";
|
|
92
92
|
import { detachBlockedSentence, makeDetachGate, runtimeFromFacts } from "./netns-keeper.mjs";
|
|
93
93
|
import { runLiveProbes } from "./live-probes.mjs";
|
|
@@ -95,7 +95,12 @@ import { VALKEY_PASSWORD_KEY, VALKEY_PASSWORD_HOWTO, VALKEY_PORT_KEY, isLoopback
|
|
|
95
95
|
import { urlShown, valkeyContextFromResolution, valkeyPasswordFor, valkeyUrlProblem } from "./valkey-endpoint.mjs";
|
|
96
96
|
import { SANDBOX_TOMBSTONE_STUCK_MS, isSandboxTombstone, sandboxTombstoneAge } from "./sandbox-store.mjs";
|
|
97
97
|
import { installedUnitPaths, readUnitSeam, readUnitUser } from "./service.mjs";
|
|
98
|
-
import { CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
|
|
98
|
+
import { CONTAINER_HOME, SHIPPED_IMAGE_UID, SIZE_LABEL_CPU, SIZE_LABEL_MEM } from "./container-spec.mjs";
|
|
99
|
+
import { DEFAULT_JOB_SIZE, cpuCeilingCenti, formatCpus, formatMemory, jobSizeDefaults, resolveJobSize } from "./job-size.mjs";
|
|
100
|
+
import { SUGGEST_WINDOW_DAYS, cpusText, hostCap, refusalWords, sizeRefusal, suggestSize, suggestionCall, suggestionEvidence } from "./size-suggest.mjs";
|
|
101
|
+
import { SIZING_RECORD_MAX_BYTES, readSizingRecords } from "./size-records.mjs";
|
|
102
|
+
import { CGROUP_PARENT, cgroupParentFor, operatorQuotaCommand, readQuota, reservePlan, userQuotaCommand } from "./cpu-reserve.mjs";
|
|
103
|
+
import { HOST_BUDGET_KEYS, computeHostBudget, hostBudgetSettings, largestFit, neverFits, projectBudgetRow, publishedBudget, readUserServiceLimits } from "./host-budget.mjs";
|
|
99
104
|
import { makeImagePreflight, normalizeImageId } from "./image-preflight.mjs";
|
|
100
105
|
import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, JOB_USER_FIX, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
|
|
101
106
|
import { parseSecretProfiles } from "./secret-profiles.mjs";
|
|
@@ -274,6 +279,9 @@ export async function runDoctor(shellVars = process.env, deps = {}) {
|
|
|
274
279
|
// Issue #458 (PR #463 round 2): the clock the keeper's age is judged on, epoch ms. Its own name, not `now`: `--live`
|
|
275
280
|
// pairs `now` with `delay`, and a clock that does not advance without its `delay` would never reach a deadline.
|
|
276
281
|
wallClock = Date.now,
|
|
282
|
+
// Issue #596, phase 3: the run records the size suggestions read, `(nowMs) => records`. A seam so a test decides
|
|
283
|
+
// the runs; absent, the logs directory's records of the window are read (`readSizingRecords`).
|
|
284
|
+
readRunRecords,
|
|
277
285
|
// Issue #448: `systemctl show podman.service`, read only where the local daemon is rootful Podman on this host. A seam
|
|
278
286
|
// so a test decides what the unit says; absent, it spawns systemctl through `spawn`, as the docker reads do.
|
|
279
287
|
readPodmanService,
|
|
@@ -375,7 +383,7 @@ export async function runDoctor(shellVars = process.env, deps = {}) {
|
|
|
375
383
|
return { ...(await valkeyAuthState(url, { context, withoutPassword })), passwordSet: Boolean(sent.password), from: sent.from };
|
|
376
384
|
}
|
|
377
385
|
: null;
|
|
378
|
-
const seams = { cwd, out, spawn, probeValkey, valkeyAuth: valkeyAuthSeam, readHosts, ...(modelCatalog ? { modelCatalog } : {}), ...(piModelLoader ? { piModelLoader } : {}), ...(dollarKeysExist ? { dollarKeysExist } : {}), ...(readAppliedSplit ? { readAppliedSplit } : {}), fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home, providerOracle, facts, jobUserIdentity, stat, passwd, readUnit, readEnvFile: readEnvFileShared, observationFs, jobsDirFs, jobsDirUid, valkeyOwner: valkeyOwnerSeam, isAlive, pid, runTimeouts, live: live === true, wallClock, venueChecks, userName, proxyFilesExist, proxyFileIsDirectory, ...(includeNeeds ? { includeNeeds } : {}), ...(declaredEndpoints ? { declaredEndpoints } : {}), ...(readOverlayFile ? { readOverlayFile } : {}), ...(lstatOverlayFile ? { lstatOverlayFile } : {}), ...(hostAddresses ? { hostAddresses } : {}), ...(readProxyConf ? { readProxyConf } : {}), ...(readPackagedConf ? { readPackagedProxyConf: readPackagedConf } : {}), ...(readPodmanService ? { readPodmanService } : {}), serviceEnvFile: envValues === null ? null : serviceEnvFileOf(envValues, envPath, serviceEnvLoader(platform)) };
|
|
386
|
+
const seams = { cwd, out, spawn, probeValkey, valkeyAuth: valkeyAuthSeam, readHosts, ...(modelCatalog ? { modelCatalog } : {}), ...(piModelLoader ? { piModelLoader } : {}), ...(dollarKeysExist ? { dollarKeysExist } : {}), ...(readAppliedSplit ? { readAppliedSplit } : {}), fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home, providerOracle, facts, jobUserIdentity, stat, passwd, readUnit, readEnvFile: readEnvFileShared, observationFs, jobsDirFs, jobsDirUid, valkeyOwner: valkeyOwnerSeam, isAlive, pid, runTimeouts, live: live === true, wallClock, ...(readRunRecords ? { readRunRecords } : {}), venueChecks, userName, proxyFilesExist, proxyFileIsDirectory, ...(includeNeeds ? { includeNeeds } : {}), ...(declaredEndpoints ? { declaredEndpoints } : {}), ...(readOverlayFile ? { readOverlayFile } : {}), ...(lstatOverlayFile ? { lstatOverlayFile } : {}), ...(hostAddresses ? { hostAddresses } : {}), ...(readProxyConf ? { readProxyConf } : {}), ...(readPackagedConf ? { readPackagedProxyConf: readPackagedConf } : {}), ...(readPodmanService ? { readPodmanService } : {}), serviceEnvFile: envValues === null ? null : serviceEnvFileOf(envValues, envPath, serviceEnvLoader(platform)) };
|
|
379
387
|
// Issue #471: every other service key, resolved ONCE for the whole run (the fix pass's re-collect and `--live` judge the
|
|
380
388
|
// same resolution). THE RULE (PR #474's round cap, after three rounds of trust patches): no program doctor starts is
|
|
381
389
|
// handed anything from `.env`. Every child gets this shell's own environment, the one it had before #471; a `.env`
|
|
@@ -587,7 +595,7 @@ export const ENV_FILE_READABLE_KEYS = Object.freeze(["PI_PAUSE_WINDOWS_FILE", "P
|
|
|
587
595
|
export const GITHUB_SERVICE_KEYS = Object.freeze(["GITHUB_AUTH_SOURCE", "GITHUB_APP_ID", "GITHUB_APP_INSTALLATION_ID", "GITHUB_APP_PRIVATE_KEY_PATH", "GITHUB_APP_PRIVATE_KEY"]);
|
|
588
596
|
/** Issue #471: the worker's settings doctor judges, which it read from this shell alone while the service read them from
|
|
589
597
|
* `.env`. TEMP is TMPDIR's twin in the worker's temp root; PI_CODING_AGENT_DIR is where the worker reads auth.json. */
|
|
590
|
-
export const WORKER_SERVICE_KEYS = Object.freeze(["PI_JOB_IMAGE", "PI_TRIGGERS_FILE", "PI_LOGS_DIR", "PI_SETTINGS_FILE", "PI_SESSIONS_DIR", "PI_SESSIONS_TTL_DAYS", "PI_SESSION_MAX_AGE_DAYS", "PI_SESSION_MAX_CONTEXT_PCT", "PI_SESSION_MAX_RESUME_CHAIN", "PI_GLOBAL_PI_DIR", "PI_GLOBAL_ALLOW_EXTENSIONS", "PI_FORWARD_ENV", "PI_AUTH_FROM_PI", "PI_CODING_AGENT_DIR", "PI_BACKEND_FLOOR", "PI_SECRET_PROFILES", "PI_SECRET_RESOLVER_ROOTS", "PI_WAIT_PROFILES", "PI_WAIT_AFTER_MAX_MS", "PI_SANDBOX_RETENTION_HOURS", "PI_ALLOWED_MODELS", "PI_DISPATCH_RUN_ROOTS", "GITHUB_PAT_VAR", "TEMP", ...Object.values(DOLLAR_ENV_NAMES)]);
|
|
598
|
+
export const WORKER_SERVICE_KEYS = Object.freeze(["PI_JOB_IMAGE", "PI_JOB_MEMORY", "PI_JOB_CPUS", "PI_CONCURRENCY", ...Object.values(HOST_BUDGET_KEYS), "PI_TRIGGERS_FILE", "PI_LOGS_DIR", "PI_SETTINGS_FILE", "PI_SESSIONS_DIR", "PI_SESSIONS_TTL_DAYS", "PI_SESSION_MAX_AGE_DAYS", "PI_SESSION_MAX_CONTEXT_PCT", "PI_SESSION_MAX_RESUME_CHAIN", "PI_GLOBAL_PI_DIR", "PI_GLOBAL_ALLOW_EXTENSIONS", "PI_FORWARD_ENV", "PI_AUTH_FROM_PI", "PI_CODING_AGENT_DIR", "PI_BACKEND_FLOOR", "PI_SECRET_PROFILES", "PI_SECRET_RESOLVER_ROOTS", "PI_WAIT_PROFILES", "PI_WAIT_AFTER_MAX_MS", "PI_SANDBOX_RETENTION_HOURS", "PI_ALLOWED_MODELS", "PI_DISPATCH_RUN_ROOTS", "GITHUB_PAT_VAR", "TEMP", ...Object.values(DOLLAR_ENV_NAMES)]);
|
|
591
599
|
/** Issue #471: the receiver's keys doctor judges its boot by (the receiver's unit reads the same `.env`). */
|
|
592
600
|
export const RECEIVER_SERVICE_KEYS = Object.freeze(["WEBHOOK_SECRET", "RECEIVER_PORT", "GITLAB_TOKEN", "GITLAB_URL", "GITLAB_WEBHOOK_MODE", "GITLAB_WEBHOOK_SECRET", "FORGEJO_URL", "FORGEJO_TOKEN", "FORGEJO_WEBHOOK_SECRET", "AZURE_ORG_URL", "AZURE_TOKEN", "AZURE_WEBHOOK_MODE", "AZURE_WEBHOOK_SECRET", "AZURE_WEBHOOK_HEADER"]);
|
|
593
601
|
/**
|
|
@@ -1827,9 +1835,58 @@ export async function collectChecks(shellVars, seams) {
|
|
|
1827
1835
|
? { answered: false, reason: "docker-not-found", transient: true }
|
|
1828
1836
|
: null);
|
|
1829
1837
|
checks.push(...backendChecks(env, { endpoint, daemon, fs: seams.observationFs, unit: jobUser.unit, ...(podman ? { podman: podman.observed } : {}) }));
|
|
1838
|
+
// Issue #596: the default job size, and what this host's runtime says about the bounds a size becomes.
|
|
1839
|
+
// Each venue this deployment runs, from that venue's own read: `undefined` leaves a venue out, null is "not read".
|
|
1840
|
+
// Issue #596, phase 2: the host budget a worker started now would compute, from the same two reads, and what it means
|
|
1841
|
+
// beside PI_CONCURRENCY and the project sizes in the scoped-limits file (a file that does not load adds no project).
|
|
1842
|
+
// Computed first, because its CPU budget is every job's `--cpus` and the size lines say so.
|
|
1843
|
+
const budgetView = doctorHostBudget(env, { daemon: localUsed ? (daemon ?? null) : undefined, podman: podman?.observed?.read ?? undefined, readFile: (path) => (seams.observationFs ?? { readFileSync }).readFileSync(path, "utf8"), euid: seams.jobUserIdentity?.euid ?? null });
|
|
1844
|
+
checks.push(...jobSizeChecks(env, { daemon: localUsed ? (daemon ?? null) : undefined, podman: podman?.observed?.read !== undefined && podman?.observed?.read !== null ? podman.observed.read : undefined, cpuBudgetCenti: Number.isSafeInteger(budgetView.cpuCenti) ? budgetView.cpuCenti : null }));
|
|
1845
|
+
const concurrencyHere = /^[1-9][0-9]{0,5}$/.test(String(env.PI_CONCURRENCY ?? "").trim()) ? Number(String(env.PI_CONCURRENCY).trim()) : 3;
|
|
1846
|
+
// Said again once the registry is read (below), in this place, when a worker that declares no fleet has peers.
|
|
1847
|
+
const budgetChecksArgs = { concurrency: concurrencyHere, limits: scopedLimitFacts.parseError === null ? scopedLimitFacts.limits : [], env };
|
|
1848
|
+
const budgetChecksAt = checks.length;
|
|
1849
|
+
const budgetChecksHere = hostBudgetChecks(budgetView, budgetChecksArgs);
|
|
1850
|
+
checks.push(...budgetChecksHere);
|
|
1851
|
+
// Issue #596, phase 3: one line per project with its size and what its recent runs suggest, capped at what this host
|
|
1852
|
+
// offers. The records are read only when there is a project to suggest for, and not at all when the scoped-limits
|
|
1853
|
+
// file does not load: every size read without it would be the default, not the project's, and a project that has a
|
|
1854
|
+
// row would be told to `dispatch_limit_add` one.
|
|
1855
|
+
const sizingProjects = readProjectFacts(env, fileExists).projects;
|
|
1856
|
+
if (sizingProjects.length > 0 && !budgetView.error) {
|
|
1857
|
+
if (scopedLimitFacts.parseError !== null) {
|
|
1858
|
+
checks.push({ ok: true, label: "size suggestions: off until the scoped-limits file loads (a size read without it would be the default, not the project's)" });
|
|
1859
|
+
} else {
|
|
1860
|
+
const nowMs = (typeof seams.wallClock === "function" ? seams.wallClock : Date.now)();
|
|
1861
|
+
const read = typeof seams.readRunRecords === "function" ? { records: seams.readRunRecords(nowMs), skipped: 0 } : readSizingRecords(logsDirPath(env, home), { nowMs });
|
|
1862
|
+
const f = budgetView.facts ?? {};
|
|
1863
|
+
const least = (...vs) => {
|
|
1864
|
+
const known = vs.filter((v) => Number.isSafeInteger(v) && v > 0);
|
|
1865
|
+
return known.length === 0 ? null : Math.min(...known);
|
|
1866
|
+
};
|
|
1867
|
+
const total = { memMiB: least(f.memTotalMiB, f.userMemMiB), cpuCenti: least(Number.isSafeInteger(f.hostCpus) ? f.hostCpus * 100 : null, f.userCpuCenti) };
|
|
1868
|
+
checks.push(...sizeSuggestionChecks({ projects: sizingProjects, limits: budgetChecksArgs.limits, env, records: read.records, budget: { memMiB: budgetView.memMiB, cpuCenti: budgetView.cpuCenti }, total, nowMs }));
|
|
1869
|
+
if (read.skipped > 0) checks.push({ ok: true, label: `size suggestions: ${read.skipped} run record${read.skipped === 1 ? "" : "s"} over ${SIZING_RECORD_MAX_BYTES / 1024} KiB skipped, not read` });
|
|
1870
|
+
}
|
|
1871
|
+
}
|
|
1872
|
+
// Issue #596, phase 2: the aggregate CPU reserve per venue, read (never written) the way the worker reads it.
|
|
1873
|
+
if (!budgetView.error) {
|
|
1874
|
+
const reserveReads = await doctorCpuReserve({
|
|
1875
|
+
daemon: localUsed ? (daemon ?? null) : undefined,
|
|
1876
|
+
podman: podman?.observed?.read ?? undefined,
|
|
1877
|
+
endpointLocal: endpoint?.local === true,
|
|
1878
|
+
platform: seams.jobUserIdentity?.platform ?? seams.platform ?? process.platform,
|
|
1879
|
+
run: (bin, args, { timeoutMs }) => dockerRunVia(spawn, timeoutMs, { bin })(args),
|
|
1880
|
+
image: jobImage,
|
|
1881
|
+
imagePresent: imageCode === 0,
|
|
1882
|
+
});
|
|
1883
|
+
checks.push(...cpuReserveChecks(reserveReads, budgetView.cpuCenti));
|
|
1884
|
+
}
|
|
1830
1885
|
checks.push(...jobUser.checks);
|
|
1831
1886
|
if (podman) checks.push(...podman.checks);
|
|
1832
1887
|
if (facts) facts.jobUser = jobUser.forLive;
|
|
1888
|
+
// Issue #596, phase 2: the CPU budget the parent's quota is read back against by `--live`.
|
|
1889
|
+
if (facts) facts.cpuBudgetCenti = budgetView.cpuCenti ?? null;
|
|
1833
1890
|
// Issue #355: the same answer, kept for `--live`, which decides from it whether its probes' own mounts carry `:Z`.
|
|
1834
1891
|
if (facts) facts.daemon = jobUser.daemon;
|
|
1835
1892
|
// Issue #354: which read-backs `--live` runs, and the podman venue's facts for its own.
|
|
@@ -2399,6 +2456,33 @@ export async function collectChecks(shellVars, seams) {
|
|
|
2399
2456
|
// Valkey doctor reads, a host row is another party's text, and a control byte in a name or zone must not reach the
|
|
2400
2457
|
// terminal. The registry's own charset already refuses them at the source; this is the reader not relying on it.
|
|
2401
2458
|
const peers = (fleet.hosts ?? []).map((h) => ({ ...h, name: printable(h.name), tz: h.tz ? printable(h.tz) : h.tz })).filter((h) => h.name !== workerNameOf(declaredWorkerName));
|
|
2459
|
+
if (!env.PI_WORKER_NAME && peers.length > 0) checks.splice(budgetChecksAt, budgetChecksHere.length, ...hostBudgetChecks(budgetView, { ...budgetChecksArgs, peers: true }));
|
|
2460
|
+
// Issue #596, phase 2: this host's own row, when its worker runs and publishes one. Its host budget ledger is held
|
|
2461
|
+
// against the size labels of the job containers each venue's runtime lists, read only when the row carries the ledger.
|
|
2462
|
+
const selfRow = (fleet.hosts ?? []).find((h) => printable(h.name) === workerNameOf(declaredWorkerName)) ?? null;
|
|
2463
|
+
// a worker that has not read its boot listing of the job containers left from before it started admits nothing
|
|
2464
|
+
// on that venue. Per venue since gate round 2 of phase 2: `unlisted:<venue>[,<venue>]`, each a backend name;
|
|
2465
|
+
// a bare `unlisted` (a worker of this round's first draft) is the whole host.
|
|
2466
|
+
const seedField = typeof selfRow?.budgetSeed === "string" ? selfRow.budgetSeed : "";
|
|
2467
|
+
if (seedField === "unlisted" || seedField.startsWith("unlisted:")) {
|
|
2468
|
+
const unread = seedField
|
|
2469
|
+
.slice("unlisted:".length)
|
|
2470
|
+
.split(",")
|
|
2471
|
+
.filter((v) => /^[a-z][a-z0-9-]{0,31}$/.test(v));
|
|
2472
|
+
const bins = unread.map((v) => (v === "podman" ? "`podman ps -a`" : "`docker ps -a`"));
|
|
2473
|
+
const where = unread.length === 0 ? "NO job" : `NO job on the ${unread.join(" and ")} venue${unread.length === 1 ? "" : "s"} (the other venues' jobs still run)`;
|
|
2474
|
+
checks.push({ ok: false, warn: true, label: `this host's worker admits ${where}: it could not list the job containers left from before it started there, so it cannot count what they hold (host_budget_seed_unread)`, fix: `make the runtime answer for the worker's account (${bins.length > 0 ? [...new Set(bins)].join(", ") : "`docker ps -a`, `podman ps -a`"}); the worker asks again every few seconds and starts admitting once it reads the listing` });
|
|
2475
|
+
}
|
|
2476
|
+
if (selfRow && typeof selfRow.usedMemMiB === "string" && selfRow.usedMemMiB !== "") {
|
|
2477
|
+
const listed = [];
|
|
2478
|
+
let listedAll = true;
|
|
2479
|
+
for (const bin of [...(localUsed ? ["docker"] : []), ...(podmanUsed ? ["podman"] : [])]) {
|
|
2480
|
+
const ps = await runCmdCapture(spawn, bin, [...SIZE_LABEL_PS_ARGS], { stdoutOnly: true });
|
|
2481
|
+
if (ps.code !== 0) listedAll = false;
|
|
2482
|
+
else listed.push(...parseSizeLabels(ps.output));
|
|
2483
|
+
}
|
|
2484
|
+
if (listedAll) checks.push(...budgetLedgerChecks({ ...selfRow }, listed));
|
|
2485
|
+
}
|
|
2402
2486
|
// The applied split (issue #504 part B): one GET whenever this command may talk to the Valkey, so a single host with
|
|
2403
2487
|
// no envelope that refuses every job is told why. `{ digest }`, `{ undecodable: true }` for a key that exists and does
|
|
2404
2488
|
// not decode (the worker's EXISTS still counts it as governed), or null (no split, or no answer: nothing is said).
|
|
@@ -2494,6 +2578,13 @@ export async function collectChecks(shellVars, seams) {
|
|
|
2494
2578
|
// "no projects" against healthy peers would send the operator to the wrong host.
|
|
2495
2579
|
const projectFactsHere = readProjectFacts(env, fileExists);
|
|
2496
2580
|
if (projectFactsHere.parseError === null) checks.push(...fleetProjectsChecks(projectsFingerprint(projectFactsHere.projects), peers));
|
|
2581
|
+
// Issue #596: a peer that predates version 3 keeps its last good scoped-limits file once this one is version 3, so
|
|
2582
|
+
// no edit to the file applies on it. Judged on the DECLARED version, or the derived one if that is higher (gate
|
|
2583
|
+
// round 2). SKIPPED when this host's file does not load.
|
|
2584
|
+
if (scopedLimitFacts.parseError === null) checks.push(...fleetSizeChecks(Math.max(Number(scopedLimitFacts.declaredVersion) || 0, scopedLimitsVersionFor(scopedLimitFacts.limits)), peers));
|
|
2585
|
+
// Issue #596, phase 2: each host's budget and use, and which hosts each project's size fits on, from every row that
|
|
2586
|
+
// publishes a budget (this host's own included). SKIPPED when this host's scoped-limits file does not load.
|
|
2587
|
+
if (scopedLimitFacts.parseError === null) checks.push(...fleetBudgetChecks((fleet.hosts ?? []).map((h) => ({ ...h, name: printable(h.name) })), { limits: scopedLimitFacts.limits, env }));
|
|
2497
2588
|
// Issue #504 part B: one applied split, judged on every host against its own envelope. A host whose envelope digest
|
|
2498
2589
|
// differs refuses every governed job as `envelope-mismatch`. SKIPPED when this host's file does not load, for the
|
|
2499
2590
|
// projects check's reason above.
|
|
@@ -4470,7 +4561,12 @@ function readScopedLimitFacts(env, fileExists) {
|
|
|
4470
4561
|
// guarded read was ever reached. `readFileSync` is synchronous, so no test timeout can interrupt it
|
|
4471
4562
|
// -- the failure mode is a job that never ends rather than one that goes red.
|
|
4472
4563
|
if (!statSync(path).isFile()) return { limits: [], parseError: `scoped-limits file is not a regular file: ${path}`, path };
|
|
4473
|
-
|
|
4564
|
+
const text = readFileSync(path, "utf8");
|
|
4565
|
+
const limits = parseScopedLimits(text, path);
|
|
4566
|
+
// Issue #596, gate round 2: the version the file DECLARES, beside the rows. An older worker refuses by the declared
|
|
4567
|
+
// number, so a hand-written `"version": 3` with no size field is as unreadable to it as one with a size. The parse
|
|
4568
|
+
// above already accepted this text, so this one cannot throw.
|
|
4569
|
+
return { limits, declaredVersion: JSON.parse(text).version, parseError: null, path };
|
|
4474
4570
|
} catch (e) {
|
|
4475
4571
|
return { limits: [], parseError: e?.message ?? String(e), path };
|
|
4476
4572
|
}
|
|
@@ -4847,6 +4943,25 @@ export async function fleetDollarChecks(mine, peers, { dollarKeysExist = async (
|
|
|
4847
4943
|
* Hosts are named, never a project's members or name: the registry carries a digest, so "different" is all a reader
|
|
4848
4944
|
* can know.
|
|
4849
4945
|
*/
|
|
4946
|
+
/**
|
|
4947
|
+
* The fleet's job sizes (issue #596): WARNS when this host's scoped-limits file is version 3 (it declares 3, or a row
|
|
4948
|
+
* carries a size) and a peer publishes no `limitsVersion` of 3 or more. Such a worker refuses the file only when it
|
|
4949
|
+
* LOADS it, at boot; a running one keeps its LAST GOOD file on reload (`scoped_limits_reload_invalid`), so the size and
|
|
4950
|
+
* every later edit to the file (job counts, `concurrent`, dollar caps) do not apply on it until it is upgraded and
|
|
4951
|
+
* restarted, with nothing on its jobs saying so. Nothing is said while the file is version 1 or 2.
|
|
4952
|
+
*/
|
|
4953
|
+
export function fleetSizeChecks(fileVersion, peers) {
|
|
4954
|
+
if (fileVersion < 3) return [];
|
|
4955
|
+
const old = peers.filter((h) => !(Number(h.limitsVersion) >= 3));
|
|
4956
|
+
if (old.length === 0) return [];
|
|
4957
|
+
return [{
|
|
4958
|
+
ok: false,
|
|
4959
|
+
warn: true,
|
|
4960
|
+
label: `${old.map((h) => h.name).join(", ")} ${old.length === 1 ? "predates" : "predate"} job sizes (scoped-limits version 3), while this host's file is version 3: a running worker from before keeps its last good file, so neither the size nor any later edit to the file (job counts, concurrent, dollar caps) applies on it until it is upgraded and restarted, and it refuses the file at its next start`,
|
|
4961
|
+
fix: "upgrade and restart every worker on this Valkey before scoped-limits.json is written as version 3",
|
|
4962
|
+
}];
|
|
4963
|
+
}
|
|
4964
|
+
|
|
4850
4965
|
export function fleetProjectsChecks(mine, peers) {
|
|
4851
4966
|
const checks = [];
|
|
4852
4967
|
const opinions = peers.filter((h) => typeof h.fpProjects === "string" && h.fpProjects !== "");
|
|
@@ -6173,7 +6288,7 @@ async function egressChecks(env, seams, { dockerCode, imageCode, jobImage, endpo
|
|
|
6173
6288
|
* then replaced by none; `CANARY_NO_WORKSPACE` is a path nothing creates, so if a later edit ever kept the mount, the
|
|
6174
6289
|
* run would fail on a missing source rather than bind a real directory.
|
|
6175
6290
|
*/
|
|
6176
|
-
export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, proxy, image, url, user = null, script = egressCanaryScript(url), name = egressCanaryProbe(slug, pid), httpProxy = false }) {
|
|
6291
|
+
export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, proxy, image, url, user = null, script = egressCanaryScript(url), name = egressCanaryProbe(slug, pid), httpProxy = false, size = DEFAULT_JOB_SIZE, hostCpus = null }) {
|
|
6177
6292
|
if (bin !== "podman") {
|
|
6178
6293
|
return [
|
|
6179
6294
|
"run",
|
|
@@ -6203,7 +6318,9 @@ export function egressCanaryProbeArgs({ bin = "docker", slug, pid, network, prox
|
|
|
6203
6318
|
// HOME as a podman job gets it (`resolvePodmanImageUser` always answers CONTAINER_HOME): under keep-id the job user's
|
|
6204
6319
|
// passwd entry otherwise names this host's home path, which does not exist in the image, and the canary loads pi as
|
|
6205
6320
|
// that user. No credential rides along: the canary proves the route, and a 401 from the provider is its success.
|
|
6206
|
-
|
|
6321
|
+
// Issue #596: at a job's size (the deployment's default, which doctor reads from the same settings the worker does) and
|
|
6322
|
+
// under the same `--cpus` ceiling, so the canary's container is a job's in its bounds as well as its flags.
|
|
6323
|
+
const { mounts: _placeholder, ...spec } = containerSpec({ image, name, env: { HOME: CONTAINER_HOME, ...egressEnv({ proxy, armed: true }) }, workspace: CANARY_NO_WORKSPACE, network, user, userns: "keep-id", extraFlags: ["--entrypoint", "node"], size, hostCpus });
|
|
6207
6324
|
return [...podmanArgsFromSpec({ ...spec, mounts: [] }), "-e", script];
|
|
6208
6325
|
}
|
|
6209
6326
|
|
|
@@ -6233,7 +6350,7 @@ const CANARY_NO_WORKSPACE = "/nonexistent/pi-dispatch-egress-canary-mounts-nothi
|
|
|
6233
6350
|
* `endpoints` (issue #503) are the declared model endpoints to prove on the same network after the three
|
|
6234
6351
|
* (`runEndpointProbes`), `[]` by default, so the conformance script and a deployment with none run exactly the three.
|
|
6235
6352
|
*/
|
|
6236
|
-
export async function runEgressCanary({ run, bin = "docker", proxy, image, pid = process.pid, user = null, probeRun = null, endpoints = [], gate = makeDetachGate((args, opts) => run(args, { timeoutMs: opts?.timeoutMs ?? CANARY_STEP_TIMEOUT_MS }), { bin }) }) {
|
|
6353
|
+
export async function runEgressCanary({ run, bin = "docker", proxy, image, pid = process.pid, user = null, probeRun = null, endpoints = [], size = DEFAULT_JOB_SIZE, hostCpus = null, gate = makeDetachGate((args, opts) => run(args, { timeoutMs: opts?.timeoutMs ?? CANARY_STEP_TIMEOUT_MS }), { bin }) }) {
|
|
6237
6354
|
const venue = canaryVenueFor(bin);
|
|
6238
6355
|
const probe = probeRun ?? ((args) => run(args, { timeoutMs: bin === "podman" ? PODMAN_FIRST_START_TIMEOUT_MS : RUN_TIMEOUTS.cmd }));
|
|
6239
6356
|
const checks = [];
|
|
@@ -6312,7 +6429,7 @@ export async function runEgressCanary({ run, bin = "docker", proxy, image, pid =
|
|
|
6312
6429
|
[CANARY_PROBE_SLUGS[1], "an unlisted host", "https://example.com/", false],
|
|
6313
6430
|
[CANARY_PROBE_SLUGS[2], "plain HTTP to a listed host off port 80", "http://api.anthropic.com:443/", false, egressCanaryPlainScript("http://api.anthropic.com:443/", { proxyUrl: egressProxyUrl(proxy) })],
|
|
6314
6431
|
]) {
|
|
6315
|
-
const answer = await probe(egressCanaryProbeArgs({ bin, slug, pid, network: net, proxy, image, url, user, ...(script ? { script } : {}) }));
|
|
6432
|
+
const answer = await probe(egressCanaryProbeArgs({ bin, slug, pid, network: net, proxy, image, url, user, size, hostCpus, ...(script ? { script } : {}) }));
|
|
6316
6433
|
// The script exits 0 (reached) or 3 (blocked). Anything else is the container not running it -- a name clash,
|
|
6317
6434
|
// the image, the daemon -- which is no reading at all, and must not pass for a deny.
|
|
6318
6435
|
// `code === null` is the ONE case where a container may still be RUNNING under a name we chose: the
|
|
@@ -6388,7 +6505,7 @@ export async function runEgressCanary({ run, bin = "docker", proxy, image, pid =
|
|
|
6388
6505
|
if (staleRunner) {
|
|
6389
6506
|
checks.push({ ok: false, warn: true, label: `${venue.prefix}Model endpoints: not probed, because the job image has no runner module (above), and their probes take the runner's route`, fix: "use a job image built after issue #427, then re-run doctor" });
|
|
6390
6507
|
} else {
|
|
6391
|
-
checks.push(...(await runEndpointProbes({ probe, bin, pid, network: net, proxy, image, user, endpoints, venue, unfinished })));
|
|
6508
|
+
checks.push(...(await runEndpointProbes({ probe, bin, pid, network: net, proxy, image, user, endpoints, venue, unfinished, size, hostCpus })));
|
|
6392
6509
|
}
|
|
6393
6510
|
}
|
|
6394
6511
|
} finally {
|
|
@@ -6604,7 +6721,7 @@ export function undeclaredPortNear(endpoint, endpoints) {
|
|
|
6604
6721
|
* variable as well on docker: the runner's dispatcher sends an `http://` origin to HTTP_PROXY, which docker's canary
|
|
6605
6722
|
* argv does not otherwise carry (podman's is a job's environment and has it).
|
|
6606
6723
|
*/
|
|
6607
|
-
async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user, endpoints, venue, unfinished }) {
|
|
6724
|
+
async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user, endpoints, venue, unfinished, size = DEFAULT_JOB_SIZE, hostCpus = null }) {
|
|
6608
6725
|
const checks = [];
|
|
6609
6726
|
for (const endpoint of endpointsById(endpoints)) {
|
|
6610
6727
|
const next = undeclaredPortNear(endpoint, endpoints);
|
|
@@ -6617,7 +6734,7 @@ async function runEndpointProbes({ probe, bin, pid, network, proxy, image, user,
|
|
|
6617
6734
|
];
|
|
6618
6735
|
for (const [slug, url, script] of runs) {
|
|
6619
6736
|
const name = egressEndpointProbe(slug, endpoint.id, pid);
|
|
6620
|
-
const answer = await probe(egressCanaryProbeArgs({ bin, slug, name, pid, network, proxy, image, url, user, script, httpProxy: true }));
|
|
6737
|
+
const answer = await probe(egressCanaryProbeArgs({ bin, slug, name, pid, network, proxy, image, url, user, script, httpProxy: true, size, hostCpus }));
|
|
6621
6738
|
if (answer?.code === null && answer.ended !== "error") unfinished.push(name);
|
|
6622
6739
|
checks.push(endpointProbeCheck({ slug, endpoint, answer, venue, bin, proxy, next }));
|
|
6623
6740
|
}
|
|
@@ -7359,6 +7476,451 @@ async function defaultProbeValkey(url) {
|
|
|
7359
7476
|
}
|
|
7360
7477
|
}
|
|
7361
7478
|
|
|
7479
|
+
/**
|
|
7480
|
+
* The deployment's default job size as doctor reads it (issue #596): `PI_JOB_MEMORY` and `PI_JOB_CPUS` through the
|
|
7481
|
+
* worker's own rule (`jobSizeDefaults`), or the built-in 4g and 2 when they do not parse (`jobSizeChecks` reports that as
|
|
7482
|
+
* the boot refusal it is, and the read-backs then probe the size a fixed configuration would get).
|
|
7483
|
+
*/
|
|
7484
|
+
export function doctorJobSize(env) {
|
|
7485
|
+
try {
|
|
7486
|
+
const d = jobSizeDefaults(env);
|
|
7487
|
+
return { memMiB: d.memMiB, cpuCenti: d.cpuCenti, source: d.memSet || d.cpuSet ? "env" : "default" };
|
|
7488
|
+
} catch {
|
|
7489
|
+
return DEFAULT_JOB_SIZE;
|
|
7490
|
+
}
|
|
7491
|
+
}
|
|
7492
|
+
|
|
7493
|
+
/**
|
|
7494
|
+
* The job size lines (issue #596): the default size every job without a project size gets, the `--cpus` ceiling each
|
|
7495
|
+
* venue's runtime gives every job, and WARNINGS where a bound will not hold: the ceiling is unknown (the runtime did
|
|
7496
|
+
* not answer, or gave no CPU count, so jobs run with no `--cpus`, the worker's `cpu_ceiling_unknown`), or the Docker
|
|
7497
|
+
* daemon reports `SwapLimit` or `CPUShares` false (it drops `--memory-swap` or `--cpu-shares` with a client warning
|
|
7498
|
+
* only, the worker's `size_bound_unenforced`). A setting that does not parse is a FAILURE: the worker refuses to start
|
|
7499
|
+
* on it (`loadConfig`).
|
|
7500
|
+
* `daemon` is the local venue's one `docker info` answer (null where it was not read), `undefined` where this
|
|
7501
|
+
* deployment does not run `local`; `podman` is the podman venue's one `podman info` read the same way.
|
|
7502
|
+
*/
|
|
7503
|
+
export function jobSizeChecks(env, { daemon = undefined, podman = undefined, cpuBudgetCenti = null } = {}) {
|
|
7504
|
+
let d;
|
|
7505
|
+
try {
|
|
7506
|
+
d = jobSizeDefaults(env);
|
|
7507
|
+
} catch (error) {
|
|
7508
|
+
return [{ ok: false, label: `job size does not parse: ${error.message}, so the worker REFUSES TO START`, fix: "set PI_JOB_MEMORY like 512m, 1536m or 4g and PI_JOB_CPUS like 0.5 or 2 (or unset them for 4g and 2), then re-run doctor" }];
|
|
7509
|
+
}
|
|
7510
|
+
const where = d.memSet || d.cpuSet ? "PI_JOB_MEMORY and PI_JOB_CPUS" : "the built-in default";
|
|
7511
|
+
const checks = [{ ok: true, label: `Job size: ${formatMemory(d.memMiB)} of memory with no swap beyond it, and the CPU weight of ${formatCpus(d.cpuCenti)} CPUs, per job (${where}; a project row's memory and cpus override it, docs/scoped-limits.md)` }];
|
|
7512
|
+
const venues = [];
|
|
7513
|
+
// `reason` is the reader's own token (`timeout`, `unparseable`, ...), never the runtime's text.
|
|
7514
|
+
const reasonOf = (read) => (typeof read?.reason === "string" && /^[a-z0-9-]{1,40}$/.test(read.reason) ? read.reason : "not read");
|
|
7515
|
+
if (daemon !== undefined) venues.push({ venue: "local", answered: daemon?.answered === true, reason: reasonOf(daemon), hostCpus: daemon?.answered === true ? daemon.facts?.hostCpus : null, cmd: "docker info" });
|
|
7516
|
+
if (podman !== undefined) venues.push({ venue: "podman", answered: podman?.answered === true, reason: reasonOf(podman), hostCpus: podman?.answered === true ? podman.info?.hostCpus : null, cmd: "podman info" });
|
|
7517
|
+
for (const v of venues) {
|
|
7518
|
+
// Issue #596, phase 2: the host's CPU budget is every job's `--cpus` (capped at the runtime's count) once it is known.
|
|
7519
|
+
const ceilingCenti = Number.isSafeInteger(v.hostCpus) ? cpuCeilingCenti(v.hostCpus, cpuBudgetCenti) : null;
|
|
7520
|
+
const ceiling = ceilingCenti === null ? null : formatCpus(ceilingCenti);
|
|
7521
|
+
if (ceiling !== null) {
|
|
7522
|
+
// "any ONE job": each container's quota is its own and they do not sum, so busy jobs together can still use
|
|
7523
|
+
// every core (measured, issue #596); the host budget's ledger bounds the jobs' CPU SIZES together, and a reserve
|
|
7524
|
+
// that holds across their USE is the parent cgroup the lab is measuring.
|
|
7525
|
+
checks.push({ ok: true, label: `${v.venue}: any one job may use at most ${ceiling} of this runtime's ${v.hostCpus} CPUs (--cpus); under contention a larger size gets more CPU than a smaller one (--cpu-shares)` });
|
|
7526
|
+
} else {
|
|
7527
|
+
checks.push({ ok: false, warn: true, label: `${v.venue}: ${v.answered ? `\`${v.cmd}\` gave no CPU count` : `\`${v.cmd}\` gave no answer that says its CPU count (${v.reason})`}, so the CPU ceiling is unknown and a job that runs gets no --cpus: it may use every core of the host (cpu_ceiling_unknown)`, fix: `make \`${v.cmd}\` answer for the worker's account with its CPU count, then re-run doctor; the worker reads it with every job's user` });
|
|
7528
|
+
}
|
|
7529
|
+
}
|
|
7530
|
+
const facts = daemon?.answered === true ? daemon.facts : null;
|
|
7531
|
+
if (facts?.swapLimit === false) {
|
|
7532
|
+
checks.push({ ok: false, warn: true, label: "local: the Docker daemon reports SwapLimit false, so it drops --memory-swap and a job may swap beyond its memory (size_bound_unenforced)", fix: "enable swap accounting in the kernel (cgroup v2, or swapaccount=1 on cgroup v1), restart Docker, then re-run doctor" });
|
|
7533
|
+
}
|
|
7534
|
+
if (facts?.cpuShares === false) {
|
|
7535
|
+
checks.push({ ok: false, warn: true, label: "local: the Docker daemon reports CPUShares false, so it drops --cpu-shares and jobs get no CPU weight by size (size_bound_unenforced)", fix: "enable the cpu cgroup controller for Docker (cgroup v2 with cpu delegated), restart Docker, then re-run doctor" });
|
|
7536
|
+
}
|
|
7537
|
+
return checks;
|
|
7538
|
+
}
|
|
7539
|
+
|
|
7540
|
+
/**
|
|
7541
|
+
* The host budget as doctor computes it (issue #596, phase 2, DES-HOST-BUDGET), by the worker's own functions from the
|
|
7542
|
+
* same reads the size lines use: `{ settings, jobDefault, facts, memMiB, cpuCenti, detail }`, or `{ error, jobDefault }`
|
|
7543
|
+
* when a setting does not parse (the worker refuses to start on it). `daemon` is the local venue's `docker info` answer
|
|
7544
|
+
* and `podman` the podman venue's `podman info` read, either absent; on rootless Podman the user service's `memory.max`
|
|
7545
|
+
* and `cpu.max` are read beside them (`readFile`, the observation seam), as the worker reads them. Doctor answers for the
|
|
7546
|
+
* CONFIGURATION, so this is the budget a worker started now would compute; the registry row says what a running one has.
|
|
7547
|
+
*/
|
|
7548
|
+
export function doctorHostBudget(env, { daemon = undefined, podman = undefined, readFile = null, euid = null } = {}) {
|
|
7549
|
+
let jobDefault;
|
|
7550
|
+
try {
|
|
7551
|
+
const d = jobSizeDefaults(env);
|
|
7552
|
+
jobDefault = { memMiB: d.memMiB, cpuCenti: d.cpuCenti };
|
|
7553
|
+
} catch {
|
|
7554
|
+
jobDefault = { memMiB: DEFAULT_JOB_SIZE.memMiB, cpuCenti: DEFAULT_JOB_SIZE.cpuCenti };
|
|
7555
|
+
}
|
|
7556
|
+
let settings;
|
|
7557
|
+
try {
|
|
7558
|
+
settings = hostBudgetSettings(env, jobDefault);
|
|
7559
|
+
} catch (error) {
|
|
7560
|
+
return { error: error.message, jobDefault };
|
|
7561
|
+
}
|
|
7562
|
+
const views = [];
|
|
7563
|
+
let user = {};
|
|
7564
|
+
if (daemon?.answered === true) views.push(daemon.facts);
|
|
7565
|
+
if (podman?.answered === true) {
|
|
7566
|
+
views.push(podman.info);
|
|
7567
|
+
if (podman.info?.rootless === true && typeof readFile === "function") user = readUserServiceLimits({ uid: euid, readFile });
|
|
7568
|
+
}
|
|
7569
|
+
const least = (key) => {
|
|
7570
|
+
const known = views.map((v) => v?.[key]).filter((v) => Number.isSafeInteger(v));
|
|
7571
|
+
return known.length > 0 ? Math.min(...known) : null;
|
|
7572
|
+
};
|
|
7573
|
+
const facts = { memTotalMiB: least("memTotalMiB"), hostCpus: least("hostCpus"), ...user };
|
|
7574
|
+
return { settings, jobDefault, facts, ...computeHostBudget(settings, facts, jobDefault) };
|
|
7575
|
+
}
|
|
7576
|
+
|
|
7577
|
+
/**
|
|
7578
|
+
* The aggregate CPU reserve's reads for doctor (issue #596, phase 2): per venue this deployment runs, its `reservePlan`
|
|
7579
|
+
* from the same facts the size lines use and, where a method exists, the parent's quota read the way the worker reads it
|
|
7580
|
+
* (`readQuota`: `systemctl show`, or on Docker's cgroupfs driver a read-only one-shot helper of the job image, run only
|
|
7581
|
+
* when that image is present). Doctor never WRITES a quota; the worker does at boot where it may.
|
|
7582
|
+
* `run(bin, args, { timeoutMs })` resolves `{ code, stdout, error }`.
|
|
7583
|
+
*/
|
|
7584
|
+
export async function doctorCpuReserve({ daemon = undefined, podman = undefined, endpointLocal = false, platform = process.platform, run, image, imagePresent = false }) {
|
|
7585
|
+
const venues = [];
|
|
7586
|
+
if (daemon !== undefined && daemon?.answered === true) venues.push({ venue: "local", facts: daemon.facts, endpointLocal });
|
|
7587
|
+
if (podman !== undefined && podman?.answered === true) venues.push({ venue: "podman", facts: podman.info, endpointLocal: podman.info?.serviceIsRemote === false });
|
|
7588
|
+
const out = [];
|
|
7589
|
+
for (const v of venues) {
|
|
7590
|
+
const plan = reservePlan({ ...v, platform });
|
|
7591
|
+
let read = null;
|
|
7592
|
+
if (plan.parent && plan.method === "helper" && !imagePresent) read = { ok: false, reason: "job-image-absent" };
|
|
7593
|
+
else if (plan.parent && plan.method) read = await readQuota(plan, { run, image });
|
|
7594
|
+
out.push({ plan, read, cgroupManager: v.facts?.cgroupDriver ?? v.facts?.cgroupManager ?? null });
|
|
7595
|
+
}
|
|
7596
|
+
return out;
|
|
7597
|
+
}
|
|
7598
|
+
|
|
7599
|
+
/** What each method means for an operator, said once per held line. */
|
|
7600
|
+
const RESERVE_METHOD_SAID = Object.freeze({
|
|
7601
|
+
"user-systemd": "set by the worker through this account's systemd user manager when it starts, and kept across reboots",
|
|
7602
|
+
helper: "written by the worker when it starts through a one-shot helper container of the job image; a Docker Desktop restart drops it and the worker writes it again within ten minutes",
|
|
7603
|
+
"system-systemd": "set by the operator with systemctl, and kept across reboots",
|
|
7604
|
+
});
|
|
7605
|
+
|
|
7606
|
+
/** Why a venue keeps no quota, said in the warning, with the fix beside it. `N` is the budget's percentage. */
|
|
7607
|
+
function reserveUnmanaged(why, want) {
|
|
7608
|
+
const pct = want === null ? "CPUQuota=" : `CPUQuota=${want}%`;
|
|
7609
|
+
const table = {
|
|
7610
|
+
"cgroup-v1": ["the host runs cgroup v1, where the worker keeps no quota (every venue measured is cgroup v2)", `move the host to cgroup v2, or set the parent's quota yourself: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
|
|
7611
|
+
"remote-daemon": ["the Docker daemon uses the systemd driver and is not observed on this host, so its slice cannot be read from here", `on the daemon's host, run once as root: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
|
|
7612
|
+
"rootless-docker": ["rootless Docker's slice is its own account's, which the worker does not manage", `as the daemon's account: \`systemctl --user set-property ${CGROUP_PARENT} ${pct}\``],
|
|
7613
|
+
"driver-unknown": ["the runtime did not say which cgroup driver it uses", "make `docker info` report its CgroupDriver, then re-run doctor"],
|
|
7614
|
+
"podman-rootful-remote": ["this Podman is rootful or remote, whose slice the worker does not manage", `on Podman's host, run once as root: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``],
|
|
7615
|
+
};
|
|
7616
|
+
return table[why] ?? ["the worker keeps no quota on this venue", `set it yourself: \`sudo systemctl set-property ${CGROUP_PARENT} ${pct}\``];
|
|
7617
|
+
}
|
|
7618
|
+
|
|
7619
|
+
/**
|
|
7620
|
+
* The CPU reserve lines (issue #596, phase 2): per venue, whether every job runs under the one parent cgroup and whether
|
|
7621
|
+
* its quota is the host's CPU budget, so all jobs TOGETHER leave the reserve free. WARNINGS, never failures: without the
|
|
7622
|
+
* quota jobs still share the parent (which already keeps a large job from starving the egress proxy and Valkey), and on
|
|
7623
|
+
* a systemd host only root can set it, so the line prints the one command. `reads` is `doctorCpuReserve`'s answer and
|
|
7624
|
+
* `cpuCenti` the budget doctor computed (an integer, `Infinity` for off, null for unknown).
|
|
7625
|
+
*/
|
|
7626
|
+
export function cpuReserveChecks(reads, cpuCenti) {
|
|
7627
|
+
const checks = [];
|
|
7628
|
+
for (const { plan, read, cgroupManager } of reads) {
|
|
7629
|
+
const v = plan.venue;
|
|
7630
|
+
if (!plan.parent) {
|
|
7631
|
+
checks.push({ ok: false, warn: true, label: `${v}: jobs run without the ${CGROUP_PARENT} parent cgroup (Podman uses the ${cgroupManager ?? "unknown"} cgroup manager here), so each job's CPU weight is capped at 1024: the egress proxy and Valkey get a fair share of the CPU, not a reserve`, fix: "run the worker as a systemd user service with linger on, so Podman uses the systemd cgroup manager, then re-run doctor" });
|
|
7632
|
+
continue;
|
|
7633
|
+
}
|
|
7634
|
+
if (cpuCenti === null || cpuCenti === undefined) {
|
|
7635
|
+
checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs yet: the CPU budget is unknown, so no quota is kept on ${CGROUP_PARENT} until it is (jobs still share the parent)`, fix: "see the host budget line above" });
|
|
7636
|
+
continue;
|
|
7637
|
+
}
|
|
7638
|
+
const want = cpuCenti === Infinity ? null : cpuCenti;
|
|
7639
|
+
const pct = want === null ? "none" : `${formatCpus(want)} CPUs`;
|
|
7640
|
+
if (!plan.method) {
|
|
7641
|
+
const [why, fix] = reserveUnmanaged(plan.why, want);
|
|
7642
|
+
if (want === null) checks.push({ ok: true, label: `${v}: the CPU budget is off (PI_HOST_CPU_BUDGET=off), so no quota is set on ${CGROUP_PARENT}; ${why}` });
|
|
7643
|
+
else checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs: every job runs under ${CGROUP_PARENT}, but ${why}`, fix });
|
|
7644
|
+
continue;
|
|
7645
|
+
}
|
|
7646
|
+
const fixFor = (target) =>
|
|
7647
|
+
plan.method === "system-systemd"
|
|
7648
|
+
? `run once, as root (persistent across reboots): \`${operatorQuotaCommand(target)}\``
|
|
7649
|
+
: plan.method === "user-systemd"
|
|
7650
|
+
? `the worker sets it when it starts; start or restart it, or run as the worker's account: \`${userQuotaCommand(target)}\``
|
|
7651
|
+
: "the worker writes it when it starts and re-checks it every ten minutes; start or restart the worker";
|
|
7652
|
+
if (!read?.ok) {
|
|
7653
|
+
checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs could be confirmed: the quota of ${CGROUP_PARENT} was not readable (${read?.reason ?? "not read"})`, fix: fixFor(want) });
|
|
7654
|
+
continue;
|
|
7655
|
+
}
|
|
7656
|
+
const has = read.cpuCenti === null ? "no CPU quota" : `a quota of ${formatCpus(read.cpuCenti)} CPUs`;
|
|
7657
|
+
if (read.cpuCenti === want) {
|
|
7658
|
+
if (want === null) checks.push({ ok: true, label: `${v}: the CPU budget is off (PI_HOST_CPU_BUDGET=off), so no quota is set on ${CGROUP_PARENT}: jobs share the parent and together may use every core` });
|
|
7659
|
+
else checks.push({ ok: true, label: `${v}: every job runs under ${CGROUP_PARENT}, whose quota is ${pct}, the host's CPU budget, so all jobs together leave the reserve free (${RESERVE_METHOD_SAID[plan.method]})` });
|
|
7660
|
+
continue;
|
|
7661
|
+
}
|
|
7662
|
+
if (want === null) checks.push({ ok: false, warn: true, label: `${v}: the CPU budget is off, but ${CGROUP_PARENT} still has ${has}, so jobs together are held to it`, fix: fixFor(null) });
|
|
7663
|
+
else checks.push({ ok: false, warn: true, label: `${v}: no host CPU reserve across jobs: ${CGROUP_PARENT} has ${has}, not the CPU budget of ${pct}`, fix: fixFor(want) });
|
|
7664
|
+
}
|
|
7665
|
+
return checks;
|
|
7666
|
+
}
|
|
7667
|
+
|
|
7668
|
+
/** A memory budget or size for a line: `28g`, `7936m`, `off`, or `unknown`. */
|
|
7669
|
+
function budgetMemShown(memMiB) {
|
|
7670
|
+
return memMiB === Infinity ? "off" : Number.isSafeInteger(memMiB) ? formatMemory(memMiB) : "unknown";
|
|
7671
|
+
}
|
|
7672
|
+
/** A CPU budget or size for a line: `7`, `3.5`, `off`, or `unknown`. */
|
|
7673
|
+
function budgetCpuShown(cpuCenti) {
|
|
7674
|
+
return cpuCenti === Infinity ? "off" : Number.isSafeInteger(cpuCenti) ? formatCpus(cpuCenti) : "unknown";
|
|
7675
|
+
}
|
|
7676
|
+
|
|
7677
|
+
/**
|
|
7678
|
+
* Every project row's size, as `[{ id, size, hostShare, minJobs }]` in file order: the project rows of the limits that
|
|
7679
|
+
* set `memory` or `cpus`, each resolved by the worker's own function (the deployment's settings fill a field the row
|
|
7680
|
+
* leaves out). A project whose row sets no size runs at the default and is not listed.
|
|
7681
|
+
*/
|
|
7682
|
+
export function projectSizes(limits, env) {
|
|
7683
|
+
const rows = (Array.isArray(limits) ? limits : []).filter((l) => isProjectScope(l?.scope) && (typeof l.memory === "string" || (l.cpus !== null && l.cpus !== undefined)));
|
|
7684
|
+
const sizes = [];
|
|
7685
|
+
for (const row of rows) {
|
|
7686
|
+
const id = row.scope.slice("project:".length);
|
|
7687
|
+
try {
|
|
7688
|
+
const size = resolveJobSize({ project: id, limits, env });
|
|
7689
|
+
sizes.push({ id, size, ...projectBudgetRow(limits, id) });
|
|
7690
|
+
} catch {
|
|
7691
|
+
// A size the worker refuses is reported by the scoped-limits lines; it fits nowhere and is left out here.
|
|
7692
|
+
}
|
|
7693
|
+
}
|
|
7694
|
+
return sizes;
|
|
7695
|
+
}
|
|
7696
|
+
|
|
7697
|
+
/**
|
|
7698
|
+
* The host budget lines (issue #596, phase 2): the budget and where each half comes from, which of it and
|
|
7699
|
+
* `PI_CONCURRENCY` binds first, and WARNINGS for what the budget will refuse or cannot keep: a project size larger than
|
|
7700
|
+
* the budget (`job-size-exceeds-host` on this host's own queue) or than its `hostShare` of it (`job-size-exceeds-share`), a project whose
|
|
7701
|
+
* `minJobs` times its size is more than its `hostShare` of the budget, and all projects' minimums together above the
|
|
7702
|
+
* budget. Warnings, never failures: the budget is per host, and on a fleet a forge job waits for a host it fits on. A
|
|
7703
|
+
* setting that does not parse is a FAILURE, because the worker refuses to start on it.
|
|
7704
|
+
*/
|
|
7705
|
+
export function hostBudgetChecks(view, { concurrency = 3, limits = [], env = {}, peers = false } = {}) {
|
|
7706
|
+
if (view.error) return [{ ok: false, label: `host budget does not parse: ${view.error}, so the worker REFUSES TO START`, fix: "set PI_HOST_MEMORY_BUDGET and PI_HOST_CPU_BUDGET to auto, off or a value such as 64g or 12, and PI_HOST_RESERVE_MEMORY and PI_HOST_RESERVE_CPUS to auto or a value (or unset all four for auto), then re-run doctor" }];
|
|
7707
|
+
const { memMiB, cpuCenti, detail, settings } = view;
|
|
7708
|
+
// a worker without PI_WORKER_NAME drains no host queue and declares no fleet, so it refuses every never-fits job.
|
|
7709
|
+
const multiHost = Boolean(env?.PI_WORKER_NAME);
|
|
7710
|
+
// Why such a worker refuses: alone, there is no other host; beside peers in the registry, it simply declares no fleet.
|
|
7711
|
+
const noFleet = (code) => `every job of it is refused before anything is spent (${code}): ${peers ? "this worker declares no fleet (no PI_WORKER_NAME), so it refuses jobs a bigger peer could run" : "with no PI_WORKER_NAME this host declares no fleet, so there is no other host to wait for"}`;
|
|
7712
|
+
const nameFix = peers ? ", or set PI_WORKER_NAME on this worker so a job of it on the shared queue waits for a peer it fits on" : "";
|
|
7713
|
+
const memWhy = settings.memory.mode === "off" ? "off: PI_HOST_MEMORY_BUDGET" : settings.memory.mode === "value" ? "PI_HOST_MEMORY_BUDGET" : detail.memTotalMiB === null ? "auto" : `auto: ${formatMemory(detail.memTotalMiB)} here, ${formatMemory(detail.memReserveMiB)} kept for the host${detail.memFloored ? ", raised to one job of the default size" : ""}`;
|
|
7714
|
+
const cpuWhy = settings.cpus.mode === "off" ? "off: PI_HOST_CPU_BUDGET" : settings.cpus.mode === "value" ? "PI_HOST_CPU_BUDGET" : detail.cpuTotalCenti === null ? "auto" : `auto: ${formatCpus(detail.cpuTotalCenti)} here, ${formatCpus(detail.cpuReserveCenti)} kept for the host${detail.cpuFloored ? ", raised to one job of the default size" : ""}`;
|
|
7715
|
+
const checks = [{ ok: true, label: `Host budget: memory ${budgetMemShown(memMiB)} (${memWhy}), CPUs ${budgetCpuShown(cpuCenti)} (${cpuWhy}); a job starts only when its size fits beside what already runs on this host` }];
|
|
7716
|
+
// the CPU half is a reservation in the budget's arithmetic and a weight at the runtime, and an operator
|
|
7717
|
+
// sizing a project by "it only needs the cores when it is busy" must know the budget does not see it that way.
|
|
7718
|
+
if (cpuCenti !== Infinity) checks.push({ ok: true, label: "The budget counts each job's cpus as CPU reserved for it, although the runtime uses them as a weight (a busy job may use idle cores beyond them): so a job's cpus must fit the CPU budget beside what runs, even on an idle host" });
|
|
7719
|
+
const unknown = [memMiB === null ? "memory" : null, cpuCenti === null ? "CPU count" : null].filter(Boolean);
|
|
7720
|
+
if (unknown.length > 0) {
|
|
7721
|
+
checks.push({ ok: false, warn: true, label: `host budget: the runtime gave no ${unknown.join(" or ")}, so a worker holds no job back on ${unknown.length === 2 ? "either" : "it"} until it does (host_budget_unknown)`, fix: "make the runtime's info answer for the worker's account (`docker info`, `podman info`), or set PI_HOST_MEMORY_BUDGET and PI_HOST_CPU_BUDGET to values, then re-run doctor" });
|
|
7722
|
+
}
|
|
7723
|
+
// How many jobs of the DEFAULT size the budget holds at once, against PI_CONCURRENCY: whichever is smaller binds.
|
|
7724
|
+
const per = (budget, size) => (budget === null || budget === Infinity ? Infinity : Math.floor(budget / size));
|
|
7725
|
+
const fitDefault = Math.min(per(memMiB, view.jobDefault.memMiB), per(cpuCenti, view.jobDefault.cpuCenti));
|
|
7726
|
+
if (fitDefault === Infinity) checks.push({ ok: true, label: `PI_CONCURRENCY (${concurrency}) is the only bound on how many jobs run at once here: the budget is ${unknown.length > 0 ? "not known yet" : "off"}` });
|
|
7727
|
+
else if (concurrency <= fitDefault) checks.push({ ok: true, label: `PI_CONCURRENCY (${concurrency}) binds first: the budget holds ${fitDefault} job${fitDefault === 1 ? "" : "s"} of the default size (${formatMemory(view.jobDefault.memMiB)}, ${formatCpus(view.jobDefault.cpuCenti)} CPUs) at once` });
|
|
7728
|
+
else checks.push({ ok: true, label: `The host budget binds first: it holds ${fitDefault} job${fitDefault === 1 ? "" : "s"} of the default size (${formatMemory(view.jobDefault.memMiB)}, ${formatCpus(view.jobDefault.cpuCenti)} CPUs) at once, fewer than PI_CONCURRENCY (${concurrency}); bigger sizes fit fewer` });
|
|
7729
|
+
const budget = { memMiB, cpuCenti };
|
|
7730
|
+
const sizes = projectSizes(limits, env);
|
|
7731
|
+
let minMem = 0;
|
|
7732
|
+
let minCpu = 0;
|
|
7733
|
+
for (const p of sizes) {
|
|
7734
|
+
const shown = `${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs`;
|
|
7735
|
+
const misfit = neverFits(p.size, budget, p.hostShare);
|
|
7736
|
+
if (misfit === "host") {
|
|
7737
|
+
checks.push({ ok: false, warn: true, label: `project ${p.id}: its job size (${shown}) is larger than this host's budget (${budgetMemShown(memMiB)}, ${budgetCpuShown(cpuCenti)} CPUs), so ${multiHost ? "a job of it on this host's own queue is refused before anything is spent (job-size-exceeds-host), and one on the shared queue waits for a host it fits on" : noFleet("job-size-exceeds-host")}`, fix: `lower project:${p.id}'s memory or cpus in scoped-limits.json, or raise this host's budget${nameFix}` });
|
|
7738
|
+
} else if (misfit === "share") {
|
|
7739
|
+
checks.push({ ok: false, warn: true, label: `project ${p.id}: its job size (${shown}) is larger than its hostShare (${p.hostShare}%) of this host's budget, so ${multiHost ? "a job of it on this host's own queue is refused before anything is spent (job-size-exceeds-share), and one on the shared queue waits for a host it fits on" : noFleet("job-size-exceeds-share")}`, fix: `raise project:${p.id}'s hostShare or lower its size in scoped-limits.json${nameFix}` });
|
|
7740
|
+
}
|
|
7741
|
+
if (p.minJobs > 0) {
|
|
7742
|
+
minMem += p.minJobs * p.size.memMiB;
|
|
7743
|
+
minCpu += p.minJobs * p.size.cpuCenti;
|
|
7744
|
+
const share = p.hostShare ?? 100;
|
|
7745
|
+
const room = largestFit(budget, share);
|
|
7746
|
+
const over = (need, cap) => Number.isSafeInteger(cap) && need > cap;
|
|
7747
|
+
if (misfit === null && (over(p.minJobs * p.size.memMiB, room.memMiB) || over(p.minJobs * p.size.cpuCenti, room.cpuCenti))) {
|
|
7748
|
+
checks.push({ ok: false, warn: true, label: `project ${p.id}: minJobs ${p.minJobs} of its size (${shown}) is more than its hostShare (${share}%) of this host's budget, so this host can never keep room for all of them at once`, fix: `lower project:${p.id}'s minJobs or size, or raise its hostShare or this host's budget` });
|
|
7749
|
+
}
|
|
7750
|
+
}
|
|
7751
|
+
}
|
|
7752
|
+
const overAll = (need, cap) => Number.isSafeInteger(cap) && need > cap;
|
|
7753
|
+
if (overAll(minMem, memMiB) || overAll(minCpu, cpuCenti)) {
|
|
7754
|
+
checks.push({ ok: false, warn: true, label: `the projects' minJobs together (${formatMemory(minMem)}, ${formatCpus(minCpu)} CPUs) are more than this host's budget (${budgetMemShown(memMiB)}, ${budgetCpuShown(cpuCenti)} CPUs), so this host cannot keep every minimum at once: the oldest waiting jobs are served first`, fix: "lower some projects' minJobs in scoped-limits.json, or raise this host's budget" });
|
|
7755
|
+
}
|
|
7756
|
+
return checks;
|
|
7757
|
+
}
|
|
7758
|
+
|
|
7759
|
+
/**
|
|
7760
|
+
* The fleet's budgets (issue #596, phase 2), from the registry rows (this host's own included): one line per host that
|
|
7761
|
+
* publishes a budget (its budget, what its jobs hold, what its holds keep, and the largest project size that fits it),
|
|
7762
|
+
* one line per sized project naming the hosts it fits on (a WARNING when none does: its jobs on the shared queue wait,
|
|
7763
|
+
* never refused, until a host it fits on is live; a host restarting is missing from the registry for that while), and a
|
|
7764
|
+
* WARNING per host whose budget is below the projects' minJobs
|
|
7765
|
+
* together. Nothing when no host publishes a budget (workers from before it).
|
|
7766
|
+
*/
|
|
7767
|
+
export function fleetBudgetChecks(rows, { limits = [], env = {} } = {}) {
|
|
7768
|
+
const hosts = (Array.isArray(rows) ? rows : []).map((row) => ({ name: row.name, budget: publishedBudget(row), row })).filter((h) => h.budget.memMiB !== null && h.budget.cpuCenti !== null);
|
|
7769
|
+
if (hosts.length === 0) return [];
|
|
7770
|
+
const sizes = projectSizes(limits, env);
|
|
7771
|
+
const checks = [];
|
|
7772
|
+
const int = (v) => (typeof v === "string" && /^[0-9]{1,15}$/.test(v) ? Number(v) : null);
|
|
7773
|
+
for (const h of hosts) {
|
|
7774
|
+
const fitting = sizes.filter((p) => neverFits(p.size, h.budget, p.hostShare) === null).sort((a, b) => b.size.memMiB - a.size.memMiB || b.size.cpuCenti - a.size.cpuCenti);
|
|
7775
|
+
const largest = fitting.length > 0 ? `largest project size that fits: ${formatMemory(fitting[0].size.memMiB)}, ${formatCpus(fitting[0].size.cpuCenti)} CPUs (${fitting[0].id})` : sizes.length > 0 ? "no project's size fits" : "no project sets a size";
|
|
7776
|
+
const usedMem = int(h.row.usedMemMiB);
|
|
7777
|
+
const usedCpu = int(h.row.usedCpuCenti);
|
|
7778
|
+
const used = usedMem !== null && usedCpu !== null ? `, in use ${usedMem === 0 ? "0" : formatMemory(usedMem)} and ${formatCpus(usedCpu)} CPUs` : "";
|
|
7779
|
+
checks.push({ ok: true, label: `Host ${h.name}: budget ${budgetMemShown(h.budget.memMiB)} and ${budgetCpuShown(h.budget.cpuCenti)} CPUs${used}; ${largest}` });
|
|
7780
|
+
let minMem = 0;
|
|
7781
|
+
let minCpu = 0;
|
|
7782
|
+
for (const p of sizes) {
|
|
7783
|
+
minMem += p.minJobs * p.size.memMiB;
|
|
7784
|
+
minCpu += p.minJobs * p.size.cpuCenti;
|
|
7785
|
+
}
|
|
7786
|
+
const over = (need, cap) => Number.isSafeInteger(cap) && need > cap;
|
|
7787
|
+
if (over(minMem, h.budget.memMiB) || over(minCpu, h.budget.cpuCenti)) {
|
|
7788
|
+
checks.push({ ok: false, warn: true, label: `Host ${h.name}: the projects' minJobs together (${formatMemory(minMem)}, ${formatCpus(minCpu)} CPUs) are more than its budget, so it cannot keep every minimum at once`, fix: "lower some projects' minJobs, or raise that host's budget" });
|
|
7789
|
+
}
|
|
7790
|
+
}
|
|
7791
|
+
for (const p of sizes) {
|
|
7792
|
+
const on = hosts.filter((h) => neverFits(p.size, h.budget, p.hostShare) === null).map((h) => h.name);
|
|
7793
|
+
if (on.length > 0) checks.push({ ok: true, label: `Project ${p.id} (${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs) fits on: ${on.join(", ")}` });
|
|
7794
|
+
else checks.push({ ok: false, warn: true, label: `Project ${p.id} (${formatMemory(p.size.memMiB)}, ${formatCpus(p.size.cpuCenti)} CPUs) fits on no live host's budget, so its jobs on the shared queue wait (they are never refused for it) until a host it fits on is live`, fix: `if a host it fits on is restarting, wait for it; else lower project:${p.id}'s size, or raise a host's budget` });
|
|
7795
|
+
}
|
|
7796
|
+
return checks;
|
|
7797
|
+
}
|
|
7798
|
+
|
|
7799
|
+
/**
|
|
7800
|
+
* One line per project in projects.json with its job size and what its recent runs suggest (issue #596, phase 3,
|
|
7801
|
+
* DES-SIZE-SUGGESTIONS), from the one pure `suggestSize` the panel and the insights page also call: "fits", "not enough
|
|
7802
|
+
* runs", or a suggestion naming the exact `dispatch_limit_edit` call (`dispatch_limit_add` for a project with no row)
|
|
7803
|
+
* that applies it. A RAISE is a warning with the call as its fix; a lowering is a fact line carrying the call.
|
|
7804
|
+
*
|
|
7805
|
+
* Every raise is CAPPED at what this host offers the project (`hostCap`: its budget per dimension, the project's
|
|
7806
|
+
* `hostShare` of it where the row has one, or with the budget off or unknown the runtime's memory and CPU count, `total`): where the cap binds the line says the project's runs need more than
|
|
7807
|
+
* this host offers, and where the size already is the cap it says so and offers no call. Nothing here advises growing
|
|
7808
|
+
* the host's budget: that budget is what the host promised every other project. The call is offered exactly when this
|
|
7809
|
+
* host's admission would accept the suggested pair (`sizeRefusal`, the one rule the panel and the insights page also
|
|
7810
|
+
* use): a pair it would refuse for ever is flagged, naming the dimension that does not fit, and carries NO call
|
|
7811
|
+
* (applying it would trade one refusal for the same refusal); with the budget off or unknown nothing is refused, so a
|
|
7812
|
+
* size above the runtime's own memory or CPU count is only noted beside its call. Facts (memory pressure at the limit, the
|
|
7813
|
+
* host's CPU ceiling) ride along as information, with no call. Nothing here applies a size: the numbers come from
|
|
7814
|
+
* inside the jobs' containers, and an operator confirms the call.
|
|
7815
|
+
*/
|
|
7816
|
+
export function sizeSuggestionChecks({ projects = [], limits = [], env = {}, records = [], budget = null, total = null, nowMs }) {
|
|
7817
|
+
const checks = [];
|
|
7818
|
+
for (const project of Array.isArray(projects) ? projects : []) {
|
|
7819
|
+
const id = project?.id;
|
|
7820
|
+
// the project's hostShare of an integer budget: a job above it is refused here (`job-size-exceeds-share`), so a
|
|
7821
|
+
// raise is capped at the share, never at the whole budget, or the line would offer a call to a size that never runs.
|
|
7822
|
+
const share = projectBudgetRow(limits, id).hostShare;
|
|
7823
|
+
const cap = hostCap(budget, total, share);
|
|
7824
|
+
let current;
|
|
7825
|
+
try {
|
|
7826
|
+
current = resolveJobSize({ project: id, limits, env });
|
|
7827
|
+
} catch {
|
|
7828
|
+
continue; // a refused PI_JOB_MEMORY or PI_JOB_CPUS fails the size lines above; the worker does not start
|
|
7829
|
+
}
|
|
7830
|
+
const s = suggestSize({ project: id, records, current, cap, now: nowMs });
|
|
7831
|
+
const words = suggestionEvidence(s);
|
|
7832
|
+
const size = `${formatMemory(current.memMiB)}, ${cpusText(current.cpuCenti)}`;
|
|
7833
|
+
const call = suggestionCall(s, limits);
|
|
7834
|
+
const facts = [words.memoryFact, words.cpuFact].filter(Boolean);
|
|
7835
|
+
const factTail = facts.length > 0 ? `; ${facts.join("; ")}` : "";
|
|
7836
|
+
const held = s.memory.suggested === null && s.memory.held !== null;
|
|
7837
|
+
if (call === null && !held) {
|
|
7838
|
+
const both = s.memory.reason === "not-enough-runs" && s.cpu.reason === "not-enough-runs";
|
|
7839
|
+
const why = both ? `not enough runs to suggest a size yet (${words.memory} in the last ${SUGGEST_WINDOW_DAYS} days)` : `fits its runs (memory: ${s.memory.reason === "fits" ? words.memory : `${s.memory.reason}, ${words.memory}`}; CPUs: ${s.cpu.reason === "fits" ? words.cpu : `${s.cpu.reason}, ${words.cpu}`})`;
|
|
7840
|
+
checks.push({ ok: true, label: `project ${id}: size ${size}${both ? ": " : " "}${why}${factTail}` });
|
|
7841
|
+
continue;
|
|
7842
|
+
}
|
|
7843
|
+
const parts = [];
|
|
7844
|
+
if (s.memory.suggested) parts.push(`memory ${formatMemory(s.memory.suggested)} (${s.memory.reason}: ${words.memory}${s.memory.held === "cap" ? `; ${words.memoryHeld}` : ""})`);
|
|
7845
|
+
if (s.cpu.suggested) parts.push(`${cpusText(s.cpu.suggested)} (${s.cpu.reason}: ${words.cpu})`);
|
|
7846
|
+
const heldText = held ? `${words.memory}; ${words.memoryHeld}` : "";
|
|
7847
|
+
// THE ONE RULE (`sizeRefusal`, shared with the panel and the insights page): the call is withheld exactly when this
|
|
7848
|
+
// host's admission would refuse the suggested pair for ever (job-size-exceeds-host or -share), whichever dimension
|
|
7849
|
+
// does it. With the budget off or unknown admission refuses nothing, so the call is offered, and a size above the
|
|
7850
|
+
// runtime's own memory or CPU count is only flagged.
|
|
7851
|
+
const pair = { memMiB: s.memory.suggested ?? current.memMiB, cpuCenti: s.cpu.suggested ?? current.cpuCenti };
|
|
7852
|
+
const refusal = call === null ? null : sizeRefusal(pair, [budget], share);
|
|
7853
|
+
const totals = refusal !== null ? [] : [s.memory.overBudget ? `memory ${formatMemory(s.memory.suggested)} is above ${formatMemory(cap.memMiB)}, this host's own memory` : null, s.cpu.overBudget ? `${cpusText(s.cpu.suggested)} ${s.cpu.suggested === 100 ? "is" : "are"} above ${cpusText(cap.cpuCenti)}, this host's own CPU count` : null].filter(Boolean);
|
|
7854
|
+
const over = refusal !== null ? [refusalWords(refusal, s, share, budget)] : [];
|
|
7855
|
+
const note = totals.length > 0 ? ` (note: ${totals.join(" and ")}; its budget is off or unknown, so the worker admits it)` : "";
|
|
7856
|
+
const raise = (s.memory.suggested ?? 0) > current.memMiB || (s.cpu.suggested ?? 0) > current.cpuCenti;
|
|
7857
|
+
const suggests = parts.length > 0 ? `its runs in the last ${SUGGEST_WINDOW_DAYS} days suggest ${parts.join(" and ")}` : "";
|
|
7858
|
+
const label = `project ${id}: size ${size}; ${[heldText ? `in the last ${SUGGEST_WINDOW_DAYS} days ${heldText}` : "", suggests].filter(Boolean).join("; ")}${over.length > 0 ? `, but ${over.join(" and ")}, so a job of it would never fit this host` : ""}${note}${factTail}`;
|
|
7859
|
+
// a size that would never fit is never offered as a call: applying it would turn a refusal into the same refusal.
|
|
7860
|
+
const apply = call === null ? "" : over.length > 0 ? "" : `apply it in the admin panel with ${call} (an operator confirms it; nothing applies a size by itself)`;
|
|
7861
|
+
const never = over.length > 0 ? "no call is offered: a job of that size would never fit this host" : "";
|
|
7862
|
+
if (held) {
|
|
7863
|
+
const above = Number.isSafeInteger(s.memory.cap) && current.memMiB > s.memory.cap;
|
|
7864
|
+
const none = s.memory.held === "largest" ? (above ? "no larger memory is offered: the size is already above what this host offers" : "no larger memory is offered: no larger size fits this host") : "no memory call is offered while the largest size this host offers is unknown (its budget is off or unknown and its runtime's memory was not read)";
|
|
7865
|
+
checks.push({ ok: false, warn: true, label, fix: apply ? `${none}; for the rest, ${apply}` : never ? `${none}; ${never}` : none });
|
|
7866
|
+
} else if (raise || over.length > 0 || totals.length > 0) {
|
|
7867
|
+
checks.push({ ok: false, warn: true, label, fix: apply || never });
|
|
7868
|
+
} else {
|
|
7869
|
+
checks.push({ ok: true, label: `${label}; ${apply.replace(/ \(an operator confirms it; nothing applies a size by itself\)$/, "")}` });
|
|
7870
|
+
}
|
|
7871
|
+
}
|
|
7872
|
+
return checks;
|
|
7873
|
+
}
|
|
7874
|
+
|
|
7875
|
+
/** The `ps` that lists this runtime's job containers with their two size labels (issue #596, phase 2). */
|
|
7876
|
+
export const SIZE_LABEL_PS_ARGS = Object.freeze(["ps", "--filter", `name=${JOB_NAME_PREFIX}`, "--format", `{{.Names}}\t{{.Label "${SIZE_LABEL_MEM}"}}\t{{.Label "${SIZE_LABEL_CPU}"}}`]);
|
|
7877
|
+
|
|
7878
|
+
/**
|
|
7879
|
+
* The running job containers' size labels, from `SIZE_LABEL_PS_ARGS`' output: `[{ name, memMiB, cpuCenti }]`, a label
|
|
7880
|
+
* that is absent or not an integer read as null. Only names in the job namespace (the filter is a substring match).
|
|
7881
|
+
*/
|
|
7882
|
+
export function parseSizeLabels(stdout) {
|
|
7883
|
+
const int = (v) => (/^[1-9][0-9]{0,8}$/.test(v ?? "") ? Number(v) : null);
|
|
7884
|
+
return String(stdout ?? "")
|
|
7885
|
+
.split("\n")
|
|
7886
|
+
.map((line) => line.split("\t"))
|
|
7887
|
+
.filter(([name]) => typeof name === "string" && name.startsWith(JOB_NAME_PREFIX))
|
|
7888
|
+
.map(([name, mem, cpu]) => ({ name, memMiB: int(mem?.trim()), cpuCenti: int(cpu?.trim()) }));
|
|
7889
|
+
}
|
|
7890
|
+
|
|
7891
|
+
/**
|
|
7892
|
+
* The ledger held against what runs (issue #596, phase 2): this host's registry row says what its worker's budget
|
|
7893
|
+
* counts (`usedMemMiB`, `usedCpuCenti`, orphans included); the job containers' labels say what runs. A WARNING when they
|
|
7894
|
+
* differ (a job starting or ending between the two reads differs for a moment, so the fix says to re-run first), and
|
|
7895
|
+
* when a job container carries no size label (one started by a worker from before the labels). Nothing without a row
|
|
7896
|
+
* that publishes the two fields.
|
|
7897
|
+
*/
|
|
7898
|
+
export function budgetLedgerChecks(selfRow, containers) {
|
|
7899
|
+
const int = (v) => (typeof v === "string" && /^[0-9]{1,15}$/.test(v) ? Number(v) : null);
|
|
7900
|
+
const usedMem = int(selfRow?.usedMemMiB);
|
|
7901
|
+
const usedCpu = int(selfRow?.usedCpuCenti);
|
|
7902
|
+
if (usedMem === null || usedCpu === null || !Array.isArray(containers)) return [];
|
|
7903
|
+
const labelled = containers.filter((c) => c.memMiB !== null && c.cpuCenti !== null);
|
|
7904
|
+
const unlabelled = containers.length - labelled.length;
|
|
7905
|
+
const mem = labelled.reduce((sum, c) => sum + c.memMiB, 0);
|
|
7906
|
+
const cpu = labelled.reduce((sum, c) => sum + c.cpuCenti, 0);
|
|
7907
|
+
const checks = [];
|
|
7908
|
+
if (mem === usedMem && cpu === usedCpu) {
|
|
7909
|
+
checks.push({ ok: true, label: `Host budget ledger matches the running job containers (${containers.length} running, ${mem === 0 ? "0" : formatMemory(mem)} and ${formatCpus(cpu)} CPUs)` });
|
|
7910
|
+
} else {
|
|
7911
|
+
checks.push({ ok: false, warn: true, label: `Host budget ledger holds ${usedMem === 0 ? "0" : formatMemory(usedMem)} and ${formatCpus(usedCpu)} CPUs, while the running job containers are labelled ${mem === 0 ? "0" : formatMemory(mem)} and ${formatCpus(cpu)} CPUs`, fix: "re-run doctor: a job starting or ending between the two reads differs for a moment. A difference that stays is a container the worker does not count, or one whose stop failed (it keeps its hold until the runtime says it is gone)" });
|
|
7912
|
+
}
|
|
7913
|
+
// NAMED, so an operator can find each one (`pi-job-<id>`, a job id and never payload text). A worker that
|
|
7914
|
+
// started beside one counts it at the largest size a project row or the default could have started it at, capped
|
|
7915
|
+
// at the budget, so the ledger line above differs while it runs.
|
|
7916
|
+
if (unlabelled > 0) {
|
|
7917
|
+
const names = containers.filter((c) => c.memMiB === null || c.cpuCenti === null).map((c) => c.name);
|
|
7918
|
+
const shown = names.length > 5 ? `${names.slice(0, 5).join(", ")} and ${names.length - 5} more` : names.join(", ");
|
|
7919
|
+
checks.push({ ok: false, warn: true, label: `${unlabelled} running job container${unlabelled === 1 ? " carries" : "s carry"} no size label (${shown}), so the ledger cannot be checked against ${unlabelled === 1 ? "it" : "them"} (started by a worker from before the host budget); a worker that found ${unlabelled === 1 ? "it" : "them"} at its start counts each at the largest size a project may run at, capped at the budget, until ${unlabelled === 1 ? "it is" : "they are"} gone`, fix: "nothing to do: the label is on every container a current worker starts; stop one early with `docker stop <name>` (or `podman stop`) to give its room back sooner" });
|
|
7920
|
+
}
|
|
7921
|
+
return checks;
|
|
7922
|
+
}
|
|
7923
|
+
|
|
7362
7924
|
/**
|
|
7363
7925
|
* WHERE this deployment's jobs run, and what that place actually guarantees (issue #227).
|
|
7364
7926
|
*
|
|
@@ -8333,6 +8895,12 @@ export async function liveChecks(env, seams, facts) {
|
|
|
8333
8895
|
const relabel = relabelsPrivateMounts(facts.daemon?.answered ? facts.daemon.facts : null, facts.endpoint, ids.platform ?? seams.platform);
|
|
8334
8896
|
const result = await runLiveProbes({
|
|
8335
8897
|
image: facts.jobImage ?? jobImageOf(env).image,
|
|
8898
|
+
// Issue #596: the deployment's default size, and the `--cpus` ceiling from the same `docker info` answer, so the
|
|
8899
|
+
// probe is built at the size a job gets and reads back memory, swap, weight and ceiling against it.
|
|
8900
|
+
size: doctorJobSize(env),
|
|
8901
|
+
hostCpus: facts.daemon?.answered ? (facts.daemon.facts?.hostCpus ?? null) : null,
|
|
8902
|
+
// Issue #596, phase 2: the parent's quota is read back against the CPU budget doctor computed above.
|
|
8903
|
+
cpuBudgetCenti: facts.cpuBudgetCenti ?? null,
|
|
8336
8904
|
endpoint: facts.endpoint,
|
|
8337
8905
|
// Asked again right before the first probe command, through the same resolver as the collection's read.
|
|
8338
8906
|
resolveEndpoint: makeDockerEndpointResolver({ run: dockerRunVia(spawn) }),
|
|
@@ -8411,6 +8979,14 @@ function readBackChecks({ venue, bin, result, user, ids, relabel, facts, userFix
|
|
|
8411
8979
|
return { ok: false, label: `${prefix}: ${v.property} does NOT hold -- declared ${declared}, observed: ${v.detail}`, fix };
|
|
8412
8980
|
}),
|
|
8413
8981
|
);
|
|
8982
|
+
// Issue #596, phase 2: where the runtime put the probe (the jobs' parent cgroup) and, where readable, the parent's
|
|
8983
|
+
// quota against the CPU budget. A line of its own: the reserve is not one of the declared properties.
|
|
8984
|
+
const parent = result.cgroupParent;
|
|
8985
|
+
if (parent) {
|
|
8986
|
+
if (!parent.ok && !parent.warn) checks.push({ ok: false, label: `${prefix}: cgroup parent does NOT hold -- ${parent.detail}`, fix: `the worker builds every job under the ${CGROUP_PARENT} parent (--cgroup-parent); a runtime that records another parent or places the container elsewhere keeps no CPU reserve across jobs: check the runtime's cgroup driver (\`${bin} info\`) and re-run \`pi-dispatch doctor --live\`` });
|
|
8987
|
+
else if (parent.warn) checks.push({ ok: false, warn: true, label: `${prefix}: cgroup parent: ${parent.detail}`, fix: "see the CPU reserve lines above for what keeps the host's reserve on this venue" });
|
|
8988
|
+
else checks.push({ ok: true, label: `${prefix}: cgroup parent holds (${parent.detail})` });
|
|
8989
|
+
}
|
|
8414
8990
|
checks.push(...noteChecks());
|
|
8415
8991
|
// What a green read-back does NOT mean, on a line of its own so a row of ✓ is never read as more than it is.
|
|
8416
8992
|
// Each sentence says only what DID happen: a probe that was not read back has its own line above saying why, and
|
|
@@ -8470,10 +9046,17 @@ export async function podmanLiveChecks(env, seams, facts) {
|
|
|
8470
9046
|
const readInfo = makePodmanInfoReader({ run: dockerRunVia(spawn, PODMAN_INFO_TIMEOUT_MS, { bin: "podman" }) });
|
|
8471
9047
|
const run = liveRunVia(spawn, { bin: "podman" });
|
|
8472
9048
|
const image = facts.jobImage ?? jobImageOf(env).image;
|
|
8473
|
-
const canary = await podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, announce: (line) => out(`\nread back on podman: ${line}\n`) });
|
|
9049
|
+
const canary = await podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, size: doctorJobSize(env), announce: (line) => out(`\nread back on podman: ${line}\n`) });
|
|
8474
9050
|
const egress = { armed: podman.egress.armed, results: canary.results, proxy: podman.egress.proxy, proxyRunning: podman.egress.proxyRunning, keeperBlocked: podman.egress.keeperBlocked ?? null };
|
|
8475
9051
|
const result = await runLiveProbes({
|
|
8476
9052
|
image,
|
|
9053
|
+
// Issue #596: as the docker read-back, from this account's own `podman info`.
|
|
9054
|
+
size: doctorJobSize(env),
|
|
9055
|
+
hostCpus: podman.info?.hostCpus ?? null,
|
|
9056
|
+
// Issue #596, phase 2: the parent as this venue's jobs get it (none where Podman's cgroup manager is not systemd),
|
|
9057
|
+
// and its quota against the CPU budget doctor computed above.
|
|
9058
|
+
cgroupParent: cgroupParentFor({ podman: true, cgroupManager: podman.info?.cgroupManager ?? null }),
|
|
9059
|
+
cpuBudgetCenti: facts.cpuBudgetCenti ?? null,
|
|
8477
9060
|
endpoint: podman.info,
|
|
8478
9061
|
resolveEndpoint: async () => {
|
|
8479
9062
|
const again = await readInfo();
|
|
@@ -8533,7 +9116,7 @@ export async function podmanLiveChecks(env, seams, facts) {
|
|
|
8533
9116
|
* judges a pid against THIS process table and the canary's containers must start where the section looked, and both of
|
|
8534
9117
|
* those are false the moment CONTAINER_HOST points elsewhere. The re-ask costs one spawn and only with egress armed.
|
|
8535
9118
|
*/
|
|
8536
|
-
async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, announce = () => {} }) {
|
|
9119
|
+
async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive, size = DEFAULT_JOB_SIZE, announce = () => {} }) {
|
|
8537
9120
|
const none = { checks: [], results: [] };
|
|
8538
9121
|
if (podman.egress?.armed === false) return none;
|
|
8539
9122
|
// Issue #458: on Podman 4.x without the keeper, the canary's own teardown (and the stale sweep's detach) is the step
|
|
@@ -8559,7 +9142,7 @@ async function podmanEgressCanary({ podman, readInfo, run, image, pid, isAlive,
|
|
|
8559
9142
|
const endpoints = podman.egress.endpoints ?? [];
|
|
8560
9143
|
const more = endpoints.length > 0 ? `; then three per declared model endpoint (${endpointsById(endpoints).map((e) => e.id).join(", ")}), named ${EGRESS_ENDPOINT_PROBE_PREFIX}<probe>-<id>-${pid}, removed the same way` : "";
|
|
8561
9144
|
announce(`starting ${probes.slice(0, -1).join(", ")} and ${probes.at(-1)} from ${image} (as the job user ${podman.user}) on the --internal network ${egressCanaryNetwork(pid)}, with ${podman.egress.proxy} attached, to read the egress allowlist back; all three are removed when the canary ends${more}`);
|
|
8562
|
-
const canary = await runEgressCanary({ run, bin: "podman", proxy: podman.egress.proxy, image, pid, user: podman.user, gate, endpoints });
|
|
9145
|
+
const canary = await runEgressCanary({ run, bin: "podman", proxy: podman.egress.proxy, image, pid, user: podman.user, gate, endpoints, size, hostCpus: again.info?.hostCpus ?? null });
|
|
8563
9146
|
return { checks: [...checks, ...canary.checks], results: canary.results };
|
|
8564
9147
|
}
|
|
8565
9148
|
|