@edgehero/pi-dispatch 1.10.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +29 -2
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +363 -13
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1348 -326
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
@@ -0,0 +1,1168 @@
1
+ /**
2
+ * THE `podman` BACKEND: the worker account's own ROOTLESS Podman, through the real `podman` CLI (issue #354,
3
+ * `DES-PODMAN-NATIVE-ROOTLESS-BACKEND`).
4
+ *
5
+ * `backends.mjs` says what the venue guarantees; this module is what makes it true. It is `local`'s machinery with the
6
+ * runtime's two seams set (PR #425): `bin: "podman"` on every spawn, and `buildPodmanRunArgs`, which always emits
7
+ * `--user <euid>:<egid> --userns=keep-id`. Nothing here re-implements a run, a stop, a preflight or a sweep; a podman
8
+ * behaviour that differs from docker's is either handled in the shared code (the disconnect trap in `egress.mjs`) or
9
+ * named as a residual in the design entry.
10
+ *
11
+ * ROOTLESS ONLY, and every refusal is a measurement rather than caution (Podman 5.8.1, Fedora 44, 2026-09-25):
12
+ * - rootful `podman run --userns=keep-id` is NOT refused by Podman: exit 0, an identity uid map, and the process gains
13
+ * supplementary group 0. So the venue refuses unless `podman info` says rootless (`podman-rootful`).
14
+ * - a remote service (`CONTAINER_HOST`, `--remote`) runs the job's bind sources and `-e` values on another machine
15
+ * (`podman-remote`), and a root worker would be uid 0 on the host (`worker-is-root`).
16
+ * - only Linux was measured; Podman machine on macOS and Windows was not (`podman-platform`).
17
+ *
18
+ * THE FACTS COME FROM ONE READ, `podman info --format json`, cached once it answers: it decides the job user, whether
19
+ * mounts are relabelled, and, with this host's files, the three observations the table's words rest on. Rootless-ness
20
+ * and remoteness cannot change under a running worker without a restart (they are this account's and this process's
21
+ * environment), so re-reading them per job would only add a spawn. The FILES an observation reads are re-read per job:
22
+ * mounts.conf and containers.conf, because an operator creating the empty override must not need a restart, and the
23
+ * account's user-manager cgroup (issue #453), because that manager can stop under a running worker (linger turned off,
24
+ * the last login session ending) and the next job must then be judged without its bounds. One residual rides the
25
+ * cache: the cgroup manager `podman info` reported at the first answered read is kept, so a user bus that appears or
26
+ * disappears later is not seen until a restart.
27
+ *
28
+ * `backends.mjs` stays a leaf, so this module imports it and never the other way round.
29
+ */
30
+
31
+ import { readdirSync, readFileSync, statSync } from "node:fs";
32
+ import { homedir } from "node:os";
33
+ import { JOB_NAME_PREFIX, execDockerBounded, jobContainerName, makeReaper, makeStopContainer } from "./backend-local.mjs";
34
+ import { BACKENDS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_CONF_WIDENS_JOB, PODMAN_NETWORK_HELPER_KEYS, PODMAN_WIDENING_KEYS, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
35
+ import { CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
36
+ import { buildPodmanRunArgs } from "./docker-run.mjs";
37
+ import { DEFAULT_EGRESS_PROXY, makeEgressPreflight } from "./egress.mjs";
38
+ import { NETNS_KEEPER, NETNS_KEEPER_FORMAT, QUADLET_FILES, STARTED_AT_FORMAT, judgeNetnsKeeper, netnsKeeperRemedy, podmanNeedsNetnsKeeper } from "./podman-stack.mjs";
39
+ import { makeImagePreflight } from "./image-preflight.mjs";
40
+ import { DAEMON_FACTS_TIMEOUT_MS } from "./job-user.mjs";
41
+ import { PODMAN_INFO_ARGS, parsePodmanInfo } from "./daemon-facts.mjs";
42
+
43
+ // Moved to the leaf `daemon-facts.mjs` (issue #452, gate round 3) and re-exported, so every importer keeps its path.
44
+ export { PODMAN_INFO_ARGS, parsePodmanInfo };
45
+ import { makeRunContainer } from "./run-container.mjs";
46
+ import { MOUNT_KEY, MOUNT_KEY_SAYS, PODMAN_HOOKS_DIRS, TRANSIENT_READ_ERRORS, confFilesIn, confKeyFinding, confWidening, stripStockBlocks, widenKeyPattern, fipsFinding, hooksFinding, unreadFileFinding } from "./runtime-observations.mjs";
47
+
48
+ /**
49
+ * `docker info`'s bound, reused: `podman info` also walks the store, and a per-job `unknown` retries the job, so a bound a
50
+ * busy host routinely misses would retry every job forever.
51
+ */
52
+ /**
53
+ * The bound on starting a container of the job image on this venue when nothing else bounds it (doctor's and the
54
+ * conformance script's live read-back). Rootless Podman on an overlay store that cannot shift ids makes the FIRST
55
+ * keep-id run of an image copy its layers to the mapped ids: 27.1 s for the job image on Fedora 44 (btrfs backing,
56
+ * "Supports shifting: false"), then 152 ms, against 115 ms without keep-id (measured, Podman 5.8.1). The read-back's
57
+ * 20 s step bound read that first run as a probe that never started.
58
+ */
59
+ export const PODMAN_FIRST_START_TIMEOUT_MS = 120_000;
60
+
61
+ export const PODMAN_INFO_TIMEOUT_MS = DAEMON_FACTS_TIMEOUT_MS;
62
+
63
+ // `podman info --format json` is pretty-printed and carries the registries and store paths: tens of KiB, not one line.
64
+ const PODMAN_INFO_MAX_BUFFER = 1024 * 1024;
65
+
66
+ /**
67
+ * The exits `podman run` spells "the runner never ran" (measured): 125 is podman itself (a name in use, an absent image
68
+ * under `--pull=never`), 126 and 127 crun's not-executable and not-found when no `--init` wraps the entrypoint. The job
69
+ * argv carries `--init`, under which catatonit reports both as exit 1: indistinguishable from the runner's own infra exit
70
+ * except by a stderr line the job could print too, so NOT normalised (a residual the design entry names).
71
+ */
72
+ export const PODMAN_NEVER_STARTED_EXITS = Object.freeze([125, 126, 127]);
73
+
74
+ /**
75
+ * `async () => ({ answered: true, info } | { answered: false, reason, transient })`. `run(args)` is the seam: a bounded
76
+ * spawn of `podman` resolving `{ code, stdout, stderr }` (or `execDockerBounded`'s `{ code, stdout, error }`).
77
+ *
78
+ * TWO DETERMINATE answers, the rest transient, on `makeDaemonFactsReader`'s rule: no podman binary (`podman-not-found`),
79
+ * and a clean exit that parses to nothing (`unparseable`). A non-zero exit, a timeout or a signal is a Podman that may
80
+ * answer next time, and a worker unit with `RestartPreventExitStatus=2` must not be stranded by one that is merely slow.
81
+ * No CLI text is read or logged: a remote service's error names its URL.
82
+ */
83
+ export function makePodmanInfoReader({ run = (args) => execDockerBounded(args, { bin: "podman", timeoutMs: PODMAN_INFO_TIMEOUT_MS, maxBuffer: PODMAN_INFO_MAX_BUFFER }) } = {}) {
84
+ return async function readPodmanInfo() {
85
+ let result;
86
+ try {
87
+ result = await run(PODMAN_INFO_ARGS);
88
+ } catch (err) {
89
+ result = { code: null, stdout: "", error: err };
90
+ }
91
+ const error = result?.error ?? null;
92
+ if (error?.code === "ENOENT") return { answered: false, reason: "podman-not-found", transient: false };
93
+ if (error || result?.code !== 0) {
94
+ const reason = error?.timedOut || error?.killed ? "timeout"
95
+ : typeof error?.signal === "string" ? `signal-${error.signal.toLowerCase()}`
96
+ : typeof result?.code === "number" ? `exit-${result.code}`
97
+ : typeof error?.code === "number" ? `exit-${error.code}`
98
+ : "spawn-failed";
99
+ return { answered: false, reason, transient: true };
100
+ }
101
+ const info = parsePodmanInfo(result.stdout);
102
+ return info ? { answered: true, info } : { answered: false, reason: "unparseable", transient: false };
103
+ };
104
+ }
105
+
106
+ const CACHED = Symbol("podman-info-cached");
107
+
108
+ /**
109
+ * `readInfo` with its first ANSWERED result kept, and concurrent callers sharing one read. An unanswered read is never
110
+ * kept: a Podman that timed out once must be asked again, or every later job would retry on a stale failure. Idempotent,
111
+ * so the boot wiring can wrap a reader once and hand the same one to the bundle and to its own boot read.
112
+ */
113
+ export function cachedPodmanInfo(readInfo) {
114
+ if (readInfo?.[CACHED]) return readInfo;
115
+ let kept = null;
116
+ let inFlight = null;
117
+ const cached = async () => {
118
+ if (kept) return kept;
119
+ if (inFlight) return inFlight;
120
+ inFlight = (async () => {
121
+ try {
122
+ const read = await readInfo();
123
+ if (read?.answered === true && read.info) kept = read;
124
+ return read;
125
+ } catch {
126
+ return { answered: false, reason: "spawn-failed", transient: true };
127
+ }
128
+ })();
129
+ try {
130
+ return await inFlight;
131
+ } finally {
132
+ inFlight = null;
133
+ }
134
+ };
135
+ cached[CACHED] = true;
136
+ // Issue #452, gate round 4: the kept ANSWERED read, without reading, for a job's teardown.
137
+ cached.peek = () => kept;
138
+ return cached;
139
+ }
140
+
141
+ /**
142
+ * The system containers.conf files and drop-in directories rootless Podman reads (containers/common v0.67.0
143
+ * `systemConfigs`): the vendor and `/etc` files, `/etc/containers/containers.conf.d`, and for a uid above 0
144
+ * `/etc/containers/containers.rootless.conf`, its `.d` and the `.d/<uid>` beneath it (added in `podmanConfFiles`).
145
+ * Measured for issue #448 (gate round 1 of PR #473) with a drop-in in each place: Podman 5.8.1 read
146
+ * `containers.rootless.conf`, which this list missed until then, and neither Podman read
147
+ * `/usr/share/containers/containers.conf.d` or `/usr/share/containers/containers.rootless.conf.d`, so they are not on it.
148
+ * Podman 4.9.3 read none of the `rootless` places; they stay, since a superset only refuses more.
149
+ */
150
+ export const PODMAN_ROOTLESS_CONF_FILES = Object.freeze(["/usr/share/containers/containers.conf", "/etc/containers/containers.conf", "/etc/containers/containers.rootless.conf"]);
151
+ export const PODMAN_ROOTLESS_CONF_DIRS = Object.freeze(["/etc/containers/containers.conf.d", "/etc/containers/containers.rootless.conf.d"]);
152
+ /** The system-wide mounts.conf a rootless Podman falls back to when the user has none (the user's one overrides it). */
153
+ export const PODMAN_SYSTEM_MOUNTS_CONF = "/etc/containers/mounts.conf";
154
+
155
+ /**
156
+ * A key that takes a container out of its cgroup (`cgroups = "disabled"` or `"no-conmon"`), under which rootless Podman
157
+ * accepts `--pids-limit` and `--memory` and applies neither (measured with the flag; the key sets the same option).
158
+ * Any value withholds credit: only the default reads the bounds back. Matched as `MOUNT_KEY` is, in any case and any
159
+ * TOML spelling; `cgroupns` and `cgroup_manager` do not match (the key must end at `cgroups`).
160
+ */
161
+ export const CGROUPS_KEY = /^(?!\s*#).*?(?:^|[\s.{,"'])["']?cgroups["']?\s*=/im;
162
+
163
+ /**
164
+ * The containers.conf files THIS account's Podman reads, as `{ files }` or `{ finding }`. `CONTAINERS_CONF` replaces the
165
+ * whole list and `CONTAINERS_CONF_OVERRIDE` adds one, and neither is chased: set, the answer is `false`, never a credit
166
+ * from files Podman may not read. `XDG_CONFIG_HOME` moves the user's own directory, as Podman's `GetConfigHome` does.
167
+ */
168
+ function podmanConfFiles({ fs, home, env, euid }) {
169
+ for (const name of ["CONTAINERS_CONF", "CONTAINERS_CONF_OVERRIDE"]) {
170
+ if (typeof env?.[name] === "string" && env[name] !== "") return { finding: { value: false, evidence: `${name} is set, so which containers.conf Podman reads is not the list this check reads` } };
171
+ }
172
+ if (typeof home !== "string" || !home.startsWith("/")) return { finding: { value: false, evidence: "the worker account's home is not known, so its own containers.conf could not be read" } };
173
+ if (!Number.isInteger(euid)) return { finding: { value: false, evidence: "the worker's uid is not known, so its per-uid containers.conf drop-ins could not be read" } };
174
+ // env-internal XDG_CONFIG_HOME: Podman's own variable, read only to find the containers.conf the podman CLI this worker
175
+ // spawns reads, never a pi-dispatch setting.
176
+ const configHome = typeof env?.XDG_CONFIG_HOME === "string" && env.XDG_CONFIG_HOME.startsWith("/") ? env.XDG_CONFIG_HOME : `${home}/.config`;
177
+ return confFilesIn(fs, {
178
+ files: [...PODMAN_ROOTLESS_CONF_FILES, `${configHome}/containers/containers.conf`],
179
+ dirs: [...PODMAN_ROOTLESS_CONF_DIRS, `/etc/containers/containers.rootless.conf.d/${euid}`, `${configHome}/containers/containers.conf.d`],
180
+ });
181
+ }
182
+
183
+ /**
184
+ * `podmanAddsNoMounts` from this host's files, as `{ value, evidence }`. The mounts.conf that WINS decides: the user's
185
+ * `$HOME/.config/containers/mounts.conf` when it exists (Podman builds that path from `HOME`, not `XDG_CONFIG_HOME`),
186
+ * else `/etc/containers/mounts.conf`, and an EMPTY winner suppresses `/usr/share/containers/mounts.conf`'s `/run/secrets`
187
+ * default (measured rootless). Then the same containers.conf keys, OCI hooks and FIPS rule the rootful observation
188
+ * reads, over the files a rootless Podman reads.
189
+ */
190
+ function observePodmanMounts({ fs, home, env, euid }) {
191
+ const fips = fipsFinding(fs);
192
+ if (fips) return fips;
193
+ if (typeof home !== "string" || !home.startsWith("/")) return { value: false, evidence: "the worker account's home is not known, so its own mounts.conf could not be read" };
194
+ const userMounts = `${home}/.config/containers/mounts.conf`;
195
+ let winner = null;
196
+ for (const path of [userMounts, PODMAN_SYSTEM_MOUNTS_CONF]) {
197
+ let size;
198
+ try {
199
+ size = fs.statSync(path).size;
200
+ } catch (error) {
201
+ if (error?.code === "ENOENT") continue;
202
+ return unreadFileFinding(path, error?.code ?? "error");
203
+ }
204
+ winner = { path, size };
205
+ break;
206
+ }
207
+ if (!winner) return { value: false, evidence: `neither ${userMounts} nor ${PODMAN_SYSTEM_MOUNTS_CONF} exists, so Podman mounts the default list (/run/secrets on Fedora and RHEL)` };
208
+ if (winner.size !== 0) return { value: false, evidence: `${winner.path} is not empty, so Podman mounts what it lists` };
209
+ const listed = podmanConfFiles({ fs, home, env, euid });
210
+ if (listed.finding) return listed.finding;
211
+ const conf = confKeyFinding(fs, listed.files, { key: MOUNT_KEY, says: MOUNT_KEY_SAYS });
212
+ if (conf) return conf;
213
+ const hooks = hooksFinding(fs, PODMAN_HOOKS_DIRS);
214
+ if (hooks) return hooks;
215
+ return { value: true, evidence: `${winner.path} is empty, no containers.conf this account's Podman reads sets volumes, mounts, devices or hooks, and no OCI hook is installed` };
216
+ }
217
+
218
+ /**
219
+ * The cgroup of this account's systemd user manager, whose `cgroup.controllers` says which controllers are delegated to
220
+ * it. Measured (issue #453, Fedora 44): the path is absent while `user@<uid>.service` is inactive and present while it
221
+ * runs, which is systemd's own behaviour (the unit's cgroup is removed when it stops).
222
+ */
223
+ export function podmanUserManagerControllersPath(euid) {
224
+ return `/sys/fs/cgroup/user.slice/user-${euid}.slice/user@${euid}.service/cgroup.controllers`;
225
+ }
226
+
227
+ /** What `observePodmanBounds` says when no systemd user manager runs for the account (issue #453); doctor keys its fix on the cause. */
228
+ export function podmanNoUserManagerEvidence(euid) {
229
+ return `no systemd user manager is running for this account (${podmanUserManagerControllersPath(euid)} does not exist), so Podman leaves a job's pids, memory and cpu bounds unapplied, whichever cgroup manager it uses: with linger off the manager runs only while the account has a login session, and a \`sudo -iu\` shell starts none (measured). On a host without systemd managing cgroups under user.slice this path never exists, and the venue gives no credit there`;
230
+ }
231
+
232
+ /** Whether a `/proc/self/cgroup` text puts this process inside the account's user manager (a systemd user service). */
233
+ function insideUserManager(text, euid) {
234
+ const line = String(text ?? "").split("\n").find((l) => l.startsWith("0::"));
235
+ return typeof line === "string" && line.slice(3).startsWith(`/user.slice/user-${euid}.slice/user@${euid}.service/`);
236
+ }
237
+
238
+ /**
239
+ * `podmanBoundsDelegated` from the info and this host's files: rootless, cgroup v2, a RUNNING systemd user manager for
240
+ * this account with `pids`, `memory` and `cpu` delegated to it, Podman able to put its containers under that manager,
241
+ * and no containers.conf taking containers out of their cgroup. Rootful Podman is refused as a venue, so it earns
242
+ * nothing here either.
243
+ *
244
+ * WHAT DECIDES IT (issue #453, measured on Fedora 44 with rootless Podman 5.8.1, round-446 M0-e and the gate's L rows):
245
+ * - no user manager (linger off, a `sudo -iu` shell): the container landed in the caller's root-owned session scope,
246
+ * `pids.max`, `memory.max` and `cpu.max` `max`, exit 0, whichever cgroup manager was configured;
247
+ * - a user manager running AND Podman reaching it (`podman info` reports the `systemd` cgroup manager): applied;
248
+ * - a user manager running but NOT reachable over D-Bus (no user bus socket, a DBUS_SESSION_BUS_ADDRESS pointing
249
+ * nowhere, a system unit with `User=` and no user bus): Podman fell back to `cgroupfs`, the container landed in the
250
+ * caller's own root-owned cgroup, unbounded (gate L1, L2, L3);
251
+ * - `cgroupfs` from a process already inside `user@<uid>.service` (a user unit, with or without `Delegate=`): applied.
252
+ * So credit needs the manager's cgroup with the three controllers, and then either the `systemd` cgroup manager or this
253
+ * process inside that manager. `podman info`'s `host.cgroupControllers` decides nothing: it is the caller's own cgroup,
254
+ * which listed all five with nothing applied. Fail-closed where not measured, or where measured applied but not
255
+ * provable from here: an explicit `cgroupfs` from a shell outside the manager with a reachable bus applied (M0-e E3c)
256
+ * and is not credited; no user manager with the caller in a cgroup of its own that the account owns (a system unit with
257
+ * `User=` and `Delegate=yes`, linger off) was not measured and is not credited.
258
+ */
259
+ function observePodmanBounds(info, { fs, home, env, euid }) {
260
+ if (info.rootless !== true) return { value: false, evidence: "podman info does not report a rootless Podman, which is the only kind this venue runs on" };
261
+ if (info.cgroupVersion !== "v2") return { value: false, evidence: `podman info reports cgroup ${info.cgroupVersion ?? "version unknown"}, and rootless bounds need cgroup v2` };
262
+ if (!Number.isInteger(euid) || euid < 0) return { value: false, evidence: "the worker's uid is not known, so whether its systemd user manager runs could not be read" };
263
+ const path = podmanUserManagerControllersPath(euid);
264
+ let text;
265
+ try {
266
+ text = String(fs.readFileSync(path, "utf8"));
267
+ } catch (error) {
268
+ if (error?.code === "ENOENT") return { value: false, cause: "no-user-manager", evidence: podmanNoUserManagerEvidence(euid) };
269
+ return unreadFileFinding(path, error?.code ?? "error");
270
+ }
271
+ const controllers = text.split(/\s+/).filter((c) => /^[a-z][a-z0-9_]{0,31}$/.test(c));
272
+ const missing = ["pids", "memory", "cpu"].filter((c) => !controllers.includes(c));
273
+ // Measured (gate 456, Fedora 44, Podman 5.8.1, a manager delegated only memory and pids): a job carrying `--cpus`
274
+ // FAILED to start, crun exit 126 "controller `cpu` is not available", and one without it ran with no cpu.max at all.
275
+ // Only the cpu case was measured, so the sentence says what a job gets without claiming more for the other two.
276
+ if (missing.length > 0) return { value: false, evidence: `the ${missing.join(", ")} cgroup ${missing.length === 1 ? "controller is" : "controllers are"} not delegated to this account's systemd user manager (user@${euid}.service), so a job cannot have that bound: measured for cpu, Podman 5.8.1 refuses to start a container carrying --cpus (exit 126, "controller \`cpu\` is not available")` };
277
+ if (info.cgroupManager !== "systemd") {
278
+ let inside = false;
279
+ try {
280
+ inside = insideUserManager(fs.readFileSync("/proc/self/cgroup", "utf8"), euid);
281
+ } catch (error) {
282
+ if (error?.code !== "ENOENT") return unreadFileFinding("/proc/self/cgroup", error?.code ?? "error");
283
+ }
284
+ if (!inside) {
285
+ return {
286
+ value: false,
287
+ cause: "user-manager-unreachable",
288
+ evidence: `${info.cgroupManager === null ? "podman info does not say which cgroup manager Podman uses" : `Podman is using the ${info.cgroupManager} cgroup manager, not systemd`}, and this worker is not running inside the account's user manager (user@${euid}.service), so whether a job's pids, memory and cpu bounds are applied is not observed from here. Measured: where a configured systemd manager fell back to cgroupfs because Podman could not reach the user manager over D-Bus (no user bus socket, a DBUS_SESSION_BUS_ADDRESS pointing nowhere), the container landed in the worker's own cgroup unbounded; an explicit cgroupfs with the bus reachable had its bounds applied, and nothing read here tells the two apart`,
289
+ };
290
+ }
291
+ }
292
+ const listed = podmanConfFiles({ fs, home, env, euid });
293
+ if (listed.finding) return listed.finding;
294
+ const conf = confKeyFinding(fs, listed.files, { key: CGROUPS_KEY, says: "sets a cgroups key, which can run every container outside its cgroup with its bounds unapplied" });
295
+ if (conf) return conf;
296
+ return {
297
+ value: true,
298
+ controllers,
299
+ evidence: `rootless Podman on cgroup v2, with this account's systemd user manager (user@${euid}.service) running and the pids, memory and cpu controllers delegated to it, ${info.cgroupManager === "systemd" ? "Podman using the systemd cgroup manager" : "this worker running inside that manager"}, and no containers.conf sets cgroups`,
300
+ };
301
+ }
302
+
303
+ /**
304
+ * A containers.conf key that WIDENS what a job reaches, and that no flag on the job's own command line takes back (issue
305
+ * #428, measured on Fedora 44, rootless Podman 5.8.1, pasta, 2026-09-26):
306
+ * - `pasta_options`: Podman puts these BEFORE its own options in pasta's argv, for the job's `--network=private` AND for
307
+ * the rootless netns pasta behind every bridge network (a job's own non-internal bridge, the egress proxy's). Mapping
308
+ * the host's loopback (`--map-host-loopback 169.254.1.2`, `--map-gw`) handed an egress-off job, a bridged job and the
309
+ * proxy the host's 127.0.0.1 services, the job queue's Valkey among them; `-T 6379` made the job's own
310
+ * 127.0.0.1:6379 the host's, and Podman then drops its own `-T none`.
311
+ * - `network_cmd_options`: slirp4netns's (`allow_host_loopback=true` under `default_rootless_network_cmd="slirp4netns"`,
312
+ * the rootless default before Podman 5, which was not measured) opened the same three through 10.0.2.2.
313
+ * - `annotations`: `run.oci.keep_original_groups=1` kept the account's supplementary groups inside the job, and a
314
+ * root:podman 0640 host file mounted into it became readable. `dockerExtra` refuses the flag; the conf set it anyway.
315
+ * - `env`, in any table: `[engine] env` is Podman's OWN environment, and `CONTAINERS_CONF_OVERRIDE` there made it read
316
+ * a containers.conf this check never reads (measured: pasta then got `--map-host-loopback`), as `CONTAINERS_CONF`,
317
+ * `XDG_CONFIG_HOME` or `HOME` could; `[containers] env` adds variables to every job past the worker's own closed
318
+ * environment. Both are account-wide and no argv takes them back. `env_host` is NOT this key (the pattern ends at
319
+ * `env`), and needs no refusal: `--env-host=false` is in `PODMAN_PINNED_FLAGS`.
320
+ * - `helper_binaries_dir` and `network_cmd_path` (issue #450, gate round 1): where Podman finds pasta, slirp4netns and
321
+ * its other helpers, and which slirp4netns it runs. Either one swaps the program behind every job's network for
322
+ * another: measured, a `helper_binaries_dir` naming a directory with a wrapper in it first ran the rootless network
323
+ * as `podpasta`, and a plain bridge container then reached the host's loopback. Neither is set in the stock files
324
+ * (Fedora 44's and Ubuntu 24.04's carry both commented out).
325
+ * - issue #448, measured with the venue's own argv on rootless Podman 5.8.1 and 4.9.3, each key alone in the account's
326
+ * own containers.conf: ten more reached the job, and are in the list for that (`PODMAN_WIDENING_KEYS` in backends.mjs
327
+ * says what each did); the rest measured were inert under the argv's own pins (`PODMAN_ROOTLESS_INERT_KEYS`). The
328
+ * vendor's own `default_sysctls` block, uncommented in the stock file both distributions ship, is accepted exactly
329
+ * as it ships (`STOCK_CONF_BLOCKS`), since refusing it would refuse every stock host.
330
+ * REFUSED ON PRESENCE, whatever the value, as `CGROUPS_KEY` is, and for a stronger reason: no argv pins it back. Rejected:
331
+ * pinning `--network=pasta:--map-host-loopback,none` (pasta takes the last mapping, so it cancels the first two, but a
332
+ * conf `-T` survives it, measured); a per-job bridge for an egress-off job (the shared rootless netns pasta reads the
333
+ * same options, measured open); a floor observation (with egress off no declared property covers what a job reaches on
334
+ * the host, so no floor would ever ask on the deployments at risk); and judging the options' VALUES (an allowlist of
335
+ * pasta flags is a parser for another program's argv, and the next release's flag is the one it misses). An egress-armed
336
+ * job on its `--internal` network was closed in every row (no default route, and `--cap-drop=ALL` means it cannot add
337
+ * one), but the venue is refused whatever `PI_EGRESS` says: `annotations` widens a job whatever its network, and the
338
+ * proxy's own bridge takes the same pasta options (open under the loopback mappings, measured). Matched as `MOUNT_KEY` is: any letter case, bare or
339
+ * quoted, dotted (`network.pasta_options`) or in an inline table, never on a whole-line comment, never inside a longer
340
+ * key name. The one capture group is the key, so the refusal names the key it found.
341
+ */
342
+ export const WIDENING_KEY = widenKeyPattern(PODMAN_WIDENING_KEYS);
343
+
344
+ /** `PODMAN_WIDENING_KEYS` as a sentence names them: "a, b, c or d". */
345
+ const WIDENING_KEYS_LISTED = `${PODMAN_WIDENING_KEYS.slice(0, -1).join(", ")} or ${PODMAN_WIDENING_KEYS.at(-1)}`;
346
+
347
+ /** What each widening key does, completing the sentence that names the file it was found in. */
348
+ const WIDENING_KEY_SAYS = Object.freeze({
349
+ pasta_options: "sets pasta_options, which Podman hands to the pasta behind every job's network, where a host-loopback mapping (--map-host-loopback, --map-gw, -T) gives the job this host's 127.0.0.1 services",
350
+ network_cmd_options: "sets network_cmd_options, which Podman hands to slirp4netns behind every job's network, where allow_host_loopback=true gives the job this host's 127.0.0.1 services",
351
+ annotations: "sets annotations, which Podman adds to every container, where run.oci.keep_original_groups=1 keeps this account's supplementary groups inside the job",
352
+ env: "sets env, which under [engine] is Podman's own environment (where CONTAINERS_CONF_OVERRIDE, CONTAINERS_CONF, XDG_CONFIG_HOME or HOME moves which containers.conf it reads) and under [containers] adds variables to every job",
353
+ helper_binaries_dir: "sets helper_binaries_dir, which is where Podman finds the pasta and slirp4netns behind every job's network, so another program can stand in for them",
354
+ network_cmd_path: "sets network_cmd_path, which names the slirp4netns Podman runs behind every job's network, so another program can stand in for it",
355
+ default_sysctls: "sets default_sysctls, which Podman sets in every job (any value but the vendor's own ping_group_range block)",
356
+ default_ulimits: "sets default_ulimits, which Podman sets on every job's processes",
357
+ seccomp_profile: "sets seccomp_profile, which replaces the seccomp filter of every job",
358
+ init_path: "sets init_path, which names the binary that runs as every job's PID 1",
359
+ dns_servers: "sets dns_servers, which writes the nameservers of every job with no network of its own",
360
+ dns_options: "sets dns_options, which writes every job's resolver options",
361
+ dns_searches: "sets dns_searches, which writes every job's resolver search list",
362
+ base_hosts_file: "sets base_hosts_file, which names the file every job's /etc/hosts starts from",
363
+ oom_score_adj: "sets oom_score_adj, which Podman applies to every job's processes",
364
+ privileged: "sets privileged, which gives every job a full capability bounding set, no seccomp filter, the host's devices and an unconfined SELinux label",
365
+ label: "sets label, which with false runs every job unconfined by SELinux (spc_t, measured)",
366
+ cgroup_conf: "sets cgroup_conf, which writes cgroup files of every job past its own bounds (pids.max=max outlasted --pids-limit, measured)",
367
+ host_containers_internal_ip: "sets host_containers_internal_ip, which names the address every job reaches as host.containers.internal",
368
+ runtimes: "sets runtimes, which as the [engine.runtimes] table names the OCI runtime binary that creates every job (a wrapper there ran for every job, measured)",
369
+ conmon_path: "sets conmon_path, which names the conmon that monitors every job (a wrapper there ran for every job, measured)",
370
+ cgroups: "sets cgroups, which with disabled runs every job outside its cgroup with its pids and memory bounds unapplied (measured)",
371
+ umask: "sets umask, which every job's processes start with, so what a job writes to this host is as open as it says (measured)",
372
+ });
373
+
374
+ // The cause a widening containers.conf refuses the venue under, defined in the leaf (backends.mjs) and re-exported here.
375
+ export { PODMAN_CONF_WIDENS_JOB };
376
+
377
+ /**
378
+ * A containers.conf this account's Podman reads that widens what a job reaches, as `{ cause, key, evidence }`, else
379
+ * `null` (issue #428). Read over `podmanConfFiles`, the chain the observations read, per call, so removing the key needs
380
+ * no restart. `key` is null when no key was found but the chain could not be read whole: `CONTAINERS_CONF` or
381
+ * `CONTAINERS_CONF_OVERRIDE` set, an unknown home or uid, a file or drop-in directory that exists and cannot be read, or
382
+ * a spelling this check does not decode (an escaped key, a non-ASCII character, a multi-line string). Each of those
383
+ * REFUSES too, on the observations' rule (a drop-in nobody could see must not read as none), and is DETERMINATE: a retry
384
+ * reads the same bytes. The one exception is a read that failed for a moment (`TRANSIENT_READ_ERRORS`: out of
385
+ * descriptors, an I/O error), which comes back with `transient: true` and is retried, never refused, on
386
+ * `observationRefusalIsTransient`'s rule. `podman info` is not an input.
387
+ *
388
+ * THEN THE LIVE NETWORK (issue #450), only when no containers.conf refused: a clean chain says what the NEXT rootless
389
+ * network starts with, not what the running one carries, and that one keeps the options it started with until every
390
+ * bridge container of the account stops (`podmanNetnsWidening`). Here rather than beside each caller, so the boot, the
391
+ * per-job pre-spend check, the sandbox and doctor cannot disagree about it: they all call this one function.
392
+ */
393
+ export function podmanConfWidening({ fs, home, env, euid, runRoot }) {
394
+ const listed = podmanConfFiles({ fs, home, env, euid });
395
+ // The chain-agnostic scan (issue #448) the rootful local check shares, with no `unread` list: every file this account's
396
+ // Podman reads is one the worker can read, so one it cannot is refused, as it always was.
397
+ // The vendor's own `default_sysctls` block is accepted as it ships (issue #448), as the rootful check accepts it: the
398
+ // stock file is on this chain too, and refusing it would refuse every stock host.
399
+ const found = listed.finding ?? confWidening(fs, listed.files, { keys: PODMAN_WIDENING_KEYS, says: WIDENING_KEY_SAYS, strip: stripStockBlocks });
400
+ if (!found) return podmanNetnsWidening({ fs, euid, runRoot });
401
+ if (found.value === null) return { cause: PODMAN_CONF_WIDENS_JOB, key: null, evidence: found.evidence, transient: true };
402
+ return { cause: PODMAN_CONF_WIDENS_JOB, key: found.key ?? null, evidence: found.evidence, ...(found.spelling ? { spelling: found.spelling } : {}) };
403
+ }
404
+
405
+ /**
406
+ * The operator text for a `podmanConfWidening` finding: the boot refusal, doctor's fix and the per-job log line, never a
407
+ * forge comment (it names a host path). Names the file and key, says to remove it, and names the trade-off: a setting an
408
+ * operator wanted account-wide (a pasta MTU, say) now goes on their own containers' command line instead.
409
+ */
410
+ export function podmanConfRefusal(found) {
411
+ return `${found?.transient ? "Not read yet" : "Refused"}: ${found?.evidence ?? "the containers.conf chain was not read"}; ${podmanConfFix(found)} (issue #${found?.live ? 450 : 428}).`;
412
+ }
413
+
414
+ /**
415
+ * How to reset this account's rootless network (issue #450), measured on Podman 5.8.1 and 4.9.3: it lives until the LAST
416
+ * running container on a bridge network stops, and a bridge container started meanwhile joins it as it is, so a restart
417
+ * of one container at a time never resets it. The shipped Quadlet units are named, the worker's first so its queue does
418
+ * not go under a running job, and the keeper through its unit: a `podman stop` of its container is undone a second
419
+ * later by `Restart=always`, and a keeper back while another bridge container still runs rejoins the network as it is
420
+ * (measured on 5.8.1 and 4.9.3). A unit the account does not have is reported "not loaded" and the others still stop
421
+ * and start (measured, exit 5). A container this project did not start is the operator's to find (`podman ps`).
422
+ */
423
+ export const ROOTLESS_NETNS_RESET = `stop every running container of this account that is on a bridge network, all of them at once, then start them again, since the rootless network they share lives until the last of them stops and a container started meanwhile joins it as it is: with this project's units, systemctl --user stop pi-dispatch-worker.service ${QUADLET_FILES.proxy.unit} ${QUADLET_FILES.keeper.unit} ${QUADLET_FILES.valkey.unit}, then podman stop any other container \`podman ps\` still lists, then systemctl --user start ${QUADLET_FILES.valkey.unit} ${QUADLET_FILES.keeper.unit} ${QUADLET_FILES.proxy.unit} pi-dispatch-worker.service (a worker installed at system scope is stopped and started with sudo systemctl stop and start pi-dispatch-worker.service instead; for containers started by hand, podman stop them all, then podman start them). Stop the keeper with systemctl, not podman stop: its unit starts it again a second later, and it then rejoins the network as it is while any other bridge container still runs. A unit this account does not have is reported as not loaded, and the others still stop and start`;
424
+
425
+ /** `PODMAN_NETWORK_HELPER_KEYS` (backends.mjs) as a set: the only keys whose remedy resets the rootless network. */
426
+ const NETWORK_HELPER_KEYS = new Set(PODMAN_NETWORK_HELPER_KEYS);
427
+
428
+ /** The remedy half of `podmanConfRefusal`, alone, for doctor's fix line. */
429
+ export function podmanConfFix(found) {
430
+ if (found?.live && !found.transient) {
431
+ return found.key
432
+ ? `${ROOTLESS_NETNS_RESET}. A worker this stopped at boot exits 2 and stays down until that start brings it back; a running one reads this network again before every podman job and admits the next once it no longer carries the option`
433
+ : "the podman venue reads /proc to find this account's rootless network and the options it runs with; run the worker where /proc is mounted and lists the worker account's own processes";
434
+ }
435
+ if (found?.key && !NETWORK_HELPER_KEYS.has(found.key)) {
436
+ // Gate round 1 of PR #473: a key Podman applies per container needs no network reset; the next job reads the file.
437
+ return `remove that key from that file; the next podman job runs once it is gone, since Podman reads this account's containers.conf for every container it starts. The podman venue refuses any containers.conf this account's Podman reads that sets ${WIDENING_KEYS_LISTED}, whatever the value, because no flag on a job's command line takes it back. A setting you need for your own containers goes on their own command line or Quadlet unit instead, not account-wide`;
438
+ }
439
+ return found?.key
440
+ ? `remove that key from that file, then ${ROOTLESS_NETNS_RESET}. The podman venue refuses any containers.conf this account's Podman reads that sets ${WIDENING_KEYS_LISTED}, whatever the value, because no flag on a job's command line takes it back, and it refuses a rootless network still running with such an option after the key is gone. A setting you need for your own containers (a pasta MTU, say) goes on their own command line (--network=pasta:...) or Quadlet unit instead, not account-wide`
441
+ : found?.transient
442
+ ? "the read failed for a moment, not for a reason in the file; a job refused this way is retried once (the queue's second attempt) and a boot exits to be restarted, so if it recurs, fix what the host ran out of (file descriptors, memory, a failing disk)"
443
+ : found?.spelling
444
+ ? `rewrite ${found.spelling === "escaped" ? "that key without a backslash escape" : "that line in plain ASCII with no \"\"\" or ''' multi-line string"}: this check reads a containers.conf only in plain ASCII with plain keys, since a non-ASCII case fold, a multi-line string or an escape was measured hiding a key from it that Podman honoured, and it refuses what it cannot read rather than guess`
445
+ : `the podman venue must read every containers.conf this account's Podman reads to know that none sets ${WIDENING_KEYS_LISTED}; make that file or directory readable by the worker's account, or unset the variable for it`;
446
+ }
447
+
448
+ /**
449
+ * Where Podman 5 RECORDS this account's rootless network helper (issue #450, measured on 5.8.1 with pasta, and with
450
+ * slirp4netns under `default_rootless_network_cmd`): the file holds the helper's host pid, and the helper's own argv
451
+ * names the network under the same directory (`--netns <dir>/rootless-netns` for pasta, the positional path after
452
+ * `--netns-type=path` for slirp4netns). `runRoot` is `podman info`'s `store.runRoot` (`/run/user/<uid>/containers` by
453
+ * default, measured; wherever `XDG_RUNTIME_DIR` or a storage.conf moves it otherwise). No job can write there.
454
+ */
455
+ export function rootlessNetnsRecord(runRoot) {
456
+ const dir = `${runRoot}/networks/rootless-netns`;
457
+ return { pidFile: `${dir}/rootless-netns-conn.pid`, netns: `${dir}/rootless-netns` };
458
+ }
459
+
460
+ /**
461
+ * Podman 4.9.3's rootless network, as its slirp4netns's argv names it: `--netns-type=path
462
+ * <XDG_RUNTIME_DIR>/netns/rootless-netns-<hex>` (measured). 4.9.3 does keep a record, a
463
+ * `<engine tmp_dir>/rootless-netns/rootless-netns-slirp4netns.pid` (measured present while the helper runs), but `podman
464
+ * info` does not report that directory (measured: no key names `libpod/tmp`), so on 4.x the helper is found by this path
465
+ * instead, and only among processes in
466
+ * the worker's OWN pid namespace (`NSpid` with one field, which a job's `--pid=private` process never has, and which the
467
+ * worker reads from `/proc/<pid>/status` with no ptrace check a job could deny).
468
+ */
469
+ const ROOTLESS_NETNS_4X = /\/netns\/rootless-netns-[0-9a-f]+$/;
470
+
471
+ /**
472
+ * Which kind of helper `argv` is, from its SHAPE, never its name: a renamed binary (a `helper_binaries_dir` that puts
473
+ * another program first, measured running as `podpasta`) keeps Podman's argv. slirp4netns is the one given
474
+ * `--netns-type`; anything else Podman starts for the rootless network is pasta.
475
+ */
476
+ export function rootlessNetnsKind(argv) {
477
+ return argv.slice(1).some((t) => t === "--netns-type" || t.startsWith("--netns-type=")) ? "slirp4netns" : "pasta";
478
+ }
479
+
480
+ /** The values `argv` carries, each option's `=` value as well as each bare token. */
481
+ function argvValues(argv) {
482
+ return argv.slice(1).map((t) => (t.startsWith("-") && t.includes("=") ? t.slice(t.indexOf("=") + 1) : t));
483
+ }
484
+ /**
485
+ * What in a rootless network helper's argv gives a job this host's own services (issue #450), as phrases completing
486
+ * "the rootless network still ...", empty when nothing does. Keyed on what Podman 5.8.1 and 4.9.3 were MEASURED to put
487
+ * there when a containers.conf widened it, which is not always the option the conf named:
488
+ * - `--map-host-loopback`, in the space or the `=` form, and any prefix pasta's getopt_long takes for it (measured:
489
+ * Fedora 44's pasta 2026-01-20 ran with `--map-h 169.254.1.2` and `--tcp-n 6379`);
490
+ * - Podman's own `--no-map-gw` MISSING: a conf `--map-gw` never shows in the argv, Podman drops `--no-map-gw` instead
491
+ * (so issue #450's "look for --map-gw" would have missed it);
492
+ * - a `-T`/`--tcp-ns` (and `-U`/`--udp-ns`) other than `none`, attached or in a short cluster too, or Podman's own
493
+ * `-T none` MISSING, which a conf `-T <port>` makes Podman drop. `-U` was not measured widened; it gets the same
494
+ * rule because pasta's default for it is not `none` either, and every measured argv carried `-U none`;
495
+ * - on slirp4netns, Podman's own `--disable-host-loopback` MISSING, which `allow_host_loopback=true` removes.
496
+ * Presence of the known narrowing options, not a list of harmless ones: an option this does not know is not judged,
497
+ * which is why the containers.conf check beside it still refuses every value of the key.
498
+ */
499
+ export function rootlessNetnsWidening(kind, argv) {
500
+ const args = argv.slice(1);
501
+ if (kind === "slirp4netns") return args.includes("--disable-host-loopback") ? [] : ["lacks Podman's own --disable-host-loopback, which allow_host_loopback=true removes, so 10.0.2.2 there is this host's 127.0.0.1"];
502
+ let mapsLoopback = false;
503
+ let noMapGw = false;
504
+ const ns = { T: [], U: [] };
505
+ args.forEach((t, i) => {
506
+ if (t.startsWith("--")) {
507
+ const eq = t.indexOf("=");
508
+ const name = eq < 0 ? t : t.slice(0, eq);
509
+ const value = eq < 0 ? args[i + 1] : t.slice(eq + 1);
510
+ // getopt_long takes an unambiguous prefix: `--map-h` is --map-host-loopback, `--tcp-n` is --tcp-ns.
511
+ if (name === "--no-map-gw") noMapGw = true;
512
+ else if (name.length >= 7 && "--map-host-loopback".startsWith(name)) mapsLoopback = true;
513
+ else if (name.length >= 7 && "--tcp-ns".startsWith(name)) ns.T.push(value);
514
+ else if (name.length >= 7 && "--udp-ns".startsWith(name)) ns.U.push(value);
515
+ } else if (/^-[^-]/.test(t)) {
516
+ for (const letter of ["T", "U"]) {
517
+ const at = t.indexOf(letter);
518
+ if (at > 0) ns[letter].push(t.length > at + 1 ? t.slice(at + 1) : args[i + 1]);
519
+ }
520
+ }
521
+ });
522
+ const found = [];
523
+ if (mapsLoopback) found.push("carries --map-host-loopback, which maps this host's 127.0.0.1 into it");
524
+ if (!noMapGw) found.push("lacks Podman's own --no-map-gw, which a --map-gw option removes, so its gateway address is this host");
525
+ for (const [letter, proto] of [["T", "TCP"], ["U", "UDP"]]) {
526
+ if (ns[letter].length === 0) found.push(`lacks Podman's own -${letter} none, so pasta forwards to this host's loopback ${proto} ports`);
527
+ else if (ns[letter].some((v) => v !== "none")) found.push(`carries a -${letter} other than none, which forwards to this host's loopback ${proto} ports`);
528
+ }
529
+ return found;
530
+ }
531
+
532
+ /**
533
+ * This account's running rootless network helpers, from Podman's own record (issue #450), as `{ helpers: [{ pid, kind,
534
+ * widened }] }` or `{ unread: { path, code } }`. Never found by process name, and never by a path a job could forge:
535
+ * 1. Podman 5: the pid in `rootlessNetnsRecord(runRoot).pidFile`. That process is the helper when it is still this
536
+ * uid's and its argv names that record's pid file or network (so a pid the kernel has since handed to another
537
+ * process is not trusted). A missing file is no helper: Podman removes it when the last bridge container stops
538
+ * (measured), and with it missing under a running helper Podman itself could not start a job (TUNSETIFF,
539
+ * measured). A file whose pid is gone (a killed helper leaves it, measured) is no helper either.
540
+ * 2. Podman 4.x, only when there is no such record: every process of this uid in the worker's own pid namespace
541
+ * whose argv carries `--netns-type=path <...>/netns/rootless-netns-<hex>` (`ROOTLESS_NETNS_4X`).
542
+ * A process gone mid-read (ENOENT, ESRCH) is skipped; a status denied under `hidepid` is only ever another account's
543
+ * and is skipped. What decides it and cannot be read REFUSES, the containers.conf chain's rule: the record (a
544
+ * `runRoot` that is not an absolute path, a pid file that exists and cannot be read or holds no pid), a `/proc` that
545
+ * exists and cannot be listed, or the argv of a process this rule has already taken for the helper. A transient errno
546
+ * anywhere is the retry. A `/proc` that does not exist is not read, as a missing drop-in directory is not.
547
+ *
548
+ * UNCACHED, measured: the 4.x scan reads every `/proc/<pid>/status` synchronously, and with about 2200 processes took
549
+ * a median of 24.5 ms (max 40.8) on Fedora 44 and 19.1 ms (max 31.9) on Ubuntu 24.04; with a 5.x record it reads three
550
+ * files. A cache would be a window in which a network widened since reads as narrow.
551
+ */
552
+ export function observeRootlessNetns({ fs, euid, runRoot }) {
553
+ if (typeof runRoot !== "string" || !runRoot.startsWith("/")) return { unread: { path: "podman info's store.runRoot", code: "unreported" } };
554
+ const gone = (code) => code === "ENOENT" || code === "ESRCH";
555
+ const record = rootlessNetnsRecord(runRoot);
556
+ let recorded;
557
+ try {
558
+ recorded = String(fs.readFileSync(record.pidFile, "utf8")).trim();
559
+ } catch (error) {
560
+ if (error?.code !== "ENOENT") return { unread: { path: record.pidFile, code: error?.code ?? "error" } };
561
+ }
562
+ if (recorded !== undefined) {
563
+ if (!/^[1-9][0-9]{0,9}$/.test(recorded)) return { unread: { path: record.pidFile, code: "no-pid" } };
564
+ const proc = readProcess(fs, recorded, euid);
565
+ if (proc.unread) return proc;
566
+ if (!proc.argv) return { helpers: [] };
567
+ const values = argvValues(proc.argv);
568
+ if (!values.includes(record.pidFile) && !values.includes(record.netns)) return { helpers: [] };
569
+ // A pid the kernel handed on after the helper died, to a process that names the record (gate round 2 of PR #469:
570
+ // a job's, after a pasta crash left the file behind): not the helper.
571
+ const age = startedByRecord(fs, record.pidFile, recorded, proc.nspid);
572
+ if (age.unread) return age;
573
+ if (!age.trusted) return { helpers: [] };
574
+ const kind = rootlessNetnsKind(proc.argv);
575
+ return { helpers: [{ pid: Number(recorded), kind, widened: rootlessNetnsWidening(kind, proc.argv) }] };
576
+ }
577
+ let entries;
578
+ try {
579
+ entries = fs.readdirSync("/proc");
580
+ } catch (error) {
581
+ if (error?.code === "ENOENT") return { helpers: [] };
582
+ return { unread: { path: "/proc", code: error?.code ?? "error" } };
583
+ }
584
+ const helpers = [];
585
+ for (const entry of entries) {
586
+ const pid = String(entry);
587
+ if (!/^\d+$/.test(pid)) continue;
588
+ const proc = readProcess(fs, pid, euid, { ownNamespaceOnly: true });
589
+ if (proc.unread) return proc;
590
+ if (!proc.argv) continue;
591
+ const argv = proc.argv;
592
+ const at = argv.findIndex((t, i) => t === "--netns-type=path" || (t === "--netns-type" && argv[i + 1] === "path"));
593
+ if (at < 0 || !argv.slice(at + 1).some((t) => ROOTLESS_NETNS_4X.test(t))) continue;
594
+ helpers.push({ pid: Number(pid), kind: "slirp4netns", widened: rootlessNetnsWidening("slirp4netns", argv) });
595
+ }
596
+ return { helpers };
597
+ }
598
+
599
+ /**
600
+ * One process, as `{ argv }` when it is this uid's (and, with `ownNamespaceOnly`, in the worker's own pid namespace:
601
+ * one `NSpid` field), `{}` when it is not, or is gone, or `{ unread }` when a read failed for a moment or its argv, once
602
+ * it is this uid's, could not be read. A status denied (EACCES, EPERM) is another account's under `hidepid`.
603
+ */
604
+ function readProcess(fs, pid, euid, { ownNamespaceOnly = false } = {}) {
605
+ let status;
606
+ try {
607
+ status = String(fs.readFileSync(`/proc/${pid}/status`, "utf8"));
608
+ } catch (error) {
609
+ if (TRANSIENT_READ_ERRORS.has(error?.code)) return { unread: { path: `/proc/${pid}/status`, code: error.code } };
610
+ return {};
611
+ }
612
+ const uid = /^Uid:\s+(\d+)/m.exec(status)?.[1];
613
+ if (uid === undefined || Number(uid) !== euid) return {};
614
+ const nspidLine = /^NSpid:\s+(.*)$/m.exec(status)?.[1];
615
+ const nspid = nspidLine === undefined ? null : nspidLine.trim().split(/\s+/);
616
+ if (ownNamespaceOnly && (nspid === null || nspid.length !== 1)) return {};
617
+ let cmdline;
618
+ try {
619
+ cmdline = String(fs.readFileSync(`/proc/${pid}/cmdline`, "utf8"));
620
+ } catch (error) {
621
+ if (error?.code === "ENOENT" || error?.code === "ESRCH") return {};
622
+ return { unread: { path: `/proc/${pid}/cmdline`, code: error?.code ?? "error" } };
623
+ }
624
+ const argv = cmdline.split("\0");
625
+ if (argv.at(-1) === "") argv.pop();
626
+ return argv.length > 0 ? { argv, nspid } : {};
627
+ }
628
+
629
+ /**
630
+ * The units `/proc/<pid>/stat`'s start time is in: USER_HZ, which the kernel fixes at 100 for what `/proc` reports on
631
+ * every architecture Podman ships for (x86_64, aarch64, ppc64le, s390x; alpha's 1024 is not one), whatever CONFIG_HZ is.
632
+ */
633
+ export const PROC_USER_HZ = 100;
634
+
635
+ /**
636
+ * How much later than its pid file's mtime the recorded process may have started and still be the one that wrote it
637
+ * (gate round 2 of PR #469). The helper writes the file after it starts (pasta's own `--pid`, Podman's write for
638
+ * slirp4netns), so its true start is never later; the slack is for the arithmetic: `btime` is whole seconds, rounded
639
+ * down, which only makes the start read EARLIER; a start is in 10 ms ticks; and NTP may slew the wall clock by at most
640
+ * 500 ppm between the write and this read, 1.8 s over an hour. 2 s covers those and no more, since a pid recycled
641
+ * within it would need the helper to die within 2 s of starting.
642
+ */
643
+ export const RECORD_START_TOLERANCE_MS = 2_000;
644
+
645
+ /**
646
+ * Whether the recorded process is the one Podman's record was written for, as `{ trusted }` or `{ unread }`: it started
647
+ * no later than the pid file's mtime (`/proc/<pid>/stat` field 22 in `PROC_USER_HZ` ticks after `/proc/stat`'s `btime`,
648
+ * within `RECORD_START_TOLERANCE_MS`). A later start is a recycled pid, UNLESS the process is shaped as no job's process
649
+ * can be: `NSpid` with exactly one field (the worker's own pid namespace, 5.8.1's slirp4netns, measured), or exactly
650
+ * two ending in 1 (PID 1 of a namespace DIRECTLY beneath the worker's, 5.8.1's pasta, measured). A job's own
651
+ * namespace is one level down too, but its PID 1 is always its `--init` process (`/run/podman-init -- <the image's
652
+ * entrypoint>`, measured), whose argv names no record because the worker passes no command after the image (a job's
653
+ * command travels in its environment) and the image is operator-authored: a job can put no string there, though an
654
+ * image that bakes the record's path into its own entrypoint could (PR #469's gate round 4); and a namespace a job
655
+ * makes for itself (`unshare -Urpf` succeeds in a job-shaped container, PR #469's gate round 3) is two levels down, so
656
+ * its PID 1 has three or more fields and is not trusted. Measured on 5.8.1 with
657
+ * pasta and with slirp4netns: the helper started 228 to 353 ms before its record's mtime. That
658
+ * keeps a wall-clock step (which moves `btime` and not the file's mtime) from hiding the real helper: a step is judged
659
+ * on the helper's shape, never read as "no helper". Anything that decides it and cannot be read refuses, named.
660
+ */
661
+ function startedByRecord(fs, pidFile, pid, nspid) {
662
+ const unread = (path, error) => ({ unread: { path, code: typeof error === "string" ? error : (error?.code ?? "error") } });
663
+ let mtimeMs;
664
+ try {
665
+ mtimeMs = fs.statSync(pidFile)?.mtimeMs;
666
+ } catch (error) {
667
+ if (error?.code === "ENOENT") return { trusted: false };
668
+ return unread(pidFile, error);
669
+ }
670
+ if (!Number.isFinite(mtimeMs)) return unread(pidFile, "no-mtime");
671
+ let stat;
672
+ try {
673
+ stat = String(fs.readFileSync(`/proc/${pid}/stat`, "utf8"));
674
+ } catch (error) {
675
+ if (error?.code === "ENOENT" || error?.code === "ESRCH") return { trusted: false };
676
+ return unread(`/proc/${pid}/stat`, error);
677
+ }
678
+ // Field 22, counted after the `(comm)` field, which may itself hold spaces and parentheses.
679
+ const ticks = stat.slice(stat.lastIndexOf(")") + 1).trim().split(/\s+/)[19];
680
+ if (!/^\d+$/.test(ticks ?? "")) return unread(`/proc/${pid}/stat`, "no-start-time");
681
+ let btime;
682
+ try {
683
+ btime = /^btime\s+(\d+)$/m.exec(String(fs.readFileSync("/proc/stat", "utf8")))?.[1];
684
+ } catch (error) {
685
+ return unread("/proc/stat", error);
686
+ }
687
+ if (btime === undefined) return unread("/proc/stat", "no-btime");
688
+ const startedMs = (Number(btime) + Number(ticks) / PROC_USER_HZ) * 1000;
689
+ if (startedMs <= mtimeMs + RECORD_START_TOLERANCE_MS) return { trusted: true };
690
+ return { trusted: Array.isArray(nspid) && (nspid.length === 1 || (nspid.length === 2 && nspid[1] === "1")) };
691
+ }
692
+
693
+ /**
694
+ * The live half of `podmanConfWidening` (issue #450): this account's running rootless network, when it still carries
695
+ * an option that gives a job this host's services, as a `podman-conf-widens-job` finding with `live: true`, else
696
+ * `null`. The SAME cause, not a sibling: it is the same widening, a containers.conf key's, still in force after the key
697
+ * went, and the cause is a fixed enum the forge comment and the run record already carry; the processor's comment for
698
+ * a `live` finding says it is the running network, not the configuration. `key` is the conf key that gave the helper
699
+ * that option (`pasta_options` for pasta, `network_cmd_options` for slirp4netns). No helper running is no live network:
700
+ * the next bridge container starts one from the conf as it is now, which the chain above has just judged.
701
+ *
702
+ * `runRoot` is `podman info`'s `store.runRoot`: a path, or `null` for an answered info that reported none, which REFUSES
703
+ * (named). `undefined` is an info read that has not answered, and is not judged here: that same read leaves the job user
704
+ * undecided (`unknown`), which retries the job before any container, and the next judgement has the answer.
705
+ */
706
+ export function podmanNetnsWidening({ fs, euid, runRoot }) {
707
+ if (!Number.isInteger(euid) || runRoot === undefined) return null;
708
+ const seen = observeRootlessNetns({ fs, euid, runRoot });
709
+ if (seen.unread) {
710
+ const { path, code } = seen.unread;
711
+ const evidence = code === "unreported"
712
+ ? "podman info reports no store.runRoot, so where Podman records this account's running rootless network is not known"
713
+ : code === "no-pid"
714
+ ? `${path} holds no pid, so which process is this account's running rootless network is not known`
715
+ : code === "no-mtime" || code === "no-start-time" || code === "no-btime"
716
+ ? `${path} gives no ${code === "no-mtime" ? "modification time" : code === "no-btime" ? "boot time" : "start time"}, so whether the recorded process is the one Podman's record was written for is not known`
717
+ : `${path} could not be read (${code}), so whether this account's running rootless network still carries an option that widens a job is not known`;
718
+ return TRANSIENT_READ_ERRORS.has(code) ? { cause: PODMAN_CONF_WIDENS_JOB, key: null, live: true, evidence, transient: true } : { cause: PODMAN_CONF_WIDENS_JOB, key: null, live: true, evidence };
719
+ }
720
+ const widened = seen.helpers.find((h) => h.widened.length > 0);
721
+ if (!widened) return null;
722
+ return {
723
+ cause: PODMAN_CONF_WIDENS_JOB,
724
+ key: widened.kind === "pasta" ? "pasta_options" : "network_cmd_options",
725
+ live: true,
726
+ evidence: `this account's running rootless network (${widened.kind}, pid ${widened.pid}), which every container on a bridge network shares, the egress proxy's among them, still ${widened.widened.join(", and ")}: it keeps the options it started with, whatever containers.conf says now`,
727
+ };
728
+ }
729
+
730
+ /**
731
+ * Every podman observation for one info read, as `observeHost`'s `{ observations, evidence, reasons }`.
732
+ *
733
+ * THREE ANSWERS, NEVER TWO, on `runtime-observations.mjs`'s rule: a TRANSIENT unanswered read is `null` for all three
734
+ * (a floor retries) with the read's reason; a DETERMINATE one (no podman on PATH, an answer nothing parses) is `false`
735
+ * (a floor refuses). The mounts answer rests on the read too: which mounts.conf wins depends on Podman being rootless.
736
+ * `fs` is `{ statSync, readFileSync, readdirSync }`; `home`, `env` and `euid` are the account whose Podman runs the jobs.
737
+ */
738
+ export function observePodman({ read, fs = { statSync, readFileSync, readdirSync }, home = homedir(), env = process.env, euid = process.geteuid?.() } = {}) {
739
+ const names = [PODMAN_BOUNDS_DELEGATED, PODMAN_ADDS_NO_MOUNTS, PODMAN_SERVICE_LOCAL];
740
+ if (!(read?.answered === true && read.info)) {
741
+ const determinate = read && read.answered === false && read.transient === false;
742
+ const value = determinate ? false : null;
743
+ const evidence = determinate
744
+ ? read.reason === "podman-not-found" ? "no podman CLI was found on PATH" : `podman info answered in a shape nothing here reads (${read.reason ?? "unparseable"})`
745
+ : `podman info was not read (${read?.reason ?? "not asked"})`;
746
+ return {
747
+ observations: Object.fromEntries(names.map((n) => [n, value])),
748
+ evidence: Object.fromEntries(names.map((n) => [n, evidence])),
749
+ reasons: value === null ? Object.fromEntries(names.map((n) => [n, read?.reason ?? "not-read"])) : {},
750
+ };
751
+ }
752
+ const { info } = read;
753
+ const bounds = observePodmanBounds(info, { fs, home, env, euid });
754
+ const mounts = info.rootless === true ? observePodmanMounts({ fs, home, env, euid }) : { value: false, evidence: "podman info does not report a rootless Podman, whose own mounts.conf this check reads" };
755
+ const service = info.serviceIsRemote === false
756
+ ? { value: true, evidence: "podman info reports serviceIsRemote false" }
757
+ : { value: false, evidence: info.serviceIsRemote === true ? "podman info reports serviceIsRemote true (CONTAINER_HOST, --remote or a service destination)" : "podman info does not say whether its service is remote" };
758
+ return {
759
+ observations: { [PODMAN_BOUNDS_DELEGATED]: bounds.value, [PODMAN_ADDS_NO_MOUNTS]: mounts.value, [PODMAN_SERVICE_LOCAL]: service.value },
760
+ evidence: { [PODMAN_BOUNDS_DELEGATED]: bounds.evidence, [PODMAN_ADDS_NO_MOUNTS]: mounts.evidence, [PODMAN_SERVICE_LOCAL]: service.evidence },
761
+ // The controllers delegated to the account's user manager when the bounds hold (issue #453), for doctor's line: the
762
+ // fact the answer rests on, never `podman info`'s caller-cgroup list.
763
+ ...(bounds.value === true ? { boundsControllers: bounds.controllers } : {}),
764
+ // Which miss it was, when it was the user manager's, so doctor gives that fix and no other.
765
+ ...(bounds.cause ? { boundsCause: bounds.cause } : {}),
766
+ // With an answered read, a `null` here is a host file that could not be read for a moment (`unreadFileFinding`),
767
+ // and its reason says so, so the retry names a file rather than "unknown".
768
+ reasons: {
769
+ ...(bounds.value === null ? { [PODMAN_BOUNDS_DELEGATED]: bounds.reason ?? "file-unread" } : {}),
770
+ ...(mounts.value === null ? { [PODMAN_ADDS_NO_MOUNTS]: mounts.reason ?? "file-unread" } : {}),
771
+ },
772
+ };
773
+ }
774
+
775
+ /**
776
+ * An observation miss that rests on a read that did not answer, as the preflight's `{ unavailable, reason }`, plus
777
+ * `message` (the evidence, which names the path) when that read was a HOST FILE, so the retry's words say which file
778
+ * rather than blaming the runtime. Shared with `local`'s preflight in start.mjs.
779
+ */
780
+ export function unavailableFor(observed, name) {
781
+ const reason = observed?.reasons?.[name] ?? "unknown";
782
+ return reason === "file-unread" ? { unavailable: true, reason, message: observed?.evidence?.[name] ?? null } : { unavailable: true, reason };
783
+ }
784
+
785
+ /** The three podman answers as one string, so a caller logs them only when they change. */
786
+ export function podmanObservationKey(observed) {
787
+ return `${observed.observations[PODMAN_BOUNDS_DELEGATED]}|${observed.observations[PODMAN_ADDS_NO_MOUNTS]}|${observed.observations[PODMAN_SERVICE_LOCAL]}`;
788
+ }
789
+
790
+ // The fixed operator texts per cause on this venue: the boot refusal, doctor and the per-job log. OUT of job-user.mjs's
791
+ // JOB_USER_FIX on purpose, since that map is pinned to `local`'s causes by backends-doc.test.mjs and is `local`'s
792
+ // vocabulary (a rootless DAEMON is a refusal there and the only kind this venue runs on). No text carries CLI output, a
793
+ // path or a URL: a refusal that repeats what podman printed can repeat a remote service's credentials.
794
+ export const PODMAN_JOB_USER_FIX = Object.freeze({
795
+ "podman-platform": "the podman venue runs only on Linux (Podman machine on macOS and Windows was not measured); use the local venue with Docker Desktop, or run the worker on a Linux host",
796
+ "podman-not-found": "no podman CLI was found on the worker's PATH; install Podman for the worker account, or remove podman from PI_BACKENDS",
797
+ "podman-unreadable": "podman info answered with something no rule can read, so which uid a job may run as is unknown; check that `podman info --format json` works as the worker account",
798
+ "podman-remote": "podman runs its containers through a remote service (CONTAINER_HOST, --remote or a containers.conf service destination), where a job's mounts and secrets are another machine's; unset it for the worker account",
799
+ "podman-rootful": "podman is not rootless for this account, and the podman venue runs only on rootless Podman (rootful keep-id adds the root group); run the worker as an unprivileged account, or use rootful Podman through the local venue's Docker API route (docs/podman.md)",
800
+ "worker-is-root": "the worker runs as root, and a job must not run as root (nonRoot); run the worker as an unprivileged account with its own rootless Podman",
801
+ "root-group": "the worker's primary group is gid 0, and a podman job runs with that group; run the worker with an unprivileged primary group",
802
+ "any-uid-unsupported": "the job image does not declare `anyUid` (`dev.pi-dispatch.capabilities`), so it cannot run as this worker's own uid, which the podman venue always uses; rebuild it from a release that has this feature, or run the worker as uid 1001",
803
+ });
804
+
805
+ /**
806
+ * The causes that stop a worker whose DEFAULT venue is `podman` from booting: facts about the platform, the CLI and the
807
+ * worker's identity, which no podman job can get past. A plain Set read with `.has`, as `BOOT_REFUSING_JOB_USER_CAUSES`.
808
+ * `podman-unreadable` describes one answer, and the group and image rules one job, so they refuse per job.
809
+ */
810
+ export const PODMAN_BOOT_REFUSING_CAUSES = new Set(["podman-platform", "podman-not-found", "podman-remote", "podman-rootful", "worker-is-root"]);
811
+
812
+ /** The operator-facing refusal for a podman cause or decision. */
813
+ export function podmanJobUserRefusal(causeOrDecision) {
814
+ const cause = typeof causeOrDecision === "string" ? causeOrDecision : causeOrDecision?.cause;
815
+ return `Refused: ${Object.hasOwn(PODMAN_JOB_USER_FIX, cause ?? "") ? PODMAN_JOB_USER_FIX[cause] : "the job user could not be decided"} (issue #354).`;
816
+ }
817
+
818
+ /**
819
+ * Who a podman job runs as, decided from the info read (`read`, `readInfo()`'s answer) and the worker's own ids. Returns
820
+ * `{ mode: "worker", user, relabel }`, `{ mode: "unmappable", cause }` or `{ mode: "unknown", reason }`. The rows are
821
+ * ORDERED, and the order is the contract:
822
+ * 1. not Linux: `podman-platform`, before any read, since nothing Podman could say changes it;
823
+ * 2. an unanswered read: `unknown` when transient (retried, never a boot exit), else `podman-not-found` or
824
+ * `podman-unreadable`;
825
+ * 3. a remote service (or one that does not say it is local): `podman-remote`;
826
+ * 4. not rootless (or not saying): `podman-rootful`;
827
+ * 5. no process ids: `unknown`; the worker is root: `worker-is-root`;
828
+ * 6. else the worker's own `<euid>:<egid>`, with `relabel` true only where Podman reports SELinux on (the PR #424 rule:
829
+ * `:Z` on the job's own per-job directories, and only there).
830
+ */
831
+ export function decidePodmanJobUser({ platform = process.platform, euid, egid, read } = {}) {
832
+ const unmappable = (cause) => ({ mode: "unmappable", user: null, relabel: false, cause, reason: null });
833
+ const unknown = (reason) => ({ mode: "unknown", user: null, relabel: false, cause: null, reason });
834
+ if (platform !== "linux") return unmappable("podman-platform");
835
+ if (!(read?.answered === true && read.info)) {
836
+ if (read?.answered === false && read.transient === false) return unmappable(read.reason === "podman-not-found" ? "podman-not-found" : "podman-unreadable");
837
+ return unknown(read?.reason ?? "no-podman-info");
838
+ }
839
+ const { info } = read;
840
+ if (info.serviceIsRemote !== false) return unmappable("podman-remote");
841
+ if (info.rootless !== true) return unmappable("podman-rootful");
842
+ if (!Number.isInteger(euid) || !Number.isInteger(egid)) return unknown("no-process-ids");
843
+ if (euid === 0) return unmappable("worker-is-root");
844
+ return { mode: "worker", user: `${euid}:${egid}`, relabel: info.selinux === true, cause: null, reason: null };
845
+ }
846
+
847
+ /**
848
+ * The per-image half, for a podman job about to run: `{ user, home, relabel }`, `{ refused, cause }` or `{ unavailable,
849
+ * reason }`. The job ALWAYS runs as the worker's uid (keep-id needs `--user`, measured: without it the image's user runs
850
+ * with `/job` unreadable), so unlike `local` there is no "image's own user" answer. `anyUid` is required unless that uid
851
+ * is the image's own 1001; a primary group of 0 is refused here, before the builder would throw on it (`assertJobUser`).
852
+ */
853
+ export function resolvePodmanImageUser(decision, { capabilities = [], euid, egid } = {}) {
854
+ if (decision?.mode === "unmappable") return { refused: "job-user-unmappable", cause: decision.cause };
855
+ if (decision?.mode !== "worker") return { unavailable: true, reason: decision?.reason ?? "unknown" };
856
+ if (euid !== SHIPPED_IMAGE_UID && (!Array.isArray(capabilities) || !capabilities.includes("anyUid"))) return { refused: "job-image-any-uid-unsupported", cause: "any-uid-unsupported" };
857
+ if (egid === 0) return { refused: "job-user-unmappable", cause: "root-group" };
858
+ return { user: decision.user, home: CONTAINER_HOME, relabel: decision.relabel === true };
859
+ }
860
+
861
+ /**
862
+ * The venue half of a podman job's pre-spend preflight, over one info read (`read`, `readInfo()`'s answer), shaped
863
+ * exactly as the bundle's `observationPreflight` answers: `{ ok: true, podman }`, `{ refused, message, observations }`
864
+ * for an answered floor miss, `{ unavailable, reason }` when the miss rests only on a read that did not answer. Beside
865
+ * `ok`, a refused identity rides as `jobUserRefused` and a widening containers.conf as `podmanConfRefused` (`{ reason,
866
+ * key, message }`, issue #428); the processor refuses either before its image preflight.
867
+ *
868
+ * EXPORTED AND SHARED (issue #429) because a sandbox opened on this venue must be refused for exactly what a job is
869
+ * refused for, in the same order, and "the same order" is a property a second copy loses first. The bundle wraps it
870
+ * with its own log line (`onObserved`, called once per judgement that reached the observations) and nothing else.
871
+ *
872
+ * ORDER, and each step is before the next for a reason:
873
+ * 1. the IDENTITY, from this same read. A venue whose job user is refused (rootful, remote, no podman) fails the
874
+ * observations too, and judging them first told every job naming it to fix a mounts.conf or delegate controllers
875
+ * when the one fix is its identity's. Handed back as `jobUserRefused`, which the processor acts on BEFORE its
876
+ * image preflight: that preflight asks the same Podman, so behind it a missing podman was retried forever and a
877
+ * rootful one without the image in its store was blamed on the image. No image can change an unmappable answer
878
+ * (`resolvePodmanImageUser` refuses it before reading any capability). An undecided read (`unknown`) is still
879
+ * judged here and is retried, never refused.
880
+ * 2. the account's own containers.conf (issue #428), before the floor: it refuses whatever the floor says, since
881
+ * with egress off no declared property covers what the job reaches on the host, so a floor could never ask for it
882
+ * on the deployments at risk. Re-read per call, so removing the key needs no restart.
883
+ * 3. the observations against `backendFloor`.
884
+ */
885
+ export function judgePodmanVenue({ read, platform = process.platform, euid, egid, fs = { statSync, readFileSync, readdirSync }, home = homedir(), env = process.env, backendFloor = {}, onObserved = () => {} } = {}) {
886
+ const decision = decidePodmanJobUser({ platform, euid, egid, read });
887
+ if (decision.mode === "unmappable") return { ok: true, podman: read, jobUserRefused: { refused: "job-user-unmappable", cause: decision.cause } };
888
+ // `runRoot` from this same read (issue #450): `undefined` while it has not answered, when the undecided job user retries.
889
+ const widened = podmanConfWidening({ fs, home, env, euid, runRoot: read?.answered === true && read.info ? (read.info.runRoot ?? null) : undefined });
890
+ // A transient read rides the same field with `transient: true`, which the processor retries rather than refuses.
891
+ if (widened) return { ok: true, podman: read, podmanConfRefused: { reason: PODMAN_CONF_WIDENS_JOB, key: widened.key, message: podmanConfRefusal(widened), ...(widened.live ? { live: true } : {}), ...(widened.transient ? { transient: true, evidence: widened.evidence } : {}) } };
892
+ const observed = observePodman({ read, fs, home, env, euid });
893
+ onObserved(observed);
894
+ const args = { backends: [PODMAN_BACKEND], backendFloor, observations: observed.observations, evidence: observed.evidence };
895
+ const [refusal] = observationRefusals(args);
896
+ if (!refusal) return { ok: true, podman: read };
897
+ const missed = [...new Set(unobservedFloor(args.backends, args.backendFloor, args.observations).map((m) => m.observedBy))];
898
+ if (observationRefusalIsTransient(args)) return unavailableFor(observed, missed[0]);
899
+ return { refused: true, message: refusal, observations: missed };
900
+ }
901
+
902
+ /**
903
+ * A `promisify(execFile)`-shaped runner over a `spawn`-shaped one: resolves `{ stdout, stderr }` on exit 0, rejects with
904
+ * `{ code, stdout, stderr }` otherwise, so the reaper and the stop can be driven by the same fake child a test hands the
905
+ * run. A spawn `error` (no binary) rejects with that error, as `execFile` does.
906
+ */
907
+ export function execViaSpawn(spawnFn) {
908
+ return (cmd, args) =>
909
+ new Promise((resolve, reject) => {
910
+ let stdout = "";
911
+ let stderr = "";
912
+ let child;
913
+ try {
914
+ child = spawnFn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
915
+ } catch (err) {
916
+ reject(err);
917
+ return;
918
+ }
919
+ child.stdout?.on?.("data", (c) => {
920
+ stdout += String(c);
921
+ });
922
+ child.stderr?.on?.("data", (c) => {
923
+ stderr += String(c);
924
+ });
925
+ child.on("error", reject);
926
+ child.on("close", (code) => {
927
+ if (code === 0) resolve({ stdout, stderr });
928
+ else reject(Object.assign(new Error(`${cmd} exited ${code}`), { code, stdout, stderr }));
929
+ });
930
+ });
931
+ }
932
+
933
+ /**
934
+ * The podman venue's boot reaper: `makeReaper` with `bin: "podman"`, so the sweep enumerates, removes and cleans networks
935
+ * in the store the jobs actually ran in, and keeps its tri-state (a failed enumeration is `{ reaped: false }`). Built at
936
+ * boot, early, like `local`'s, because the boot sweep runs before the bundle exists. `exec` (execFile-shaped) or
937
+ * `spawnFn` (spawn-shaped, adapted) are the test seams.
938
+ */
939
+ export function makePodmanReaper({ log, exec, spawnFn } = {}) {
940
+ return makeReaper({ log: log ?? (() => {}), bin: "podman", ...(exec ? { exec } : spawnFn ? { exec: execViaSpawn(spawnFn) } : {}) });
941
+ }
942
+
943
+ /** The keys `makePodmanBackend` takes; anything else is refused, since a misspelt config key would be silently dropped. */
944
+ export const PODMAN_BACKEND_OPTIONS = Object.freeze([
945
+ // makeRunContainer's deployment inputs, as start.mjs hands the local one.
946
+ "image",
947
+ "hostEnv",
948
+ "egress",
949
+ "egressProxy",
950
+ "openJobLog",
951
+ "globalPiDir",
952
+ "allowGlobalExtensions",
953
+ "packagePaths",
954
+ "forwardEnv",
955
+ "authFromPi",
956
+ "forgeHosts",
957
+ "onOutput",
958
+ // the floor the per-job observation preflight judges, and the boot-built reaper
959
+ "backendFloor",
960
+ "reap",
961
+ // the facts, the identity and the files the observations and the job user come from
962
+ "readInfo",
963
+ "platform",
964
+ "euid",
965
+ "egid",
966
+ "fs",
967
+ "home",
968
+ "env",
969
+ "log",
970
+ // seams
971
+ "spawnFn",
972
+ "exec",
973
+ "makeRunContainer",
974
+ "makeImagePreflight",
975
+ "makeEgressPreflight",
976
+ "makeStopContainer",
977
+ ]);
978
+
979
+ /** The keeper read's bound: one `podman inspect`, on a pre-spend gate, like the proxy's own read beside it. */
980
+ export const NETNS_KEEPER_READ_TIMEOUT_MS = 10_000;
981
+
982
+ /**
983
+ * The egress preflight on this venue, with the rootless network keeper (issue #458) added to what it checks. On Podman
984
+ * 4.x (or a version `podman info` did not give) a job whose proxy is up but whose keeper does not hold would start,
985
+ * spend, and get 503 from the proxy as soon as any earlier job's teardown ran: measured, every egress job after the
986
+ * first. So it is `{ unavailable, keeper }`, an INFRA retry before anything is spent, carrying the sentence that names
987
+ * the keeper and what to run. "Holds" is `judgeNetnsKeeper` with the clock and the proxy's start (PR #463 round 2):
988
+ * running on its own bridge for at least 3 s (a crash loop reads as running for moments), and not started more than
989
+ * the grace (15 s) after the proxy, since a keeper that restarted while the proxy ran may have let a teardown cut the
990
+ * proxy's route out, which only a proxy restart repairs and nothing outside can see. A keeper that is only too young
991
+ * (issue #476) is still not held, but the answer carries `young`, so its caller waits it out rather than failing on it.
992
+ * On 5.x nothing is read. Nothing is cached across jobs but what the bundle already caches (`podman info`, for the
993
+ * version): the keeper and the proxy's start are read on every armed job, two bounded `podman inspect`s.
994
+ */
995
+ export function keeperPreflight(proxyPreflight, { armed, proxy, info, spawnFn = null, readKeeper = null, now = Date.now }) {
996
+ const read =
997
+ readKeeper ??
998
+ (spawnFn
999
+ ? (args) =>
1000
+ execViaSpawn(spawnFn)("podman", args).then(
1001
+ (r) => ({ code: 0, stdout: r.stdout }),
1002
+ (err) => ({ code: typeof err?.code === "number" ? err.code : null, stdout: err?.stdout ?? "" }),
1003
+ )
1004
+ : (args) => execDockerBounded(args, { bin: "podman", timeoutMs: NETNS_KEEPER_READ_TIMEOUT_MS }));
1005
+ return async (...args) => {
1006
+ const result = await proxyPreflight(...args);
1007
+ if (!armed || result?.ok !== true) return result;
1008
+ const answered = await info();
1009
+ const version = answered?.answered === true ? answered.info?.version : null;
1010
+ if (!podmanNeedsNetnsKeeper(version)) return result;
1011
+ const name = result.proxy ?? proxy ?? DEFAULT_EGRESS_PROXY;
1012
+ const keeperRead = await read(["inspect", NETNS_KEEPER_FORMAT, NETNS_KEEPER]);
1013
+ const proxyRead = await read(["inspect", STARTED_AT_FORMAT, name]);
1014
+ const proxyStarted = proxyRead?.code === 0 ? String(proxyRead.stdout ?? "").trim() : "";
1015
+ const keeper = judgeNetnsKeeper(keeperRead, { now: now(), proxyStartedMs: /^\d+$/.test(proxyStarted) ? Number(proxyStarted) : null });
1016
+ if (keeper.holds) return result;
1017
+ const on = typeof version === "string" && version.trim() ? `Podman ${version.trim()}` : "a Podman of unreported version";
1018
+ const cause = keeper.restartProxy ? `the rootless network keeper ${NETNS_KEEPER} ${keeper.problem} (${on}, issue #458)` : `the rootless network keeper ${NETNS_KEEPER} ${keeper.problem}, and on ${on} a job network's teardown would then cut the egress proxy's route out (issue #458)`;
1019
+ return {
1020
+ unavailable: name,
1021
+ // The two halves on their own (issue #452, gate round 2), for a caller that is not about a job: the sandbox opener.
1022
+ cause,
1023
+ remedy: netnsKeeperRemedy(keeper, name),
1024
+ // The same facts without a job in them, for the worker's boot line (PR #463 round 3).
1025
+ keeperAtBoot: `${cause}, so every egress job is retried rather than started until this is fixed. To fix it, ${netnsKeeperRemedy(keeper, name)}`,
1026
+ keeper: `${cause}, so this job is retried rather than started. To fix it, ${netnsKeeperRemedy(keeper, name)}`,
1027
+ // Issue #476: a keeper whose only fault is its age (`{ startedMs, ageMs, waitMs }`), which the boot and the
1028
+ // sandbox opener wait out and a job is held for, without an attempt.
1029
+ ...(keeper.young ? { young: keeper.young } : {}),
1030
+ // The judge's own words alone, which follow the keeper's name, for the crash-loop sentence a held job may end on.
1031
+ problem: keeper.problem,
1032
+ };
1033
+ };
1034
+ }
1035
+
1036
+ /**
1037
+ * The podman bundle: `{ name, declares, neverStartedExits, containerName, namePrefix, binds, runContainer,
1038
+ * imagePreflight, egressPreflight, stopContainer, reap, jobUserPreflight, observationPreflight }`.
1039
+ *
1040
+ * BUILT here from PR #425's factories with `bin: "podman"` (and `buildPodmanRunArgs` for the run), rather than taken
1041
+ * built as `makeLocalBackend` does, because the pairing IS the adapter: a run under one binary and a stop, a preflight or
1042
+ * a sweep under another is the mixed venue every `bin` seam comment warns about. The factory seams exist for the tests.
1043
+ *
1044
+ * `reap` is the boot-built `makePodmanReaper` (the sweep runs before the bundle exists); omitted, one is built here.
1045
+ * `readInfo` is wrapped by `cachedPodmanInfo`, so a boot read through the same wrapped reader is shared with every job.
1046
+ */
1047
+ export function makePodmanBackend(opts = {}) {
1048
+ for (const key of Object.keys(opts ?? {})) {
1049
+ if (!PODMAN_BACKEND_OPTIONS.includes(key)) throw new Error(`backend "${PODMAN_BACKEND}": unknown option ${JSON.stringify(key)} (takes ${PODMAN_BACKEND_OPTIONS.join(", ")})`);
1050
+ }
1051
+ const {
1052
+ image,
1053
+ hostEnv = process.env,
1054
+ egress = false,
1055
+ egressProxy,
1056
+ openJobLog,
1057
+ globalPiDir = null,
1058
+ allowGlobalExtensions = true,
1059
+ packagePaths = [],
1060
+ forwardEnv = [],
1061
+ authFromPi = false,
1062
+ forgeHosts = {},
1063
+ onOutput,
1064
+ backendFloor = {},
1065
+ reap,
1066
+ readInfo = makePodmanInfoReader(),
1067
+ platform = process.platform,
1068
+ euid = process.geteuid?.(),
1069
+ egid = process.getegid?.(),
1070
+ fs = { statSync, readFileSync, readdirSync },
1071
+ home = homedir(),
1072
+ env = process.env,
1073
+ log = () => {},
1074
+ spawnFn,
1075
+ exec,
1076
+ makeRunContainer: makeRunContainerFn = makeRunContainer,
1077
+ makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
1078
+ makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
1079
+ makeStopContainer: makeStopContainerFn = makeStopContainer,
1080
+ } = opts ?? {};
1081
+ if (reap !== undefined && typeof reap !== "function") throw new Error(`backend "${PODMAN_BACKEND}": reap must be a function (makePodmanReaper)`);
1082
+ const spawnSeam = spawnFn ? { spawnFn } : {};
1083
+ const execSeam = exec ? { exec } : spawnFn ? { exec: execViaSpawn(spawnFn) } : {};
1084
+ const info = cachedPodmanInfo(readInfo);
1085
+
1086
+ const runContainer = makeRunContainerFn({
1087
+ image,
1088
+ hostEnv,
1089
+ egress,
1090
+ egressProxy,
1091
+ ...(openJobLog ? { openJobLog } : {}),
1092
+ ...(onOutput ? { onOutput } : {}),
1093
+ globalPiDir,
1094
+ allowGlobalExtensions,
1095
+ packagePaths,
1096
+ forwardEnv,
1097
+ authFromPi,
1098
+ forgeHosts,
1099
+ neverStartedExits: PODMAN_NEVER_STARTED_EXITS,
1100
+ bin: "podman",
1101
+ buildArgs: buildPodmanRunArgs,
1102
+ // Issue #452, gate round 4: the teardown's detach gate uses the `podman info` this venue admitted jobs on, never a
1103
+ // read of its own; before any answered read it falls back to one. A refused teardown is logged with its token.
1104
+ teardownRuntime: () => {
1105
+ const kept = info.peek?.();
1106
+ return kept?.info ? { podman: true, rootless: kept.info.rootless, version: kept.info.version } : undefined;
1107
+ },
1108
+ log,
1109
+ ...spawnSeam,
1110
+ });
1111
+
1112
+ // Logged only when the answer CHANGES, as `local`'s `runtime_observed` and `job_user` are: per job otherwise.
1113
+ let observedSaid = null;
1114
+ let jobUserSaid = null;
1115
+
1116
+ // The floor's podman half, per job and pre-spend: `judgePodmanVenue` over this job's read, with the observation line
1117
+ // logged only when the answer CHANGES. The judgement itself is shared with a sandbox opened on this venue (issue #429).
1118
+ const observationPreflight = async () =>
1119
+ judgePodmanVenue({
1120
+ read: await info(),
1121
+ platform,
1122
+ euid,
1123
+ egid,
1124
+ fs,
1125
+ home,
1126
+ env,
1127
+ backendFloor,
1128
+ onObserved: (observed) => {
1129
+ if (podmanObservationKey(observed) !== observedSaid) {
1130
+ observedSaid = podmanObservationKey(observed);
1131
+ log("podman_observed", { ...observed.observations, changed: true });
1132
+ }
1133
+ },
1134
+ });
1135
+
1136
+ const jobUserPreflight = async (_job, { capabilities = [], observed } = {}) => {
1137
+ const read = observed?.podman ?? (await info());
1138
+ const decision = decidePodmanJobUser({ platform, euid, egid, read });
1139
+ const said = `${decision.mode}|${decision.user}|${decision.cause}|${decision.reason}|${decision.relabel}`;
1140
+ if (said !== jobUserSaid) {
1141
+ jobUserSaid = said;
1142
+ log("job_user", { backend: PODMAN_BACKEND, mode: decision.mode, user: decision.user, cause: decision.cause, reason: decision.reason });
1143
+ }
1144
+ const chosen = resolvePodmanImageUser(decision, { capabilities, euid, egid });
1145
+ // Issue #429: the store this job's container lives in rides beside its user, so a retained run records it and a
1146
+ // sandbox or the retention sweep can tell another store's empty answer from "not open".
1147
+ const store = read?.answered === true ? read.info?.graphRoot : null;
1148
+ return chosen.user && typeof store === "string" ? { ...chosen, store } : chosen;
1149
+ };
1150
+
1151
+ return {
1152
+ name: PODMAN_BACKEND,
1153
+ // The table's own frozen words, never re-typed (`makeLocalBackend`'s reason).
1154
+ declares: BACKENDS[PODMAN_BACKEND].declares,
1155
+ neverStartedExits: PODMAN_NEVER_STARTED_EXITS,
1156
+ containerName: jobContainerName,
1157
+ namePrefix: JOB_NAME_PREFIX,
1158
+ // Bind mounts (`-v`, the shared builder), so `/job`'s read-only is the kernel's, as on `local`.
1159
+ binds: true,
1160
+ runContainer,
1161
+ imagePreflight: makeImagePreflightFn({ image, bin: "podman", ...spawnSeam }),
1162
+ egressPreflight: keeperPreflight(makeEgressPreflightFn({ proxy: egressProxy, armed: egress, bin: "podman", ...spawnSeam }), { armed: egress, proxy: egressProxy, info, spawnFn }),
1163
+ stopContainer: makeStopContainerFn({ bin: "podman", ...execSeam }),
1164
+ reap: reap ?? makePodmanReaper({ log, ...execSeam }),
1165
+ jobUserPreflight,
1166
+ observationPreflight,
1167
+ };
1168
+ }