@edgehero/pi-dispatch 1.10.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +32 -5
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +387 -17
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1395 -268
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
@@ -0,0 +1,1020 @@
1
+ /**
2
+ * `doctor --live`: the backend declarations READ BACK off a real container (issue #278, INT-LIVE-PROBE-CONTRACT).
3
+ *
4
+ * `doctor` prints what the backend table DECLARES, and the conformance harness checks what a probe REPORTS, but
5
+ * nothing in this repository asked a running container whether the words were true. This module does, for the
6
+ * `local` backend on this host, and says exactly how far that reaches.
7
+ *
8
+ * EVERY CONTAINER IS BUILT BY THE SAME BUILDER A JOB IS. `buildDockerRunArgs` with every isolation flag, an EMPTY
9
+ * environment, fixture directories for every conditional mount, and the job user; what each adds through `extraFlags`
10
+ * is `-d` and an entrypoint. A hand-written argv would read back a container nobody runs; these differ from a job's
11
+ * only in what they execute and what they are given, both of which are stated, and in the job's `--cidfile` (issue
12
+ * #345), which only a job's never-started exit reads. There are four kinds: the READING container (`sleep`, then a
13
+ * `cat /proc/self/mountinfo` and the status and write scripts), the PINNING container (an absent image), the EPHEMERAL
14
+ * pair (two runs under one name, issue #344) and, only with the egress policy armed, two PEERS on their own
15
+ * `--internal` job networks behind the proxy (issue #344). The egress canary in `doctor.mjs` (`runEgressCanary`, run on
16
+ * each venue's own runtime since issue #431) is the one reading that does NOT come from this module (its containers need
17
+ * the proxy's allowlist), folded in from its own `readBack`.
18
+ *
19
+ * NO SPAWN OF ITS OWN. Every docker step goes through `run(args, { timeoutMs })`, every filesystem call through `fs`,
20
+ * and every wait through `now`/`delay`, so the whole sequence -- the teardown on every failure path included -- is
21
+ * driven by tests without Docker. The job networks are built by `egress.mjs`'s own `createJobNetworkWith` over that
22
+ * same runner, so a peer's network is a job's network, not a second copy of the sequence.
23
+ *
24
+ * WHAT IT PROVES, AND ONLY THAT. The verdicts are about THE FIXTURE on THIS daemon with PI_JOB_IMAGE: not the
25
+ * operator's own folder, not an image a trigger names, not a remote venue (which reads back through the conformance
26
+ * harness's `readBack` probe instead), and never on a docker CLI not observed local, where bind paths and `.Mounts`
27
+ * would describe another machine.
28
+ */
29
+
30
+ import { READ_BACK_BY_A_LIVE_PROBE } from "./backend-conformance.mjs";
31
+ import { containerSpec } from "./container-spec.mjs";
32
+ import { ISOLATION_FLAGS, buildDockerRunArgs } from "./docker-run.mjs";
33
+ import { DEFAULT_EGRESS_PROXY, EGRESS_PROXY_PORT, createJobNetworkWith, networkEndpoints, networkNameFor, removeNetworkOrSay } from "./egress.mjs";
34
+ import { detachBlockedSentence, makeDetachGate } from "./netns-keeper.mjs";
35
+
36
+ /**
37
+ * The namespace every live-probe object carries: every container name, the peer networks and the fixture directory. OUTSIDE the
38
+ * boot reapers' `pi-job-` filter, the sandbox tooling's `pi-sandbox-` and the loopback `pi-dispatch-valkey`, so no
39
+ * sweep of theirs can touch a probe and no probe name can be mistaken for one of theirs.
40
+ */
41
+ export const LIVE_PREFIX = "pi-dispatch-live-";
42
+
43
+ /** Each docker step's bound. A seam for the tests; the sleep below is derived from it, never typed twice. */
44
+ export const LIVE_STEP_TIMEOUT_MS = 20_000;
45
+
46
+ /**
47
+ * The docker steps that run while a container must still be alive: four for the reading container (inspect, mountinfo
48
+ * exec, status exec, write exec), and three for the peers (the control before, the attempt, the control after; peer2's
49
+ * inspect runs while peer1 is already waiting, well inside one step's bound).
50
+ */
51
+ const STEPS_WHILE_ALIVE = 4;
52
+
53
+ /**
54
+ * How long the probe container sleeps: every step that needs it alive at its full bound, plus a margin for the
55
+ * spawn itself. DERIVED, so a longer step bound cannot leave a container that exits under the last read, and a
56
+ * Ctrl-C mid-probe leaves a STARTED one that removes itself (`--rm`) within that window rather than a literal 300
57
+ * seconds. One interrupted before it started has no `--rm` to run; the next run's sweep removes it.
58
+ */
59
+ export function liveSleepSeconds(stepTimeoutMs = LIVE_STEP_TIMEOUT_MS) {
60
+ return Math.ceil((STEPS_WHILE_ALIVE * stepTimeoutMs) / 1000) + 30;
61
+ }
62
+
63
+ /**
64
+ * How long a container run with `--rm` may take to be gone after it exits before its survival is a finding (issue
65
+ * #344). Measured removal: at most 51 ms over twenty runs on rootful Docker and 32 ms on rootful Podman, polled every
66
+ * 10 ms; 278 ms on Docker Desktop and about 120 ms on both rootful daemons through this module's own 100 ms polls.
67
+ * 10 s is over thirty times the slowest. A container still listed at the deadline FAILS only when it is stopped
68
+ * (`exited`, `dead`, or Podman's `stopped`), and is not read back while it is still running or being removed.
69
+ */
70
+ export const LIVE_REMOVAL_DEADLINE_MS = 10_000;
71
+ const LIVE_REMOVAL_POLL_MS = 100;
72
+
73
+ /** The port a peer listens on. Unprivileged, fixed, and inside a container that has nothing else listening. */
74
+ export const PEER_PORT = 47431;
75
+
76
+ /**
77
+ * The names one run uses. `pid` and a random nonce, so two concurrent `--live` runs never share one. Every CONTAINER
78
+ * kind is a key here, and the stale-container sweep derives its match from these keys, so a kind added here is a
79
+ * kind the sweep removes.
80
+ */
81
+ export function liveNames(pid, nonce) {
82
+ return {
83
+ probe: `${LIVE_PREFIX}probe-${pid}-${nonce}`,
84
+ pin: `${LIVE_PREFIX}pin-${pid}-${nonce}`,
85
+ ephemeral: `${LIVE_PREFIX}ephemeral-${pid}-${nonce}`,
86
+ peer1: `${LIVE_PREFIX}peer1-${pid}-${nonce}`,
87
+ peer2: `${LIVE_PREFIX}peer2-${pid}-${nonce}`,
88
+ fixturePrefix: `${LIVE_PREFIX}${pid}-`,
89
+ };
90
+ }
91
+
92
+ /** The container kinds `liveNames` makes, derived from it rather than typed a second time. */
93
+ export const LIVE_CONTAINER_KINDS = Object.freeze(Object.keys(liveNames(0, "0")).filter((k) => k !== "fixturePrefix"));
94
+
95
+ /** A reference no registry can serve (`.invalid` is reserved, RFC 6761), with the nonce so it is never cached. */
96
+ export function absentImageRef(nonce) {
97
+ return `pi-dispatch-live-probe.invalid/absent:${nonce}`;
98
+ }
99
+
100
+ /** Every conditional mount a job can get, as fixture subdirectories of one root. */
101
+ export function liveFixture(root) {
102
+ return { jobDir: `${root}/job`, workspace: `${root}/workspace`, outboxDir: `${root}/outbox`, sessionDir: `${root}/session`, globalPiDir: `${root}/global` };
103
+ }
104
+
105
+ /**
106
+ * The builder options every live-probe container shares: the fixture mounts, no network (a peer sets its own), no
107
+ * environment at all, and the
108
+ * job user a job on this host would get (issue #341). `user` rides the builder's own field, so the probe is `--user`
109
+ * exactly where a job is; there is still no `-e`, not even HOME, because the probe runs no pi and reads no home.
110
+ */
111
+ //
112
+ // `relabel` (issue #355) is a job's too: where a job's own mounts carry `:Z`, so do the probe's, because a probe mounted
113
+ // the way no job is would read back a container no job gets. The fixture is doctor's own directory, so its workspace
114
+ // is relabelled like a forge job's clone (`workspaceOwned`); its global overlay directory never is, exactly as a job's.
115
+ function probeOptions({ image, name, fixture, user = null, relabel = false }) {
116
+ return { image, name, env: {}, network: "none", user, relabel: relabel === true, workspaceOwned: true, ...fixture };
117
+ }
118
+
119
+ // `buildArgs` (issue #354) is the RUNTIME's job builder, never a second one: each probe below is built by whichever
120
+ // builder that venue's jobs are, so a podman venue's probe carries its `--userns=keep-id` exactly where its jobs do.
121
+ // The default is docker's, so every argv here is byte-for-byte what it was.
122
+
123
+ /** The probe container's argv: the job builder's, detached, with `sleep <derived seconds>` as its whole program. */
124
+ export function liveProbeRunArgs({ image, name, fixture, sleepSeconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
125
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sleep"] }), String(sleepSeconds)];
126
+ }
127
+
128
+ /**
129
+ * The pinning probe's argv: the job builder's, detached, against an image this host does not have. Detached so a
130
+ * container that WAS created prints the ID it is removed by; with `--pull=never` in the builder none should be.
131
+ */
132
+ export function pinningProbeRunArgs({ name, nonce, fixture, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
133
+ return buildArgs({ ...probeOptions({ image: absentImageRef(nonce), name, fixture, user, relabel }), extraFlags: ["-d"] });
134
+ }
135
+
136
+ /**
137
+ * One ephemeral run's argv (issue #344): the job builder's, detached, running EPHEMERAL_SCRIPT with the nonce and the
138
+ * run's number. The same NAME both times, because "a job id run twice" is the question.
139
+ */
140
+ export function ephemeralRunArgs({ image, name, fixture, nonce, run, user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
141
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), extraFlags: ["-d", "--entrypoint", "sh"] }), "-c", EPHEMERAL_SCRIPT, "sh", nonce, String(run)];
142
+ }
143
+
144
+ /**
145
+ * One peer's argv (issue #344): the job builder's, detached, on its OWN job network, running PEER_SCRIPT, which answers
146
+ * every connection with the nonce for `seconds`. Built with `network` set exactly as a job with egress armed is.
147
+ */
148
+ export function peerRunArgs({ image, name, fixture, network, nonce, seconds = liveSleepSeconds(), user = null, relabel = false, buildArgs = buildDockerRunArgs }) {
149
+ return [...buildArgs({ ...probeOptions({ image, name, fixture, user, relabel }), network, extraFlags: ["-d", "--entrypoint", "node"] }), "--eval", PEER_SCRIPT, nonce, String(PEER_PORT), String(seconds)];
150
+ }
151
+
152
+ /**
153
+ * What the probe runs inside, as ONE constant script: no value from the host is interpolated into it. `cat` rather
154
+ * than `grep`, so a job image without grep still answers, and the cgroup v1 paths are read where v2's are absent.
155
+ */
156
+ export const STATUS_SCRIPT = [
157
+ "cat /proc/1/status",
158
+ 'if [ -f /sys/fs/cgroup/cgroup.controllers ]; then echo "cgroup:v2"; echo "pids.max:$(cat /sys/fs/cgroup/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory.max 2>/dev/null)";',
159
+ 'else echo "cgroup:v1"; echo "pids.max:$(cat /sys/fs/cgroup/pids/pids.max 2>/dev/null)"; echo "memory.max:$(cat /sys/fs/cgroup/memory/memory.limit_in_bytes 2>/dev/null)"; fi',
160
+ ].join("\n");
161
+
162
+ /**
163
+ * The write probe, as a job uses its mounts (issue #341): read and traverse the `0700` job dir (`[ -r ]` and `[ -x ]`,
164
+ * shell builtins, the runner's own `access(R_OK|X_OK)`, so no image binary answers for it), then write the
165
+ * workspace, the outbox and the session. The nonce rides argv as `$1`, never the script text; `-w` answers
166
+ * writability without prose, and the FIRST mount that fails is named by its fixed path, never by an error message.
167
+ */
168
+ export const WRITE_SCRIPT = [
169
+ '[ -r /job ] && [ -x /job ] || { echo job-unreadable; exit 0; }',
170
+ 'for d in /workspace /outbox /session; do [ -w "$d" ] || { echo "not-writable $d"; exit 0; }; done',
171
+ 'printf %s "$1" > /workspace/.pi-dispatch-live-probe && printf %s "$1" > /outbox/.pi-dispatch-live-probe && printf %s "$1" > /session/.pi-dispatch-live-probe && echo wrote',
172
+ ].join("\n");
173
+
174
+ /**
175
+ * The ephemeral run's script (issue #344). A marker in the container's OWN `/tmp` says whether this filesystem was
176
+ * used before; the nonce, or the word `residue`, lands in the shared fixture workspace under the run's number, so the
177
+ * host reads both what ran and what it found. Positional values only: nothing from the host is in the text.
178
+ */
179
+ export const EPHEMERAL_SCRIPT = [
180
+ 'if [ -e /tmp/.pi-dispatch-live-ephemeral ]; then printf residue > "/workspace/.pi-dispatch-live-ephemeral-$2"; exit 0; fi',
181
+ // ONE list, so a /tmp this user cannot write leaves NO workspace marker, which is "not read back", rather than a
182
+ // marker that says a residue check ran when it could not have.
183
+ 'printf %s "$1" > /tmp/.pi-dispatch-live-ephemeral && printf %s "$1" > "/workspace/.pi-dispatch-live-ephemeral-$2"',
184
+ ].join("\n");
185
+
186
+ /**
187
+ * A peer's program (issue #344), for `node --eval`: a TCP listener on every address that answers each connection with
188
+ * the nonce, for `seconds`, then exits. `node` because the job image has it and a shell cannot listen portably; its
189
+ * arguments are `process.argv.slice(1)` under `--eval` (measured in the job image).
190
+ */
191
+ export const PEER_SCRIPT = [
192
+ 'const [nonce, port, seconds] = process.argv.slice(1);',
193
+ 'const net = require("node:net");',
194
+ 'const listen = (host) => net.createServer((s) => s.end(nonce + "\\n")).on("error", () => host === "::" && listen("0.0.0.0")).listen(Number(port), host);',
195
+ 'listen("::");',
196
+ 'setTimeout(() => process.exit(0), Number(seconds) * 1000);',
197
+ ].join("\n");
198
+
199
+ /**
200
+ * The connection attempt (issue #344), for `node --eval` inside a peer: `nonce` then `host:port` targets (`[v6]:port`
201
+ * for an IPv6 literal), each tried at once with a 3 s bound, one output line per target. `reached` means the nonce
202
+ * came back, `connected` that something accepted without it, and anything else is the error code in lower case or
203
+ * `timeout`: a word, never a message.
204
+ */
205
+ export const CONNECT_SCRIPT = [
206
+ 'const [nonce, ...targets] = process.argv.slice(1);',
207
+ 'const net = require("node:net");',
208
+ 'const one = (t) => new Promise((resolve) => {',
209
+ ' const m = /^\\[(.*)\\]:(\\d+)$/.exec(t) || /^([^:]+):(\\d+)$/.exec(t);',
210
+ ' if (!m) return resolve(t + " unparsed");',
211
+ ' let data = ""; let done = false; let s = null;',
212
+ ' const finish = (r) => { if (done) return; done = true; if (s) s.destroy(); resolve(t + " " + r); };',
213
+ ' s = net.connect({ host: m[1], port: Number(m[2]) });',
214
+ ' s.setTimeout(3000, () => finish(data.includes(nonce) ? "reached" : s.connecting ? "timeout" : "connected"));',
215
+ ' s.on("connect", () => setTimeout(() => finish(data.includes(nonce) ? "reached" : "connected"), 500));',
216
+ ' s.on("data", (d) => { data += d; if (data.includes(nonce)) finish("reached"); });',
217
+ ' s.on("error", (e) => finish(String((e && e.code) || "error").toLowerCase()));',
218
+ '});',
219
+ 'Promise.all(targets.map(one)).then((lines) => console.log(lines.join("\\n")));',
220
+ ].join("\n");
221
+
222
+ /**
223
+ * CONNECT_SCRIPT's output, as a Map of target to result word. Lines it did not print are simply absent, and so is a
224
+ * line whose word is not a plain lower-case word: the image's `node` prints these, and nothing else reaches a verdict.
225
+ */
226
+ export function parseConnectResults(output) {
227
+ const results = new Map();
228
+ for (const line of String(output ?? "").split(/\r?\n/)) {
229
+ const trimmed = line.trim();
230
+ const at = trimmed.lastIndexOf(" ");
231
+ if (at > 0 && /^[a-z0-9_]{1,40}$/.test(trimmed.slice(at + 1))) results.set(trimmed.slice(0, at), trimmed.slice(at + 1));
232
+ }
233
+ return results;
234
+ }
235
+
236
+ /** The fields of `/proc/1/status` and the cgroup lines the verdicts read, or `null` for each one absent. */
237
+ export function parseStatus(output) {
238
+ // `[ \t]*`, not `\s*`: an empty value (`pids.max:` on a host that could not read it) must not swallow the newline
239
+ // and read the NEXT line as its value.
240
+ const field = (name) => new RegExp(`^${name.replace(/\./g, "\\.")}:[ \\t]*(.*)$`, "m").exec(String(output ?? ""))?.[1]?.trim() ?? null;
241
+ const uid = field("Uid");
242
+ return {
243
+ uids: uid === null ? null : uid.split(/\s+/),
244
+ capBnd: field("CapBnd"),
245
+ noNewPrivs: field("NoNewPrivs"),
246
+ cgroup: field("cgroup"),
247
+ pidsMax: field("pids.max"),
248
+ memoryMax: field("memory.max"),
249
+ };
250
+ }
251
+
252
+ /** The pids bound the builder passes, read off the imported flags rather than restated. */
253
+ export function expectedPidsLimit(flags = ISOLATION_FLAGS) {
254
+ const flag = flags.find((f) => typeof f === "string" && f.startsWith("--pids-limit="));
255
+ return flag ? Number(flag.slice("--pids-limit=".length)) : null;
256
+ }
257
+
258
+ /** The memory bound in bytes, from the spec's own default (`4g`), never a literal. */
259
+ export function expectedMemoryBytes(memory = containerSpec({ image: "i", name: "n", workspace: "/w" }).memory) {
260
+ const m = /^(\d+)([kmg]?)$/i.exec(String(memory));
261
+ if (!m) return null;
262
+ return Number(m[1]) * { "": 1, k: 1024, m: 1024 ** 2, g: 1024 ** 3 }[m[2].toLowerCase()];
263
+ }
264
+
265
+ const verdict = (property, ok, detail, extra = {}) => ({ property, ok, detail, ...extra });
266
+ const notReadBack = (property, why) => verdict(property, false, `not read back: ${why}`, { warn: true });
267
+
268
+ /**
269
+ * ISOLATION, read from PID 1 (docker-init under `--init`, running as the job user). `CapBnd` must be ZERO and
270
+ * `NoNewPrivs` 1: `CapEff` alone is zero for any non-root image with or without `--cap-drop=ALL` (measured: the
271
+ * raw image reads CapEff 0 and CapBnd 00000000a80425fb), so a check on it would pass a container with no boundary.
272
+ * The pids and memory bounds come from the imported flags and the spec's default. On cgroup v2 a literal `max` is a
273
+ * FAILURE (the bound was not applied, as on rootless docker without delegation); a bound that could not be read at
274
+ * all is "not read back", never a pass.
275
+ */
276
+ export function isolationVerdict(status, { pidsLimit = expectedPidsLimit(), memoryBytes = expectedMemoryBytes() } = {}) {
277
+ if (!status || status.capBnd === null || status.noNewPrivs === null) return notReadBack("isolation", "PID 1's status did not show CapBnd and NoNewPrivs");
278
+ const failures = [];
279
+ if (!/^0+$/.test(status.capBnd)) failures.push(`CapBnd ${status.capBnd} (the capability bounding set is not empty)`);
280
+ if (status.noNewPrivs !== "1") failures.push(`NoNewPrivs ${status.noNewPrivs}`);
281
+ const bound = (name, got, want) => {
282
+ if (got === null || got === "") return `${name} not readable`;
283
+ if (got === "max") {
284
+ failures.push(`${name} is max (the bound was not applied)`);
285
+ return null;
286
+ }
287
+ if (Number(got) !== want) failures.push(`${name} ${got}, expected ${want}`);
288
+ return null;
289
+ };
290
+ const unread = [bound("pids.max", status.pidsMax, pidsLimit), bound("memory.max", status.memoryMax, memoryBytes)].filter(Boolean);
291
+ if (failures.length > 0) return verdict("isolation", false, failures.join("; "));
292
+ if (unread.length > 0) return notReadBack("isolation", `${unread.join(" and ")} (cgroup ${status.cgroup ?? "unknown"}); CapBnd 0 and NoNewPrivs 1 did hold`);
293
+ return verdict("isolation", true, `CapBnd 0, NoNewPrivs 1, pids.max ${pidsLimit}, memory.max ${memoryBytes}`);
294
+ }
295
+
296
+ /** NON-ROOT: all four Uid fields (real, effective, saved, filesystem) nonzero, on the process the job would be. */
297
+ export function nonRootVerdict(status) {
298
+ const uids = status?.uids;
299
+ if (!Array.isArray(uids) || uids.length !== 4 || uids.some((u) => !/^\d+$/.test(u))) return notReadBack("nonRoot", "PID 1's status did not show four Uid fields");
300
+ if (uids.some((u) => u === "0")) return verdict("nonRoot", false, `Uid ${uids.join(" ")} (the job would run as root)`);
301
+ return verdict("nonRoot", true, `Uid ${uids.join(" ")}`);
302
+ }
303
+
304
+ /**
305
+ * Mount points `/proc/self/mountinfo` may show inside a job container that no `.Mounts` entry lists, because the runtime
306
+ * makes them for every container (issue #345, measured with the builder's argv on Docker Desktop, rootful Docker 27.5.1
307
+ * and rootful Podman 5.8.2): the root, the three files docker writes per container, each runtime's init binary
308
+ * (`/usr/sbin/docker-init` and `/run/podman-init` measured; `/sbin/docker-init`, where older Docker mounts it, is not),
309
+ * and Podman's `.containerenv`. EXACT paths, never prefixes, so `/run/secrets` beside `/run/.containerenv` is not allowed.
310
+ */
311
+ export const MOUNTINFO_ALLOWED_EXACT = Object.freeze(["/", "/etc/resolv.conf", "/etc/hostname", "/etc/hosts", "/usr/sbin/docker-init", "/sbin/docker-init", "/run/podman-init", "/run/.containerenv"]);
312
+
313
+ /**
314
+ * Kernel filesystems a container always has, allowed with everything beneath them BY PATH SEGMENT (the path itself or
315
+ * `<tree>/...`): the proc masks (`/proc/kcore`, and Podman's `/proc/interrupts`), `/dev/pts`, `/dev/shm`, `/dev/mqueue`,
316
+ * cgroup v1's per-controller mounts and `/sys/firmware`. A string prefix would allow `/devices`; a segment does not.
317
+ */
318
+ export const MOUNTINFO_ALLOWED_TREES = Object.freeze(["/proc", "/dev", "/sys"]);
319
+
320
+ /**
321
+ * Filesystem types a mount under those trees must NOT have: a disk or a network filesystem, or a host share, is a host
322
+ * directory bound there, never one of the runtime's own masks (measured: the masks are `proc`, `tmpfs` and `sysfs` on
323
+ * Docker, and `overlay` for Podman's `/proc/scsi` and `/sys/firmware`, so an allow-list of kernel types would fail
324
+ * Podman). A deny-list, so a host path bound under a tree from an `overlay` or `tmpfs` source still passes (a residual).
325
+ */
326
+ export const MOUNTINFO_HOST_FILESYSTEM = /^(?:ext[234]|xfs|btrfs|zfs|f2fs|vfat|exfat|ntfs3?|nfs4?|cifs|smb3|9p|virtiofs|fakeowner|fuseblk|fuse(?:\..+)?)$/;
327
+
328
+ /**
329
+ * The mount points in a `/proc/self/mountinfo` body: field five of every line that has the `-` separator, with the
330
+ * kernel's octal escapes (`\040` for a space) decoded. Lines of any other shape are skipped; an empty answer is `[]`.
331
+ */
332
+ export function mountPointsOf(mountinfo) {
333
+ return mountEntriesOf(mountinfo).map((entry) => entry.point);
334
+ }
335
+
336
+ /** `mountPointsOf` with each point's filesystem type (the field after the `-`), as `{ point, fstype }`. */
337
+ export function mountEntriesOf(mountinfo) {
338
+ const entries = [];
339
+ for (const line of String(mountinfo ?? "").split(/\r?\n/)) {
340
+ const fields = line.trim().split(" ");
341
+ const separator = fields.indexOf("-", 6);
342
+ if (fields.length < 10 || separator < 0) continue;
343
+ const point = fields[4].replace(/\\([0-7]{3})/g, (_, octal) => String.fromCharCode(parseInt(octal, 8)));
344
+ if (point.startsWith("/")) entries.push({ point, fstype: fields[separator + 1] ?? "" });
345
+ }
346
+ return entries;
347
+ }
348
+
349
+ /**
350
+ * THE MOUNT SET, from `docker inspect .Mounts`, keyed by destination and read-write flag. A COUNT would pass a
351
+ * container whose `/job` became writable while another mount went missing; so each declared destination must be
352
+ * present with its own RW, nothing else may be mounted, and three sources are refused outright wherever they
353
+ * appear: the docker socket, the operator's home directory or any ancestor of it, and the shared session store.
354
+ *
355
+ * `mountinfo` is what `/proc/self/mountinfo` said inside the container (issue #345): `undefined` when the caller did not
356
+ * read it, `null` when the read failed. `.Mounts` is the daemon's own list, and a runtime can mount things into every
357
+ * container that it never lists there (rootful Podman's `/run/secrets`, measured), so a mount point inside the container
358
+ * that is neither declared nor on the runtime's own short list fails `runtime-mount`.
359
+ *
360
+ * `bin` (issue #354) names the CLI whose `inspect` was read, in the words only: a podman venue's verdict saying "docker
361
+ * inspect" would send an operator to a daemon that never saw the container. Docker's words are the default, so every
362
+ * local verdict is byte-for-byte what it was.
363
+ */
364
+ export function mountSetVerdict(inspectOutput, { expected, home = null, sessionsDir = null, mountinfo = undefined, bin = "docker" }) {
365
+ let mounts;
366
+ try {
367
+ mounts = JSON.parse(String(inspectOutput ?? "").trim());
368
+ } catch {
369
+ return notReadBack("mountSet", `${bin} inspect did not return the mounts as JSON`);
370
+ }
371
+ if (!Array.isArray(mounts)) return notReadBack("mountSet", `${bin} inspect did not return a mount list`);
372
+ const failures = [];
373
+ const byDestination = new Map(mounts.map((m) => [m?.Destination, m]));
374
+ for (const want of expected) {
375
+ const got = byDestination.get(want.container);
376
+ if (!got) failures.push(`${want.container} is missing`);
377
+ else if (got.RW !== !want.readOnly) failures.push(`${want.container} is ${got.RW ? "writable" : "read-only"}, declared ${want.readOnly ? "read-only" : "writable"}`);
378
+ }
379
+ const declared = new Set(expected.map((m) => m.container));
380
+ for (const m of mounts) {
381
+ if (!declared.has(m?.Destination)) failures.push(`${m?.Destination} is mounted and nothing declares it`);
382
+ const source = String(m?.Source ?? "");
383
+ const under = (dir) => source !== "" && (source === dir || dir.startsWith(`${source.replace(/\/+$/, "")}/`));
384
+ if (/docker\.sock$/.test(source) || /docker\.sock$/.test(String(m?.Destination ?? ""))) failures.push(`the docker socket is mounted (${m?.Destination})`);
385
+ // Issue #354: Podman's API socket is the same key to the same house (rootless, the worker account's whole store and
386
+ // every container it runs), so it is refused on sight wherever it appears, on either venue.
387
+ else if (/podman\.sock$/.test(source) || /podman\.sock$/.test(String(m?.Destination ?? ""))) failures.push(`the podman socket is mounted (${m?.Destination})`);
388
+ if (home && (source === "/" || under(home))) failures.push(`${m?.Destination} mounts the home directory or an ancestor of it`);
389
+ if (sessionsDir && source !== "" && (source === sessionsDir || source.startsWith(`${sessionsDir.replace(/\/+$/, "")}/`) || under(sessionsDir))) failures.push(`${m?.Destination} mounts the shared session store or an ancestor of it`);
390
+ }
391
+ if (failures.length > 0) return verdict("mountSet", false, failures.join("; "));
392
+ const declaredList = `${expected.map((m) => `${m.container}${m.readOnly ? ":ro" : ""}`).join(", ")} and nothing else`;
393
+ if (mountinfo === undefined) return verdict("mountSet", true, declaredList);
394
+ const entries = mountinfo === null ? [] : mountEntriesOf(mountinfo);
395
+ if (entries.length === 0) return notReadBack("mountSet", `${bin} inspect lists ${declaredList}, but /proc/self/mountinfo could not be read inside the container, so a mount the runtime adds without listing it was not checked`);
396
+ const underTree = (point) => MOUNTINFO_ALLOWED_TREES.some((tree) => point === tree || point.startsWith(`${tree}/`));
397
+ const allowed = ({ point, fstype }) => declared.has(point) || MOUNTINFO_ALLOWED_EXACT.includes(point) || (underTree(point) && !MOUNTINFO_HOST_FILESYSTEM.test(fstype));
398
+ // Printable only: a mount point is the container's text, and a verdict detail reaches the operator's terminal.
399
+ const extra = [...new Set(entries.filter((e) => !allowed(e)).map((e) => e.point))].map((p) => p.replace(/[^\x20-\x7e]/g, "?"));
400
+ if (extra.length > 0) {
401
+ return verdict("mountSet", false, `${extra.join(", ")} ${extra.length === 1 ? "is" : "are"} mounted inside the container, and neither ${bin} inspect nor the job lists ${extra.length === 1 ? "it" : "them"}`, { cause: "runtime-mount" });
402
+ }
403
+ return verdict("mountSet", true, `${declaredList}, in ${bin} inspect and in /proc/self/mountinfo`);
404
+ }
405
+
406
+ /**
407
+ * LOCAL FOLDERS: a nonce written inside `/workspace` and read on the host, which is what "edited in place" means.
408
+ * A folder the job user cannot write, and a write that does not show up on the host, are different failures with
409
+ * different fixes, so they are told apart -- by `[ -w ]` inside, never by an error message.
410
+ *
411
+ * Issue #341 widens it to every mount a job uses, with the fixture in a job's real modes: a `0700` job dir the job
412
+ * user cannot list (`job-unreadable`), an outbox or session it cannot write (`mount-not-writable`), and a nonce the
413
+ * host sees owned by a uid other than this shell's (`not-yours`), which is a worker that could not clean up after
414
+ * its own job. The owner check is skipped where there is no uid to compare (`euid` undefined, as on Windows) or the
415
+ * host could not stat the file.
416
+ */
417
+ export function localFoldersVerdict({ code, stdout, hostRead, nonce, hostOwner = null, euid = undefined, unseen = "/workspace" }) {
418
+ if (code !== 0) return notReadBack("localFolders", "the write probe did not run in the container");
419
+ const said = String(stdout ?? "").trim();
420
+ if (said === "job-unreadable") return verdict("localFolders", false, "the job user cannot list a 0700 job directory this shell created, so no job here can read its own inputs", { cause: "job-unreadable" });
421
+ if (said === "not-writable /workspace") return verdict("localFolders", false, "the job user cannot write a bind-mounted host folder, so a local-folder job cannot edit its folder in place", { cause: "not-writable" });
422
+ if (said === "not-writable /outbox" || said === "not-writable /session") return verdict("localFolders", false, `the job user cannot write ${said.slice("not-writable ".length)}, a mount every job of that kind writes`, { cause: "mount-not-writable" });
423
+ if (said !== "wrote") return notReadBack("localFolders", "the write probe gave no answer");
424
+ if (hostRead !== nonce) return verdict("localFolders", false, `a file written inside ${unseen} is not visible on the host, so that mount is not the folder a job edits in place`, { cause: "not-visible" });
425
+ if (typeof euid === "number" && typeof hostOwner === "number" && hostOwner !== euid) {
426
+ return verdict("localFolders", false, `a file the job wrote is owned by uid ${hostOwner} on the host, not by this shell's uid ${euid}, so the worker could not remove what a job leaves`, { cause: "not-yours" });
427
+ }
428
+ return verdict("localFolders", true, `every mount a job uses was used as the job user, and the files written inside /workspace, /outbox and /session were read back on the host${typeof euid === "number" && typeof hostOwner === "number" ? `, owned by this shell's uid ${euid}` : ""}`);
429
+ }
430
+
431
+ /**
432
+ * IMAGE PINNING: the builder's argv against an image this host does not have. It holds only if the run was
433
+ * refused (nonzero), the daemon said `No such image`, docker did NOT try a pull (`Unable to find image` is the
434
+ * CLI's announcement of one), and the image is still absent afterwards. A refusal in words this check does not
435
+ * know is "not read back" rather than a pass.
436
+ *
437
+ * Issue #354: rootless Podman 5.8.1 refuses the same `--pull=never` run with `Error: <ref>: image not known` (exit
438
+ * 125, measured), so those words are the absent-image refusal too; without them every podman venue would abstain
439
+ * here forever. Podman announces a pull as `Trying to pull <ref>...`, which fails like docker's announcement does:
440
+ * a pull attempted and then refused is still not pinning. Any other refusal still abstains. `bin` names the CLI in the
441
+ * details only (docker's by default, so local's are unchanged); the words recognised are both runtimes' either way.
442
+ */
443
+ export function imagePinningVerdict({ code, output, stillAbsent, bin = "docker" }) {
444
+ const text = String(output ?? "");
445
+ if (code === null || code === undefined) return notReadBack("imagePinning", "the pinning probe did not run");
446
+ if (code === 0) return verdict("imagePinning", false, "a container started from an image this host does not have");
447
+ if (/Unable to find image|Trying to pull/i.test(text)) return verdict("imagePinning", false, `${bin} tried to pull an image this host does not have`);
448
+ if (stillAbsent === false) return verdict("imagePinning", false, "an image this host did not have is present after the run");
449
+ if (stillAbsent !== true) return notReadBack("imagePinning", "whether the image is still absent could not be read");
450
+ if (!/No such image|image not known/i.test(text)) return notReadBack("imagePinning", `${bin} refused the run with a message this check does not recognise`);
451
+ return verdict("imagePinning", true, "an absent image was refused without a pull");
452
+ }
453
+
454
+ /**
455
+ * EGRESS, folded in from `doctor.mjs`'s canary: two containers on a job-shaped network behind the proxy, one that
456
+ * must reach the provider and one that must not reach an unlisted host. `results` is `[{ want, reached }]` from the
457
+ * canary's `readBack`; anything short of both readings -- the policy off, the proxy down, the canary skipped -- is
458
+ * "not read back", never a pass.
459
+ *
460
+ * Each venue hands in its OWN canary's readings: docker's from doctor's egress lines, the podman venue's from the canary
461
+ * its `--live` runs under Podman (issue #431), so "see the egress lines above" names lines about the proxy that venue's
462
+ * jobs use. The `unread` override issue #354 added for a podman venue with no canary is gone with that gap.
463
+ */
464
+ export function egressVerdict({ armed, results, keeperBlocked = null }) {
465
+ if (armed === null) return notReadBack("egress", "PI_EGRESS could not be read (see the .env check above)");
466
+ if (armed !== true) return notReadBack("egress", "PI_EGRESS is off, so there is no policy to read back");
467
+ // Issue #458: the podman venue on Podman 4.x without its keeper runs no canary, since its teardown would break the proxy.
468
+ if (keeperBlocked) return notReadBack("egress", `the egress canary was not run, because ${keeperBlocked} (see the keeper line above)`);
469
+ if (!Array.isArray(results) || results.length < 2) return notReadBack("egress", "the egress canary did not run both probes (see the egress lines above)");
470
+ // A WRONG reading fails first, whatever else is missing: an unlisted host that was reached is a finding even when
471
+ // the provider probe did not run, and reporting it as merely unread would pass doctor over it.
472
+ const wrong = results.filter((r) => typeof r.reached === "boolean" && r.reached !== r.want);
473
+ if (wrong.length > 0) return verdict("egress", false, wrong.map((r) => (r.want ? "the provider was not reached" : "an unlisted host was reached")).join("; "));
474
+ if (results.some((r) => typeof r.reached !== "boolean")) return notReadBack("egress", "an egress probe did not run to an answer (see the egress lines above)");
475
+ return verdict("egress", true, "the provider was reached and an unlisted host was not");
476
+ }
477
+
478
+ /**
479
+ * EPHEMERAL (issue #344): two runs under ONE name, each detached with `--rm`, each waited on until `docker ps -a` no
480
+ * longer lists it. `first`/`second` are `{ started, id, removal: { state, ms } }` (`state` one of `gone`, `exited`
481
+ * (still listed, stopped, past the deadline), `present` (still listed, not stopped), `unanswered`), plus `nameHeld` on
482
+ * the second when its run was refused while a container held the name; `markers` are what each run left in the
483
+ * workspace. A finding needs positive evidence; anything that did not run to an answer is not read back. `bin` names the
484
+ * CLI whose `ps` was polled, in the words only (issue #354).
485
+ */
486
+ export function ephemeralVerdict({ first, second, markers = {}, nonce, bin = "docker" }) {
487
+ const deadline = `${LIVE_REMOVAL_DEADLINE_MS / 1000} s`;
488
+ const unfinished = (which, run) => {
489
+ if (run?.removal?.state === "present") return notReadBack("ephemeral", `the ${which} run was still listed and not stopped after ${deadline}, so its removal was not seen`);
490
+ if (run?.removal?.state === "unanswered") return notReadBack("ephemeral", `${bin} ps did not answer while the ${which} run was being waited on`);
491
+ return null;
492
+ };
493
+ if (!first?.started) return notReadBack("ephemeral", "the first ephemeral run did not start");
494
+ if (first.removal?.state === "exited") return verdict("ephemeral", false, `the first run's container was still listed, stopped, ${deadline} after it started: a job's container outlives the job`, { cause: "survived" });
495
+ const firstUnfinished = unfinished("first", first);
496
+ if (firstUnfinished) return firstUnfinished;
497
+ if (second?.nameHeld) return verdict("ephemeral", false, "a second run under the same name was refused because a container still held the name after the first was gone: a job id cannot run twice", { cause: "name-held" });
498
+ if (second?.started && second.id && first.id && second.id === first.id) return verdict("ephemeral", false, "the second run was given the first run's container", { cause: "reused" });
499
+ if (markers.second === "residue") return verdict("ephemeral", false, "the second run found the first run's /tmp marker: a container's filesystem outlived it", { cause: "residue" });
500
+ if (second?.removal?.state === "exited") return verdict("ephemeral", false, `the second run's container was still listed, stopped, ${deadline} after it started: a job's container outlives the job`, { cause: "survived" });
501
+ const secondUnfinished = unfinished("second", second);
502
+ if (secondUnfinished) return secondUnfinished;
503
+ if (!second?.started) return notReadBack("ephemeral", "the second ephemeral run did not start");
504
+ if (markers.first !== nonce || markers.second !== nonce) return notReadBack("ephemeral", "a run left no marker in the workspace, so whether it ran its script is not known");
505
+ return verdict("ephemeral", true, `two runs under one name each removed themselves (in ${first.removal.ms} ms and ${second.removal.ms} ms), and the second found nothing of the first`);
506
+ }
507
+
508
+ /**
509
+ * JOB-TO-JOB ISOLATION (issue #344): two peers, each on its own `--internal` job network behind the proxy, the way two
510
+ * concurrent jobs get them. `fromPeer1` is peer1's results for the proxy (`proxyTarget`) and every name and address of
511
+ * peer2 (`peerTargets`, `peerAddresses` the address subset); `controlBefore` and `controlAfter` are peer2's results
512
+ * against its own addresses, run BEFORE and AFTER peer1's attempt. A peer reached is a FAILURE, and wins over every
513
+ * missing reading. A block counts only when: at least one ADDRESS was tried (a name alone proves DNS scoping, not
514
+ * routing); peer1 reached the proxy (a network that reaches nothing blocks everything); and peer2 answered itself
515
+ * both before the attempt (else a refused connection may be a listener not yet up, which is the opposite of a block)
516
+ * and after it (so it stayed up throughout).
517
+ *
518
+ * `bin` (issue #354) names the runtime in the words only. Unarmed, a podman job is on podman's default network, which
519
+ * rootless is not a bridge at all (pasta), so the other runtime says "default network" rather than borrowing docker's
520
+ * noun; docker's sentence is unchanged.
521
+ */
522
+ export function jobToJobIsolationVerdict({ bin = "docker", armed, proxyRunning, keeperBlocked = null, networksCreated, peersStarted, proxyTarget, peerTargets = [], peerAddresses = [], fromPeer1 = new Map(), controlBefore = new Map(), controlAfter = new Map() }) {
523
+ if (armed === null) return notReadBack("jobToJobIsolation", "PI_EGRESS could not be read (see the .env check above)");
524
+ if (armed !== true) return notReadBack("jobToJobIsolation", `PI_EGRESS is off, so jobs share ${bin === "docker" ? "docker's default bridge" : `${bin}'s default network`} by design and there is no per-job network to read back`);
525
+ if (proxyRunning === false) return notReadBack("jobToJobIsolation", "the egress proxy is not running, so no job-shaped network could be built");
526
+ if (keeperBlocked) return notReadBack("jobToJobIsolation", `no peer networks were built, because ${keeperBlocked} (see the keeper line above)`);
527
+ if (networksCreated !== true) return notReadBack("jobToJobIsolation", "the peer networks could not be created");
528
+ if (peersStarted !== true) return notReadBack("jobToJobIsolation", "a peer container did not start");
529
+ const reached = peerTargets.filter((t) => fromPeer1.get(t) === "reached");
530
+ if (reached.length > 0) return verdict("jobToJobIsolation", false, `one job reached another across their own networks, at ${reached.join(", ")}`, { cause: "reached" });
531
+ if (peerAddresses.length === 0) return notReadBack("jobToJobIsolation", `${bin} inspect gave peer2 no address on its network, so only names could be tried and a name proves nothing about routing`);
532
+ if (peerTargets.some((t) => !fromPeer1.has(t)) || !fromPeer1.has(proxyTarget)) return notReadBack("jobToJobIsolation", "the connection attempt did not answer for every target");
533
+ if (!["connected", "reached"].includes(fromPeer1.get(proxyTarget))) return notReadBack("jobToJobIsolation", `peer1 could not reach the proxy (${fromPeer1.get(proxyTarget)}), so an unreached peer proves nothing`);
534
+ // Keyed by the addresses peer1 was given, never by whatever the control printed: a control that answered one of
535
+ // two addresses says nothing about the one peer1 was refused on.
536
+ const answered = (control) => peerAddresses.every((address) => control.get(address) === "reached");
537
+ if (!answered(controlBefore)) return notReadBack("jobToJobIsolation", "peer2 did not answer its own addresses before the attempt, so a refused connection may be a listener not yet up");
538
+ if (!answered(controlAfter)) return notReadBack("jobToJobIsolation", "peer2 did not answer its own addresses after the attempt, so its listener was not up throughout");
539
+ const accepted = peerTargets.filter((t) => fromPeer1.get(t) === "connected");
540
+ if (accepted.length > 0) return notReadBack("jobToJobIsolation", `something accepted a connection at ${accepted.join(", ")} without the peer's nonce, which is neither reached nor refused`);
541
+ return verdict("jobToJobIsolation", true, `peer1 reached the proxy and none of peer2's ${peerTargets.length} name(s) and address(es) (${peerTargets.map((t) => `${t} ${fromPeer1.get(t)}`).join(", ")}); peer2 answered itself before and after`);
542
+ }
543
+
544
+ /**
545
+ * Wait for a container to be gone (issue #344): `docker ps -a --no-trunc --filter id=` until it lists nothing, or the
546
+ * deadline passes. `docker wait --condition=removed` would be one call, and the docker CLI has no such flag (checked
547
+ * on 27.4.0). Returns `{ state, ms }` with `state` `gone`, `exited`, `present` or `unanswered`.
548
+ */
549
+ export async function awaitRemoved({ step, id, now, delay, deadlineMs = LIVE_REMOVAL_DEADLINE_MS, pollMs = LIVE_REMOVAL_POLL_MS }) {
550
+ const start = now();
551
+ for (;;) {
552
+ const listed = await step(["ps", "-a", "--no-trunc", "--filter", `id=${id}`, "--format", "{{.ID}} {{.State}}"]);
553
+ if (listed?.code !== 0) return { state: "unanswered", ms: now() - start };
554
+ const line = String(listed.stdout ?? "").split(/\r?\n/).map((l) => l.trim()).find((l) => l.startsWith(id));
555
+ if (!line) return { state: "gone", ms: now() - start };
556
+ // Stopped, in either daemon's words: Docker's `exited` and `dead`, and Podman's `stopped` for a container that
557
+ // exited and was never cleaned up (measured on rootful Podman 5.8.2 as the state before `exited`).
558
+ if (now() - start >= deadlineMs) return { state: /^(exited|dead|stopped)$/i.test(line.split(/\s+/)[1] ?? "") ? "exited" : "present", ms: now() - start };
559
+ await delay(pollMs);
560
+ }
561
+ }
562
+
563
+ /**
564
+ * The first line the runtime printed on a failed run, stderr first, as one printable line of at most 300 characters,
565
+ * or null. Control bytes become spaces, so nothing a runtime prints can move an operator's cursor.
566
+ */
567
+ export function runtimeErrorLine(result) {
568
+ for (const stream of [result?.stderr, result?.stdout]) {
569
+ const line = String(stream ?? "")
570
+ .split(/\r?\n/)
571
+ .map((l) => l.replace(/[\u0000-\u001f\u007f]/g, " ").trim())
572
+ .find((l) => l.length > 0);
573
+ if (line) return line.length > 300 ? `${line.slice(0, 300)}...` : line;
574
+ }
575
+ return null;
576
+ }
577
+
578
+ /**
579
+ * The names and addresses a peer answers at on one network, from `docker inspect --format={{json .NetworkSettings.Networks}}`:
580
+ * `{ targets, addresses }`, where `addresses` are the `IPAddress` and `GlobalIPv6Address` targets only, taken from
581
+ * those FIELDS rather than recognised by shape (a short container ID that starts with a digit is a name).
582
+ */
583
+ export function peerTargetsOf(inspectOutput, { network, name }) {
584
+ let networks;
585
+ try {
586
+ networks = JSON.parse(String(inspectOutput ?? "").trim());
587
+ } catch {
588
+ return { targets: [], addresses: [] };
589
+ }
590
+ const on = networks?.[network];
591
+ if (!on || typeof on !== "object") return { targets: [], addresses: [] };
592
+ const hosts = new Set([name]);
593
+ for (const n of [...(Array.isArray(on.DNSNames) ? on.DNSNames : []), ...(Array.isArray(on.Aliases) ? on.Aliases : [])]) if (typeof n === "string" && n) hosts.add(n);
594
+ const addresses = [];
595
+ if (typeof on.IPAddress === "string" && on.IPAddress) addresses.push(`${on.IPAddress}:${PEER_PORT}`);
596
+ if (typeof on.GlobalIPv6Address === "string" && on.GlobalIPv6Address) addresses.push(`[${on.GlobalIPv6Address}]:${PEER_PORT}`);
597
+ return { targets: [...[...hosts].map((h) => `${h}:${PEER_PORT}`), ...addresses], addresses };
598
+ }
599
+
600
+ /**
601
+ * The whole sequence. Returns `{ ran, reason, verdicts, notes, swept, ranAs }`: `ran` false with a `reason` when nothing
602
+ * was read back; the verdicts in READ_BACK_BY_A_LIVE_PROBE's order otherwise; `notes` for a teardown that failed, on
603
+ * either path; `swept` for what an interrupted earlier run left and this one removed, on either path too.
604
+ *
605
+ * Order: precondition, a FRESH endpoint read, the announcement, the sweeps, the fixture, then one PHASE per kind of
606
+ * container (reading, pinning, ephemeral, peers), each removing what it started before the next begins. ONE ownership
607
+ * list holds every container and network this run made, and a `finally` removes whatever is left in it: containers
608
+ * first, BY THE ID `run -d` printed (by the pid-and-nonce name only when no ID came back), then networks, then the
609
+ * fixture. A container already SEEN gone is never removed again.
610
+ */
611
+ export async function runLiveProbes({
612
+ image,
613
+ endpoint,
614
+ resolveEndpoint = null,
615
+ dockerReachable,
616
+ imagePresent,
617
+ jobsDir,
618
+ home = null,
619
+ sessionsDir = null,
620
+ egress,
621
+ pid,
622
+ nonce,
623
+ run,
624
+ fs,
625
+ isAlive,
626
+ announce = () => {},
627
+ stepTimeoutMs = LIVE_STEP_TIMEOUT_MS,
628
+ // The bound on each container START alone, the step bound unless a venue needs more: rootless Podman's first keep-id
629
+ // run of an image copies its layers to the mapped ids, which took 27 s for the job image (measured; later runs 0.15 s).
630
+ startTimeoutMs = stepTimeoutMs,
631
+ user = null,
632
+ relabel = false,
633
+ euid = undefined,
634
+ now = () => Date.now(),
635
+ delay = (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
636
+ removalDeadlineMs = LIVE_REMOVAL_DEADLINE_MS,
637
+ // Issue #354, the runtime seams. `buildArgs` is the venue's job builder (see the four builders above); `bin` is the
638
+ // CLI `run` spawns, used here only to NAME the commands a note tells the operator to run, since `run` is the
639
+ // caller's and already knows its binary. `isLocal` is the gate on both endpoint reads: docker's is the observed
640
+ // endpoint, a podman venue's is `serviceIsRemote === false`, which has no `.local` to read. The gate still decides
641
+ // on the value it is handed and must answer exactly `true`; anything else runs nothing.
642
+ buildArgs = buildDockerRunArgs,
643
+ bin = "docker",
644
+ isLocal = (observed) => observed?.local === true,
645
+ // Issue #452, gate round 3: the runtime as this pass already read it (`{ podman, rootless, version } | null`), for the
646
+ // detach gate, so doctor asks the daemon once per run. Absent, the gate reads it itself.
647
+ readRuntime = null,
648
+ }) {
649
+ const notRun = (reason) => ({ ran: false, reason, verdicts: [], notes: [], swept: [] });
650
+ const notLocal = notRun(`this shell's ${bin} CLI is not observed to point at this host, so bind paths and .Mounts would describe another machine; nothing was run`);
651
+ if (isLocal(endpoint) !== true) return notLocal;
652
+ if (dockerReachable !== true) return notRun(`the ${bin === "docker" ? "Docker" : bin} daemon did not answer, so no container was run`);
653
+ if (imagePresent !== true) return notRun(`the job image ${image} is not present, so no container was run`);
654
+ // ASKED AGAIN, immediately before the first command. The answer above came from the start of doctor's run, and
655
+ // prompts and slow checks sit between the two; a `docker context use` in that window would otherwise send every
656
+ // read below to another machine and report it as this one (the per-job re-read in `start.mjs`, for the same reason).
657
+ if (typeof resolveEndpoint === "function" && isLocal(await resolveEndpoint()) !== true) return notLocal;
658
+
659
+ const names = liveNames(pid, nonce);
660
+ const notes = [];
661
+ // `opts` only from the detach gate's runtime read, which asks with the facts readers' own bound (issue #452).
662
+ const step = (args, opts) => run(args, { timeoutMs: opts?.timeoutMs ?? stepTimeoutMs });
663
+ const proxy = typeof egress?.proxy === "string" && egress.proxy ? egress.proxy : DEFAULT_EGRESS_PROXY;
664
+ // The peers run only where a job would get its own network: the policy armed and its proxy not known to be down.
665
+ // Issue #458: and not while the podman venue's keeper is missing on Podman 4.x, where the peers' teardown (detaching
666
+ // the proxy from their networks) is exactly what cuts the proxy's route out.
667
+ // Issue #452, gate round 3: the same question on EVERY venue, through the one detach gate every detach below goes
668
+ // through anyway: `local` can be a rootless Podman 4.x too (reached through `podman-docker`, or its Docker API), where
669
+ // no keeper read of doctor's ran. One read for this pass, shared by the peers' teardown and the stale sweep.
670
+ const gate = makeDetachGate(step, { bin, ...(typeof readRuntime === "function" ? { readRuntime } : {}) });
671
+ let keeperBlocked = egress?.keeperBlocked || null;
672
+ if (!keeperBlocked && egress?.armed === true && egress?.proxyRunning !== false) {
673
+ const why = await gate({ running: true });
674
+ if (why) keeperBlocked = detachBlockedSentence(why, bin);
675
+ }
676
+ const peersWanted = egress?.armed === true && egress?.proxyRunning !== false && !keeperBlocked;
677
+ const networkOf = { peer1: networkNameFor(names.peer1), peer2: networkNameFor(names.peer2) };
678
+ // SHOWN BEFORE IT HAPPENS (REQ-DEPLOYMENT-BOOTSTRAP): every host mutation this makes is named, with where it lives
679
+ // and that it goes away, before a sweep or the fixture touches anything.
680
+ announce(
681
+ `starting ${names.probe}, ${names.pin} and ${names.ephemeral} (twice) from ${image} (no environment, ${user ? `as the job user ${user}` : "as the image's own user"})` +
682
+ (peersWanted ? `, and ${names.peer1} and ${names.peer2} on their own --internal networks ${networkOf.peer1} and ${networkOf.peer2}, with ${proxy} attached to both` : "") +
683
+ `, with a fixture under ${jobsDir}; all of them are removed when the read-back ends, as is anything an interrupted earlier run left`,
684
+ );
685
+ const swept = [...sweepStaleFixtures({ jobsDir, pid, fs, isAlive }), ...(await sweepStaleContainers({ step, pid, isAlive })), ...(await sweepStaleNetworks({ step, pid, isAlive, notes, bin, keeperBlocked, gate }))];
686
+
687
+ const owned = [];
688
+ const networks = [];
689
+ const start = async (what, name, args) => {
690
+ const entry = { what, name, id: null, done: false };
691
+ owned.push(entry);
692
+ const result = await run(args, { timeoutMs: startTimeoutMs });
693
+ // The ID is taken whenever one was printed, whatever the exit: a CLI killed or timed out after the create can
694
+ // leave a container that never started, which `--rm` does not remove.
695
+ entry.id = containerIdOf(result);
696
+ return { result, entry };
697
+ };
698
+ const release = async (entry) => {
699
+ if (entry.done) return;
700
+ entry.done = true;
701
+ if (entry.id !== null) {
702
+ const removed = await step(["rm", "-f", entry.id]);
703
+ if (removed?.code !== 0) notes.push(`the ${entry.what} ${entry.id.slice(0, 12)} could not be removed: ${bin} rm -f ${entry.id}`);
704
+ } else {
705
+ // No ID came back. The name carries this run's pid and nonce, so no other run's container can answer to it;
706
+ // usually there is nothing by that name and docker says so, which is not worth a line.
707
+ await step(["rm", "-f", entry.name]);
708
+ }
709
+ };
710
+ const dropNetwork = async (entry) => {
711
+ if (entry.done) return;
712
+ entry.done = true;
713
+ // The wording rule this used to carry inline now lives in `egress.mjs` beside the proxy name and the
714
+ // network suffix, because three more sweeps needed the same rule and a second copy is the one nobody
715
+ // updates when a third runtime words it differently. The call sequence is byte-for-byte what it was:
716
+ // disconnect the proxy, `network rm` without `-f`, and on a failure one `network inspect` whose answer
717
+ // decides between silence and a note.
718
+ // `bin` only spells the command a note tells the operator to type (issue #354); `step` already knows its binary.
719
+ const outcome = await removeNetworkOrSay(step, { network: entry.name, detach: [proxy], bin, gate });
720
+ if (!outcome.removed) notes.push(`the network ${entry.name} could not be removed: ${outcome.command}`);
721
+ };
722
+ const makeFixture = (base) => {
723
+ const fixture = liveFixture(base);
724
+ // A JOB'S MODES, not friendlier ones (issue #341). A job's dir is a `0700` mkdtemp and its session dir `0700`;
725
+ // this fixture used to chmod every directory `0755`, which is how a probe passed on hosts where every job
726
+ // failed. The rest take the default mode, as a job's clone and outbox do. Still created EMPTY.
727
+ for (const [key, dir] of Object.entries(fixture)) {
728
+ fs.mkdirSync(dir, { recursive: true, ...(key === "jobDir" || key === "sessionDir" ? { mode: 0o700 } : {}) });
729
+ }
730
+ return fixture;
731
+ };
732
+
733
+ let root = null;
734
+ try {
735
+ // A jobs dir this shell cannot write (another user's, `PI_JOBS_DIR=""`, a path that is a file) is a reason not
736
+ // to run, said as one, never an exception that takes the rest of doctor's output with it. `root` is set before
737
+ // the realpath so a directory mkdtemp did make is still removed below when a later call fails.
738
+ let fixture;
739
+ try {
740
+ // 0700 for what this creates (issue #464): the worker's own `ensureJobsDir` makes the per-account root so, and a
741
+ // probe run first must not leave it wider.
742
+ fs.mkdirSync(jobsDir, { recursive: true, mode: 0o700 });
743
+ root = fs.mkdtempSync(`${jobsDir}/${names.fixturePrefix}`);
744
+ root = fs.realpathSync(root);
745
+ fixture = makeFixture(root);
746
+ } catch (err) {
747
+ // The SAME `notes` array the `finally` below pushes into, so a removal that fails there is still reported.
748
+ return { ran: false, reason: `the fixture could not be created under ${JSON.stringify(jobsDir)} (${err?.code ?? "error"}), so no container was run`, verdicts: [], notes, swept };
749
+ }
750
+
751
+ // --- the reading container: mounts, status, writes ---
752
+ const reading = await start("probe container", names.probe, liveProbeRunArgs({ image, name: names.probe, fixture, sleepSeconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
753
+ const probeId = reading.entry.id;
754
+ if (reading.result?.code !== 0 || probeId === null) {
755
+ // The runtime's own words, when it printed any (issue #453, gate round 1): "did not start" alone left the operator
756
+ // to reproduce the run to learn why (a missing controller, a name in use, an image the store lacks).
757
+ const said = runtimeErrorLine(reading.result);
758
+ return { ran: false, reason: `the probe container did not start${said ? ` (${bin} said: ${said})` : ""}, so nothing was read back`, verdicts: [], notes, swept };
759
+ }
760
+
761
+ const expected = containerSpec(probeOptions({ image, name: names.probe, fixture, user, relabel })).mounts;
762
+ const inspected = await step(["inspect", "--format={{json .Mounts}}", probeId]);
763
+ // Issue #345: the mount table as the container itself sees it, by a constant `cat`, for what `.Mounts` does not list.
764
+ const mountinfo = await step(["exec", probeId, "cat", "/proc/self/mountinfo"]);
765
+ const mountSet = inspected?.code === 0 ? mountSetVerdict(inspected.stdout, { expected, home, sessionsDir, mountinfo: mountinfo?.code === 0 ? mountinfo.stdout : null, bin }) : notReadBack("mountSet", `${bin} inspect did not answer`);
766
+
767
+ const statusRun = await step(["exec", probeId, "sh", "-c", STATUS_SCRIPT]);
768
+ const status = statusRun?.code === 0 ? parseStatus(statusRun.stdout) : null;
769
+ const isolation = status ? isolationVerdict(status) : notReadBack("isolation", "the status probe did not run in the container");
770
+ const nonRoot = status ? nonRootVerdict(status) : notReadBack("nonRoot", "the status probe did not run in the container");
771
+
772
+ const written = await step(["exec", probeId, "sh", "-c", WRITE_SCRIPT, "sh", nonce]);
773
+ const readBack = (dir) => {
774
+ try {
775
+ return fs.readFileSync(`${dir}/.pi-dispatch-live-probe`, "utf8");
776
+ } catch {
777
+ return null;
778
+ }
779
+ };
780
+ // The workspace nonce is the check; the outbox's and the session's must match it too, or a write the container
781
+ // reported is not one the host can see. A read that worked with a stat that did not leaves the owner unchecked.
782
+ const reads = [["/workspace", readBack(fixture.workspace)], ["/outbox", readBack(fixture.outboxDir)], ["/session", readBack(fixture.sessionDir)]];
783
+ const unseen = reads.find(([, r]) => r !== nonce)?.[0] ?? "/workspace";
784
+ const hostRead = reads.every(([, r]) => r === nonce) ? nonce : null;
785
+ let hostOwner = null;
786
+ try {
787
+ hostOwner = fs.statSync(`${fixture.workspace}/.pi-dispatch-live-probe`).uid;
788
+ } catch {
789
+ hostOwner = null;
790
+ }
791
+ const localFolders = localFoldersVerdict({ code: written?.code, stdout: written?.stdout, hostRead, nonce, hostOwner, euid, unseen });
792
+ await release(reading.entry);
793
+
794
+ // --- the pinning container: an image this host does not have ---
795
+ const pinning = await start("pinning container", names.pin, pinningProbeRunArgs({ name: names.pin, nonce, fixture, user, relabel, buildArgs }));
796
+ const after = await step(["image", "inspect", absentImageRef(nonce)]);
797
+ const stillAbsent = after?.code === 0 ? false : typeof after?.code === "number" ? true : null;
798
+ const imagePinning = imagePinningVerdict({ code: pinning.result?.code, output: `${pinning.result?.stdout ?? ""}${pinning.result?.stderr ?? ""}`, stillAbsent, bin });
799
+ await release(pinning.entry);
800
+
801
+ // --- the ephemeral pair (issue #344): one name, two runs, each waited on until it is gone ---
802
+ const runEphemeral = async (n) => {
803
+ const { result, entry } = await start(`ephemeral container (run ${n})`, names.ephemeral, ephemeralRunArgs({ image, name: names.ephemeral, fixture, nonce, run: n, user, relabel, buildArgs }));
804
+ const started = result?.code === 0 && entry.id !== null;
805
+ // A HELD NAME is the daemon refusing the create for the name, in its own words (measured: Docker "Conflict. ...
806
+ // is already in use", Podman "that name is already in use"). Not a listed container: after the first run was
807
+ // seen gone, the only container listed under this run's name is this run's own failed one.
808
+ // A run that printed an ID made its own container, so whatever failed after that was not the name.
809
+ const nameHeld = !started && entry.id === null && /already in use/i.test(`${result?.stdout ?? ""}${result?.stderr ?? ""}`);
810
+ const removal = entry.id === null ? null : await awaitRemoved({ step, id: entry.id, now, delay, deadlineMs: removalDeadlineMs });
811
+ // SEEN GONE: never removed again, so a later `rm -f` cannot land on a new container that took its ID.
812
+ if (removal?.state === "gone") entry.done = true;
813
+ else await release(entry);
814
+ return { started, id: entry.id, nameHeld, removal };
815
+ };
816
+ const first = await runEphemeral(1);
817
+ const second = first.started && first.removal?.state === "gone" ? await runEphemeral(2) : null;
818
+ const marker = (n) => {
819
+ try {
820
+ return fs.readFileSync(`${fixture.workspace}/.pi-dispatch-live-ephemeral-${n}`, "utf8");
821
+ } catch {
822
+ return null;
823
+ }
824
+ };
825
+ const ephemeral = ephemeralVerdict({ first, second, markers: { first: marker(1), second: marker(2) }, nonce, bin });
826
+
827
+ // --- the peers (issue #344): two job networks, two peers, one attempt each way ---
828
+ let jobToJobIsolation = jobToJobIsolationVerdict({ bin, armed: egress?.armed, proxyRunning: egress?.proxyRunning, keeperBlocked });
829
+ if (peersWanted) {
830
+ const reading = { bin, armed: true, proxyRunning: egress?.proxyRunning, networksCreated: false, peersStarted: false };
831
+ const peerNetworks = [];
832
+ const peers = [];
833
+ try {
834
+ for (const key of ["peer1", "peer2"]) {
835
+ // OWNED BEFORE THE CREATE, as a container is before its run: a create that timed out may still have made it.
836
+ const entry = { name: networkOf[key], done: false };
837
+ networks.push(entry);
838
+ peerNetworks.push(entry);
839
+ if (!(await createJobNetworkWith(step, { network: networkOf[key], proxy, bin }))) break;
840
+ entry.created = true;
841
+ }
842
+ reading.networksCreated = peerNetworks.length === 2 && peerNetworks.every((e) => e.created);
843
+ if (reading.networksCreated) {
844
+ const ids = {};
845
+ for (const key of ["peer1", "peer2"]) {
846
+ let peerFixture;
847
+ try {
848
+ peerFixture = makeFixture(`${root}/${key}`);
849
+ } catch {
850
+ break;
851
+ }
852
+ const { result, entry } = await start(`${key} container`, names[key], peerRunArgs({ image, name: names[key], fixture: peerFixture, network: networkOf[key], nonce, seconds: liveSleepSeconds(stepTimeoutMs), user, relabel, buildArgs }));
853
+ peers.push(entry);
854
+ if (result?.code !== 0 || entry.id === null) break;
855
+ ids[key] = entry.id;
856
+ }
857
+ reading.peersStarted = Boolean(ids.peer1 && ids.peer2);
858
+ if (reading.peersStarted) {
859
+ const described = await step(["inspect", "--format={{json .NetworkSettings.Networks}}", ids.peer2]);
860
+ const { targets, addresses } = described?.code === 0 ? peerTargetsOf(described.stdout, { network: networkOf.peer2, name: names.peer2 }) : { targets: [], addresses: [] };
861
+ reading.peerTargets = targets;
862
+ reading.peerAddresses = addresses;
863
+ reading.proxyTarget = `${proxy}:${EGRESS_PROXY_PORT}`;
864
+ // The control brackets the attempt: peer2 answering its own addresses BEFORE proves a refused connection is
865
+ // not a listener still starting, and AFTER proves it stayed up.
866
+ const control = async () => {
867
+ if (addresses.length === 0) return new Map();
868
+ const self = await step(["exec", ids.peer2, "node", "--eval", CONNECT_SCRIPT, nonce, ...addresses]);
869
+ return self?.code === 0 ? parseConnectResults(self.stdout) : new Map();
870
+ };
871
+ reading.controlBefore = await control();
872
+ const attempt = await step(["exec", ids.peer1, "node", "--eval", CONNECT_SCRIPT, nonce, reading.proxyTarget, ...targets]);
873
+ reading.fromPeer1 = attempt?.code === 0 ? parseConnectResults(attempt.stdout) : new Map();
874
+ reading.controlAfter = await control();
875
+ }
876
+ }
877
+ } finally {
878
+ // Peers first, THEN their networks: a network with a container still on it is not removed without -f, and
879
+ // this never uses -f.
880
+ for (const entry of peers) await release(entry);
881
+ for (const entry of peerNetworks) await dropNetwork(entry);
882
+ }
883
+ jobToJobIsolation = jobToJobIsolationVerdict(reading);
884
+ }
885
+
886
+ const byProperty = { isolation, ephemeral, mountSet, egress: egressVerdict(egress ?? {}), jobToJobIsolation, imagePinning, nonRoot, localFolders };
887
+ // The uid PID 1 actually ran as, for doctor's job-user line: the decision it was given, read back.
888
+ const ranAs = Array.isArray(status?.uids) && /^\d+$/.test(status.uids[1] ?? "") ? Number(status.uids[1]) : null;
889
+ return { ran: true, verdicts: READ_BACK_BY_A_LIVE_PROBE.map((p) => byProperty[p]), notes, swept, ranAs };
890
+ } finally {
891
+ for (const entry of owned) await release(entry);
892
+ for (const entry of networks) await dropNetwork(entry);
893
+ if (root !== null) {
894
+ try {
895
+ fs.rmSync(root, { recursive: true, force: true });
896
+ } catch {
897
+ notes.push(`the fixture ${root} could not be removed`);
898
+ }
899
+ }
900
+ }
901
+ }
902
+
903
+ /** The container ID `docker run -d` printed: the last line of stdout that is one, else `null`. */
904
+ export function containerIdOf(result) {
905
+ const lines = String(result?.stdout ?? "").split(/\r?\n/).map((l) => l.trim()).filter(Boolean);
906
+ const last = lines.at(-1);
907
+ return last && /^[0-9a-f]{12,64}$/.test(last) ? last : null;
908
+ }
909
+
910
+ /**
911
+ * A fixture left by a `--live` run that did not reach its `finally` (a Ctrl-C). Only an entry of EXACTLY the shape
912
+ * `mkdtemp` makes (the prefix, a PID, six characters), only a real directory (a symlink is never followed or
913
+ * removed), and only one whose PID is no longer alive, so a concurrent run's fixture is never pulled out from under
914
+ * its container. The shape is NARROW, not unique: a directory an operator happened to name `pi-dispatch-live-<n>-xxxxxx`
915
+ * under the jobs dir, with no live process of that PID, is removed as well, which is why the jobs dir is not a place
916
+ * for anything else and why what is removed is said. PIDs are this host's: a `--live` run from inside a container
917
+ * that shares the host's docker numbers its own. Best effort: a sweep that fails is not a reason not to probe.
918
+ */
919
+ export function sweepStaleFixtures({ jobsDir, pid, fs, isAlive }) {
920
+ let entries;
921
+ try {
922
+ entries = fs.readdirSync(jobsDir);
923
+ } catch {
924
+ return [];
925
+ }
926
+ const swept = [];
927
+ for (const entry of entries) {
928
+ const m = new RegExp(`^${LIVE_PREFIX}(\\d+)-[A-Za-z0-9]{6}$`).exec(entry);
929
+ if (!m || Number(m[1]) === pid || isAlive(Number(m[1]))) continue;
930
+ try {
931
+ const st = fs.lstatSync(`${jobsDir}/${entry}`);
932
+ if (!st.isDirectory() || st.isSymbolicLink()) continue;
933
+ fs.rmSync(`${jobsDir}/${entry}`, { recursive: true, force: true });
934
+ swept.push(`fixture ${entry}`);
935
+ } catch {
936
+ // left for the next run
937
+ }
938
+ }
939
+ return swept;
940
+ }
941
+
942
+ /**
943
+ * A live-probe container left by a run interrupted before its container STARTED, which `--rm` never removes
944
+ * (measured: a SIGINT during `docker run -d` leaves one in `created`), or one a peer phase never reached the end of.
945
+ * Only names of exactly one of `liveNames`' container kinds (derived, never retyped) and a PID no longer alive,
946
+ * removed by ID. Best effort, like the fixture sweep.
947
+ */
948
+ export async function sweepStaleContainers({ step, pid, isAlive }) {
949
+ const listed = await step(["ps", "-a", "--filter", `name=${LIVE_PREFIX}`, "--format", "{{.ID}} {{.Names}}"]);
950
+ if (listed?.code !== 0) return [];
951
+ const swept = [];
952
+ const shape = new RegExp(`^${LIVE_PREFIX}(?:${LIVE_CONTAINER_KINDS.join("|")})-(\\d+)-[0-9a-f]+$`);
953
+ for (const line of String(listed.stdout ?? "").split(/\r?\n/)) {
954
+ const [id, name] = line.trim().split(/\s+/);
955
+ const m = shape.exec(name ?? "");
956
+ if (!m || !/^[0-9a-f]{12,64}$/.test(id ?? "") || Number(m[1]) === pid || isAlive(Number(m[1]))) continue;
957
+ if ((await step(["rm", "-f", id]))?.code === 0) swept.push(`container ${name}`);
958
+ }
959
+ return swept;
960
+ }
961
+
962
+ /**
963
+ * A peer network an interrupted run left (issue #344): only a name of exactly a peer network's shape (the peer kinds
964
+ * of `liveNames`, derived) and a PID no longer alive. What is attached to it decides what happens: a live-probe
965
+ * CONTAINER still on it means a run that is not over (a PID from another namespace reads as dead here), so nothing is
966
+ * touched; otherwise every endpoint still attached (a proxy, whatever `PI_EGRESS_PROXY` named when that run started)
967
+ * is detached, each one named in what is reported, and the network is removed WITHOUT `-f`. One that stays is a note
968
+ * carrying the command, never silence.
969
+ */
970
+ export async function sweepStaleNetworks({ step, pid, isAlive, notes = [], bin = "docker", keeperBlocked = null, gate = makeDetachGate(step, { bin }) }) {
971
+ const listed = await step(["network", "ls", "--filter", `name=${LIVE_PREFIX}`, "--format", "{{.Name}}"]);
972
+ if (listed?.code !== 0) return [];
973
+ const swept = [];
974
+ const peerKinds = LIVE_CONTAINER_KINDS.filter((k) => k.startsWith("peer"));
975
+ const shape = new RegExp(`^${LIVE_PREFIX}(?:${peerKinds.join("|")})-(\\d+)-[0-9a-f]+-net$`);
976
+ const probeContainer = new RegExp(`^${LIVE_PREFIX}(?:${LIVE_CONTAINER_KINDS.join("|")})-\\d+-`);
977
+ for (const line of String(listed.stdout ?? "").split(/\r?\n/)) {
978
+ const name = line.trim();
979
+ const m = shape.exec(name);
980
+ if (!m || Number(m[1]) === pid || isAlive(Number(m[1]))) continue;
981
+ // Through the shared reader since issue #357, which is where the fail-closed rule lives: a `.Containers`
982
+ // that renders as `null` used to parse to `{}` here and read as "nothing attached", which would detach
983
+ // and remove a network that still had members. Unreadable is now unreadable, and this sweep's own
984
+ // silence on it is unchanged -- it is best effort, and what it CAN read it still says.
985
+ const { ok, names: running, parked = [] } = await networkEndpoints(step, name, { bin });
986
+ if (!ok) continue;
987
+ // RESIDUAL ON DOCKER, recorded under issue #337 where it was measured rather than left for the next reader:
988
+ // `.Containers` lists RUNNING endpoints, so a probe in `created` state is not in `running`, and on that
989
+ // daemon the `network rm` below would then SUCCEED and leave that container unable to start. Unguarded
990
+ // there on purpose rather than by oversight: this sweep only looks at a network whose owning pid is DEAD,
991
+ // and a dead process has no launch in flight. The sandbox sweep, whose owner may be very much alive, does
992
+ // carry the guard.
993
+ //
994
+ // CLOSED ON PODMAN (issue #452), where the read also returns `parked`, the members in every other state:
995
+ // Podman's `network rm` refuses while ANY of them remains (measured on 4.9.3 and 5.8.1), so a stopped proxy
996
+ // left on a peer network kept it forever, one ⚠ per run. A parked probe counts as a probe, and the rest are
997
+ // detached with the running ones. Docker's read carries no `parked`, so its pass is what it was.
998
+ const attached = [...running, ...parked];
999
+ if (attached.some((n) => probeContainer.test(n))) continue;
1000
+ // Issue #458: a detach is the trigger there, so a stale network with anything still on it is left and SAID; one
1001
+ // with nothing attached is removed as before, since `network rm` alone does not tear the helper down (measured).
1002
+ if (keeperBlocked && attached.length > 0) {
1003
+ notes.push(`the stale network ${name} was left with ${attached.join(", ")} attached, because ${keeperBlocked}; the next \`pi-dispatch doctor --live\` with the keeper running removes it`);
1004
+ continue;
1005
+ }
1006
+ // Through the ONE detach helper (issue #452, gate round 3), which asks the gate for the RUNNING members first; a
1007
+ // refusal leaves the network whole and is said like the keeper block above.
1008
+ const outcome = await removeNetworkOrSay(step, { network: name, detach: attached, running, bin, gate });
1009
+ if (outcome.blocked) {
1010
+ notes.push(`the stale network ${name} was left with ${attached.join(", ")} attached, because ${detachBlockedSentence(outcome.blocked, bin)}; the next \`pi-dispatch doctor --live\` with the keeper running removes it`);
1011
+ continue;
1012
+ }
1013
+ if (outcome.absent) continue;
1014
+ // Every endpoint detached is SAID, the proxy included: a container this sweep did not make may be among them.
1015
+ const detached = outcome.detached.length > 0 ? ` (after detaching ${outcome.detached.join(", ")})` : "";
1016
+ if (outcome.removed) swept.push(`network ${name}${detached}`);
1017
+ else notes.push(`the stale network ${name}${detached} could not be removed: ${bin} network rm ${name}`);
1018
+ }
1019
+ return swept;
1020
+ }