@edgehero/pi-dispatch 1.10.2 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +300 -148
- package/README.md +50 -0
- package/deploy/com.pi-dispatch.worker.plist +9 -3
- package/deploy/docker-compose.yml +49 -16
- package/deploy/egress-proxy.conf +32 -2
- package/deploy/nssm-install.cmd +12 -6
- package/deploy/pi-dispatch-egress-out.network +10 -0
- package/deploy/pi-dispatch-egress-proxy.container +50 -0
- package/deploy/pi-dispatch-netns-keeper.container +80 -0
- package/deploy/pi-dispatch-netns-keeper.network +18 -0
- package/deploy/pi-dispatch-valkey.container +51 -0
- package/deploy/pi-dispatch-valkey.network +16 -0
- package/deploy/receiver.service +6 -0
- package/deploy/worker-env-wrapper.cmd +11 -0
- package/deploy/worker-env-wrapper.sh +60 -34
- package/deploy/worker.service +18 -8
- package/package.json +14 -4
- package/src/azure-host.mjs +19 -0
- package/src/azure-identity.mjs +18 -2
- package/src/backend-conformance.mjs +71 -18
- package/src/backend-local.mjs +637 -21
- package/src/backend-podman.mjs +1168 -0
- package/src/backend-registry.mjs +86 -3
- package/src/backends.mjs +489 -37
- package/src/branch.mjs +7 -2
- package/src/cancel-cli.mjs +174 -0
- package/src/cancel-state.mjs +125 -0
- package/src/cli.mjs +188 -90
- package/src/config.mjs +503 -43
- package/src/connection.mjs +374 -8
- package/src/container-spec.mjs +102 -7
- package/src/daemon-facts.mjs +167 -0
- package/src/deployment-venue.mjs +158 -0
- package/src/docker-run.mjs +146 -15
- package/src/doctor.mjs +4701 -414
- package/src/egress-conf-copy.mjs +166 -0
- package/src/egress-proxy-state.mjs +151 -0
- package/src/egress.mjs +455 -25
- package/src/entry.mjs +27 -0
- package/src/env-allowlist.mjs +222 -40
- package/src/env-file.mjs +1869 -33
- package/src/exit-code.mjs +15 -0
- package/src/flow-gate.mjs +5 -3
- package/src/forgejo-host.mjs +19 -0
- package/src/forgejo-identity.mjs +21 -2
- package/src/get-token.mjs +67 -18
- package/src/git-dirty.mjs +9 -1
- package/src/git-hardening.mjs +33 -0
- package/src/github-app-setup.mjs +29 -12
- package/src/github-prompt.mjs +4 -1
- package/src/gitlab-host.mjs +19 -0
- package/src/gitlab-identity.mjs +19 -2
- package/src/host-registry.mjs +32 -5
- package/src/identity.mjs +29 -4
- package/src/image-preflight.mjs +46 -11
- package/src/image-ref.mjs +21 -0
- package/src/index.mjs +387 -17
- package/src/init.mjs +197 -38
- package/src/job-user.mjs +252 -0
- package/src/json-duplicates.mjs +204 -0
- package/src/live-probes.mjs +1020 -0
- package/src/materialize.mjs +4 -11
- package/src/netns-keeper.mjs +264 -0
- package/src/on-failure.mjs +119 -0
- package/src/outbox.mjs +7 -0
- package/src/podman-stack.mjs +1304 -0
- package/src/prepare-github.mjs +6 -6
- package/src/prepare-local.mjs +51 -17
- package/src/prepare.mjs +27 -6
- package/src/processor.mjs +505 -26
- package/src/provider-key.mjs +41 -0
- package/src/provider-steering.mjs +144 -0
- package/src/queue.mjs +35 -8
- package/src/redact.mjs +84 -0
- package/src/reserved-env.mjs +7 -3
- package/src/retention-sweep.mjs +178 -0
- package/src/run-container.mjs +181 -14
- package/src/run-history.mjs +105 -16
- package/src/runtime-observations.mjs +1152 -0
- package/src/runtime-settings.mjs +13 -8
- package/src/sandbox-cli.mjs +100 -95
- package/src/sandbox-store.mjs +612 -45
- package/src/sandbox.mjs +1459 -37
- package/src/schedules.mjs +16 -3
- package/src/secret-profiles.mjs +2 -1
- package/src/secrets.mjs +23 -6
- package/src/service-env.mjs +247 -0
- package/src/service.mjs +618 -28
- package/src/session-store.mjs +678 -53
- package/src/start.mjs +1395 -268
- package/src/transient.mjs +240 -0
- package/src/triggers-file.mjs +71 -15
- package/src/triggers.mjs +176 -19
- package/src/up.mjs +1399 -85
- package/src/valkey-auth.mjs +529 -0
- package/src/valkey-endpoint.mjs +367 -0
- package/src/watch-closer.mjs +158 -0
package/src/sandbox.mjs
CHANGED
|
@@ -1,18 +1,31 @@
|
|
|
1
1
|
import { execFile, spawn } from "node:child_process";
|
|
2
|
-
import { existsSync } from "node:fs";
|
|
2
|
+
import { existsSync, lstatSync, readdirSync, readFileSync, statSync } from "node:fs";
|
|
3
|
+
import { basename, join } from "node:path";
|
|
4
|
+
import { release as osRelease } from "node:os";
|
|
3
5
|
import { promisify } from "node:util";
|
|
6
|
+
import { execDockerBounded, makeDockerEndpointResolver } from "./backend-local.mjs";
|
|
7
|
+
import { PODMAN_CONF_WIDENS_JOB, PODMAN_JOB_USER_FIX, decidePodmanJobUser, judgePodmanVenue, keeperPreflight, makePodmanInfoReader, resolvePodmanImageUser } from "./backend-podman.mjs";
|
|
8
|
+
import { DEFAULT_BACKEND, PODMAN_BACKEND, UNATTRIBUTED_BACKEND, parseBackendFloor, parseBackendList } from "./backends.mjs";
|
|
4
9
|
import { configError } from "./config.mjs";
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
10
|
+
import { assertJobUser, CONTAINER_HOME, SHIPPED_IMAGE_UID } from "./container-spec.mjs";
|
|
11
|
+
import { buildDockerRunArgs, buildPodmanRunArgs, insideDir } from "./docker-run.mjs";
|
|
12
|
+
import { makeImagePreflight } from "./image-preflight.mjs";
|
|
13
|
+
import { decideJobUser, JOB_USER_FIX, makeDaemonFactsReader, relabelsPrivateMounts, resolveImageUser, socketFacts } from "./job-user.mjs";
|
|
14
|
+
import { DEFAULT_EGRESS_PROXY, NETWORK_SUFFIX, createJobNetwork, egressArmed, egressEnv, egressProxyName, networkEndpoints, networkExists, networkNameFor, removeJobNetwork, removeNetworkOrSay } from "./egress.mjs";
|
|
15
|
+
import { NETNS_KEEPER, makeDetachGate, runtimeFromFacts } from "./netns-keeper.mjs";
|
|
16
|
+
import { NETNS_KEEPER_START } from "./podman-stack.mjs";
|
|
17
|
+
import { makePodmanServiceReader, observeRootfulConf, rootfulConfRefusal } from "./runtime-observations.mjs";
|
|
18
|
+
import { isSandboxTombstone, readManifest, readRetained, sandboxDeadline, sandboxEntryName } from "./sandbox-store.mjs";
|
|
8
19
|
|
|
9
20
|
/**
|
|
10
21
|
* sandbox.mjs -- the operator session's container shape (INT-SANDBOX-CONTRACT).
|
|
11
22
|
*
|
|
12
|
-
* A SECOND container shape, deliberately not a second copy of the first. The argv comes from
|
|
13
|
-
* `buildDockerRunArgs`
|
|
14
|
-
*
|
|
15
|
-
* this
|
|
23
|
+
* A SECOND container shape, deliberately not a second copy of the first. The argv comes from the run's VENUE's own
|
|
24
|
+
* job builder (`buildDockerRunArgs` on `local`, `buildPodmanRunArgs` on `podman`, `SANDBOX_LAUNCHERS` below) through
|
|
25
|
+
* its `extraFlags` seam, so `ISOLATION_FLAGS`, `--memory` and `--cpus` (and on podman keep-id and
|
|
26
|
+
* `PODMAN_PINNED_FLAGS`) reach this container BY CONSTRUCTION: a future change to the boundary cannot land on job
|
|
27
|
+
* containers and miss this one, which is the whole reason for reusing the builder rather than writing a leaner argv
|
|
28
|
+
* here.
|
|
16
29
|
*
|
|
17
30
|
* What differs from a job, and every difference is the point:
|
|
18
31
|
* - `-i -t --entrypoint bash`. `INT-CONTAINER-RUNTIME-CONTRACT` says "No TTY (`-it` absent)" and stays
|
|
@@ -20,7 +33,9 @@ import { readManifest } from "./sandbox-store.mjs";
|
|
|
20
33
|
* - NO CREDENTIALS. Not the minted forge token, not the provider key, not one forwarded host variable.
|
|
21
34
|
* `buildContainerEnv` is deliberately NOT reused: it writes the mint into that forge's variable names
|
|
22
35
|
* (env-allowlist.mjs) and throws outright when no provider credential resolves, so a credential-free
|
|
23
|
-
* container cannot be produced from it. The env here is two variables
|
|
36
|
+
* container cannot be produced from it. The env here is two variables about the terminal, plus the four
|
|
37
|
+
* proxy variables when an egress policy is armed, plus `HOME=/home/pi` beside `--user` when the run had a job
|
|
38
|
+
* user (issue #341).
|
|
24
39
|
* - No `/outbox`, no `/session`, no `/opt/pi-global`. The agent is not running; there is nothing to
|
|
25
40
|
* chain, no transcript to continue and no overlay to layer.
|
|
26
41
|
*
|
|
@@ -29,17 +44,68 @@ import { readManifest } from "./sandbox-store.mjs";
|
|
|
29
44
|
|
|
30
45
|
const exec = promisify(execFile);
|
|
31
46
|
|
|
47
|
+
/**
|
|
48
|
+
* The venues a sandbox opens on, and how each one spells its container (issue #429): the CLI every spawn for that
|
|
49
|
+
* venue goes through, and the job argv builder the session's argv is built by. Keyed by the venue name the manifest
|
|
50
|
+
* records, so a run is reopened ONLY in the runtime that ran it: a podman run under docker would reproduce none of it
|
|
51
|
+
* (another store, no keep-id, another uid map), and the reverse is as wrong.
|
|
52
|
+
*
|
|
53
|
+
* A TABLE rather than a `bin` read off the backend bundles, and that alternative was the obvious one: the bundles are
|
|
54
|
+
* built by the worker at boot, and the CLI and the admin panel never build them (they would construct a
|
|
55
|
+
* `runContainer`, a reaper and a `podman info` reader to open a shell). What the sandbox needs from a venue is two
|
|
56
|
+
* facts, and both are the same values the bundles are built with (`bin: "podman"` and `buildPodmanRunArgs` in
|
|
57
|
+
* `makePodmanBackend`); `sandbox.test.mjs` pins the pairing. A venue not in this table has no sandbox, by
|
|
58
|
+
* construction: nothing here can reopen a run under a runtime it did not run in.
|
|
59
|
+
*
|
|
60
|
+
* Not "any venue declaring `remote: false`": a future non-remote venue on a third runtime would pass that and be
|
|
61
|
+
* reopened under one of these two. Such a venue must add its own row, deliberately.
|
|
62
|
+
*/
|
|
63
|
+
export const SANDBOX_LAUNCHERS = Object.freeze({
|
|
64
|
+
[DEFAULT_BACKEND]: Object.freeze({ bin: "docker", build: buildDockerRunArgs }),
|
|
65
|
+
[PODMAN_BACKEND]: Object.freeze({ bin: "podman", build: buildPodmanRunArgs }),
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
/** The launcher for `venue`, or null. `Object.hasOwn`, so `"toString"` is not a venue. */
|
|
69
|
+
export function sandboxLauncher(venue) {
|
|
70
|
+
return typeof venue === "string" && Object.hasOwn(SANDBOX_LAUNCHERS, venue) ? SANDBOX_LAUNCHERS[venue] : null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* The blessed venues a sandbox can open on, in `PI_BACKENDS` order: the venues this host both runs and has a launcher
|
|
75
|
+
* for. The retention reaper asks each one which sandboxes are open, and `--list` draws its RUNNING column from them.
|
|
76
|
+
*/
|
|
77
|
+
export function sandboxVenues(blessed) {
|
|
78
|
+
return (Array.isArray(blessed) ? blessed : []).filter((venue) => sandboxLauncher(venue) !== null);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* `{ blessed, backendFloor }` as a sandbox opened from `env` sees them, through the worker's own two parsers. For the
|
|
83
|
+
* admin panel, which deliberately never calls `loadConfig`; the CLI has the parsed config already. THROWS on a
|
|
84
|
+
* malformed `PI_BACKENDS` or `PI_BACKEND_FLOOR`, exactly as the worker refuses to boot on one: a typo must never read
|
|
85
|
+
* as "podman is not blessed" (a refusal an operator then chases in the wrong place) or as "no floor" (a podman sandbox
|
|
86
|
+
* opened without the observations the deployment asks for). Read from THIS process's environment and never a
|
|
87
|
+
* deployment's `.env`, which `OQ-038` records for the panel.
|
|
88
|
+
*/
|
|
89
|
+
export function sandboxVenuePolicy(env) {
|
|
90
|
+
return { blessed: parseBackendList(env?.PI_BACKENDS), backendFloor: parseBackendFloor(env?.PI_BACKEND_FLOOR) };
|
|
91
|
+
}
|
|
92
|
+
|
|
32
93
|
/**
|
|
33
94
|
* The name namespace, and it is load-bearing. The boot reaper filters `name=pi-job-`
|
|
34
|
-
* (`makeReaper` in
|
|
95
|
+
* (`makeReaper` in backend-local.mjs) and docker matches that as a SUBSTRING, so a sandbox must not contain it --
|
|
35
96
|
* otherwise a worker restart kills the shell an operator is sitting in. `pi-sandbox-` is outside that
|
|
36
97
|
* filter on purpose, and a test pins it.
|
|
37
98
|
*/
|
|
38
99
|
export const SANDBOX_NAME_PREFIX = "pi-sandbox-";
|
|
39
100
|
|
|
40
|
-
/**
|
|
101
|
+
/**
|
|
102
|
+
* `pi-sandbox-<jobId>`. `sanitizeJobId` already maps to `[A-Za-z0-9._-]`, which is a legal docker name. Through
|
|
103
|
+
* `sandboxEntryName` (issue #446), the retained directory's own name, so the id a runtime reports for this container is
|
|
104
|
+
* the name the retention sweep holds: an id whose sanitized form starts with `.` is retained under `_...`, and a
|
|
105
|
+
* container named off the other spelling would never hold its own directory.
|
|
106
|
+
*/
|
|
41
107
|
export function sandboxContainerName(jobId) {
|
|
42
|
-
return `${SANDBOX_NAME_PREFIX}${
|
|
108
|
+
return `${SANDBOX_NAME_PREFIX}${sandboxEntryName(jobId)}`;
|
|
43
109
|
}
|
|
44
110
|
|
|
45
111
|
/**
|
|
@@ -70,8 +136,10 @@ function inPortRange(n) {
|
|
|
70
136
|
}
|
|
71
137
|
|
|
72
138
|
/**
|
|
73
|
-
* Build the `
|
|
139
|
+
* Build the `run` argv for one operator session (excluding the leading "docker" or "podman").
|
|
74
140
|
*
|
|
141
|
+
* @param venue the venue the run used (`SANDBOX_LAUNCHERS`); its builder builds this argv. Defaults to
|
|
142
|
+
* `local`, so every caller from before issue #429 gets the argv it always did, byte for byte.
|
|
75
143
|
* @param image the image the original run used, from its manifest
|
|
76
144
|
* @param name `pi-sandbox-<jobId>`
|
|
77
145
|
* @param workspace host path mounted /workspace:rw -- the retained clone, or the operator's own folder
|
|
@@ -81,16 +149,53 @@ function inPortRange(n) {
|
|
|
81
149
|
* @param idleSeconds bash's own TMOUT; 0 omits it
|
|
82
150
|
* @param network this session's own egress network (REQ-EGRESS-ALLOWLIST); null = the default bridge
|
|
83
151
|
* @param egressEnv the proxy variables that go with it, or {} when no policy is armed
|
|
152
|
+
* @param user "<uid>:<gid>" the run had (issue #341), or null for the image's own USER
|
|
153
|
+
* @param home CONTAINER_HOME beside `user`, and required with it
|
|
154
|
+
* @param relabel true where the job's own mounts carried `:Z` (issue #355), so the retained ones do again
|
|
155
|
+
* @param workspaceOwned true when `workspace` is the retained clone (the worker's own), false for an operator's folder
|
|
84
156
|
*/
|
|
85
|
-
export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish = [], term, idleSeconds = 0, network = null, egressEnv: proxyEnv = {} }) {
|
|
86
|
-
|
|
157
|
+
export function buildSandboxRunArgs({ venue = DEFAULT_BACKEND, image, name, workspace, jobDir, publish = [], term, idleSeconds = 0, network = null, egressEnv: proxyEnv = {}, user = null, home = null, relabel = false, workspaceOwned = false }) {
|
|
158
|
+
// Thrown, not defaulted to docker: a caller naming a venue this file has no launcher for is assembling a session
|
|
159
|
+
// in a runtime nobody chose, which is the one mistake the table exists to make impossible.
|
|
160
|
+
const launcher = sandboxLauncher(venue);
|
|
161
|
+
if (!launcher) throw new Error(`buildSandboxRunArgs: no sandbox launcher for venue ${JSON.stringify(venue)} (have ${Object.keys(SANDBOX_LAUNCHERS).join(", ")})`);
|
|
162
|
+
// Issue #341: the job path's pairing, for the same measured reason (a uid with no passwd entry gets HOME=/ or
|
|
163
|
+
// HOME=/workspace), so a sandbox shell as that uid can write its own home.
|
|
164
|
+
if (user !== null && home !== CONTAINER_HOME) {
|
|
165
|
+
throw new Error(`buildSandboxRunArgs: a user (${user}) must be paired with HOME=${CONTAINER_HOME}`);
|
|
166
|
+
}
|
|
167
|
+
// Issue #362, and it throws where `openSandbox` returns a refusal because the two answer different
|
|
168
|
+
// questions: that one is an operator's mistake with a fix, this one is a caller assembling an argv that
|
|
169
|
+
// cannot mean what it says.
|
|
170
|
+
//
|
|
171
|
+
// It refuses on ANY network rather than on an internal one, and that is deliberate rather than loose:
|
|
172
|
+
// this builder is handed a NAME and cannot see the flags the network was created with. What it can rely
|
|
173
|
+
// on is that every network this project puts a session on is created `--internal` by `createJobNetwork`.
|
|
174
|
+
// A `-p` on some other user-defined network does work (measured), so this is a refusal about this
|
|
175
|
+
// project's shapes and not a claim about docker.
|
|
176
|
+
if (publish.length > 0 && network !== null) {
|
|
177
|
+
throw new Error(`buildSandboxRunArgs: a published port cannot be paired with a session network (${network}) -- every network this project puts a session on is created --internal, where docker accepts -p and binds nothing`);
|
|
178
|
+
}
|
|
179
|
+
// The venue's own JOB builder, so on podman `--userns=keep-id`, `PODMAN_PINNED_FLAGS` and, with no session network,
|
|
180
|
+
// `--network=private` arrive exactly as a job's do. The last one is load-bearing rather than tidy: a containers.conf
|
|
181
|
+
// `netns = "host"` puts a container launched with no `--network` on the host's network namespace (measured under
|
|
182
|
+
// issue #354), so a podman sandbox with egress off must name its network as a job does. `buildPodmanRunArgs` also
|
|
183
|
+
// refuses a null `user`, which `decideSandboxJobUser`'s podman branch never answers.
|
|
184
|
+
return launcher.build({
|
|
185
|
+
user,
|
|
87
186
|
image,
|
|
88
187
|
name,
|
|
89
188
|
workspace,
|
|
90
189
|
jobDir,
|
|
91
|
-
//
|
|
92
|
-
//
|
|
93
|
-
//
|
|
190
|
+
// Issue #355: the job path's rule, by the same builder. The retained job dir is relabelled again for this one
|
|
191
|
+
// container (the run that labelled it is gone); the workspace only when it is the retained clone, never an
|
|
192
|
+
// operator's folder, which a private label would take away from every other container.
|
|
193
|
+
relabel,
|
|
194
|
+
workspaceOwned,
|
|
195
|
+
// The terminal's two variables, and neither is a credential. TERM so the shell renders; TMOUT so a
|
|
196
|
+
// forgotten session closes itself. HOME beside `--user` and the proxy variables below are the rest.
|
|
197
|
+
// `buildDockerRunArgs` skips undefined, so an unset TERM or a disabled idle timeout emits nothing rather than
|
|
198
|
+
// an empty string.
|
|
94
199
|
// A sandbox joins the SAME kind of network a job did, by the same builder, so the boundary cannot
|
|
95
200
|
// land on job containers and miss this one. Leaving sandboxes on the default bridge was the tempting
|
|
96
201
|
// alternative and it is the wrong one: it reads as a convenience (install a missing dependency while
|
|
@@ -101,15 +206,17 @@ export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish =
|
|
|
101
206
|
env: {
|
|
102
207
|
TERM: term || undefined,
|
|
103
208
|
TMOUT: idleSeconds > 0 ? String(idleSeconds) : undefined,
|
|
209
|
+
// Beside `--user` only (issue #341). Not a credential either.
|
|
210
|
+
HOME: user !== null ? home : undefined,
|
|
104
211
|
// Still NO CREDENTIALS, and that clause is untouched: a proxy URL is not a credential, and
|
|
105
|
-
// buildContainerEnv is still not reused here. The env is two variables about the terminal
|
|
106
|
-
// when a policy is armed,
|
|
212
|
+
// buildContainerEnv is still not reused here. The env is two variables about the terminal, HOME
|
|
213
|
+
// beside `--user`, and, when a policy is armed, four about the network.
|
|
107
214
|
...proxyEnv,
|
|
108
215
|
},
|
|
109
|
-
// Ahead of the env and the mounts, and well ahead of the image, which
|
|
110
|
-
// the final positional. `--entrypoint` also clears the image's CMD; this repo's Dockerfile sets
|
|
216
|
+
// Ahead of the env and the mounts, and well ahead of the image, which the builder keeps as
|
|
217
|
+
// the final positional. The same four tokens on both venues, through the same `dockerExtra` allow-list. `--entrypoint` also clears the image's CMD; this repo's Dockerfile sets
|
|
111
218
|
// none, so `bash` runs bare and `-it` makes it interactive, in the baked WORKDIR as the baked
|
|
112
|
-
// non-root USER.
|
|
219
|
+
// non-root USER, or as `user` when the run had one.
|
|
113
220
|
extraFlags: ["-i", "-t", "--entrypoint", "bash", ...publish],
|
|
114
221
|
});
|
|
115
222
|
}
|
|
@@ -127,8 +234,14 @@ export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish =
|
|
|
127
234
|
* TIMED, unlike the boot container reaper's own `docker ps`: an unreachable daemon does not fail the CLI
|
|
128
235
|
* fast, it blocks, and this call sits in front of a worker that has not started draining yet.
|
|
129
236
|
*/
|
|
130
|
-
export async function listRunningSandboxes({ execFn = exec } = {}) {
|
|
131
|
-
|
|
237
|
+
export async function listRunningSandboxes({ execFn = exec, bin = "docker", signal } = {}) {
|
|
238
|
+
// `bin` (issue #429) is ONE runtime's CLI: a podman sandbox is invisible to `docker ps` and the reverse, so the
|
|
239
|
+
// retention reaper asks the runtime each retained run records (`makeSandboxRuntimeWatch`), and `--list` each blessed
|
|
240
|
+
// one. `--format {{.Names}}` reads on both.
|
|
241
|
+
// `signal` (issue #446) lets the opener's post-launch look abandon an ask the moment the shell returns: `execFile`
|
|
242
|
+
// kills the child and rejects on abort, so an exited shell never waits out this timeout. Passed only when given,
|
|
243
|
+
// so every other caller's options are what they always were.
|
|
244
|
+
const { stdout } = await execFn(bin, ["ps", "--filter", `name=${SANDBOX_NAME_PREFIX}`, "--format", "{{.Names}}"], { timeout: 5000, ...(signal ? { signal } : {}) });
|
|
132
245
|
return stdout
|
|
133
246
|
.split("\n")
|
|
134
247
|
.map((n) => n.trim())
|
|
@@ -136,33 +249,1340 @@ export async function listRunningSandboxes({ execFn = exec } = {}) {
|
|
|
136
249
|
.map((n) => n.slice(SANDBOX_NAME_PREFIX.length));
|
|
137
250
|
}
|
|
138
251
|
|
|
252
|
+
/**
|
|
253
|
+
* What the retention reaper asks before it deletes anything, and which runtimes its network sweep visits: `{
|
|
254
|
+
* listRunning, sweepNetworks }` for `makeSandboxReaper` (issue #429, review rounds 1 and 2).
|
|
255
|
+
*
|
|
256
|
+
* PER RETAINED RUN, NEVER PER DEPLOYMENT. For each retained directory it reads the venue the run's manifest records
|
|
257
|
+
* and asks THAT venue's runtime (its launcher's `bin`) which sandboxes are open, and `listRunning` answers the ids the
|
|
258
|
+
* pass must HOLD: every id a runtime reports open, and every retained run whose runtime could not answer (no CLI, a
|
|
259
|
+
* daemon that is down, a timeout, any error). Held means kept this pass, directory and network both; the next pass asks
|
|
260
|
+
* again.
|
|
261
|
+
*
|
|
262
|
+
* WHY NOT THE WORKER'S `PI_BACKENDS`, which the first version of this used and a review refuted by reproduction: the
|
|
263
|
+
* opener's blessing comes from the OPENER's environment (`sandboxVenueRefusal`), the reaper's from the worker's, and
|
|
264
|
+
* `OQ-038` records that the two routinely differ. A worker with `PI_BACKENDS=podman` asked only podman, while an
|
|
265
|
+
* operator shell with no `PI_BACKENDS` opened a run retained earlier under docker, and the pass deleted the directory
|
|
266
|
+
* under the shell. The runtime a run opens in is a property of the RUN, so the question is asked of the run.
|
|
267
|
+
*
|
|
268
|
+
* A RUN THIS CANNOT PLACE is asked of EVERY runtime present (review round 2): a manifest that cannot be read or parsed
|
|
269
|
+
* right now (an `EMFILE`, a rewrite caught mid-write), one naming a venue with no launcher, and (PR #466 gate round 1)
|
|
270
|
+
* a directory with no manifest file at all, which a sandbox opened before its manifest went can still have mounted.
|
|
271
|
+
* Such a run is held when any runtime present cannot answer or reports it open, and swept only once every one has
|
|
272
|
+
* answered "not open". Skipping it was the defect: the reaper's own expiry then re-read the manifest, found it
|
|
273
|
+
* unreadable, and deleted a directory an open sandbox was using with no runtime asked. Holding it forever was the other
|
|
274
|
+
* wrong answer, since nothing would ever sweep it. With no runtime present at all there is nothing on this host that
|
|
275
|
+
* could hold it open.
|
|
276
|
+
*
|
|
277
|
+
* A PODMAN RUN RECORDS ITS STORE (`podmanStore`, `podman info`'s graphRoot) and is held while the podman this asks uses
|
|
278
|
+
* another, or cannot say which it uses (review round 2, measured on Podman 5.8.1): rootless `podman ps -a` over another
|
|
279
|
+
* store (another HOME, XDG_DATA_HOME or storage.conf) answers exit 0 with an EMPTY list, which read as "not open" and
|
|
280
|
+
* deleted the directory under a sandbox opened from that store. The store is read once per pass, only when a podman run
|
|
281
|
+
* recorded one. A run from before the key is asked as it always was.
|
|
282
|
+
*
|
|
283
|
+
* FAIL CLOSED PER DIRECTORY, not per pass: a stale docker CLI with no daemon holds the docker runs and nothing else, so
|
|
284
|
+
* a podman-only host still sweeps its podman runs, and a host with no docker CLI at all asks docker only if a retained
|
|
285
|
+
* run says it ran there. A manifest with no `backend` key predates venue attribution and ran on `local`, which is how
|
|
286
|
+
* `sandboxVenueRefusal` opens it, so docker is asked for it. THROWS only when the retention root itself cannot be
|
|
287
|
+
* listed, which skips the whole pass, on `listRunningSandboxes`' rule.
|
|
288
|
+
*
|
|
289
|
+
* THE NETWORK SWEEP visits every runtime that has been PRESENT since this worker started: one a blessed venue names, or
|
|
290
|
+
* one a retained run has recorded in any pass. Cumulative on purpose, so the network of the LAST run a runtime held is
|
|
291
|
+
* still visited after that run's directory is gone. One sweeper per runtime, built once, and one failing does not stop
|
|
292
|
+
* the other. A host that blesses neither `local` nor `podman` and has retained nothing from either spawns neither CLI,
|
|
293
|
+
* which is the promise `start.mjs` makes for a host without Docker.
|
|
294
|
+
*
|
|
295
|
+
* Every log line is the family's `sandbox_reaper_skipped` (`OQ-007`'s one grep) with a fixed reason token and no CLI
|
|
296
|
+
* text and no path: `runtime-unanswered` per runtime with the number of directories it held, and `podman-store-mismatch`
|
|
297
|
+
* per directory held for its store.
|
|
298
|
+
*/
|
|
299
|
+
export function makeSandboxRuntimeWatch({ sandboxDir, blessed = [], fs = { lstatSync, readdirSync, readFileSync }, list = listRunningSandboxes, readPodmanStore = defaultPodmanStore, makeSweeper = makeSandboxNetworkSweeper, proxy = DEFAULT_EGRESS_PROXY, log = () => {} } = {}) {
|
|
300
|
+
const blessedBins = new Set(sandboxVenues(blessed).map((venue) => sandboxLauncher(venue).bin));
|
|
301
|
+
// Launcher-table order, so the asks and the log lines read the same on every pass.
|
|
302
|
+
const allBins = [...new Set(Object.values(SANDBOX_LAUNCHERS).map((l) => l.bin))];
|
|
303
|
+
const present = new Set(blessedBins);
|
|
304
|
+
const sweepers = new Map();
|
|
305
|
+
async function listRunning(pass = {}) {
|
|
306
|
+
// The reaper's ONE read per directory (`pass.reads`, review round 3): the watch places each run by exactly the
|
|
307
|
+
// read expiry will decide on. Standing alone (no pass handed in) it lists and reads for itself, the same way.
|
|
308
|
+
let names = pass?.names;
|
|
309
|
+
let reads = pass?.reads;
|
|
310
|
+
if (!Array.isArray(names) || !(reads instanceof Map)) {
|
|
311
|
+
try {
|
|
312
|
+
// A tombstone is a run already decided for deletion (issue #446), never one to place.
|
|
313
|
+
names = fs.readdirSync(sandboxDir).filter((name) => !isSandboxTombstone(name));
|
|
314
|
+
} catch (err) {
|
|
315
|
+
if (err?.code !== "ENOENT") throw err;
|
|
316
|
+
names = [];
|
|
317
|
+
}
|
|
318
|
+
reads = new Map(names.map((name) => [name, readRetained(fs, join(sandboxDir, name))]));
|
|
319
|
+
}
|
|
320
|
+
const byBin = new Map();
|
|
321
|
+
const unplaced = [];
|
|
322
|
+
const stores = [];
|
|
323
|
+
for (const name of names) {
|
|
324
|
+
const read = reads.get(name);
|
|
325
|
+
// A transient failure is the REAPER's hold, whatever any runtime says (nothing about the run is known). A
|
|
326
|
+
// manifest that does not parse, or cannot be read for good, is a run this cannot place, and so, since PR #466
|
|
327
|
+
// gate round 1, is a directory with NO manifest file: the reaper's `no-manifest` rule deletes it, and a
|
|
328
|
+
// `pi-sandbox-<name>` container can still have it mounted (measured: one running over a directory whose
|
|
329
|
+
// manifest was removed was deleted under it, the runtime never asked). Every runtime present is asked.
|
|
330
|
+
if (!read || read.transient) continue;
|
|
331
|
+
if (read.absent || !Object.hasOwn(read, "manifest")) {
|
|
332
|
+
unplaced.push(name);
|
|
333
|
+
continue;
|
|
334
|
+
}
|
|
335
|
+
const manifest = read.manifest;
|
|
336
|
+
const launcher = sandboxLauncher(sandboxVenueOf(manifest));
|
|
337
|
+
if (!launcher) {
|
|
338
|
+
unplaced.push(name);
|
|
339
|
+
continue;
|
|
340
|
+
}
|
|
341
|
+
if (!byBin.has(launcher.bin)) byBin.set(launcher.bin, []);
|
|
342
|
+
byBin.get(launcher.bin).push(name);
|
|
343
|
+
if (launcher.bin === "podman" && typeof manifest?.podmanStore === "string") stores.push([name, manifest.podmanStore]);
|
|
344
|
+
}
|
|
345
|
+
for (const bin of byBin.keys()) present.add(bin);
|
|
346
|
+
const held = new Set();
|
|
347
|
+
// Every runtime present is asked when a run could not be placed; otherwise only the runtimes a run recorded.
|
|
348
|
+
for (const bin of allBins) {
|
|
349
|
+
const entries = byBin.get(bin) ?? [];
|
|
350
|
+
const asked = entries.length > 0 || (unplaced.length > 0 && present.has(bin));
|
|
351
|
+
if (!asked) continue;
|
|
352
|
+
try {
|
|
353
|
+
for (const id of await list({ bin })) held.add(id);
|
|
354
|
+
} catch {
|
|
355
|
+
for (const entry of [...entries, ...unplaced]) held.add(entry);
|
|
356
|
+
log("sandbox_reaper_skipped", { reason: "runtime-unanswered", runtime: bin, held: entries.length + unplaced.length });
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
if (stores.length > 0) {
|
|
360
|
+
let current = null;
|
|
361
|
+
try {
|
|
362
|
+
current = await readPodmanStore();
|
|
363
|
+
} catch {
|
|
364
|
+
current = null;
|
|
365
|
+
}
|
|
366
|
+
for (const [name, recorded] of stores) {
|
|
367
|
+
if (current === recorded || held.has(name)) continue;
|
|
368
|
+
held.add(name);
|
|
369
|
+
log("sandbox_reaper_skipped", { entry: name, reason: "podman-store-mismatch" });
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
return [...held];
|
|
373
|
+
}
|
|
374
|
+
/**
|
|
375
|
+
* Whether ONE retained run's sandbox is open right now (issue #446, gate round 1): the reaper asks it immediately
|
|
376
|
+
* before renaming the run aside, because the pass's `listRunning` answer can be much older by then. Asked of the
|
|
377
|
+
* run's own runtime, or of every runtime present for a run it cannot place; THROWS when a runtime cannot answer,
|
|
378
|
+
* which the reaper holds. A directory with no manifest file is one it cannot place (PR #466 gate round 1): no NEW
|
|
379
|
+
* open can be on it (`resolveSandbox` refuses it), but one opened before its manifest went can still have it
|
|
380
|
+
* mounted, so every runtime present is asked, as `listRunning` asks.
|
|
381
|
+
*/
|
|
382
|
+
async function isOpen({ name, read } = {}) {
|
|
383
|
+
if (!read) return false;
|
|
384
|
+
const manifest = Object.hasOwn(read, "manifest") ? read.manifest : null;
|
|
385
|
+
const launcher = manifest ? sandboxLauncher(sandboxVenueOf(manifest)) : null;
|
|
386
|
+
const bins = launcher ? [launcher.bin] : allBins.filter((bin) => present.has(bin));
|
|
387
|
+
for (const bin of bins) if ((await list({ bin })).includes(name)) return true;
|
|
388
|
+
return false;
|
|
389
|
+
}
|
|
390
|
+
async function sweepNetworks(args) {
|
|
391
|
+
const bins = allBins.filter((bin) => present.has(bin));
|
|
392
|
+
if (bins.length === 0) return { swept: [], notes: [] };
|
|
393
|
+
for (const bin of bins) if (!sweepers.has(bin)) sweepers.set(bin, makeSweeper({ bin, proxy }));
|
|
394
|
+
return combineSandboxNetworkSweepers(bins.map((bin) => ({ runtime: bin, sweep: sweepers.get(bin) })))(args);
|
|
395
|
+
}
|
|
396
|
+
return { listRunning, sweepNetworks, isOpen };
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
/** The store this account's Podman uses right now, from one bounded `podman info`, or null when it cannot say. */
|
|
400
|
+
async function defaultPodmanStore() {
|
|
401
|
+
const read = await makePodmanInfoReader()();
|
|
402
|
+
return read?.answered === true ? (read.info?.graphRoot ?? null) : null;
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
/**
|
|
406
|
+
* The sweep's runner for one runtime: bounded, both streams, never throws. `execDockerBounded` already settles on its
|
|
407
|
+
* own timer and kills with SIGKILL, which matters here for the reason `retention-sweep.mjs` records -- this
|
|
408
|
+
* loop runs on a timer beside draining jobs, and `execFile`'s own `timeout` only signals and then still waits
|
|
409
|
+
* for `close`, so a CLI wedged on a dead socket never settles. Its rejection carries `stderr` on the error,
|
|
410
|
+
* which the "network is not there" rule needs and the bounded shape does not surface on its own.
|
|
411
|
+
*/
|
|
412
|
+
export function boundedRuntime(bin, { execFileFn } = {}) {
|
|
413
|
+
// `opts` (issue #452, gate round 3): the detach gate's runtime read asks with the facts readers' own bound (15 s,
|
|
414
|
+
// 1 MiB), which a `podman info` body needs; every network verb keeps the 10 s and 64 KiB it always had. `stderr` is
|
|
415
|
+
// the CLI's own, which `execDockerBounded` now returns: `execFile` hands it to the callback, never onto the error, so
|
|
416
|
+
// reading `error.stderr` got an empty string and the podman-docker `.Containers` fallback never fired here.
|
|
417
|
+
return async function bounded(args, opts = {}) {
|
|
418
|
+
// `withStderr`: matched by the "not found" and the podman-docker fallback rules, never logged.
|
|
419
|
+
const { code, stdout, stderr } = await execDockerBounded(args, { timeoutMs: opts.timeoutMs ?? 10_000, ...(opts.maxBuffer ? { maxBuffer: opts.maxBuffer } : {}), bin, withStderr: true, ...(execFileFn ? { execFileFn } : {}) });
|
|
420
|
+
return { code, stdout: String(stdout ?? ""), stderr: String(stderr ?? "") };
|
|
421
|
+
};
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
/**
|
|
425
|
+
* A network THIS project made for an operator session: the exact shape the producer builds. `docker`'s
|
|
426
|
+
* `--filter name=` is a SUBSTRING match, so the listing alone is not a namespace -- measured on 27.4.0 while
|
|
427
|
+
* fixing the same defect in the boot reaper (issue #357), where it returns an operator's own
|
|
428
|
+
* `my-pi-job-notes`. The capture is the sanitised job id, which is also what the retained directories and
|
|
429
|
+
* `listRunningSandboxes` are keyed by, so the three compare without a second grammar.
|
|
430
|
+
*
|
|
431
|
+
* Exported only so its test pins THIS constant. A test that rebuilds the pattern from the same two
|
|
432
|
+
* prefixes asserts a property of a string the test wrote, which is the drift a hand-written copy always
|
|
433
|
+
* has; the behavioural half of the same guard is the foreign names in the sweep's own fixtures.
|
|
434
|
+
*/
|
|
435
|
+
export const SANDBOX_NETWORK_SHAPE = new RegExp(`^${SANDBOX_NAME_PREFIX}(.*)${NETWORK_SUFFIX}$`);
|
|
436
|
+
|
|
437
|
+
/**
|
|
438
|
+
* The container states that leave a session network free. An ALLOWLIST rather than a denylist, so a state a
|
|
439
|
+
* future daemon adds is hands off by default: this decides whether something gets removed, and the safe
|
|
440
|
+
* direction is a leftover surviving one more pass. `--rm` means an exited sandbox is normally gone already.
|
|
441
|
+
*
|
|
442
|
+
* PODMAN'S `.State` VOCABULARY IS UNMEASURED, and that is a stated limit rather than a guess (issue #363).
|
|
443
|
+
* Docker 27.4.0 accepts `created`, `running`, `paused`, `exited`, `restarting`, `removing` and `dead` as
|
|
444
|
+
* `--filter status=` values, all lowercase and single-word (`status=bogus` is refused as an invalid filter). If a runtime renders `stopped` where docker renders `exited`, every leftover network on that
|
|
445
|
+
* host is held back forever behind a `sandbox-present` note: one network per run, named in the log, and the
|
|
446
|
+
* safe direction of the two. Adding a word on a guess is the other direction, removing networks on a
|
|
447
|
+
* vocabulary nobody has read, which is exactly what an allowlist is for.
|
|
448
|
+
*/
|
|
449
|
+
const SWEEPABLE_CONTAINER_STATES = new Set(["exited", "dead"]);
|
|
450
|
+
|
|
451
|
+
/**
|
|
452
|
+
* Whether `docker ps -a`'s output holds a container of OURS for `id` in a state that is not finished.
|
|
453
|
+
* `--filter name=` is as loose here as it is for networks, so an operator's own `my-pi-sandbox-notes` comes
|
|
454
|
+
* back from a filter on `pi-sandbox-notes` and must never be sliced into the id vocabulary: the comparison
|
|
455
|
+
* is against the whole name the producer builds.
|
|
456
|
+
*/
|
|
457
|
+
function containerHolds(stdout, id) {
|
|
458
|
+
const mine = `${SANDBOX_NAME_PREFIX}${id}`;
|
|
459
|
+
for (const line of String(stdout ?? "").split("\n")) {
|
|
460
|
+
const [name, state] = line.trim().split("\t");
|
|
461
|
+
// WHOLE name, never a prefix. `--filter name=` is not even a substring match, it is an UNANCHORED
|
|
462
|
+
// REGEX (measured: `name=pi-sandbox-a.c` returns `pi-sandbox-abc`, and `sanitizeJobId` permits `.`),
|
|
463
|
+
// so on this daemon `name=pi-sandbox-abc` comes back with an operator's `my-pi-sandbox-abc` AND with
|
|
464
|
+
// another run's `pi-sandbox-abcdef`, which is not an invented id shape (`gh-1` beside `gh-12` collides
|
|
465
|
+
// exactly so). Comparing the whole name the producer builds is what makes all of that harmless.
|
|
466
|
+
if (name !== mine) continue;
|
|
467
|
+
// Lowercased for the reason `networkAbsentInDaemonWords` is case-insensitive: this reads ANOTHER
|
|
468
|
+
// tool's rendering, and a runtime that capitalised it would hold every leftover back forever.
|
|
469
|
+
if (!SWEEPABLE_CONTAINER_STATES.has(String(state ?? "").trim().toLowerCase())) return true;
|
|
470
|
+
}
|
|
471
|
+
return false;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
/** Whether `ps -a`'s output holds a container of ours for `id` by its WHOLE name, in a finished state only. */
|
|
475
|
+
function containerFinished(stdout, id) {
|
|
476
|
+
const mine = `${SANDBOX_NAME_PREFIX}${id}`;
|
|
477
|
+
return String(stdout ?? "")
|
|
478
|
+
.split("\n")
|
|
479
|
+
.map((line) => line.trim().split("\t"))
|
|
480
|
+
.some(([name, state]) => name === mine && SWEEPABLE_CONTAINER_STATES.has(String(state ?? "").trim().toLowerCase()));
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
/**
|
|
484
|
+
* The default `retained`, and it THROWS rather than answering "nothing is retained". Its sibling default in
|
|
485
|
+
* `sandbox-store.mjs` can be a no-op because a missing sweeper means no sweep, which is safe; a missing
|
|
486
|
+
* directory listing means a sweep that ignores every retained run, which is the #277 harm with no log line.
|
|
487
|
+
* A caller that has no listing to give must say so by handing in `() => []`.
|
|
488
|
+
*/
|
|
489
|
+
function missingRetained() {
|
|
490
|
+
throw new Error("sweepSandboxNetworks: `retained` is required, and an empty listing must be passed deliberately");
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
/**
|
|
494
|
+
* Remove session networks whose run is gone (issue #337).
|
|
495
|
+
*
|
|
496
|
+
* WHAT THIS IS NOT. Issue #277 withdrew removing a network at OPEN time: two opens of the same run overlap
|
|
497
|
+
* more easily than that check assumed, and the second removed the first's network after the first had
|
|
498
|
+
* created it, disconnecting the proxy from a live shell. That decision stands. This is a different mechanism
|
|
499
|
+
* with a different input -- a background sweep on the retention reaper, keyed on the directory listing the
|
|
500
|
+
* pass STARTED with, which no open in progress can be absent from because `resolveSandbox` refuses a job
|
|
501
|
+
* whose directory is gone. Nothing here runs while an operator is opening anything.
|
|
502
|
+
*
|
|
503
|
+
* THE DANGEROUS VERB IS `detach`, NOT `rm`, for a RUNNING container. Measured on docker 27.4.0: a
|
|
504
|
+
* `network rm` of a network a running container is on FAILS ("has active endpoints"), so docker itself is
|
|
505
|
+
* the backstop for the removal there. What docker will not stop is stripping the proxy off that session,
|
|
506
|
+
* which is exactly the #277 harm, so the endpoint check gates what goes into the DETACH list, and the test
|
|
507
|
+
* asserts no `network disconnect` is issued rather than asserting the network survived -- the weaker
|
|
508
|
+
* assertion would pass on docker's refusal alone. The backstop does NOT extend to a container in `created`
|
|
509
|
+
* state, where the `rm` succeeds and leaves that container unable to start ever, which is why the last call
|
|
510
|
+
* before the removal is a `docker ps -a` for this id and not something inferred from the endpoint list.
|
|
511
|
+
*
|
|
512
|
+
* ORDER, since three of the four reads here only work in one arrangement: candidates first (`network ls`),
|
|
513
|
+
* then every piece of evidence that protects one, freshest last. An open creates its network BEFORE its
|
|
514
|
+
* container and AFTER the directory that made it legal, so evidence read before the candidate listing can be
|
|
515
|
+
* older than the thing it must protect.
|
|
516
|
+
*
|
|
517
|
+
* Returns `{ swept, notes }` rather than logging, so the reaper owns the log vocabulary and this stays a
|
|
518
|
+
* pure-ish function over its runner.
|
|
519
|
+
*/
|
|
520
|
+
export function makeSandboxNetworkSweeper({ bin = "docker", run = boundedRuntime(bin), proxy = DEFAULT_EGRESS_PROXY } = {}) {
|
|
521
|
+
// `bin` (issue #429) is the one runtime this sweeper lists, inspects and removes in: a sweep that listed under one
|
|
522
|
+
// CLI and removed under another would be the mixed venue every `bin` seam comment warns about. One sweeper per venue
|
|
523
|
+
// a sandbox can open on (`combineSandboxNetworkSweepers`). Podman's `ps -a` renders `{{.State}}`; its endpoint read is
|
|
524
|
+
// `networkEndpoints`' Podman path (issue #452: 4.9 renders no `.Containers`), whose `names` are the running and paused
|
|
525
|
+
// members exactly as docker's are, so every guard below reads the same on both. A stopped member on Podman is not in
|
|
526
|
+
// `names` and still makes the `rm` fail, which the failed-`rm` branch below handles for this run's own container and
|
|
527
|
+
// reports for anything else. Its state WORDS are this file's stated residual, below.
|
|
528
|
+
return async function sweepSandboxNetworks({ running = new Set(), keep = new Set(), retained = missingRetained, blocked = new Set() } = {}) {
|
|
529
|
+
// ONE detach gate per pass (issue #452, gate round 3): its runtime read is made once, however many networks.
|
|
530
|
+
const gate = makeDetachGate(run, { bin });
|
|
531
|
+
// CANDIDATES FIRST, then every piece of evidence that protects one. The order is the point, not an
|
|
532
|
+
// accident of writing: a network is created BEFORE the container that joins it and AFTER the directory
|
|
533
|
+
// that made the open legal, so evidence read before this listing can be older than the thing it is
|
|
534
|
+
// meant to protect. Read the other way round, anything that appears after this listing is simply not a
|
|
535
|
+
// candidate this pass.
|
|
536
|
+
const listed = await run(["network", "ls", "--filter", `name=${SANDBOX_NAME_PREFIX}`, "--format", "{{.Name}}"]);
|
|
537
|
+
// A listing that did not answer is a FAULT, not a verdict about any network, and the two carry
|
|
538
|
+
// different names on `OQ-007`'s property: the reaper turns `failed` into its family's
|
|
539
|
+
// `sandbox_reaper_skipped` line, while `notes` are per-network outcomes of a pass that ran. Reporting
|
|
540
|
+
// this as a note would say "one network was not reaped" about a look that saw none.
|
|
541
|
+
if (listed?.code !== 0) return { swept: [], notes: [], failed: "network-list-failed" };
|
|
542
|
+
|
|
543
|
+
// The retained directories as they are NOW, unioned by the caller with the listing its pass began
|
|
544
|
+
// with. Each half covers what the other cannot, and this is the half that has to be read here rather
|
|
545
|
+
// than handed in: `retainJobDir` creates a directory at job END, in this same process, so a run can be
|
|
546
|
+
// retained and opened while the pass is still going. A read that throws leaves this function, which is
|
|
547
|
+
// deliberate: half a keep set is worse than no sweep.
|
|
548
|
+
for (const name of retained()) keep.add(name);
|
|
549
|
+
|
|
550
|
+
const swept = [];
|
|
551
|
+
const notes = [];
|
|
552
|
+
for (const name of String(listed.stdout ?? "").split("\n").map((n) => n.trim()).filter(Boolean)) {
|
|
553
|
+
const m = SANDBOX_NETWORK_SHAPE.exec(name);
|
|
554
|
+
if (!m) continue; // the filter is not the namespace
|
|
555
|
+
const id = m[1];
|
|
556
|
+
// Either means the run is still reachable, and both are ordinary rather than notable: a shell is
|
|
557
|
+
// open on it, or its workspace is still retained and the next open will want this network's name
|
|
558
|
+
// free anyway. A line per retained run per pass would be noise.
|
|
559
|
+
if (running.has(id) || keep.has(id)) {
|
|
560
|
+
// SILENT for an ordinary retained run: a line per retained run per pass is noise, and that run's
|
|
561
|
+
// network is wanted. SAID when the directory that retains it could not be REMOVED this pass,
|
|
562
|
+
// which is not going to resolve by itself -- one note per stuck directory per pass, the same
|
|
563
|
+
// frequency as the `sandbox_reaper_skipped` line it pairs with, so nothing new is noisy (#363).
|
|
564
|
+
if (blocked.has(id)) notes.push({ network: name, reason: "directory-not-removed" });
|
|
565
|
+
continue;
|
|
566
|
+
}
|
|
567
|
+
const { ok, names, absent, parked = [] } = await networkEndpoints(run, name, { bin });
|
|
568
|
+
if (absent) continue;
|
|
569
|
+
if (!ok) {
|
|
570
|
+
notes.push({ network: name, reason: "unreadable" });
|
|
571
|
+
continue;
|
|
572
|
+
}
|
|
573
|
+
// One guard gates the DETACH: a session container attached means an operator may be inside it.
|
|
574
|
+
if (names.some((n) => n.startsWith(SANDBOX_NAME_PREFIX))) {
|
|
575
|
+
notes.push({ network: name, reason: "sandbox-attached" });
|
|
576
|
+
continue;
|
|
577
|
+
}
|
|
578
|
+
// ONE guard, asked LATE and asked TWICE (issue #363). It answers the question the two above cannot:
|
|
579
|
+
// is there a container of ours for this id in a state that is not finished? Measured on docker
|
|
580
|
+
// 27.4.0, a container between `docker create` and `docker start` is in `created` state, where
|
|
581
|
+
// `docker ps` does not list it, `network inspect` does not list it as an endpoint, AND `network rm`
|
|
582
|
+
// SUCCEEDS -- after which `docker start` fails with "network not found" and that sandbox can never
|
|
583
|
+
// run. The daemon backstops a RUNNING endpoint and nothing else.
|
|
584
|
+
//
|
|
585
|
+
// PASSED DOWN rather than asked once here, and what actually changed is worth stating exactly,
|
|
586
|
+
// because the obvious summary is wrong. The guard was ALREADY the statement immediately before the
|
|
587
|
+
// detach loop, so the first disconnect was zero commands after a fresh answer then and now. What
|
|
588
|
+
// moved is the `rm`: it sat k+1 commands out and is now always one. The issue says the detach is
|
|
589
|
+
// the act the guard does not DIRECTLY protect, and directly is the right word -- an earlier version
|
|
590
|
+
// of this comment said "not at all", which the command counts refute.
|
|
591
|
+
//
|
|
592
|
+
// ORDER IS THE POINT, and an earlier draft read this once at the TOP of the pass, which is the one
|
|
593
|
+
// placement that cannot work: an open creates its network BEFORE its container, so a snapshot taken
|
|
594
|
+
// before the candidate listing is older than the thing it has to protect, and a review pass drove
|
|
595
|
+
// exactly that -- 486 ms of exposure, the network removed and the operator's `docker run` dead with
|
|
596
|
+
// a 125.
|
|
597
|
+
//
|
|
598
|
+
// RESIDUAL, stated rather than implied: disconnect number i is still i commands after the guard. k
|
|
599
|
+
// is 1 in every shape this project produces (the proxy), and a per-endpoint re-ask would double the
|
|
600
|
+
// pass for a window no production shape opens. A check-then-act still has a gap; this makes it the
|
|
601
|
+
// width of one command instead of two plus k.
|
|
602
|
+
let lateReason = null;
|
|
603
|
+
const stillClear = async () => {
|
|
604
|
+
const held = await run(["ps", "-a", "--filter", `name=${SANDBOX_NAME_PREFIX}${id}`, "--format", "{{.Names}}\t{{.State}}"]);
|
|
605
|
+
// Fail CLOSED on both, and with their own tokens: the two ways this can answer badly have
|
|
606
|
+
// different causes and different fixes, and an operator grepping the log should not have to
|
|
607
|
+
// guess which one did not answer.
|
|
608
|
+
if (held?.code !== 0) {
|
|
609
|
+
lateReason = "containers-unreadable";
|
|
610
|
+
return false;
|
|
611
|
+
}
|
|
612
|
+
// Only a FINISHED container frees the network: `--rm` means an exited sandbox is normally gone
|
|
613
|
+
// already, so one still listed is abnormal and its network is a leftover either way. Every other
|
|
614
|
+
// state, and anything a future daemon adds, is hands off. Unlike the `keep` and `running` skips
|
|
615
|
+
// this one is SAID, because a container stuck in `created` would otherwise hold its network back
|
|
616
|
+
// forever with nothing on the host naming it.
|
|
617
|
+
if (containerHolds(held.stdout, id)) {
|
|
618
|
+
lateReason = "sandbox-present";
|
|
619
|
+
return false;
|
|
620
|
+
}
|
|
621
|
+
return true;
|
|
622
|
+
};
|
|
623
|
+
// A STOPPED egress proxy is detached too, as the boot reaper and the canary sweep detach one (issue #452, gate
|
|
624
|
+
// round 1). On Podman it is in `parked`, and Podman's `rm` refuses while it is attached (measured on 4.9.3 and
|
|
625
|
+
// 5.8.1), so leaving it kept a dead session's network forever, `rm-failed` on every pass. Only the proxy this
|
|
626
|
+
// worker is configured with, by name: any other stopped member is an operator's, and is named below instead.
|
|
627
|
+
// Docker's read carries no `parked`, so its pass is what it was.
|
|
628
|
+
const detach = [...names, ...parked.filter((n) => n === proxy)];
|
|
629
|
+
// Through the detach gate, as every detach is (issue #452 with #458, gate round 3): `names` are the RUNNING members,
|
|
630
|
+
// so a stopped proxy asks nothing; a refusal leaves the network whole and is said with the gate's token; a later
|
|
631
|
+
// pass with the keeper holding removes it. On both venues, since `local` can be a rootless Podman too.
|
|
632
|
+
const outcome = await removeNetworkOrSay(run, { network: name, detach, running: names, stillClear, bin, gate });
|
|
633
|
+
if (outcome.blocked) {
|
|
634
|
+
notes.push({ network: name, reason: outcome.blocked });
|
|
635
|
+
continue;
|
|
636
|
+
}
|
|
637
|
+
if (outcome.aborted) {
|
|
638
|
+
// `restored`/`lost` rather than `detached`: an endpoint put back was not removed by this pass,
|
|
639
|
+
// and one that could not be put back is off a network someone may be using, which is the half an
|
|
640
|
+
// operator has to act on.
|
|
641
|
+
notes.push({
|
|
642
|
+
network: name,
|
|
643
|
+
reason: lateReason,
|
|
644
|
+
...(outcome.restored?.length > 0 ? { restored: outcome.restored } : {}),
|
|
645
|
+
...(outcome.lost?.length > 0 ? { lost: outcome.lost } : {}),
|
|
646
|
+
});
|
|
647
|
+
continue;
|
|
648
|
+
}
|
|
649
|
+
if (outcome.absent) continue;
|
|
650
|
+
if (outcome.removed) swept.push({ network: name, detached: outcome.detached });
|
|
651
|
+
else {
|
|
652
|
+
// Podman keeps an EXITED container attached to its network (measured on 5.8.1: `network rm` then fails every
|
|
653
|
+
// pass, and the network is never reaped). So on a failed `rm` only, this run's OWN container, by its whole
|
|
654
|
+
// name and only in a finished state, is removed and the `rm` tried once more. Only on the failure path, so a
|
|
655
|
+
// pass that removes the network first time issues exactly the commands it always did; never `-f`, never a
|
|
656
|
+
// container in any other state, never another name (`containerFinished`'s whole-name rule).
|
|
657
|
+
const own = `${SANDBOX_NAME_PREFIX}${id}`;
|
|
658
|
+
const look = await run(["ps", "-a", "--filter", `name=${own}`, "--format", "{{.Names}}\t{{.State}}"]);
|
|
659
|
+
let retried = false;
|
|
660
|
+
if (look?.code === 0 && containerFinished(look.stdout, id) && (await run(["rm", own]))?.code === 0) {
|
|
661
|
+
retried = (await run(["network", "rm", name]))?.code === 0;
|
|
662
|
+
}
|
|
663
|
+
if (retried) swept.push({ network: name, detached: outcome.detached, removedContainer: own });
|
|
664
|
+
else {
|
|
665
|
+
// WHAT HOLDS IT, named (issue #452, gate round 1): `detached: []` on every pass said nothing an operator
|
|
666
|
+
// could act on. Read again now, after everything this pass did, through the same reader; bounded, as the
|
|
667
|
+
// boot reaper bounds its own list, because this goes into a log line. Container names only, which are the
|
|
668
|
+
// runtime's own vocabulary; an unreadable answer says so rather than naming nobody.
|
|
669
|
+
const after = await networkEndpoints(run, name, { bin });
|
|
670
|
+
const holding = after.ok ? [...after.names, ...(after.parked ?? [])] : null;
|
|
671
|
+
notes.push({
|
|
672
|
+
network: name,
|
|
673
|
+
reason: "rm-failed",
|
|
674
|
+
detached: outcome.detached,
|
|
675
|
+
...(holding === null ? { holding: "unreadable" } : { holding: holding.slice(0, 5), more: holding.length > 5 ? holding.length - 5 : 0 }),
|
|
676
|
+
});
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
// Between networks, for `retention-sweep.mjs`'s reason: this loop runs on a timer beside draining
|
|
680
|
+
// jobs, and `index.mjs` runs with `maxStalledCount: 0` against BullMQ's 30s lock.
|
|
681
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
682
|
+
}
|
|
683
|
+
return { swept, notes };
|
|
684
|
+
};
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
/**
|
|
688
|
+
* One sweep over every runtime's session networks (issue #429): `entries` is `[{ runtime, sweep }]`, each sweeper's `{
|
|
689
|
+
* swept, notes }` concatenated in order, and a listing that failed recorded as `{ reason, runtime }` in `failures`, EVERY
|
|
690
|
+
* one and not only the first (review round 2), so the reaper's log names which runtime did not answer. `failed` stays
|
|
691
|
+
* the first reason, for a caller that reads only that.
|
|
692
|
+
*
|
|
693
|
+
* Every sweeper is given the SAME `running`, `keep`, `retained` and `blocked`: the union of what every runtime holds is
|
|
694
|
+
* conservative in the one direction that matters (an id held in either runtime keeps its network in both), and `keep`
|
|
695
|
+
* is filled by each sweeper's own fresh read, which only ever adds. One runtime's failed listing does not stop the
|
|
696
|
+
* other's pass: nothing it would remove depends on the failed runtime, whose own networks are simply not candidates
|
|
697
|
+
* this pass.
|
|
698
|
+
*/
|
|
699
|
+
export function combineSandboxNetworkSweepers(entries) {
|
|
700
|
+
return async function sweepSandboxNetworks(args) {
|
|
701
|
+
const swept = [];
|
|
702
|
+
const notes = [];
|
|
703
|
+
const failures = [];
|
|
704
|
+
for (const { runtime, sweep } of entries) {
|
|
705
|
+
const outcome = await sweep(args);
|
|
706
|
+
swept.push(...(outcome?.swept ?? []));
|
|
707
|
+
notes.push(...(outcome?.notes ?? []));
|
|
708
|
+
if (outcome?.failed) failures.push({ reason: outcome.failed, runtime });
|
|
709
|
+
}
|
|
710
|
+
return failures.length > 0 ? { swept, notes, failed: failures[0].reason, failures } : { swept, notes };
|
|
711
|
+
};
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
/**
|
|
715
|
+
* The default `keeperCheck` (issue #452, gate round 2): the worker's own job preflight for the keeper (`keeperPreflight`,
|
|
716
|
+
* over this account's `podman info` and the keeper's and proxy's reads), with the proxy taken as up, since a missing
|
|
717
|
+
* proxy already fails the open at network creation with its own message. `null` when the keeper holds or is not needed.
|
|
718
|
+
* A keeper that is only too young is waited out once (issue #476), not refused.
|
|
719
|
+
*/
|
|
720
|
+
export async function sandboxKeeperCheck({ proxy, info = makePodmanInfoReader(), readKeeper = null, now = Date.now, sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)) } = {}) {
|
|
721
|
+
const ask = keeperPreflight(async () => ({ ok: true, proxy }), { armed: true, proxy, info, now, ...(readKeeper ? { readKeeper } : {}) });
|
|
722
|
+
let answer = await ask();
|
|
723
|
+
// Issue #476: a keeper whose only fault is its age is WAITED OUT, once and bounded (at most the minimum age plus the
|
|
724
|
+
// margin, 4 s), then judged again, rather than refused: an open right after the stack started would otherwise be sent
|
|
725
|
+
// to restart a keeper that holds. One still young after the wait restarted in it, and is refused as before.
|
|
726
|
+
if (answer?.young && typeof answer.young === "object") {
|
|
727
|
+
// `waitMs` is the judge's, bounded there (`netnsKeeperYoungWaitMs`: at most the minimum age plus the margin).
|
|
728
|
+
await sleep(answer.young.waitMs);
|
|
729
|
+
answer = await ask();
|
|
730
|
+
}
|
|
731
|
+
if (!answer?.unavailable) return null;
|
|
732
|
+
return {
|
|
733
|
+
refused: "netns-keeper-not-holding",
|
|
734
|
+
message: `${answer.cause}, so this sandbox is not opened: closing it tears its network down under the proxy, which is that same teardown. To fix it, ${answer.remedy}`,
|
|
735
|
+
};
|
|
736
|
+
}
|
|
737
|
+
|
|
738
|
+
/**
|
|
739
|
+
* Whether a retained run can be re-opened HERE, judged by the venue it ran in (issue #277). `null` when it
|
|
740
|
+
* can, else `{ refused, message }`. Exported for the admin panel, which asks before advertising the key.
|
|
741
|
+
*
|
|
742
|
+
* WHY THIS IS PER JOB. A sandbox opens a shell on THIS host, in the runtime the job ran in, against the job's retained
|
|
743
|
+
* directory, so it can reproduce only a run whose container was built here. The check used to be deployment-wide
|
|
744
|
+
* (refuse every sandbox when any blessed venue was remote) because the command takes a job id and could not learn its
|
|
745
|
+
* venue; the manifest now records it, and the answer belongs to the job.
|
|
746
|
+
*
|
|
747
|
+
* HELD MEANS A VENUE THIS FILE HAS A LAUNCHER FOR (`SANDBOX_LAUNCHERS`: `local` through the docker CLI, `podman`
|
|
748
|
+
* through this account's rootless podman), AND ONE `blessed` NAMES (issue #429). `blessed` is `PI_BACKENDS` as the
|
|
749
|
+
* CALLER's process reads it: the CLI's `loadConfig`, the panel's own environment (`sandboxVenuePolicy`). A shell whose
|
|
750
|
+
* `PI_BACKENDS` does not bless a venue has said it does not run that runtime, so reopening a run there would spawn a CLI
|
|
751
|
+
* that environment took out of service. It is NOT what keeps an open shell's directory from the retention reaper, and
|
|
752
|
+
* an earlier version of this comment said it was: the opener's `PI_BACKENDS` and the worker's routinely differ
|
|
753
|
+
* (`OQ-038`), so the reaper asks the runtime each retained run records instead, whatever either blesses
|
|
754
|
+
* (`makeSandboxRuntimeWatch`). It defaults to `local` alone, which is what `PI_BACKENDS` unset means, so a caller from
|
|
755
|
+
* before issue #429 refuses and admits exactly as it did.
|
|
756
|
+
*
|
|
757
|
+
* Not "any venue declaring `remote: false`": a future non-remote venue on a third runtime would pass that and be
|
|
758
|
+
* reopened under one of these two, reproducing a run from a runtime it never ran in. Such a venue must add its own
|
|
759
|
+
* launcher row, deliberately.
|
|
760
|
+
*
|
|
761
|
+
* A MANIFEST WITH NO `backend` KEY predates venue attribution and ran on `local` (`UNATTRIBUTED_BACKEND`).
|
|
762
|
+
* A key that is PRESENT but not a name is refused: `typeof` first, because the table's `backendFor(null)`
|
|
763
|
+
* returns `local`, and a null stamp means the venue was never known, not that it was local.
|
|
764
|
+
*/
|
|
765
|
+
export function sandboxVenueRefusal({ jobId, manifest, blessed = [DEFAULT_BACKEND] }) {
|
|
766
|
+
const venue = sandboxVenueOf(manifest);
|
|
767
|
+
if (typeof venue !== "string" || venue === "") {
|
|
768
|
+
return { refused: "venue-unreachable", message: `the manifest for ${jobId} names no backend, so this host cannot tell whether the run happened here` };
|
|
769
|
+
}
|
|
770
|
+
const launcher = sandboxLauncher(venue);
|
|
771
|
+
const held = Array.isArray(blessed) && blessed.includes(venue);
|
|
772
|
+
if (launcher && held) return null;
|
|
773
|
+
if (launcher) {
|
|
774
|
+
// A venue this file CAN open, on a host (as this process reads it) that does not bless it. The fix is an
|
|
775
|
+
// environment, and it is named with where it is read: the CLI and the panel read `PI_BACKENDS` from the process
|
|
776
|
+
// they run in, never from a deployment's `.env`, and "set it where you run this" is the whole remedy.
|
|
777
|
+
return {
|
|
778
|
+
refused: "venue-unreachable",
|
|
779
|
+
message: `${jobId} ran on the ${JSON.stringify(venue)} backend, which PI_BACKENDS in this process's environment does not bless (it reads ${JSON.stringify((Array.isArray(blessed) ? blessed : []).join(","))}), so no sandbox opens on it here: a sandbox opens only on a venue this host blesses, through that venue's own CLI (${launcher.bin}). Set PI_BACKENDS where you run this as the worker's is set; it is read from this environment, not from the deployment's .env.`,
|
|
780
|
+
};
|
|
781
|
+
}
|
|
782
|
+
return {
|
|
783
|
+
refused: "venue-unreachable",
|
|
784
|
+
// Names BOTH sides (issue #354): the venue that ran it, and the venues a sandbox opens through. "Not on this host's
|
|
785
|
+
// docker daemon" was true only while every venue but `local` was elsewhere.
|
|
786
|
+
message: `${jobId} ran on the ${JSON.stringify(venue)} backend, and a sandbox opens a shell only through a venue it has a launcher for on this host (${Object.entries(SANDBOX_LAUNCHERS).map(([name, l]) => `${JSON.stringify(name)} through the ${l.bin} CLI`).join(", ")}) against the retained directory, so it cannot reproduce that run. Open it on the venue that ran it.`,
|
|
787
|
+
};
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
/** The venue a retained manifest records: its `backend` key when present (whatever it holds), else `UNATTRIBUTED_BACKEND`. */
|
|
791
|
+
export function sandboxVenueOf(manifest) {
|
|
792
|
+
return Object.hasOwn(manifest ?? {}, "backend") ? manifest.backend : UNATTRIBUTED_BACKEND;
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
/**
|
|
796
|
+
* The refusals `resolveSandbox` decides from the MANIFEST ALONE, in the order an operator should read
|
|
797
|
+
* them. Not every manifest-only refusal on the `b` path: `decideSandboxJobUser` can refuse a run from
|
|
798
|
+
* `manifest.jobUser` too, and it stays where it is for two MEASURED reasons rather than the vaguer "it
|
|
799
|
+
* would change what the CLI refuses and when" this comment used to give (issue #367, item 5).
|
|
800
|
+
*
|
|
801
|
+
* IT IS NOT MANIFEST-ONLY. It returns `{ user: null, home: null }` on `darwin` and `win32` before it ever
|
|
802
|
+
* reads the stamp, so the SAME malformed `jobUser` refuses everywhere else and does not refuse on macOS or
|
|
803
|
+
* Windows. Measured across twelve platform strings: the split is darwin and win32 against every other
|
|
804
|
+
* value, linux, the BSDs, sunos, aix and the empty string alike, which is wider than "linux" and is why
|
|
805
|
+
* this says it that way. Folding it in would make this function's answer, and therefore whether the panel
|
|
806
|
+
* advertises `b` at all, depend on the operator's own OS for an identical run. Every other refusal here is
|
|
807
|
+
* a property of the run, and that is the whole of the argument.
|
|
808
|
+
*
|
|
809
|
+
* NOT because it is async, which is true of the function and is NOT a reason about this refusal: measured,
|
|
810
|
+
* `job-user-stamp-invalid` is reached with zero calls to the endpoint resolver, the daemon-facts reader
|
|
811
|
+
* and the image preflight, on every platform and every malformed shape. It is decided from the manifest
|
|
812
|
+
* before any await that does work. A well-formed stamp does reach the daemon, which is why this function
|
|
813
|
+
* stays async and out of a predicate called on every left and right between runs, but that is a cost of
|
|
814
|
+
* moving the WHOLE function, not of the refusal #367 item 5 is about.
|
|
815
|
+
*
|
|
816
|
+
* Extracted (issue #337) because the admin panel needs the same answer before it advertises `b`, and the
|
|
817
|
+
* alternative is the shape this file's own `openSandbox` docblock warns about: "Two callers assembling
|
|
818
|
+
* the same session from parts is how one of them drops a part." The panel had exactly that, checking the
|
|
819
|
+
* venue and not the other two, so a run whose manifest names no image was offered the key, given two
|
|
820
|
+
* lines of detail about the session it would get, and refused the moment the key was pressed.
|
|
821
|
+
*
|
|
822
|
+
* SYNCHRONOUS AND MANIFEST-ONLY is the boundary, not an accident of what fitted. The panel reads this
|
|
823
|
+
* inside a key handler, once per record; anything needing docker (a proxy that is not running, a sandbox
|
|
824
|
+
* already up) stays in `openSandbox` where it belongs and is `OQ-038`'s residual for the panel.
|
|
825
|
+
*
|
|
826
|
+
* The VENUE comes first, and that ordering is #277's: for a run from another venue the image and the
|
|
827
|
+
* workspace are symptoms, and the first refusal an operator reads should be the cause. A workspace that
|
|
828
|
+
* happens to exist at the same path on this host would otherwise pass and silently reproduce the wrong
|
|
829
|
+
* run.
|
|
830
|
+
*/
|
|
831
|
+
export function sandboxSyncRefusal({ jobId, manifest, fileExists = existsSync, blessed }) {
|
|
832
|
+
const venue = sandboxVenueRefusal({ jobId, manifest, blessed });
|
|
833
|
+
if (venue) return venue;
|
|
834
|
+
if (!manifest?.image) {
|
|
835
|
+
return { refused: "no-image", message: `the manifest for ${jobId} names no image, so the sandbox cannot reproduce the run` };
|
|
836
|
+
}
|
|
837
|
+
if (!manifest?.workspace || !fileExists(manifest.workspace)) {
|
|
838
|
+
// The common cause for a local run: the operator's folder moved or was deleted. Naming the path is
|
|
839
|
+
// the whole diagnosis, so name it.
|
|
840
|
+
return { refused: "workspace-gone", message: `the workspace for ${jobId} is no longer at ${manifest.workspace} — a local folder that moved cannot be re-opened` };
|
|
841
|
+
}
|
|
842
|
+
return null;
|
|
843
|
+
}
|
|
844
|
+
|
|
845
|
+
/**
|
|
846
|
+
* How close to its deadline a run may be and still open without `--pin` (issue #446): five minutes.
|
|
847
|
+
*
|
|
848
|
+
* WHY A MARGIN AND NOT THE DEADLINE ITSELF. The sweep holds a run whose sandbox a runtime reports open, but an open is
|
|
849
|
+
* not in any `ps` until its container starts, and the opener spends that stretch on its own runtime asks (the running
|
|
850
|
+
* check, the job user, the image, the network), each bounded in seconds. A sweep whose `ps` came before the container
|
|
851
|
+
* and whose clock (`at`, read after its asks) is past the deadline would still delete the directory under the new
|
|
852
|
+
* shell: window 1 of #446, reproduced. A run whose deadline is more than this far away cannot be past it by the time a
|
|
853
|
+
* pass that missed the container decides, so the refusal closes that window with room to spare; inside it, `--pin`
|
|
854
|
+
* writes a new deadline FIRST, which the sweep's own re-reads honour.
|
|
855
|
+
*/
|
|
856
|
+
export const SANDBOX_OPEN_GRACE_MS = 5 * 60 * 1000;
|
|
857
|
+
|
|
858
|
+
/**
|
|
859
|
+
* The refusal of a run past its deadline, or within `SANDBOX_OPEN_GRACE_MS` of it, unless the open pins it first
|
|
860
|
+
* (issue #446). `null` when the run may open.
|
|
861
|
+
*
|
|
862
|
+
* The deadline is `sandboxDeadline`'s, the sweep's own rule (the pin; else the earlier of the `retainUntil` the worker
|
|
863
|
+
* wrote and `createdAt` plus THIS opener's window), so an opener whose own PI_SANDBOX_RETENTION_HOURS is larger than
|
|
864
|
+
* the worker's is refused by the worker's deadline, not admitted by its own. The one case it cannot see, a worker
|
|
865
|
+
* whose window was lowered after the run was retained, is the post-launch look's (`openSandbox`).
|
|
866
|
+
* Shared by `resolveSandbox` and the admin panel's `readSandboxInfo`, which must not advertise `b` for a run the key
|
|
867
|
+
* press would refuse; the panel has no pin, so its refusal is this one and it names the CLI command.
|
|
868
|
+
*/
|
|
869
|
+
export function sandboxWindowRefusal({ jobId, manifest, retentionHours, at, pin = false }) {
|
|
870
|
+
if (pin) return null;
|
|
871
|
+
const { until } = sandboxDeadline(manifest, retentionHours);
|
|
872
|
+
const fix = `open it with \`pi-dispatch sandbox ${jobId} --pin\`, which extends its retention before anything starts`;
|
|
873
|
+
if (until === null) {
|
|
874
|
+
return { refused: "past-window", message: `the retained workspace for ${jobId} records no creation time, so the retention sweep deletes it on its next pass; ${fix}` };
|
|
875
|
+
}
|
|
876
|
+
if (until - at > SANDBOX_OPEN_GRACE_MS) return null;
|
|
877
|
+
const when = new Date(until).toISOString();
|
|
878
|
+
const where = until <= at ? `is past its retention window (it closed at ${when})` : `is within ${Math.round(SANDBOX_OPEN_GRACE_MS / 60000)} minutes of the end of its retention window (${when})`;
|
|
879
|
+
return {
|
|
880
|
+
refused: "past-window",
|
|
881
|
+
message: `the retained workspace for ${jobId} ${where}, so the retention sweep may delete it under the shell; ${fix}`,
|
|
882
|
+
};
|
|
883
|
+
}
|
|
884
|
+
|
|
139
885
|
/**
|
|
140
886
|
* Resolve one retained run into a launchable argv, or a NAMED refusal.
|
|
141
887
|
*
|
|
142
|
-
* Split out from the launch so both callers -- the CLI and the admin panel -- refuse identically,
|
|
888
|
+
* Split out from the launch so both callers -- the CLI and the admin panel -- refuse identically, the venue
|
|
889
|
+
* refusal included, and so
|
|
143
890
|
* the whole decision is testable without docker. Every refusal names what to do next, the posture
|
|
144
891
|
* `doctor` sets: a bare "not found" for a run the operator watched finish ten minutes ago is the least
|
|
145
892
|
* useful thing this could say.
|
|
146
893
|
*/
|
|
147
|
-
export function resolveSandbox({ jobId, sandboxDir, retentionHours, publish = [], fs, fileExists = existsSync }) {
|
|
894
|
+
export function resolveSandbox({ jobId, sandboxDir, retentionHours, publish = [], fs, fileExists = existsSync, blessed, now = Date.now, pin = false }) {
|
|
148
895
|
if (!jobId) return { refused: "no-job-id", message: "a job id is required (see `pi-dispatch sandbox --list`)" };
|
|
149
896
|
|
|
150
897
|
const manifest = readManifest({ sandboxDir, jobId, ...(fs ? { fs } : {}) });
|
|
898
|
+
if (manifest && typeof manifest.jobId === "string" && manifest.jobId !== String(jobId)) {
|
|
899
|
+
// Two ids can share a directory only by sharing a `sanitizeJobId` form (`a:b` and `a_b`), and the run in it is
|
|
900
|
+
// whichever was retained last (issue #446, gate round 1). Opening it under the other id would reproduce the wrong
|
|
901
|
+
// run, so the refusal names the run that is there; `--list` shows ids exactly as a run records them.
|
|
902
|
+
return { refused: "id-mismatch", message: `the retained workspace for ${jobId} holds run ${manifest.jobId}, not ${jobId} (two ids that differ only in characters a file name cannot hold share one directory); open it by the id \`pi-dispatch sandbox --list\` shows` };
|
|
903
|
+
}
|
|
151
904
|
if (!manifest) {
|
|
152
905
|
return retentionHours === 0
|
|
153
906
|
? { refused: "retention-off", message: "workspace retention is off — set PI_SANDBOX_RETENTION_HOURS to a positive number to make future runs resurrectable" }
|
|
154
|
-
:
|
|
907
|
+
: // Neutral about WHEN (#446 gate round 2): the window that swept it was the worker's, or the run's own recorded
|
|
908
|
+
// deadline, and this shell's `retentionHours` is neither.
|
|
909
|
+
{ refused: "absent", message: `no retained workspace for ${jobId}: it was swept at the end of its retention window, or the run predates retention (\`pi-dispatch sandbox --list\` shows what is left)` };
|
|
155
910
|
}
|
|
156
|
-
|
|
157
|
-
|
|
911
|
+
// The venue BEFORE the image and the workspace (#277): for a run from another venue those two are the
|
|
912
|
+
// symptoms, and the first refusal an operator reads should be the cause. A workspace that happens to
|
|
913
|
+
// exist at the same path on this host would otherwise pass and silently reproduce the wrong run.
|
|
914
|
+
const refusal = sandboxSyncRefusal({ jobId, manifest, fileExists, blessed });
|
|
915
|
+
if (refusal) return refusal;
|
|
916
|
+
// AFTER the run's own refusals (issue #446): a run from another venue, or with no image, cannot open pinned or not,
|
|
917
|
+
// and the operator should read that cause rather than be told to pin it.
|
|
918
|
+
const lapsed = sandboxWindowRefusal({ jobId, manifest, retentionHours, at: now(), pin });
|
|
919
|
+
if (lapsed) return lapsed;
|
|
920
|
+
|
|
921
|
+
// `venue` is the one the refusal above admitted, so it always has a launcher: every later step of the session reads
|
|
922
|
+
// its runtime from this one answer rather than re-deriving it from the manifest.
|
|
923
|
+
// `identity` (gate round 1): the retained directory's device and inode as it was resolved, so the post-launch look
|
|
924
|
+
// can tell the run's own directory from one that has REPLACED it at the same path (a retry's fresh run, a runtime's
|
|
925
|
+
// auto-created bind source on Docker). Null where the filesystem in use cannot say (an injected one without `lstatSync`).
|
|
926
|
+
//
|
|
927
|
+
// The container is named off the directory the run was FOUND in, not off the id (gate round 2): a run retained before
|
|
928
|
+
// the escape lives under its old name, and the sweep holds a run by its directory name, so a container named off
|
|
929
|
+
// the escaped id would never hold it. For every run retained since, the two are the same string.
|
|
930
|
+
return { manifest, name: `${SANDBOX_NAME_PREFIX}${basename(manifest.dir)}`, publish, venue: sandboxVenueOf(manifest), identity: dirIdentity(fs ?? { lstatSync }, manifest.dir) };
|
|
931
|
+
}
|
|
932
|
+
|
|
933
|
+
/** `{ dev, ino }` of `dir` by `lstat`, or null when the filesystem cannot say. Never throws. */
|
|
934
|
+
function dirIdentity(fs, dir) {
|
|
935
|
+
try {
|
|
936
|
+
const st = typeof fs?.lstatSync === "function" ? fs.lstatSync(dir) : null;
|
|
937
|
+
return st && Number.isFinite(st.ino) && Number.isFinite(st.dev) ? { dev: st.dev, ino: st.ino } : null;
|
|
938
|
+
} catch {
|
|
939
|
+
return null;
|
|
158
940
|
}
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
/**
|
|
944
|
+
* The egress posture a sandbox opened from `env` gets: `{ armed, proxy }`, through the same two readers the
|
|
945
|
+
* worker's config uses. THROWS on a malformed `PI_EGRESS`, exactly as the worker refuses to boot on one: a
|
|
946
|
+
* typo must never open a shell on the default bridge while the operator believes a policy is armed. For the
|
|
947
|
+
* admin panel, which deliberately never calls `loadConfig`; the CLI has the parsed config already.
|
|
948
|
+
*/
|
|
949
|
+
export function sandboxEgress(env) {
|
|
950
|
+
return { armed: egressArmed(env), proxy: egressProxyName(env) };
|
|
951
|
+
}
|
|
952
|
+
|
|
953
|
+
/**
|
|
954
|
+
* Open one retained run as an operator shell: the ONE path from a job id to a running sandbox, for the CLI
|
|
955
|
+
* and the admin panel alike (issue #277).
|
|
956
|
+
*
|
|
957
|
+
* WHY ONE FUNCTION. The CLI built the session's egress network, refused a sandbox already running and tore
|
|
958
|
+
* the network down in a finally; the panel called `resolveSandbox` and `buildSandboxRunArgs` directly and did
|
|
959
|
+
* none of it. So a sandbox opened from RUN_DETAIL ran on docker's default bridge -- the whole internet -- while
|
|
960
|
+
* `PI_EGRESS` was armed and INT-SANDBOX-CONTRACT said it lands on the network the job did. Two callers
|
|
961
|
+
* assembling the same session from parts is how one of them drops a part; this is where they now share every
|
|
962
|
+
* part that decides what the container can reach.
|
|
963
|
+
*
|
|
964
|
+
* In order: `resolveSandbox` (every refusal, the venue one and the past-window one included) -> the pin, when the
|
|
965
|
+
* caller asked for one (issue #446) -> the already-running refusal -> the argv, with this session's own network and
|
|
966
|
+
* proxy variables when armed -> create the network (a network already under this name is REFUSED and named, never
|
|
967
|
+
* removed) -> `beforeLaunch`, the caller's hook to print once the session is known to be openable, so after the last
|
|
968
|
+
* refusal (issue #462) and removed with the network if it throws -> launch, watched until
|
|
969
|
+
* the runtime lists it (issue #446) -> ask the runtime again, and remove the network in a finally unless the sandbox
|
|
970
|
+
* is still running. Returns `{ refused, message }`, or `{ code, error }` from the launch, with `detached: true` when
|
|
971
|
+
* the runtime still lists the sandbox as running after the shell returned (its network is left in place).
|
|
972
|
+
* THROWS when `egress.armed` is not a boolean.
|
|
973
|
+
*
|
|
974
|
+
* NOT here, deliberately: the terminal check and `--publish` parsing, which are about the CLI's own arguments. The
|
|
975
|
+
* pin IS here since issue #446 (`pin`), though only the CLI offers one, because WHEN it lands is the point: straight
|
|
976
|
+
* after `resolveSandbox` and before any runtime call, where `beforeLaunch` used to run it after the running ask and
|
|
977
|
+
* the job-user decision had each spent seconds of a window the sweep could close. A pin that fails REFUSES the open:
|
|
978
|
+
* the warning it used to print let a shell open over a run the sweep could take, and a `-v` bind of a directory that
|
|
979
|
+
* has gone mounts an empty one Docker creates (Podman refuses the bind, exit 125).
|
|
980
|
+
*
|
|
981
|
+
* EVERY RUNTIME STEP IS THE RUN'S VENUE'S (issue #429): the running asks, the job-user decision, the argv builder, the
|
|
982
|
+
* network's creation and removal and the launch all go through `SANDBOX_LAUNCHERS[resolved.venue]`, decided once by
|
|
983
|
+
* `resolveSandbox`. `running` is called with `{ bin }` and `launch` with `{ args, bin }`, and `beforeLaunch` is handed
|
|
984
|
+
* `runtime` so a caller's own lines (`docker attach`, `podman attach`) name the CLI that ran it. `blessed` and
|
|
985
|
+
* `backendFloor` are `PI_BACKENDS` and `PI_BACKEND_FLOOR` as the caller's process reads them; the first decides which
|
|
986
|
+
* venues open at all, the second what a podman sandbox's observations must show, exactly as for a job.
|
|
987
|
+
*/
|
|
988
|
+
export async function openSandbox({
|
|
989
|
+
jobId,
|
|
990
|
+
sandboxDir,
|
|
991
|
+
retentionHours,
|
|
992
|
+
publish = [],
|
|
993
|
+
term,
|
|
994
|
+
idleSeconds = 0,
|
|
995
|
+
egress,
|
|
996
|
+
running = listRunningSandboxes,
|
|
997
|
+
launch = launchSandbox,
|
|
998
|
+
spawnNetwork = spawn,
|
|
999
|
+
// Issue #341: `({ manifest }) => { user, home, relabel? } | { refused, message }` (`relabel`, issue #355, only ever true). Seamed like `launch`, so no test decides
|
|
1000
|
+
// a sandbox's user against a real daemon.
|
|
1001
|
+
resolveJobUser = decideSandboxJobUser,
|
|
1002
|
+
beforeLaunch = () => {},
|
|
1003
|
+
fs,
|
|
1004
|
+
fileExists,
|
|
1005
|
+
blessed,
|
|
1006
|
+
backendFloor = {},
|
|
1007
|
+
// Issue #446: `({ resolved }) => { pinned, keepUntil?, reason? }` when the caller asked for `--pin`, else null. Its
|
|
1008
|
+
// presence is also what admits a run past its window (`resolveSandbox`'s `pin`).
|
|
1009
|
+
pin = null,
|
|
1010
|
+
now = Date.now,
|
|
1011
|
+
// Issue #446: the post-launch look's two seams. `stop` removes this session's container, `pause` waits between asks.
|
|
1012
|
+
stop = stopSandbox,
|
|
1013
|
+
pause = launchWatchPause,
|
|
1014
|
+
// Issue #452, gate round 2: `({ proxy }) => null | { refused, message }`, asked before an egress-armed podman open.
|
|
1015
|
+
keeperCheck = sandboxKeeperCheck,
|
|
1016
|
+
// Issue #452, gate round 3: the detach gate this session's teardown asks, a seam for the tests; by default the one
|
|
1017
|
+
// `removeJobNetwork` builds over `spawnNetwork`, like every other teardown's.
|
|
1018
|
+
detachGate = null,
|
|
1019
|
+
// Issue #452, gate round 4: `(network, reason) => void`, told of a teardown the gate refused, which leaves the network.
|
|
1020
|
+
onNetworkKept = () => {},
|
|
1021
|
+
}) {
|
|
1022
|
+
// The posture is REQUIRED, and a boolean. Every other part of this function defaults safely; this one would
|
|
1023
|
+
// default to the open bridge, which is the dropped part this function exists to stop a caller dropping.
|
|
1024
|
+
if (typeof egress?.armed !== "boolean") {
|
|
1025
|
+
throw new Error("openSandbox: egress.armed must be a boolean -- a caller that does not say whether egress is armed must not get the default bridge");
|
|
1026
|
+
}
|
|
1027
|
+
const pinning = typeof pin === "function";
|
|
1028
|
+
const resolved = resolveSandbox({ jobId, sandboxDir, retentionHours, publish, blessed, now, pin: pinning, ...(fs ? { fs } : {}), ...(fileExists ? { fileExists } : {}) });
|
|
1029
|
+
if (resolved.refused) return resolved;
|
|
1030
|
+
// Admitted by `resolveSandbox`, so never null here.
|
|
1031
|
+
const { bin } = sandboxLauncher(resolved.venue);
|
|
1032
|
+
|
|
1033
|
+
// `--publish` AND AN ARMED POLICY ARE OPPOSITE DIRECTIONS, and docker resolves the contradiction SILENTLY
|
|
1034
|
+
// (issue #362). An armed policy puts this shell on its own `--internal` network, and a container attached
|
|
1035
|
+
// only to one publishes nothing: docker accepts `-p`, exits 0, and binds no host port. Measured on docker
|
|
1036
|
+
// 27.4.0: `docker ps --format {{.Ports}}` is empty and `docker port` prints nothing and exits 0. So the one
|
|
1037
|
+
// case the flag exists for, "start the app and click through it", did not work on a default deployment and
|
|
1038
|
+
// the CLI printed `published: ...` as though it had.
|
|
1039
|
+
//
|
|
1040
|
+
// REFUSED rather than repaired, and the alternative is named because it is the tempting one: attaching the
|
|
1041
|
+
// default bridge as a second network would make the flag work and would hand that session the whole
|
|
1042
|
+
// internet, which is the reach this session's own network exists to deny. Saying it in the CLI line and
|
|
1043
|
+
// leaving the behaviour was the third option and it keeps a flag that exits 0 and does nothing, which this
|
|
1044
|
+
// project calls the worst outcome available.
|
|
1045
|
+
//
|
|
1046
|
+
// HERE, and the position is load-bearing three ways. AFTER `resolveSandbox`, so a run that cannot be opened
|
|
1047
|
+
// at all says why first rather than being told about a flag. BEFORE the first docker ask, because a
|
|
1048
|
+
// determinate refusal must not cost a round trip. And BEFORE `beforeLaunch`, which is where the CLI prints
|
|
1049
|
+
// `published: ...`: one line later and it would print the false line and then refuse.
|
|
1050
|
+
if (resolved.publish.length > 0 && egress.armed === true) {
|
|
1051
|
+
// THE REASON DIFFERS BY RUNTIME, and the first version of this said the same thing of both, which a review measured
|
|
1052
|
+
// false (Podman 5.8.1, rootless, pasta): podman DOES publish on an `--internal` network, binding the host's
|
|
1053
|
+
// 127.0.0.1 and answering. The refusal stays on podman anyway, for the posture rather than the port: the flag is
|
|
1054
|
+
// documented as an egress-off feature on every venue, and an armed session is the one whose network is meant to
|
|
1055
|
+
// reach nothing but the proxy, so a published port there would be a second path in that no policy names. Where
|
|
1056
|
+
// the shell lands with the policy off differs too: the job's own `--network=private` on podman, a bridge on docker.
|
|
1057
|
+
const lands = bin === "podman" ? "this account's rootless podman's private network (`--network=private`, as a job with egress off)" : "docker's default bridge";
|
|
1058
|
+
const why =
|
|
1059
|
+
bin === "podman"
|
|
1060
|
+
? "this sandbox joins an `--internal` network, the session network the policy is meant to confine to its proxy, and podman would publish a host port into it anyway (it binds 127.0.0.1 there, measured), a path in that no policy names; a published port is an egress-off feature on every venue"
|
|
1061
|
+
: "this sandbox joins an `--internal` network, where docker accepts `-p`, exits 0 and binds no host port, so the flag would name a port that is not there";
|
|
1062
|
+
return {
|
|
1063
|
+
refused: "publish-needs-egress-off",
|
|
1064
|
+
message: `\`--publish\` is refused while the egress policy is armed: ${why}. Open this one with \`PI_EGRESS=0\` in the environment you run this from, and know what that buys: the shell lands on ${lands}, with the whole internet`,
|
|
1065
|
+
};
|
|
1066
|
+
}
|
|
1067
|
+
|
|
1068
|
+
// THE PIN, FIRST (issue #446): after the refusals that cost nothing, so only those come first (a refusal from a
|
|
1069
|
+
// runtime ask later, an already-running sandbox say, still leaves the pin, which is the operator's asked-for act),
|
|
1070
|
+
// and before the first runtime call. The sweep's re-reads (the fresh read, then the read through its tombstone)
|
|
1071
|
+
// honour a manifest that changed, so a pin written here holds the run against a pass already in flight; a pin
|
|
1072
|
+
// that cannot be written means the run is going or gone, and the open is refused rather than warned about.
|
|
1073
|
+
if (pinning) {
|
|
1074
|
+
let pinned;
|
|
1075
|
+
try {
|
|
1076
|
+
pinned = await pin({ resolved });
|
|
1077
|
+
} catch (err) {
|
|
1078
|
+
pinned = { pinned: false, reason: err?.message ?? "pin-failed" };
|
|
1079
|
+
}
|
|
1080
|
+
if (pinned?.pinned !== true) {
|
|
1081
|
+
const reason = pinned?.reason === "absent" ? "its retained workspace is gone, swept since it was read" : String(pinned?.reason ?? "unknown");
|
|
1082
|
+
return {
|
|
1083
|
+
refused: "pin-failed",
|
|
1084
|
+
message: `could not pin ${jobId} (${reason}), so the sandbox is not opened: an unpinned open of it could have its workspace deleted under the shell${pinned?.reason === "absent" ? "" : "; fix the cause and run it again"}`,
|
|
1085
|
+
};
|
|
1086
|
+
}
|
|
163
1087
|
}
|
|
164
1088
|
|
|
165
|
-
|
|
1089
|
+
// Whether THIS job's sandbox is running. `listRunningSandboxes` throws so the REAPER can tell "none" from
|
|
1090
|
+
// "could not ask"; here an unanswered ask costs only the early refusal (docker refuses a second container
|
|
1091
|
+
// under the same name anyway) and, after the launch, is treated as "not running" so the network is removed.
|
|
1092
|
+
// The retained directory's own name (issue #446), which is how the runtime reports this container.
|
|
1093
|
+
const id = basename(resolved.manifest.dir);
|
|
1094
|
+
// Asked of the run's OWN runtime (issue #429): a podman sandbox is not in `docker ps`, so asking docker would call
|
|
1095
|
+
// every podman session "not running", refuse nothing and tear a detached one's network down.
|
|
1096
|
+
const ask = (signal) => Promise.resolve().then(() => running(signal ? { bin, signal } : { bin })).then((ids) => ({ answered: true, live: new Set(ids) }), () => ({ answered: false, live: new Set() }));
|
|
1097
|
+
const before = await ask();
|
|
1098
|
+
if (before.live.has(id)) {
|
|
1099
|
+
return { refused: "already-running", message: `a sandbox for ${jobId} is already running; attach to it with \`${bin} attach ${resolved.name}\`, or exit it first` };
|
|
1100
|
+
}
|
|
1101
|
+
|
|
1102
|
+
// Issue #341: WHO the shell runs as, before anything is created. The daemon is this CLI's own; the uid is the
|
|
1103
|
+
// run's, off its manifest, because that uid owns the retained files.
|
|
1104
|
+
// The venue rides along (issue #429): a podman run is decided by the podman rules, from `podman info`, and refused
|
|
1105
|
+
// for what a podman job is refused for (`judgePodmanVenue`), before any network or container exists.
|
|
1106
|
+
const jobUser = await resolveJobUser({ manifest: resolved.manifest, venue: resolved.venue, backendFloor });
|
|
1107
|
+
if (jobUser?.refused) return { refused: jobUser.refused, message: jobUser.message };
|
|
1108
|
+
|
|
1109
|
+
// THE KEEPER, before anything exists (issue #452, gate round 2; #458). An egress-armed podman session ends in
|
|
1110
|
+
// `removeJobNetwork`, whose `network disconnect` of the running proxy is #458's trigger on Podman 4.x: without a
|
|
1111
|
+
// holding keeper, closing this shell would cut the proxy's route out for every job after it. So the open asks what a
|
|
1112
|
+
// job's preflight asks (`keeperPreflight`, with its age and order rules, since this admits a session that lasts) and
|
|
1113
|
+
// refuses with its reason and fix. On 5.x, and with the policy off, nothing is asked.
|
|
1114
|
+
if (egress.armed === true && bin === "podman") {
|
|
1115
|
+
const keeper = await keeperCheck({ proxy: egress.proxy });
|
|
1116
|
+
if (keeper?.refused) return keeper;
|
|
1117
|
+
}
|
|
1118
|
+
|
|
1119
|
+
// REQ-EGRESS-ALLOWLIST: this session's own network, exactly like a job's, named off its own container so the
|
|
1120
|
+
// reaper's `pi-job-` filter never touches it -- a worker restart must not tear the network out from under a
|
|
1121
|
+
// shell an operator is sitting in.
|
|
1122
|
+
const network = egress?.armed === true ? networkNameFor(resolved.name) : null;
|
|
1123
|
+
const args = buildSandboxRunArgs({
|
|
1124
|
+
venue: resolved.venue,
|
|
1125
|
+
image: resolved.manifest.image,
|
|
1126
|
+
name: resolved.name,
|
|
1127
|
+
workspace: resolved.manifest.workspace,
|
|
1128
|
+
jobDir: resolved.manifest.dir,
|
|
1129
|
+
publish: resolved.publish,
|
|
1130
|
+
term,
|
|
1131
|
+
idleSeconds,
|
|
1132
|
+
network,
|
|
1133
|
+
egressEnv: egressEnv({ proxy: egress?.proxy, armed: egress?.armed === true }),
|
|
1134
|
+
user: jobUser?.user ?? null,
|
|
1135
|
+
home: jobUser?.home ?? null,
|
|
1136
|
+
relabel: jobUser?.relabel === true,
|
|
1137
|
+
// By containment, the rule `rebaseWorkspace` already moves the retained clone by: a workspace inside the retained
|
|
1138
|
+
// job dir is the worker's own clone, one outside it is the operator's folder. Not by the manifest's `kind`, so a run
|
|
1139
|
+
// retained before a preparer moved its clone is still judged by where the files actually are.
|
|
1140
|
+
workspaceOwned: insideDir(resolved.manifest.dir, resolved.manifest.workspace),
|
|
1141
|
+
});
|
|
1142
|
+
|
|
1143
|
+
// No pre-spend gate here, deliberately: that is a MONEY gate and a sandbox spends nothing. A missing proxy
|
|
1144
|
+
// fails at network creation, in front of an operator at a terminal, which is the one place a late failure
|
|
1145
|
+
// is cheap.
|
|
1146
|
+
// In the run's runtime (issue #429): the network, the proxy's attachment and the container that joins it must all
|
|
1147
|
+
// live in ONE runtime, or `--network=` names a network the launching CLI has never heard of.
|
|
1148
|
+
if (network && !(await createJobNetwork(spawnNetwork, { network, proxy: egress.proxy, bin }))) {
|
|
1149
|
+
// A network ALREADY under this name is refused and named, never removed. It is either left by an earlier
|
|
1150
|
+
// session whose process died before its `finally` (a closed terminal, a SIGHUP -- pi's own handler exits
|
|
1151
|
+
// without unwinding), or a detached sandbox's that has since exited, or the network of an open of this
|
|
1152
|
+
// same run happening right now. Removing it automatically was tried under #277 and withdrawn: a second
|
|
1153
|
+
// open racing the first stripped the proxy from the first's live shell. Telling the two apart safely
|
|
1154
|
+
// needs more than this function can see, so the operator is told what it is and how to clear it. The
|
|
1155
|
+
// printed commands disconnect whatever is attached rather than the configured proxy, because a leftover
|
|
1156
|
+
// can carry a different one (a changed PI_EGRESS_PROXY, or two environments that disagree).
|
|
1157
|
+
// `createJobNetwork` rolls back a network it built itself, so one still present was almost always not
|
|
1158
|
+
// built here; the exception is a rollback whose own remove failed, which these commands also clear.
|
|
1159
|
+
if (await networkExists(spawnNetwork, network, { bin })) {
|
|
1160
|
+
// The member listing is per runtime (issue #452): Podman 4.9's `network inspect` renders no `.Containers`, so
|
|
1161
|
+
// the loop would disconnect nothing there, and its `network rm` refuses while a member in ANY state remains.
|
|
1162
|
+
// `ps -a --filter network=` names every member on both Podman versions (measured on 4.9.3 and 5.8.1). docker's
|
|
1163
|
+
// command is what it always was.
|
|
1164
|
+
const members = bin === "podman" ? `${bin} ps -a --filter network=${network} --format '{{.Names}}'` : `${bin} network inspect -f '{{range .Containers}}{{.Name}} {{end}}' ${network}`;
|
|
1165
|
+
return {
|
|
1166
|
+
refused: "egress-network-exists",
|
|
1167
|
+
// KEEPER FIRST on podman (issue #452, gate round 4): the loop below is a manual `network disconnect` of the running
|
|
1168
|
+
// proxy, which on Podman 4.x without the rootless network keeper is #458's trigger itself. docker's text is as it was.
|
|
1169
|
+
message: `the egress network ${network} already exists -- left by an earlier session of ${jobId} that did not clean up, or one opening right now. If \`pi-dispatch sandbox --list\` shows no sandbox running for it, ${bin === "podman" ? `first make sure the rootless network keeper is running (\`podman ps --filter name=^${NETNS_KEEPER}$\` lists it; if not, \`${NETNS_KEEPER_START}\`), because on Podman 4.x detaching the running proxy without it cuts the proxy's route out (issue #458); then ` : ""}disconnect whatever is attached and remove it: \`for c in $(${members}); do ${bin} network disconnect -f ${network} "$c"; done; ${bin} network rm ${network}\``,
|
|
1170
|
+
};
|
|
1171
|
+
}
|
|
1172
|
+
// Where the proxy comes from differs by venue: the compose file is docker-only, and on podman the proxy is
|
|
1173
|
+
// started by hand under this account's podman (docs/podman.md). On docker, `pi-dispatch up` and only that (issue #480,
|
|
1174
|
+
// PR #488's review; processor.mjs `egressProxyFix` says why a compose line here would be wrong), and a proxy
|
|
1175
|
+
// PI_EGRESS_PROXY names is the operator's own, which `up` never starts.
|
|
1176
|
+
const start = bin === "podman" ? "it runs under this account's rootless podman, started as docs/podman.md shows" : egress.proxy !== DEFAULT_EGRESS_PROXY ? `PI_EGRESS_PROXY names your own proxy, which \`pi-dispatch up\` does not start: start ${egress.proxy} yourself` : "`pi-dispatch up` from the deployment folder starts it";
|
|
1177
|
+
return {
|
|
1178
|
+
refused: "egress-network-failed",
|
|
1179
|
+
message: `could not create the egress network ${network} -- is the proxy running? ${start}. The egress setting is read from this process's environment (PI_EGRESS, PI_EGRESS_PROXY); a deployment that sets them only in its .env must export them where you run this.`,
|
|
1180
|
+
};
|
|
1181
|
+
}
|
|
1182
|
+
let detached = false;
|
|
1183
|
+
// THE POST-LAUNCH LOOK (issue #446). The refusal and the pin decide from what was on disk before the launch, and
|
|
1184
|
+
// two things can still take the run from under the new shell: the accepted residual (a worker whose window was
|
|
1185
|
+
// lowered below this opener's, whose pass missed the container), and the sweep's own rename and restore. So once
|
|
1186
|
+
// the runtime lists this sandbox, the run's own path is checked, and then KEPT checked for the life of the shell
|
|
1187
|
+
// (gate round 1: one check at listing was not enough, since a pass whose `ps` came before the container can reach
|
|
1188
|
+
// the directory later). The watching costs no runtime call: an `lstat` and a manifest read every
|
|
1189
|
+
// `SANDBOX_LAUNCH_WATCH_MS`. A run that is gone, or whose directory is no longer the one resolved (another inode at
|
|
1190
|
+
// the same path), has its container removed if that shows at the LAUNCH check, and is recorded and reported after
|
|
1191
|
+
// the shell exits if it shows later (gate round 2, below).
|
|
1192
|
+
//
|
|
1193
|
+
// CONCURRENT WITH THE SHELL, not ahead of it, and that is the stated limit: `run -it` hands the terminal over as the
|
|
1194
|
+
// container starts, and splitting it into a detached start and an attach would lose the prompt the shell prints
|
|
1195
|
+
// before the attach and changes the launch shape on two runtimes nobody has measured this on. Only a manifest file
|
|
1196
|
+
// that is NOT THERE, or a replaced directory, is a verdict: one that cannot be read for a moment, cannot be read for
|
|
1197
|
+
// good, or does not parse is not a verdict, since none of those says the run is gone. A shell that DETACHES is
|
|
1198
|
+
// no longer watched: the look ends with the shell's own return.
|
|
1199
|
+
let settled = false;
|
|
1200
|
+
let wake = () => {};
|
|
1201
|
+
const woken = new Promise((resolve) => {
|
|
1202
|
+
wake = resolve;
|
|
1203
|
+
});
|
|
1204
|
+
let swept = null;
|
|
1205
|
+
// STOPS WHEN THE SHELL RETURNS, promptly: the ask in flight is aborted (the default runner kills its `ps`), and the
|
|
1206
|
+
// look does not wait for it either way, so a `running` seam that ignores the signal cannot hold an exited shell for
|
|
1207
|
+
// the ask's own timeout.
|
|
1208
|
+
const abort = new AbortController();
|
|
1209
|
+
const shellBack = woken.then(() => null);
|
|
1210
|
+
// The injected `fs` when there is one (a test's), else the real one. One without `lstatSync` cannot look at all.
|
|
1211
|
+
const retainedFs = fs ?? { lstatSync, readFileSync };
|
|
1212
|
+
const runDir = resolved.manifest.dir;
|
|
1213
|
+
// `"swept"` (no manifest file at the run's path), `"replaced"` (a directory other than the one resolved), or null.
|
|
1214
|
+
const gone = () => {
|
|
1215
|
+
const read = readRetained(retainedFs, runDir);
|
|
1216
|
+
if (read.absent) return "swept";
|
|
1217
|
+
if (read.transient || !resolved.identity) return null;
|
|
1218
|
+
const now = dirIdentity(retainedFs, runDir);
|
|
1219
|
+
return now !== null && (now.dev !== resolved.identity.dev || now.ino !== resolved.identity.ino) ? "replaced" : null;
|
|
1220
|
+
};
|
|
1221
|
+
let lost = null;
|
|
1222
|
+
// Whether the look ever saw the container listed: a 125 WITHOUT that is the runtime refusing to start it.
|
|
1223
|
+
let everListed = false;
|
|
1224
|
+
const look = async () => {
|
|
1225
|
+
if (typeof retainedFs?.lstatSync !== "function") return;
|
|
1226
|
+
let listed = false;
|
|
1227
|
+
for (let i = 0; !listed && i < SANDBOX_LAUNCH_WATCH_TRIES; i++) {
|
|
1228
|
+
await Promise.race([pause(SANDBOX_LAUNCH_WATCH_MS), woken]);
|
|
1229
|
+
if (settled) return;
|
|
1230
|
+
const seen = await Promise.race([ask(abort.signal), shellBack]);
|
|
1231
|
+
if (settled || !seen) return;
|
|
1232
|
+
listed = seen.live.has(id);
|
|
1233
|
+
}
|
|
1234
|
+
everListed = listed;
|
|
1235
|
+
// THE LAUNCH CHECK removes the container: nothing an operator did can be in it yet, and a shell over an empty mount
|
|
1236
|
+
// is worse than none. AFTER it, a loss is RECORDED and never acted on (gate round 2): the shell may by then hold
|
|
1237
|
+
// state outside the mounts (processes, files under `/tmp`, an `apt install`), and a retry's `retainJobDir`
|
|
1238
|
+
// replacing the directory is an ordinary event that must not `rm -f` a working session. The operator is told when
|
|
1239
|
+
// the shell exits. Nothing reaps the replacing run meanwhile: this container still carries the run's name, so
|
|
1240
|
+
// every pass holds that directory as open.
|
|
1241
|
+
//
|
|
1242
|
+
// Only for a container the runtime LISTED: one never seen (its `ps` failing, or listed after the asking budget) is
|
|
1243
|
+
// not known to be the operator's fresh shell, so it is watched without being touched (gate round 3), for the whole
|
|
1244
|
+
// session, which costs an `lstat` and a manifest read per `SANDBOX_LAUNCH_WATCH_MS`.
|
|
1245
|
+
if (listed && gone()) {
|
|
1246
|
+
swept = { stopped: (await Promise.resolve().then(() => stop({ bin, name: resolved.name })).catch(() => false)) === true };
|
|
1247
|
+
return;
|
|
1248
|
+
}
|
|
1249
|
+
while (!settled) {
|
|
1250
|
+
await Promise.race([pause(SANDBOX_LAUNCH_WATCH_MS), woken]);
|
|
1251
|
+
if (settled) return;
|
|
1252
|
+
lost = gone();
|
|
1253
|
+
if (lost) return;
|
|
1254
|
+
}
|
|
1255
|
+
};
|
|
1256
|
+
try {
|
|
1257
|
+
// THE CALLER'S BANNER, AFTER THE LAST REFUSAL (issue #462): the CLI prints "opening ..." from here, and it ran before
|
|
1258
|
+
// the network was created, so a leftover network printed the banner and then refused. Inside the `try`, so a hook
|
|
1259
|
+
// that throws still has this session's network removed by the `finally`.
|
|
1260
|
+
await beforeLaunch({ resolved, args, network, runtime: bin });
|
|
1261
|
+
const watching = look().catch(() => {});
|
|
1262
|
+
let launched;
|
|
1263
|
+
try {
|
|
1264
|
+
launched = await launch({ args, bin });
|
|
1265
|
+
} finally {
|
|
1266
|
+
settled = true;
|
|
1267
|
+
abort.abort();
|
|
1268
|
+
wake();
|
|
1269
|
+
}
|
|
1270
|
+
await watching;
|
|
1271
|
+
const { code, error } = launched;
|
|
1272
|
+
if (swept) {
|
|
1273
|
+
const how = swept.stopped ? "was stopped" : `could not be stopped (\`${bin} rm -f ${resolved.name}\`)`;
|
|
1274
|
+
return {
|
|
1275
|
+
refused: "swept-at-launch",
|
|
1276
|
+
message: `the retained workspace for ${jobId} was deleted or replaced as this sandbox started, so the sandbox ${how}: its mounts were no longer the run's. \`pi-dispatch sandbox --list\` shows whether the run is still there to open again`,
|
|
1277
|
+
code: code ?? null,
|
|
1278
|
+
};
|
|
1279
|
+
}
|
|
1280
|
+
// A RUNTIME THAT REFUSED THE BIND (gate round 2, measured on podman): asked to mount a path that is not there,
|
|
1281
|
+
// podman refuses (`statfs ...: no such file or directory`, exit 125) where docker creates an empty directory. Its
|
|
1282
|
+
// words went to the terminal; the cause is the run's directory, so that launch is reported as swept rather than
|
|
1283
|
+
// left as a bare exit code. ONLY 125 and only a container never listed (gate round 3): any other exit is a shell
|
|
1284
|
+
// that ran, and a run gone by then is reported as lost below, from a check at exit.
|
|
1285
|
+
const lookable = typeof retainedFs?.lstatSync === "function";
|
|
1286
|
+
if (!error && code === 125 && !everListed && lookable && gone()) {
|
|
1287
|
+
return {
|
|
1288
|
+
refused: "swept-at-launch",
|
|
1289
|
+
message: `the retained workspace for ${jobId} was deleted or replaced as this sandbox started, so the sandbox could not start on it (the runtime refused the mount, exit ${code}). \`pi-dispatch sandbox --list\` shows whether the run is still there to open again`,
|
|
1290
|
+
code: code ?? null,
|
|
1291
|
+
};
|
|
1292
|
+
}
|
|
1293
|
+
// DETACHED, not exited: docker's detach sequence (Ctrl-P Ctrl-Q) returns with the container still running.
|
|
1294
|
+
// Tearing the network down then would strip the proxy from a live sandbox, so `docker attach` reopens a
|
|
1295
|
+
// shell with no egress at all. Leave it. An unanswered ask is NOT detached: the network is torn down, as it
|
|
1296
|
+
// always was. A network left this way outlives the sandbox, and the next open of this run names it.
|
|
1297
|
+
// Exit 0 as well: docker's detach returns 0, while a `docker run` that failed (125, a name taken by another
|
|
1298
|
+
// open of the same run with a different egress setting) must not be read as this session detaching.
|
|
1299
|
+
if (network && !error && code === 0) detached = (await ask()).live.has(id);
|
|
1300
|
+
// ONE MORE CHECK AT EXIT (gate round 3): the look can have missed a loss (it never saw the container listed, or
|
|
1301
|
+
// the loss landed after its last check), and the shell's return is when the operator is told.
|
|
1302
|
+
if (!lost && !error && lookable) lost = gone();
|
|
1303
|
+
const during = lost
|
|
1304
|
+
? {
|
|
1305
|
+
lost,
|
|
1306
|
+
message:
|
|
1307
|
+
lost === "replaced"
|
|
1308
|
+
? `the retained workspace for ${jobId} was REPLACED while this sandbox was open (a retry of the run, most likely), so what the shell had under /workspace was no longer the run on disk; nothing you saved there is in the new run's directory`
|
|
1309
|
+
: `the retained workspace for ${jobId} was DELETED while this sandbox was open (by the retention sweep, or by a retry of the run clearing it), so nothing you saved under /workspace survives; pin a run (\`--pin\`) before working in it late in its window`,
|
|
1310
|
+
}
|
|
1311
|
+
: {};
|
|
1312
|
+
return { code: code ?? null, error: error ?? null, ...(detached ? { detached: true } : {}), ...during };
|
|
1313
|
+
} finally {
|
|
1314
|
+
if (network && !detached) {
|
|
1315
|
+
// The runtime this session was ADMITTED on (issue #452, gate round 4), never a fresh read at the teardown; a
|
|
1316
|
+
// refused teardown is said with its token.
|
|
1317
|
+
await removeJobNetwork(spawnNetwork, {
|
|
1318
|
+
network,
|
|
1319
|
+
proxy: egress.proxy,
|
|
1320
|
+
bin,
|
|
1321
|
+
...(detachGate ? { gate: detachGate } : {}),
|
|
1322
|
+
...(jobUser?.runtime ? { readRuntime: async () => jobUser.runtime } : {}),
|
|
1323
|
+
onRefused: (reason) => onNetworkKept(network, reason),
|
|
1324
|
+
});
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
}
|
|
1328
|
+
|
|
1329
|
+
/** How often the post-launch look asks the runtime whether the sandbox is listed yet (issue #446). */
|
|
1330
|
+
export const SANDBOX_LAUNCH_WATCH_MS = 250;
|
|
1331
|
+
|
|
1332
|
+
/** How many times it asks before giving up the look: thirty seconds, well past a container start with `--pull=never`. */
|
|
1333
|
+
export const SANDBOX_LAUNCH_WATCH_TRIES = 120;
|
|
1334
|
+
|
|
1335
|
+
/** The look's default wait. `unref`, so a look still waiting never holds a process that is otherwise done. */
|
|
1336
|
+
function launchWatchPause(ms) {
|
|
1337
|
+
return new Promise((resolve) => setTimeout(resolve, ms).unref?.());
|
|
1338
|
+
}
|
|
1339
|
+
|
|
1340
|
+
/**
|
|
1341
|
+
* The argv that removes one session's container AT ONCE (issue #446, measured on pd-fedora in the #457 gate). A sandbox
|
|
1342
|
+
* runs an interactive bash under `--init`, which ignores SIGTERM, so `podman rm -f` without `--time=0` waits out
|
|
1343
|
+
* podman's 10 s stop timeout; the 10 000 ms bound below killed the `rm` first (rc 124 after 10004 ms) and the shell
|
|
1344
|
+
* stayed up over an emptied `/job`. `rm -f --time=0` took 106 ms. `-t, --time` is in podman 4.9.3, accepted beside
|
|
1345
|
+
* `--force` (doctor's `canaryProbeRemoval`, #431, rests on the same reading). docker's `rm -f` sends SIGKILL at once and
|
|
1346
|
+
* stays byte for byte what it was.
|
|
1347
|
+
*/
|
|
1348
|
+
export function sandboxRemovalArgs(bin, name) {
|
|
1349
|
+
return bin === "podman" ? ["rm", "-f", "--time=0", name] : ["rm", "-f", name];
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
/** Remove one session's container, bounded, never throws: true when the runtime said it did (issue #446). */
|
|
1353
|
+
export async function stopSandbox({ bin = "docker", name, run = execDockerBounded }) {
|
|
1354
|
+
try {
|
|
1355
|
+
const { code } = await run(sandboxRemovalArgs(bin, name), { timeoutMs: 10_000, bin });
|
|
1356
|
+
return code === 0;
|
|
1357
|
+
} catch {
|
|
1358
|
+
return false;
|
|
1359
|
+
}
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1362
|
+
/**
|
|
1363
|
+
* Which uid a re-opened sandbox runs as (issue #341), as `{ user, home }` or `{ refused, message }`.
|
|
1364
|
+
*
|
|
1365
|
+
* TWO SOURCES, on purpose. The DAEMON facts are this CLI's own (platform, endpoint, one `docker info`), because the
|
|
1366
|
+
* sandbox runs on whatever daemon this shell reaches. The IDENTITY is the run's, from the manifest stamp, because
|
|
1367
|
+
* that uid owns the retained files, and because the CLI's own uid need not be the worker's: `sudo pi-dispatch
|
|
1368
|
+
* sandbox` would otherwise be refused as a root worker, or would run as the wrong uid.
|
|
1369
|
+
*
|
|
1370
|
+
* A stamp that is present but malformed is REFUSED, never read as "no stamp": the manifest is host-written, so a
|
|
1371
|
+
* bad shape means something else wrote it. A run from before the stamp existed decides from the CLI's own ids.
|
|
1372
|
+
*
|
|
1373
|
+
* `venue` (issue #429) picks the rules: a `podman` run is decided by `decidePodmanSandboxJobUser` below, from `podman
|
|
1374
|
+
* info`, and everything after this line is `local`'s, unchanged.
|
|
1375
|
+
*/
|
|
1376
|
+
export async function decideSandboxJobUser(opts = {}) {
|
|
1377
|
+
if (opts?.venue === PODMAN_BACKEND) return decidePodmanSandboxJobUser(opts);
|
|
1378
|
+
return decideLocalSandboxJobUser(opts);
|
|
1379
|
+
}
|
|
1380
|
+
|
|
1381
|
+
const MALFORMED_STAMP = Object.freeze({ refused: "job-user-stamp-invalid", message: "the run's recorded job user is malformed, so the uid that owns its files is unknown; re-run the job instead" });
|
|
1382
|
+
|
|
1383
|
+
/**
|
|
1384
|
+
* The manifest's `jobUser` stamp, read: `{ malformed: true }`, `{ stamped: false }` (no stamp: a run from before it,
|
|
1385
|
+
* or a bare wiring), `{ stamped: true, user: null }` (the image's own user) or `{ stamped: true, user, uid, gid }`.
|
|
1386
|
+
* One reader for both venues, so a shape one refuses the other cannot quietly accept.
|
|
1387
|
+
*/
|
|
1388
|
+
function readJobUserStamp(stamp) {
|
|
1389
|
+
if (stamp === undefined || stamp === null) return { stamped: false };
|
|
1390
|
+
if (typeof stamp !== "object" || !("user" in stamp)) return { malformed: true };
|
|
1391
|
+
if (stamp.user === null) {
|
|
1392
|
+
if (stamp.home !== null && stamp.home !== undefined) return { malformed: true };
|
|
1393
|
+
return { stamped: true, user: null };
|
|
1394
|
+
}
|
|
1395
|
+
try {
|
|
1396
|
+
assertJobUser(stamp.user);
|
|
1397
|
+
} catch {
|
|
1398
|
+
return { malformed: true };
|
|
1399
|
+
}
|
|
1400
|
+
if (stamp.home !== CONTAINER_HOME) return { malformed: true };
|
|
1401
|
+
const [uid, gid] = stamp.user.split(":").map(Number);
|
|
1402
|
+
return { stamped: true, user: stamp.user, uid, gid };
|
|
1403
|
+
}
|
|
1404
|
+
|
|
1405
|
+
async function decideLocalSandboxJobUser({
|
|
1406
|
+
manifest,
|
|
1407
|
+
platform = process.platform,
|
|
1408
|
+
release = osRelease(),
|
|
1409
|
+
euid = process.geteuid?.(),
|
|
1410
|
+
egid = process.getegid?.(),
|
|
1411
|
+
resolveEndpoint = makeDockerEndpointResolver(),
|
|
1412
|
+
readFacts = makeDaemonFactsReader(),
|
|
1413
|
+
imageCapabilities = (image) => makeImagePreflight({ image })({}),
|
|
1414
|
+
stat,
|
|
1415
|
+
// Issue #448: the host files and `systemctl show podman.service` rootful Podman's containers.conf check reads, seamed as
|
|
1416
|
+
// the worker's are. Read only where this CLI's daemon is rootful Podman on this host.
|
|
1417
|
+
observationFs = { statSync, readFileSync, readdirSync },
|
|
1418
|
+
readPodmanService = makePodmanServiceReader(),
|
|
1419
|
+
env = process.env,
|
|
1420
|
+
} = {}) {
|
|
1421
|
+
// Issue #452, gate round 5: the facts read ONCE here even where the uid needs none of them (a VM-backed platform, an
|
|
1422
|
+
// endpoint on another machine), and carried beside the answer for the teardown's detach gate, which otherwise reads
|
|
1423
|
+
// `docker info` again at teardown and, if that read fails, keeps the network (the Docker Desktop case).
|
|
1424
|
+
const admittedOn = async () => {
|
|
1425
|
+
try {
|
|
1426
|
+
const read = await readFacts();
|
|
1427
|
+
return read?.answered ? runtimeFromFacts(read) : undefined;
|
|
1428
|
+
} catch {
|
|
1429
|
+
return undefined;
|
|
1430
|
+
}
|
|
1431
|
+
};
|
|
1432
|
+
if (platform === "darwin" || platform === "win32") return withRuntime({ user: null, home: null }, await admittedOn());
|
|
1433
|
+
const stamp = manifest?.jobUser;
|
|
1434
|
+
const read = readJobUserStamp(stamp);
|
|
1435
|
+
if (read.malformed) return { ...MALFORMED_STAMP };
|
|
1436
|
+
const identity = !read.stamped ? { euid, egid } : read.user === null ? { euid: SHIPPED_IMAGE_UID, egid: SHIPPED_IMAGE_UID } : { euid: read.uid, egid: read.gid };
|
|
1437
|
+
const endpoint = await resolveEndpoint();
|
|
1438
|
+
const daemon = endpoint?.local === false ? { answered: false, reason: "not-read", transient: true } : await readFacts();
|
|
1439
|
+
// A remote endpoint decides the uid from nothing the daemon says, but the teardown still detaches through its CLI.
|
|
1440
|
+
const remoteRuntime = endpoint?.local === false ? await admittedOn() : undefined;
|
|
1441
|
+
const socketPath = endpoint?.local === true && typeof endpoint.endpoint === "string" && endpoint.endpoint.startsWith("unix://")
|
|
1442
|
+
? endpoint.endpoint
|
|
1443
|
+
: daemon?.answered ? daemon.facts.remoteSocketPath : null;
|
|
1444
|
+
const socket = socketFacts(socketPath, stat ? { stat } : {});
|
|
1445
|
+
const decision = decideJobUser({ platform, release, ...identity, endpoint, daemon, socket });
|
|
1446
|
+
// Issue #355: this CLI's own daemon facts, the same read the uid came from, decide whether the shell's mounts carry
|
|
1447
|
+
// `:Z`. Added to the answer only when true, so every host it does not apply to returns the shape it always did.
|
|
1448
|
+
const relabel = relabelsPrivateMounts(daemon?.answered ? daemon.facts : null, endpoint, platform) ? { relabel: true } : {};
|
|
1449
|
+
// A run from before the stamp, opened with sudo: the root here is this shell's, not the worker's, so the worker's
|
|
1450
|
+
// fix text would send the operator to change the wrong thing.
|
|
1451
|
+
if (decision.cause === "worker-is-root" && (stamp === undefined || stamp === null)) {
|
|
1452
|
+
return { refused: "job-user-unmappable", message: "this run recorded no job user, so a sandbox opened as root cannot tell which uid owns its files; open it as the worker's own account (issue #341)" };
|
|
1453
|
+
}
|
|
1454
|
+
// The shared fixed texts, without the forge comment's "Refused:" lead: the CLI prints its own `error:`.
|
|
1455
|
+
if (decision.mode === "unmappable") return { refused: "job-user-unmappable", message: `${JOB_USER_FIX[decision.cause] ?? "the job user could not be decided"} (issue #341)` };
|
|
1456
|
+
if (decision.mode === "unknown") {
|
|
1457
|
+
return { refused: "job-user-unknown", message: `which uid the sandbox may run as could not be decided (${decision.reason}); is the docker daemon running?` };
|
|
1458
|
+
}
|
|
1459
|
+
// Issue #448: rootful Podman's own containers.conf reaches a sandbox exactly as it reaches a job (its `env` into the
|
|
1460
|
+
// shell's environment, its `annotations` into its groups), so a sandbox is refused for what a local job is refused
|
|
1461
|
+
// for, after the identity, as the podman venue's sandbox is (#428). A read that failed for a moment is refused in its
|
|
1462
|
+
// own words, since a sandbox has no queue to retry through. Nothing is read where the daemon is not rootful Podman here.
|
|
1463
|
+
const rootful = await observeRootfulConf({ endpoint, daemon, fs: observationFs, readService: readPodmanService, env });
|
|
1464
|
+
if (rootful?.refusal?.transient) {
|
|
1465
|
+
return { refused: "podman-conf-unread", message: `rootful Podman's containers.conf or podman.service could not be read just now, so whether it widens this sandbox is not known (${rootful.refusal.evidence}); try again` };
|
|
1466
|
+
}
|
|
1467
|
+
if (rootful?.refusal) return { refused: PODMAN_CONF_WIDENS_JOB, message: rootfulConfRefusal(rootful.refusal) };
|
|
1468
|
+
// `runtime` (issue #452, gate round 4): the facts this session was admitted on, for its teardown's detach gate.
|
|
1469
|
+
const runtime = daemon?.answered ? runtimeFromFacts(daemon) : remoteRuntime;
|
|
1470
|
+
if (decision.mode === "image") return withRuntime({ user: null, home: null, ...relabel }, runtime);
|
|
1471
|
+
const needsImage = identity.euid !== SHIPPED_IMAGE_UID;
|
|
1472
|
+
const caps = needsImage ? await imageCapabilities(manifest?.image) : { ok: true, capabilities: [] };
|
|
1473
|
+
if (needsImage && !caps?.ok) {
|
|
1474
|
+
return { refused: "job-user-image", message: `the retained image ${manifest?.image} could not be inspected, so whether it runs as another uid is unknown` };
|
|
1475
|
+
}
|
|
1476
|
+
const chosen = resolveImageUser(decision, { capabilities: caps.capabilities ?? [], euid: identity.euid, egid: identity.egid, socket });
|
|
1477
|
+
if (chosen.refused === "job-image-any-uid-unsupported") {
|
|
1478
|
+
return { refused: chosen.refused, message: `the retained image ${manifest?.image} does not declare anyUid, so it cannot run as the uid that owns this run's files (issue #341)` };
|
|
1479
|
+
}
|
|
1480
|
+
if (chosen.refused) return { refused: chosen.refused, message: `${JOB_USER_FIX[chosen.cause] ?? "the job user could not be decided"} (issue #341)` };
|
|
1481
|
+
return withRuntime({ user: chosen.user, home: chosen.home, ...relabel }, runtime);
|
|
1482
|
+
}
|
|
1483
|
+
|
|
1484
|
+
/**
|
|
1485
|
+
* The runtime a sandbox was admitted on, carried BESIDE the job-user answer rather than in it (issue #452, gate round 4):
|
|
1486
|
+
* non-enumerable, so the answer's shape, which callers compare and print, is what it always was, while `openSandbox`'s
|
|
1487
|
+
* teardown hands it to the detach gate and reads the daemon nothing more.
|
|
1488
|
+
*/
|
|
1489
|
+
function withRuntime(answer, runtime) {
|
|
1490
|
+
if (runtime !== undefined) Object.defineProperty(answer, "runtime", { value: runtime, enumerable: false });
|
|
1491
|
+
return answer;
|
|
1492
|
+
}
|
|
1493
|
+
|
|
1494
|
+
/**
|
|
1495
|
+
* Which uid a sandbox of a `podman` run runs as (issue #429), as `{ user, home, relabel? }` or `{ refused, message }`.
|
|
1496
|
+
* ALWAYS a user, as a podman job always has one: keep-id without `--user` runs the image's user with `/job` unreadable
|
|
1497
|
+
* (measured under issue #354), and `buildPodmanRunArgs` refuses the argv outright.
|
|
1498
|
+
*
|
|
1499
|
+
* THE UID IS THE ACCOUNT THAT OPENS IT, and it must be the run's. keep-id maps the host uid of the account running
|
|
1500
|
+
* `podman` into the container, and rootless Podman's store (the retained image with it) is that account's own. So,
|
|
1501
|
+
* unlike `local`, the stamp cannot choose another uid: a sandbox opened as another account would run as a uid that does
|
|
1502
|
+
* not own the retained files, from a store that may not hold the image. A stamp naming another uid is REFUSED with the
|
|
1503
|
+
* account to use, never "fixed" by passing the stamp's uid, which keep-id would map to a subordinate id that owns
|
|
1504
|
+
* nothing on the host. No stamp (a run from before it) decides from this process's ids, as `local` does.
|
|
1505
|
+
*
|
|
1506
|
+
* REFUSED FOR WHAT A JOB IS REFUSED FOR, IN THE JOB'S ORDER, by the job's own function (`judgePodmanVenue`): the
|
|
1507
|
+
* identity (not Linux, no podman, a remote service, rootful), then a containers.conf that widens a container
|
|
1508
|
+
* (`podman-conf-widens-job`, issue #428: its `pasta_options` reach a sandbox's network exactly as a job's, and its
|
|
1509
|
+
* `annotations` its groups), then the observations against `backendFloor`. A sandbox spends nothing, and these are
|
|
1510
|
+
* still not money gates: they are what the venue IS, and a shell over an agent-written workspace on a venue a job
|
|
1511
|
+
* would be refused on is the reach a sandbox exists not to widen. All of it before any network or container exists.
|
|
1512
|
+
*
|
|
1513
|
+
* sudo is refused FIRST, before `podman info` is asked: root's Podman is rootful and is not the worker's, so the
|
|
1514
|
+
* worker's fix text for a root worker would send the operator to change the wrong thing.
|
|
1515
|
+
*/
|
|
1516
|
+
async function decidePodmanSandboxJobUser({
|
|
1517
|
+
manifest,
|
|
1518
|
+
platform = process.platform,
|
|
1519
|
+
euid = process.geteuid?.(),
|
|
1520
|
+
egid = process.getegid?.(),
|
|
1521
|
+
backendFloor = {},
|
|
1522
|
+
readInfo = makePodmanInfoReader(),
|
|
1523
|
+
imageCapabilities = (image) => makeImagePreflight({ image, bin: "podman" })({}),
|
|
1524
|
+
fs,
|
|
1525
|
+
home,
|
|
1526
|
+
env,
|
|
1527
|
+
} = {}) {
|
|
1528
|
+
const read = readJobUserStamp(manifest?.jobUser);
|
|
1529
|
+
if (read.malformed) return { ...MALFORMED_STAMP };
|
|
1530
|
+
// Before any spawn: nothing podman could say changes it, and on macOS the CLI may well be a `podman machine` client.
|
|
1531
|
+
if (platform !== "linux") return { refused: "job-user-unmappable", message: `${PODMAN_JOB_USER_FIX["podman-platform"]} (issue #354)` };
|
|
1532
|
+
if (euid === 0) {
|
|
1533
|
+
return { refused: "job-user-unmappable", message: "a sandbox on the podman venue opens under the rootless Podman of the account that runs it, and root's is neither rootless nor the worker's; open it as the worker's own account (issue #429)" };
|
|
1534
|
+
}
|
|
1535
|
+
const info = await readInfo();
|
|
1536
|
+
const judged = judgePodmanVenue({ read: info, platform, euid, egid, backendFloor, ...(fs ? { fs } : {}), ...(home ? { home } : {}), ...(env ? { env } : {}) });
|
|
1537
|
+
if (judged.jobUserRefused) return { refused: "job-user-unmappable", message: `${PODMAN_JOB_USER_FIX[judged.jobUserRefused.cause] ?? "the job user could not be decided"} (issue #354)` };
|
|
1538
|
+
// A containers.conf that could not be read JUST NOW (issue #428's transient rule) is a job's retry; a sandbox has no
|
|
1539
|
+
// queue to retry through, so it is refused in its own words, naming the file, and the operator tries again.
|
|
1540
|
+
if (judged.podmanConfRefused?.transient) {
|
|
1541
|
+
return { refused: "podman-conf-unread", message: `the podman venue's containers.conf or running rootless network could not be read just now, so whether it widens this sandbox is not known (${judged.podmanConfRefused.evidence ?? "no file named"}); try again` };
|
|
1542
|
+
}
|
|
1543
|
+
if (judged.podmanConfRefused) return { refused: PODMAN_CONF_WIDENS_JOB, message: judged.podmanConfRefused.message };
|
|
1544
|
+
if (judged.unavailable) {
|
|
1545
|
+
// `file-unread` names the host file that could not be read (issue #428), which is the operator's to fix or retry.
|
|
1546
|
+
const what = judged.reason === "file-unread" && judged.message ? judged.message : `it did not answer (${judged.reason})`;
|
|
1547
|
+
return { refused: "podman-unobserved", message: `PI_BACKEND_FLOOR asks for what only an answered \`podman info\` and this host's podman files show, and ${what}; try again, and check \`podman info\` answers as this account` };
|
|
1548
|
+
}
|
|
1549
|
+
if (judged.refused) return { refused: "backend-floor", message: judged.message };
|
|
1550
|
+
const decision = decidePodmanJobUser({ platform, euid, egid, read: info });
|
|
1551
|
+
if (decision.mode !== "worker") {
|
|
1552
|
+
return { refused: "job-user-unknown", message: `which uid the sandbox may run as could not be decided (${decision.reason ?? decision.cause}); is podman answering \`podman info\` as this account?` };
|
|
1553
|
+
}
|
|
1554
|
+
// THE STORE (issue #429, review round 2, measured on Podman 5.8.1): another HOME, XDG_DATA_HOME or storage.conf is
|
|
1555
|
+
// another container store, where the run's image is absent and, sharper, where this sandbox's container would be one
|
|
1556
|
+
// the worker's `podman ps` cannot see, so the retention sweep would read "not open" and delete the directory under
|
|
1557
|
+
// it. A run that recorded its store opens only under that store. A run from before the key opens as it did.
|
|
1558
|
+
const recorded = typeof manifest?.podmanStore === "string" ? manifest.podmanStore : null;
|
|
1559
|
+
const current = info?.answered === true ? (info.info?.graphRoot ?? null) : null;
|
|
1560
|
+
if (recorded !== null && current !== recorded) {
|
|
1561
|
+
return {
|
|
1562
|
+
refused: "podman-store-mismatch",
|
|
1563
|
+
message: `this run's container store is ${recorded}, and the podman you are running uses ${current ?? "a store it did not report"}; open it as the account the worker runs as, with the worker's HOME and XDG_DATA_HOME (and no storage.conf of your own), so the worker's retention sweep can see the sandbox is open (issue #429)`,
|
|
1564
|
+
};
|
|
1565
|
+
}
|
|
1566
|
+
if (read.stamped && read.user !== decision.user) {
|
|
1567
|
+
const ran = read.user === null ? "as the image's own user, which no podman run does" : `as ${read.user}`;
|
|
1568
|
+
return {
|
|
1569
|
+
refused: "job-user-unmappable",
|
|
1570
|
+
message: `this run ran ${ran}, and a sandbox on the podman venue runs as the account that opens it (${decision.user}) under keep-id, in that account's own Podman store; open it as the account the worker runs as (issue #429)`,
|
|
1571
|
+
};
|
|
1572
|
+
}
|
|
1573
|
+
const caps = euid !== SHIPPED_IMAGE_UID ? await imageCapabilities(manifest?.image) : { ok: true, capabilities: [] };
|
|
1574
|
+
if (euid !== SHIPPED_IMAGE_UID && !caps?.ok) {
|
|
1575
|
+
return { refused: "job-user-image", message: `the retained image ${manifest?.image} could not be inspected in this account's podman store, so whether it runs as another uid is unknown` };
|
|
1576
|
+
}
|
|
1577
|
+
const chosen = resolvePodmanImageUser(decision, { capabilities: caps?.capabilities ?? [], euid, egid });
|
|
1578
|
+
if (chosen.refused === "job-image-any-uid-unsupported") {
|
|
1579
|
+
return { refused: chosen.refused, message: `the retained image ${manifest?.image} does not declare anyUid, so it cannot run as this account's uid, which the podman venue always uses (issue #354)` };
|
|
1580
|
+
}
|
|
1581
|
+
if (chosen.refused) return { refused: chosen.refused, message: `${PODMAN_JOB_USER_FIX[chosen.cause] ?? "the job user could not be decided"} (issue #354)` };
|
|
1582
|
+
if (chosen.unavailable) return { refused: "job-user-unknown", message: `which uid the sandbox may run as could not be decided (${chosen.reason}); is podman answering \`podman info\` as this account?` };
|
|
1583
|
+
// `relabel` on podman is `podman info`'s SELinux fact, the rule a podman job's own mounts follow (issue #355).
|
|
1584
|
+
// `runtime` (issue #452, gate round 4): the same read, for the session teardown's detach gate, so it reads nothing again.
|
|
1585
|
+
return withRuntime({ user: chosen.user, home: chosen.home, ...(chosen.relabel === true ? { relabel: true } : {}) }, { podman: true, rootless: info.info?.rootless ?? null, version: info.info?.version ?? null });
|
|
166
1586
|
}
|
|
167
1587
|
|
|
168
1588
|
/**
|
|
@@ -175,9 +1595,11 @@ export function resolveSandbox({ jobId, sandboxDir, retentionHours, publish = []
|
|
|
175
1595
|
* The caller owns the terminal around this: the CLI simply has one, and the admin panel brackets the call
|
|
176
1596
|
* with `tui.stop()`/`tui.start()`.
|
|
177
1597
|
*/
|
|
178
|
-
export function launchSandbox({ args, spawnFn = spawn }) {
|
|
1598
|
+
export function launchSandbox({ args, bin = "docker", spawnFn = spawn }) {
|
|
1599
|
+
// `bin` (issue #429) is the run's venue's CLI, handed in by `openSandbox`; the default keeps a caller from before it
|
|
1600
|
+
// on docker, which is what that caller built its argv for.
|
|
179
1601
|
return new Promise((resolve) => {
|
|
180
|
-
const child = spawnFn(
|
|
1602
|
+
const child = spawnFn(bin, args, { stdio: "inherit" });
|
|
181
1603
|
child.on("error", (err) => resolve({ code: null, error: err }));
|
|
182
1604
|
child.on("close", (code) => resolve({ code }));
|
|
183
1605
|
});
|