@edgehero/pi-dispatch 1.10.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +303 -150
- package/README.md +52 -0
- package/deploy/com.pi-dispatch.worker.plist +10 -4
- package/deploy/docker-compose.yml +49 -16
- package/deploy/egress-proxy.conf +32 -2
- package/deploy/nssm-install.cmd +12 -6
- package/deploy/pi-dispatch-egress-out.network +10 -0
- package/deploy/pi-dispatch-egress-proxy.container +50 -0
- package/deploy/pi-dispatch-netns-keeper.container +80 -0
- package/deploy/pi-dispatch-netns-keeper.network +18 -0
- package/deploy/pi-dispatch-valkey.container +51 -0
- package/deploy/pi-dispatch-valkey.network +16 -0
- package/deploy/receiver.service +6 -0
- package/deploy/worker-env-wrapper.cmd +12 -1
- package/deploy/worker-env-wrapper.sh +63 -37
- package/deploy/worker.service +18 -8
- package/package.json +15 -5
- package/src/azure-host.mjs +19 -0
- package/src/azure-identity.mjs +18 -2
- package/src/backend-conformance.mjs +71 -18
- package/src/backend-local.mjs +637 -21
- package/src/backend-podman.mjs +1168 -0
- package/src/backend-registry.mjs +86 -3
- package/src/backends.mjs +489 -37
- package/src/branch.mjs +7 -2
- package/src/cancel-cli.mjs +174 -0
- package/src/cancel-state.mjs +125 -0
- package/src/cli.mjs +188 -90
- package/src/config.mjs +503 -43
- package/src/connection.mjs +374 -8
- package/src/container-spec.mjs +102 -7
- package/src/daemon-facts.mjs +167 -0
- package/src/deployment-venue.mjs +158 -0
- package/src/docker-run.mjs +146 -15
- package/src/doctor.mjs +4756 -394
- package/src/egress-conf-copy.mjs +166 -0
- package/src/egress-proxy-state.mjs +151 -0
- package/src/egress.mjs +456 -25
- package/src/entry.mjs +27 -0
- package/src/env-allowlist.mjs +245 -40
- package/src/env-file.mjs +1869 -33
- package/src/exit-code.mjs +15 -0
- package/src/flow-gate.mjs +5 -3
- package/src/forgejo-host.mjs +19 -0
- package/src/forgejo-identity.mjs +21 -2
- package/src/get-token.mjs +67 -18
- package/src/git-dirty.mjs +9 -1
- package/src/git-hardening.mjs +33 -0
- package/src/github-app-setup.mjs +29 -12
- package/src/github-prompt.mjs +4 -1
- package/src/gitlab-host.mjs +19 -0
- package/src/gitlab-identity.mjs +19 -2
- package/src/host-pi.mjs +19 -3
- package/src/host-registry.mjs +29 -2
- package/src/identity.mjs +29 -4
- package/src/image-preflight.mjs +46 -11
- package/src/image-ref.mjs +21 -0
- package/src/index.mjs +363 -13
- package/src/init.mjs +197 -38
- package/src/job-user.mjs +252 -0
- package/src/json-duplicates.mjs +204 -0
- package/src/live-probes.mjs +1020 -0
- package/src/materialize.mjs +4 -11
- package/src/netns-keeper.mjs +264 -0
- package/src/on-failure.mjs +119 -0
- package/src/outbox.mjs +7 -0
- package/src/packages.mjs +2 -2
- package/src/podman-stack.mjs +1304 -0
- package/src/prepare-github.mjs +6 -6
- package/src/prepare-local.mjs +51 -17
- package/src/prepare.mjs +27 -6
- package/src/pricing.mjs +9 -5
- package/src/processor.mjs +506 -26
- package/src/provider-key.mjs +66 -0
- package/src/provider-steering.mjs +185 -0
- package/src/queue.mjs +35 -8
- package/src/redact.mjs +84 -0
- package/src/reserved-env.mjs +7 -3
- package/src/retention-sweep.mjs +178 -0
- package/src/run-container.mjs +181 -14
- package/src/run-history.mjs +105 -16
- package/src/runtime-observations.mjs +1152 -0
- package/src/runtime-settings.mjs +13 -8
- package/src/sandbox-cli.mjs +100 -95
- package/src/sandbox-store.mjs +612 -45
- package/src/sandbox.mjs +1459 -37
- package/src/schedules.mjs +16 -3
- package/src/secret-profiles.mjs +2 -1
- package/src/secrets.mjs +24 -6
- package/src/service-env.mjs +247 -0
- package/src/service.mjs +618 -28
- package/src/session-store.mjs +678 -53
- package/src/start.mjs +1348 -326
- package/src/subscriptions.mjs +7 -3
- package/src/transient.mjs +240 -0
- package/src/triggers-file.mjs +71 -15
- package/src/triggers.mjs +179 -19
- package/src/up.mjs +1399 -85
- package/src/valkey-auth.mjs +529 -0
- package/src/valkey-endpoint.mjs +367 -0
- package/src/watch-closer.mjs +158 -0
package/src/egress.mjs
CHANGED
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
|
+
import { makeDetachGate } from "./netns-keeper.mjs";
|
|
2
3
|
|
|
3
4
|
/**
|
|
4
5
|
* REQ-EGRESS-ALLOWLIST. The shipped egress policy: what a job container may talk to, expressed in the
|
|
5
6
|
* worker's own `docker run` argv rather than in a host firewall this process cannot see.
|
|
6
7
|
*
|
|
7
|
-
* This module imports nothing but `node:child_process`
|
|
8
|
+
* This module imports nothing but `node:child_process` and the two leaves under the detach gate (`netns-keeper.mjs`,
|
|
9
|
+
* `daemon-facts.mjs`, which import nothing else of this project's) -- deliberately, and for `image-preflight.mjs`'s
|
|
8
10
|
* exact reason. It holds a money gate: it decides whether a budget slot is spent, so its tests must run
|
|
9
11
|
* everywhere, unconditionally. It also owns every NAME the policy uses, so the gate that checks the proxy,
|
|
10
12
|
* the argv that joins the network and the env that points at the proxy are ONE answer by construction
|
|
@@ -35,8 +37,15 @@ import { spawn } from "node:child_process";
|
|
|
35
37
|
* the container env is a CLOSED allowlist, and the recipe's `PI_FORWARD_ENV` line names
|
|
36
38
|
* HTTPS_PROXY/HTTP_PROXY/NO_PROXY and NOT `NODE_USE_ENV_PROXY`, so the flag was set on the host and
|
|
37
39
|
* never reached the runner. Measured against the real provider through this proxy: 401 in 269ms.
|
|
38
|
-
* Hence a hostname allowlist and no address rule
|
|
39
|
-
* condition actually names
|
|
40
|
+
* Hence a hostname allowlist and no address rule that allows anything, which is the mechanism OQ-004's
|
|
41
|
+
* close condition actually names (the proxy's one address rule only denies this host's loopback and
|
|
42
|
+
* link-local addresses, issue #428).
|
|
43
|
+
* BUT THAT WAS MEASURED WITHOUT PI LOADED (issue #427). pi depends on npm `undici` (8.5.0 at the 0.80.7
|
|
44
|
+
* pin where this was measured, 8.10.2 at the 0.99.1 pin, where loading pi was re-measured to drop the
|
|
45
|
+
* env proxy the same way), whose load replaces the global dispatcher the flag installs with one that
|
|
46
|
+
* ignores the proxy variables, so in the runner the provider call went direct and every egress-armed job died at its first
|
|
47
|
+
* turn. The runner now re-installs an env-proxy dispatcher after loading pi
|
|
48
|
+
* (image/runner/src/env-proxy.mjs), and doctor's canary takes that same path instead of a plain fetch.
|
|
40
49
|
*/
|
|
41
50
|
|
|
42
51
|
/**
|
|
@@ -69,6 +78,16 @@ export function egressArmed(env) {
|
|
|
69
78
|
*/
|
|
70
79
|
export const DEFAULT_EGRESS_PROXY = "pi-dispatch-egress-proxy";
|
|
71
80
|
|
|
81
|
+
/**
|
|
82
|
+
* The proxy container this deployment names, from env: `PI_EGRESS_PROXY`, else the default. `||` rather than
|
|
83
|
+
* `??`, so an empty string falls back. The worker's config, `doctor` and the admin panel's sandbox all read it
|
|
84
|
+
* through here, so they cannot disagree about what a given environment means -- though each reads its OWN
|
|
85
|
+
* environment: the worker's comes from the deployment's .env, the panel's from wherever pi was started.
|
|
86
|
+
*/
|
|
87
|
+
export function egressProxyName(env) {
|
|
88
|
+
return env?.PI_EGRESS_PROXY || DEFAULT_EGRESS_PROXY;
|
|
89
|
+
}
|
|
90
|
+
|
|
72
91
|
/** The port squid listens on inside its container. Never published: reachable only from a job network. */
|
|
73
92
|
export const EGRESS_PROXY_PORT = 3128;
|
|
74
93
|
|
|
@@ -82,10 +101,12 @@ export const EGRESS_PROXY_PORT = 3128;
|
|
|
82
101
|
* second place for that reasoning to live and the copy that missed the next id shape would be the one
|
|
83
102
|
* nobody was looking at.
|
|
84
103
|
*
|
|
85
|
-
* It also inherits the namespace split for free. The boot reaper
|
|
86
|
-
* as a SUBSTRING,
|
|
87
|
-
*
|
|
88
|
-
*
|
|
104
|
+
* It also inherits the namespace split for free. The boot reaper narrows with `--filter name=pi-job-`, which
|
|
105
|
+
* docker matches as a SUBSTRING, and then decides on the name itself with `isJobNamespace` -- the filter is
|
|
106
|
+
* not the namespace, and since issue #357 it never was. Either way `pi-job-<id>-net` is swept and
|
|
107
|
+
* `pi-sandbox-<id>-net` is not, which is exactly the rule the container names already follow and for the
|
|
108
|
+
* same reason: a worker restart must not tear the network out from under a shell an operator is sitting in.
|
|
109
|
+
* A test pins both.
|
|
89
110
|
*/
|
|
90
111
|
export const NETWORK_SUFFIX = "-net";
|
|
91
112
|
|
|
@@ -93,6 +114,29 @@ export function networkNameFor(containerName) {
|
|
|
93
114
|
return `${containerName}${NETWORK_SUFFIX}`;
|
|
94
115
|
}
|
|
95
116
|
|
|
117
|
+
/**
|
|
118
|
+
* `pi-dispatch doctor`'s canary objects: one throwaway network per doctor PROCESS, and one probe container per
|
|
119
|
+
* direction on it. NAMED HERE rather than spelled inline in `doctor.mjs`, because after issue #350 the name has
|
|
120
|
+
* three consumers in that module -- the create, the teardown, and the anchored regex of the dead-pid sweep --
|
|
121
|
+
* and one outside it, `INT-EGRESS-POLICY-CONTRACT`'s object table.
|
|
122
|
+
*
|
|
123
|
+
* Both stay OUTSIDE `pi-job-` and `pi-sandbox-`, for the reason `NETWORK_SUFFIX` above already gives: those
|
|
124
|
+
* sweeps claim a PREFIX (the boot reaper's is `isJobNamespace`), and their `--filter` is wider still, so a
|
|
125
|
+
* canary name that fell inside either would be swept by a reaper that knows nothing about doctor.
|
|
126
|
+
*/
|
|
127
|
+
export const EGRESS_CANARY_NET_PREFIX = "pi-dispatch-egress-doctor-";
|
|
128
|
+
export const EGRESS_CANARY_PROBE_PREFIX = "pi-dispatch-egress-probe-";
|
|
129
|
+
|
|
130
|
+
/** The canary network for one doctor process. */
|
|
131
|
+
export function egressCanaryNetwork(pid) {
|
|
132
|
+
return `${EGRESS_CANARY_NET_PREFIX}${pid}`;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** One canary probe container, per direction and per doctor process. */
|
|
136
|
+
export function egressCanaryProbe(slug, pid) {
|
|
137
|
+
return `${EGRESS_CANARY_PROBE_PREFIX}${slug}-${pid}`;
|
|
138
|
+
}
|
|
139
|
+
|
|
96
140
|
/**
|
|
97
141
|
* How a container reaches the proxy: by NAME, resolved by docker's embedded DNS on the user-defined
|
|
98
142
|
* network. `docs/sandbox.md`'s recipe had to write a bare gateway IP because the DEFAULT bridge has no
|
|
@@ -139,7 +183,7 @@ export function egressEnv({ proxy = DEFAULT_EGRESS_PROXY, armed }) {
|
|
|
139
183
|
*
|
|
140
184
|
* ZERO spawns when unarmed, which is what makes a deployment without a policy pay nothing at all.
|
|
141
185
|
*
|
|
142
|
-
* The gate gets `
|
|
186
|
+
* The gate gets `Status` ("running"), not `Health`. A healthcheck is advisory and can flap; a money gate that refuses
|
|
143
187
|
* on a flapping signal silently drops real work, and one that retries on it burns the second budget slot
|
|
144
188
|
* this whole requirement exists to save. `doctor` reports health, where a human is reading.
|
|
145
189
|
*
|
|
@@ -148,14 +192,30 @@ export function egressEnv({ proxy = DEFAULT_EGRESS_PROXY, armed }) {
|
|
|
148
192
|
* party, which is not a thing to do before every job on every deployment. `doctor` does it once, when
|
|
149
193
|
* asked. What is left unproven is stated where an operator reads it rather than implied away.
|
|
150
194
|
*/
|
|
151
|
-
|
|
195
|
+
/** The proxy states that refuse a job (gate round 3's simple rule, beside `makeEgressPreflight`). */
|
|
196
|
+
// `created` among them (round-cap re-review): a container created and never started stays so until someone acts
|
|
197
|
+
// (measured on Docker Engine 29.8.1: a create that failed its mount stayed `created`, ExitCode 127).
|
|
198
|
+
export const STOPPED_PROXY_STATES = new Set(["paused", "exited", "dead", "created"]);
|
|
199
|
+
|
|
200
|
+
export function makeEgressPreflight({ proxy = DEFAULT_EGRESS_PROXY, armed = false, spawnFn = spawn, bin = "docker" } = {}) {
|
|
201
|
+
// Issue #354: `bin` is the venue's CLI, for both probes, so the proxy is looked for where the job will run.
|
|
152
202
|
return async function egressPreflight() {
|
|
153
203
|
if (!armed) return { ok: true };
|
|
154
|
-
|
|
204
|
+
// `.State.Status` "running" (issue #453): measured on Docker Engine 29.8.1, a paused and a restarting container both
|
|
205
|
+
// read `.State.Running` true, and neither carries a job's traffic; the status word tells them apart.
|
|
206
|
+
const probe = await runDocker(spawnFn, ["inspect", "--format={{.State.Status}}", proxy], true, bin);
|
|
155
207
|
if (probe.code === 0) {
|
|
156
|
-
|
|
208
|
+
const status = probe.stdout.trim();
|
|
209
|
+
// THE SIMPLE RULE (issue #453, gate round 3 and the re-review): only `running` admits; `paused`, `exited`, `dead`
|
|
210
|
+
// and `created` are a stopped proxy, which stays so until someone acts, so the job is refused; EVERY other word
|
|
211
|
+
// (restarting, stopping, removing, initialized, podman's `stopped` between restarts, one no runtime prints yet)
|
|
212
|
+
// is an infra retry with the state in its words. BullMQ gives a job two attempts, so that is one retry, then
|
|
213
|
+
// the job fails: a retry buys a moment, not a wait.
|
|
214
|
+
if (status === "running") return { ok: true, proxy };
|
|
215
|
+
if (STOPPED_PROXY_STATES.has(status)) return { proxyStopped: proxy };
|
|
216
|
+
return { unavailable: proxy, state: /^[a-z]{1,20}$/.test(status) ? status : "unreported" };
|
|
157
217
|
}
|
|
158
|
-
if ((await runDocker(spawnFn, ["info"])).code === 0) return { proxyMissing: proxy };
|
|
218
|
+
if ((await runDocker(spawnFn, ["info"], false, bin)).code === 0) return { proxyMissing: proxy };
|
|
159
219
|
return { unavailable: proxy };
|
|
160
220
|
};
|
|
161
221
|
}
|
|
@@ -170,25 +230,396 @@ export function makeEgressPreflight({ proxy = DEFAULT_EGRESS_PROXY, armed = fals
|
|
|
170
230
|
* moments ago, with these flags. `doctor` reads back the proxy's own attachments, where an operator's
|
|
171
231
|
* hand-built estate is what is being checked.
|
|
172
232
|
*/
|
|
173
|
-
export async function createJobNetwork(spawnFn, { network, proxy = DEFAULT_EGRESS_PROXY }) {
|
|
174
|
-
|
|
175
|
-
|
|
233
|
+
export async function createJobNetwork(spawnFn, { network, proxy = DEFAULT_EGRESS_PROXY, bin = "docker" }) {
|
|
234
|
+
// `bin` (issue #354) is the venue's CLI: the network, the proxy's attachment and the container that joins it must all
|
|
235
|
+
// live in ONE runtime, or the job's `--network=` names a network its daemon has never heard of.
|
|
236
|
+
return createJobNetworkWith(spawnRunner(spawnFn, bin), { network, proxy, bin });
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* `createJobNetwork` over ANY docker runner, `(args) => Promise<{ code }>` (issue #344). The job path's spawn and
|
|
241
|
+
* `doctor --live`'s bounded runner are two runners, and a probe that built its networks with a second copy of this
|
|
242
|
+
* sequence would be reading back a network no job gets. Never throws: a runner that throws is a failed step.
|
|
243
|
+
*/
|
|
244
|
+
export async function createJobNetworkWith(docker, { network, proxy = DEFAULT_EGRESS_PROXY, bin = "docker" }) {
|
|
245
|
+
if ((await runWith(docker, ["network", "create", "--internal", network]))?.code !== 0) return false;
|
|
246
|
+
if ((await runWith(docker, ["network", "connect", network, proxy]))?.code !== 0) {
|
|
176
247
|
// Roll back rather than leave a network the proxy cannot serve: a half-built policy that admits a
|
|
177
|
-
// job is worse than one that refuses it.
|
|
178
|
-
|
|
248
|
+
// job is worse than one that refuses it. The connect FAILED, so the proxy is not on it: nothing running is
|
|
249
|
+
// detached, and the gate asks nothing (issue #452).
|
|
250
|
+
await removeJobNetworkWith(docker, { network, proxy, bin, proxyRunning: false });
|
|
179
251
|
return false;
|
|
180
252
|
}
|
|
181
253
|
return true;
|
|
182
254
|
}
|
|
183
255
|
|
|
184
256
|
/**
|
|
185
|
-
*
|
|
186
|
-
*
|
|
187
|
-
|
|
257
|
+
* Whether a network by this name exists. `false` when docker says no or cannot be asked; callers use it only
|
|
258
|
+
* to explain a failure they already have, never to decide to remove anything.
|
|
259
|
+
*/
|
|
260
|
+
export async function networkExists(spawnFn, network, { bin = "docker" } = {}) {
|
|
261
|
+
return (await runDocker(spawnFn, ["network", "inspect", network], false, bin)).code === 0;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Detach the proxy and remove the network. Best-effort and never throws: it runs in a `finally`, after the
|
|
266
|
+
* container has exited (or when a network has just been built for a container that will not start), and a
|
|
267
|
+
* failure here must not change the outcome. What it leaves behind if it fails is a network, which the boot
|
|
268
|
+
* reaper tries to remove for a job (any name under the `pi-job-` prefix since issue #360, not only the
|
|
269
|
+
* `<container>-net` one this builds, detaching what is attached first since issue #357) and never for a sandbox
|
|
270
|
+
* (`pi-sandbox-`); a sandbox's next open of the same run refuses and names it for removal.
|
|
271
|
+
*/
|
|
272
|
+
export async function removeJobNetwork(spawnFn, { network, proxy = DEFAULT_EGRESS_PROXY, bin = "docker", gate = null, readRuntime = null, onRefused = null }) {
|
|
273
|
+
// `readRuntime` (issue #452, gate round 4): the runtime the job or session was ADMITTED on, so a teardown does not read
|
|
274
|
+
// it again. A fresh `docker info` at every teardown that timed out, failed or answered `ServerErrors` read as
|
|
275
|
+
// `runtime-unreadable` on Docker Engine, where no keeper exists, and left the job's network behind every time
|
|
276
|
+
// (measured); a pool of leaked networks ends in every egress job failing.
|
|
277
|
+
const runner = spawnRunner(spawnFn, bin);
|
|
278
|
+
return removeJobNetworkWith(runner, { network, proxy, bin, gate: gate ?? makeDetachGate(runner, { bin, ...(readRuntime ? { readRuntime } : {}) }), ...(onRefused ? { onRefused } : {}) });
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* `removeJobNetwork` over any docker runner (issue #344). Detaches ONLY the proxy, then `network rm` without `-f`, so a
|
|
283
|
+
* network something else is still attached to stays, rather than being pulled out from under it. Resolves whether
|
|
284
|
+
* the network is gone.
|
|
285
|
+
*
|
|
286
|
+
* Through the detach gate (issue #452, gate round 3), as every detach is: on a rootless Podman 4.x whose rootless
|
|
287
|
+
* network keeper does not hold, the proxy is NOT detached and the network is left whole (its `rm` would refuse with the
|
|
288
|
+
* proxy on it anyway), for the boot reaper to remove once the keeper holds. A job or a session is not started there in
|
|
289
|
+
* the first place (the podman venue's egress preflight and the sandbox opener refuse; `local` refuses every rootless
|
|
290
|
+
* daemon), so this is the case of a keeper that died mid-run, where leaving the network is the only safe teardown.
|
|
291
|
+
* `proxyRunning: false` (the create's rollback, whose connect failed) asks nothing.
|
|
292
|
+
*/
|
|
293
|
+
export async function removeJobNetworkWith(docker, { network, proxy = DEFAULT_EGRESS_PROXY, bin = "docker", gate = makeDetachGate(docker, { bin }), proxyRunning = true, onRefused = null }) {
|
|
294
|
+
const { blocked } = await detachEndpoints(docker, { network, endpoints: [proxy], running: proxyRunning ? [proxy] : [], gate });
|
|
295
|
+
if (blocked) {
|
|
296
|
+
// SAID, never silent (issue #452, gate round 4): a network left here is a network the caller must name.
|
|
297
|
+
if (typeof onRefused === "function") onRefused(blocked);
|
|
298
|
+
return false;
|
|
299
|
+
}
|
|
300
|
+
return (await runWith(docker, ["network", "rm", network]))?.code === 0;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* THE ONE PLACE A CONTAINER IS DETACHED FROM A NETWORK in `worker/src` (issue #452, gate round 3). Every teardown and
|
|
305
|
+
* every sweep reaches `network disconnect` through here, so the rule that makes a detach safe on a rootless Podman 4.x
|
|
306
|
+
* (`makeDetachGate`, in `netns-keeper.mjs`) cannot be forgotten by a caller: it is asked first, with whether anything
|
|
307
|
+
* RUNNING is among `endpoints` (`running`, the caller's knowledge; everything, when it does not say), and a refusal
|
|
308
|
+
* detaches NOTHING. `{ blocked, detached }`: `blocked` is the gate's token or `null`; `detached` holds only the
|
|
309
|
+
* endpoints whose disconnect exited 0, since it is printed and logged and a list of attempts would name something
|
|
310
|
+
* still attached as removed.
|
|
188
311
|
*/
|
|
189
|
-
export async function
|
|
190
|
-
|
|
191
|
-
await
|
|
312
|
+
export async function detachEndpoints(docker, { network, endpoints, running = endpoints, gate }) {
|
|
313
|
+
if (typeof gate !== "function") throw new Error("detachEndpoints: a detach gate is required (makeDetachGate)");
|
|
314
|
+
const blocked = await gate({ running: endpoints.some((e) => running.includes(e)) });
|
|
315
|
+
if (blocked) return { blocked, detached: [] };
|
|
316
|
+
const detached = [];
|
|
317
|
+
for (const endpoint of endpoints) {
|
|
318
|
+
if ((await runWith(docker, ["network", "disconnect", "-f", network, endpoint]))?.code === 0) detached.push(endpoint);
|
|
319
|
+
}
|
|
320
|
+
return { blocked: null, detached };
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* "The daemon says this network is not there", in BOTH daemons' words for a NETWORK (measured: Docker
|
|
325
|
+
* `network X not found`, Podman `unable to find network with name or ID X: network not found`). The CLI's own
|
|
326
|
+
* `context not found` must NOT match -- that is about the CLI, not the network -- and neither may an inspect
|
|
327
|
+
* that timed out or a daemon that could not be reached. `code === null` is NO ANSWER and therefore never absence.
|
|
328
|
+
*
|
|
329
|
+
* ONE COPY, because this classifies ANOTHER TOOL'S PROSE across two runtimes. The rule was earned over three
|
|
330
|
+
* review rounds in `live-probes.mjs` (issue #344), and a second copy is the one nobody updates when a third
|
|
331
|
+
* runtime words it differently. That is this file's own argument for owning the proxy name and the network
|
|
332
|
+
* suffix: a rename that lands in the producer and not in the reaper is the failure mode.
|
|
333
|
+
*
|
|
334
|
+
* Measured 2026-09-21 on docker 27.4.0: `network inspect` and `network rm` word a missing network
|
|
335
|
+
* IDENTICALLY, and with `--format` the wording is on STDERR with stdout empty. So a caller must hand in a
|
|
336
|
+
* runner that captures BOTH streams; `runDocker` below captures neither by default.
|
|
337
|
+
*/
|
|
338
|
+
export function networkAbsentInDaemonWords(result) {
|
|
339
|
+
// A NUMBER that is not zero, or nothing. `code === null` is a timeout or a launch failure, and a result
|
|
340
|
+
// with no `code` at all is a runner that answered something this rule cannot read: both are NO ANSWER,
|
|
341
|
+
// and no answer is never absence.
|
|
342
|
+
if (typeof result?.code !== "number" || result.code === 0) return false;
|
|
343
|
+
const text = `${result.stdout ?? ""}${result.stderr ?? ""}`;
|
|
344
|
+
// Podman's `network disconnect` of a container NOT on an EXISTING network ends in the same `network not found`
|
|
345
|
+
// (exit 125, measured on 5.8.1: `... is not connected to network X: network not found`). The network is there; the
|
|
346
|
+
// membership is not. Read as absence, a caller would skip the `rm` of a network that still exists, or report one
|
|
347
|
+
// gone that is not, so that wording is excluded before the match, wherever it appears in the output.
|
|
348
|
+
if (/is not connected to network/i.test(text)) return false;
|
|
349
|
+
return /network (?:\S+ )?not found/i.test(text);
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/**
|
|
353
|
+
* The container states a daemon's own member list shows for a network: `docker network inspect`'s `.Containers`
|
|
354
|
+
* lists a member in these and in no other (measured, docker 27.4.0), and so does Podman 5.8.1's (issue #452,
|
|
355
|
+
* measured: `running` and `paused` listed, `exited` and `created` not). `backend-local.mjs` re-exports it under
|
|
356
|
+
* the same name for the boot reaper's second question, which asks about the states OUTSIDE this set.
|
|
357
|
+
*
|
|
358
|
+
* An ALLOWLIST rather than a denylist, following `sandbox.mjs`: an unknown state -- a Podman rendering nothing
|
|
359
|
+
* here has measured, or one docker adds later -- is outside it, which keeps a network rather than losing it.
|
|
360
|
+
* `restarting` is NOT here, even though it can appear: whether a flapping container is in `.Containers` depends
|
|
361
|
+
* on which instant the sweep asks in, and a rule that changes answer between two runs is not a rule a sweep can
|
|
362
|
+
* act on.
|
|
363
|
+
*/
|
|
364
|
+
export const ENDPOINT_LISTED_STATES = new Set(["running", "paused"]);
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* The endpoint NAMES attached to a network, as `{ ok, names, absent }`, plus `parked` on Podman (below).
|
|
368
|
+
*
|
|
369
|
+
* TWO READS, ONE PER RUNTIME, and `bin` picks which (issue #452). On docker it is `network inspect --format
|
|
370
|
+
* {{json .Containers}}`, byte for byte what it always was. On Podman that template does not work across the
|
|
371
|
+
* versions this project runs: 5.8.1 renders the member map (with a lowercase `name`, measured under issue #354),
|
|
372
|
+
* but 4.9.3 renders no member list at all -- its inspect JSON has no such key, and the template exits 125 with
|
|
373
|
+
* `can't evaluate field Containers in type interface {}` for EVERY existing network, empty or not (measured on
|
|
374
|
+
* Ubuntu 24.04). Read that way, every leftover network on 4.9 was unreadable forever, and all four sweeps that
|
|
375
|
+
* share this reader passed over it. So on Podman the members come from `ps -a --filter network=<net> --format
|
|
376
|
+
* {{.Names}}\t{{.State}}`, which both versions answer identically (measured on 4.9.3 and 5.8.1: exit 0, one
|
|
377
|
+
* `name<TAB>state` line per member in every state; the filter matches a network by its whole NAME only, never a
|
|
378
|
+
* name prefix, and by its full id or any prefix of that id, which a sweep never passes since it names networks).
|
|
379
|
+
*
|
|
380
|
+
* THAT READ CANNOT SAY "NOT THERE", which is why Podman's path has a second step: `ps` over a network that does
|
|
381
|
+
* not exist answers exit 0 and empty on both versions (measured), exactly like an empty network, and every
|
|
382
|
+
* caller treats empty as licence to remove. `network exists` then decides it, as its exit code (0 there, 1 not
|
|
383
|
+
* there, measured on both); anything else is no answer and the read is unreadable. The network is asked about
|
|
384
|
+
* AFTER its members, so an `absent` is the freshest fact this function has.
|
|
385
|
+
*
|
|
386
|
+
* WHAT `names` MEANS IS THE SAME ON BOTH RUNTIMES: the members the daemon lists, in `ENDPOINT_LISTED_STATES`.
|
|
387
|
+
* That is what every caller was written against (docker's `.Containers`, and Podman 5.8.1's, which lists the
|
|
388
|
+
* same two states), and keeping it is what keeps each caller's own guard meaning what it says. What Podman
|
|
389
|
+
* adds is `parked`: the members in every OTHER state, in the order `ps` gave them. Docker's read cannot see
|
|
390
|
+
* those at all and carries no `parked` key, so no docker caller's input changed. Podman is where they matter,
|
|
391
|
+
* because its `network rm` without `-f` refuses while a member in ANY state remains (measured on 4.9.3 and
|
|
392
|
+
* 5.8.1, each with a lone member in each of the four states running, paused, exited and created, and with all
|
|
393
|
+
* four together), where docker's refuses only for a listed one. A caller
|
|
394
|
+
* that must see a stopped member to act correctly reads `parked`; the others can ignore it and get what they
|
|
395
|
+
* always got.
|
|
396
|
+
*
|
|
397
|
+
* FAIL CLOSED, on both paths. A member with no readable name makes the whole answer UNREADABLE (`ok: false`),
|
|
398
|
+
* never "nothing attached", because every caller treats an empty list as licence to detach and remove, and a
|
|
399
|
+
* member this parser cannot name is still a member. On Podman that is any non-empty `ps` line that is not
|
|
400
|
+
* exactly a runtime-legal name, a tab and a lowercase state word.
|
|
401
|
+
*
|
|
402
|
+
* WHAT "ATTACHED" MEANS on docker, measured end to end on docker 27.4.0 rather than assumed. `.Containers` lists
|
|
403
|
+
* RUNNING endpoints only. A running member is listed and `network rm` fails "has active endpoints"; the SAME
|
|
404
|
+
* member stopped is absent from this map AND the `rm` succeeds. So for those two states "listed" and "holds the
|
|
405
|
+
* network" agree, and a stopped container is not something a docker sweep needs to reason about.
|
|
406
|
+
*
|
|
407
|
+
* CORRECTED under issue #337: that agreement does NOT extend to a container in `created` state, and the
|
|
408
|
+
* earlier version of this comment generalised it to "listed and holds the network agree" full stop, which is
|
|
409
|
+
* false. Measured: a container created on a network but never started is absent from this map, absent from
|
|
410
|
+
* `docker ps`, and the `network rm` SUCCEEDS -- after which `docker start` fails with "network not found" and
|
|
411
|
+
* the container can never run. So the docker daemon is a backstop for a RUNNING endpoint and for nothing else,
|
|
412
|
+
* and a sweep that cares about a container being launched right now has to ask `docker ps -a --filter
|
|
413
|
+
* status=created` rather than infer it from here. On Podman the backstop covers every state (above).
|
|
414
|
+
*
|
|
415
|
+
* `local` THROUGH `podman-docker` ON PODMAN 4.9 (issue #452, gate round 1). Podman's `docker` emulation is a
|
|
416
|
+
* script that runs `podman` itself, so `bin` says docker while the answer is 4.9's: the same exit 125 and
|
|
417
|
+
* `can't evaluate field Containers`, for every existing network (measured on 4.9.3; on 5.8.1 the emulation
|
|
418
|
+
* renders the member map). That exact wording, and nothing else, sends the docker path to the Podman read
|
|
419
|
+
* through the SAME runner, which the emulation answers as Podman does (measured on 4.9.3 and 5.8.1). Its
|
|
420
|
+
* presence step is a plain `network inspect <net>` read by `networkAbsentInDaemonWords` rather than `network
|
|
421
|
+
* exists`, because the real docker CLI has no such verb and exits 1 with its usage text (measured, docker
|
|
422
|
+
* 27.4.0 and 29.7.2), which would read as absent. The real docker CLI against Podman's Docker API never takes
|
|
423
|
+
* this branch: the API renders `.Containers` with `Name` (measured against 4.9.3's and 5.8.1's services).
|
|
424
|
+
*/
|
|
425
|
+
export async function networkEndpoints(docker, network, { bin = "docker" } = {}) {
|
|
426
|
+
// HALF-PROTECTED, and the half is worth naming (issue #360, item 6). `runWith` turns a runner that THROWS
|
|
427
|
+
// into `{ code: null }`, so a throwing runner reads here as an unreadable network and every caller's guard
|
|
428
|
+
// stays cautious. The `disconnect` and `rm` in `removeNetworkOrSay` go through the same wrapper, but the
|
|
429
|
+
// callers' OWN steps around them do not, and neither does the `network ls` that produced the candidate
|
|
430
|
+
// list: on the boot reaper that one is the throwing `exec` deliberately, so a daemon that dies mid-pass
|
|
431
|
+
// still answers `{ reaped: false }`. Unreachable with the production runners, which are all non-throwing
|
|
432
|
+
// by construction; stated so a future injected runner is not assumed to be.
|
|
433
|
+
if (bin === "podman") return podmanNetworkMembers(docker, network);
|
|
434
|
+
const inspected = await runWith(docker, ["network", "inspect", "--format", "{{json .Containers}}", network]);
|
|
435
|
+
if (inspected?.code !== 0) {
|
|
436
|
+
if (/can't evaluate field Containers/.test(`${inspected?.stdout ?? ""}${inspected?.stderr ?? ""}`)) return podmanNetworkMembers(docker, network, { presence: "inspect" });
|
|
437
|
+
return { ok: false, names: [], absent: networkAbsentInDaemonWords(inspected) };
|
|
438
|
+
}
|
|
439
|
+
try {
|
|
440
|
+
const parsed = JSON.parse(String(inspected.stdout ?? "").trim() || "{}");
|
|
441
|
+
// FAIL CLOSED on anything that is not an object: a runtime rendering `.Containers` as `null` would
|
|
442
|
+
// otherwise read as "no endpoints", which makes every caller's guard vacuous rather than cautious.
|
|
443
|
+
if (parsed === null || typeof parsed !== "object") return { ok: false, names: [], absent: false };
|
|
444
|
+
const names = [];
|
|
445
|
+
for (const c of Object.values(parsed)) {
|
|
446
|
+
// `Name` first, the docker key, then Podman 5.x's `name`, kept although Podman no longer takes this path:
|
|
447
|
+
// the reader is exported, and a caller handing it a Podman rendering still gets a name, not a refusal.
|
|
448
|
+
const name = [c?.Name, c?.name].find((v) => typeof v === "string" && v !== "");
|
|
449
|
+
if (name === undefined) return { ok: false, names: [], absent: false };
|
|
450
|
+
names.push(name);
|
|
451
|
+
}
|
|
452
|
+
return { ok: true, absent: false, names };
|
|
453
|
+
} catch {
|
|
454
|
+
return { ok: false, names: [], absent: false };
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
/** A container name as both runtimes accept one (`[a-zA-Z0-9][a-zA-Z0-9_.-]*`, docker's and Podman's, measured). */
|
|
459
|
+
const RUNTIME_NAME = /^[A-Za-z0-9][A-Za-z0-9_.-]*$/;
|
|
460
|
+
|
|
461
|
+
/**
|
|
462
|
+
* `networkEndpoints` on Podman: `ps -a --filter network=`, then whether the network is there (issue #452, see there):
|
|
463
|
+
* `network exists` for the podman CLI, a plain `network inspect` for Podman reached as `docker` (`presence: "inspect"`).
|
|
464
|
+
*/
|
|
465
|
+
async function podmanNetworkMembers(podman, network, { presence = "exists" } = {}) {
|
|
466
|
+
const unreadable = { ok: false, names: [], absent: false };
|
|
467
|
+
const listed = await runWith(podman, ["ps", "-a", "--filter", `network=${network}`, "--format", "{{.Names}}\t{{.State}}"]);
|
|
468
|
+
if (listed?.code !== 0) return unreadable;
|
|
469
|
+
const names = [];
|
|
470
|
+
const parked = [];
|
|
471
|
+
for (const raw of String(listed.stdout ?? "").split("\n")) {
|
|
472
|
+
const line = raw.trim();
|
|
473
|
+
if (line === "") continue;
|
|
474
|
+
// EXACTLY two fields. Podman writes its warnings to stderr, so anything else on stdout is a rendering this
|
|
475
|
+
// parser was not measured against, and a member it cannot name is still a member.
|
|
476
|
+
const fields = line.split("\t");
|
|
477
|
+
if (fields.length !== 2) return unreadable;
|
|
478
|
+
const [name, state] = fields.map((f) => f.trim());
|
|
479
|
+
if (!RUNTIME_NAME.test(name) || !/^[a-z]+$/.test(state)) return unreadable;
|
|
480
|
+
(ENDPOINT_LISTED_STATES.has(state) ? names : parked).push(name);
|
|
481
|
+
}
|
|
482
|
+
// Only now: `ps` answers the same for an empty network and a missing one. 0 is there, 1 is not (measured on
|
|
483
|
+
// 4.9.3 and 5.8.1); 125, a timeout or a launch failure is no answer, and no answer is never absence.
|
|
484
|
+
if (presence === "inspect") {
|
|
485
|
+
const inspected = await runWith(podman, ["network", "inspect", network]);
|
|
486
|
+
if (networkAbsentInDaemonWords(inspected)) return { ok: false, names: [], absent: true };
|
|
487
|
+
if (inspected?.code !== 0) return unreadable;
|
|
488
|
+
return { ok: true, absent: false, names, parked };
|
|
489
|
+
}
|
|
490
|
+
const exists = await runWith(podman, ["network", "exists", network]);
|
|
491
|
+
if (exists?.code === 1) return { ok: false, names: [], absent: true };
|
|
492
|
+
if (exists?.code !== 0) return unreadable;
|
|
493
|
+
return { ok: true, absent: false, names, parked };
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
/**
|
|
497
|
+
* Detach what the CALLER names, then `network rm` WITHOUT `-f`, and SAY what happened rather than return a
|
|
498
|
+
* boolean a caller can drop: `{ removed, absent, detached, command }`.
|
|
499
|
+
*
|
|
500
|
+
* A SIBLING of `removeJobNetworkWith`, deliberately NOT its replacement and not built on top of it, and the
|
|
501
|
+
* direction matters both ways. Building this on that one would force-detach the proxy BEFORE anything
|
|
502
|
+
* inspected the endpoint list, which is exactly the harm a sweep's guard exists to prevent. Building that one
|
|
503
|
+
* on this would add an inspect to the job's own teardown path, which `worker/test/egress.test.mjs`'s "removeJobNetwork detaches before removing" pins
|
|
504
|
+
* as exactly two calls. So the job path keeps its two-call shape and the sweeps get their own primitive.
|
|
505
|
+
*
|
|
506
|
+
* `detach` is an EXPLICIT list because every caller decides it differently: the boot reaper detaches what it
|
|
507
|
+
* saw and nothing while a job container is still on the network, `doctor`'s canary detaches its own proxy,
|
|
508
|
+
* and a sandbox sweep leaves a session alone. A helper whose parameters ARE the decision would hide it.
|
|
509
|
+
*
|
|
510
|
+
* `command` is the one an operator would type, and it is the ONLY string a caller may print: the CLI's own
|
|
511
|
+
* error text is never surfaced, because a docker error can repeat a `DOCKER_HOST` with credentials in it
|
|
512
|
+
* (issue #339).
|
|
513
|
+
*/
|
|
514
|
+
export async function removeNetworkOrSay(docker, { network, detach = [], running = detach, stillClear = async () => true, bin = "docker", gate = makeDetachGate(docker, { bin }) }) {
|
|
515
|
+
// `stillClear` is the OTHER kind of parameter, and naming the difference is what keeps `detach` explicit.
|
|
516
|
+
// `detach` names WHICH endpoints go, which every caller decides differently, so a default there would hide
|
|
517
|
+
// a decision. `stillClear` decides NOTHING about the target: it is the caller's own guard, re-asked
|
|
518
|
+
// immediately before each destructive verb, and `false` STOPS this function rather than changing what it
|
|
519
|
+
// would have removed. It exists because the guard that protects a sandbox mid-launch is a `docker ps -a`
|
|
520
|
+
// this module cannot phrase -- the filter and the id are the caller's -- and because the interval between
|
|
521
|
+
// that guard and the `rm` was two commands wide and grew by one per endpoint (issue #363).
|
|
522
|
+
//
|
|
523
|
+
// It defaults to a no-op, so four of the five callers are byte-identical and the one that opts in does so
|
|
524
|
+
// by name, and that default is also what makes those four callers SAFE rather than merely unchanged: an
|
|
525
|
+
// aborted pass returns `command: null`, and three of them would render that into "could not be removed:
|
|
526
|
+
// null" if they ever saw the shape (the boot reaper is the exception: it logs a fixed reason token and
|
|
527
|
+
// never reads `command` at all). A caller that opts in MUST branch on `aborted` before it reads
|
|
528
|
+
// `command`. `sandbox.mjs` does; nothing else can reach it.
|
|
529
|
+
//
|
|
530
|
+
// WHY THE OTHER FOUR PASS NOTHING: `backend-local.mjs`'s boot reaper and `doctor.mjs`'s two canary calls
|
|
531
|
+
// touch objects whose owner is already gone or whose pid is DEAD, where nothing can be mid-launch, and
|
|
532
|
+
// `live-probes.mjs`'s peer sweep is best effort behind a flag an operator typed. Only the sandbox sweep
|
|
533
|
+
// has an owner who may be alive.
|
|
534
|
+
//
|
|
535
|
+
// WHAT THE SECOND ASK COSTS, and why the abort below puts back what it took. The guard before the DETACH is
|
|
536
|
+
// where it already was; what is new is the one before the `rm`, which used to sit k+1 commands out. That
|
|
537
|
+
// leaves a window the previous shape did not have: the detach can succeed and the guard then refuse, and a
|
|
538
|
+
// network whose proxy has been disconnected is a session with silently dead egress, where the old shape
|
|
539
|
+
// gave a loud `docker run` failure. Silent is the worse of the two, so anything detached on an aborted pass
|
|
540
|
+
// is reconnected.
|
|
541
|
+
if (!(await stillClear())) return { removed: false, absent: false, detached: [], command: null, aborted: true };
|
|
542
|
+
// Through the ONE detach helper (issue #452, gate round 3): a gate that refuses leaves the network whole and says so as
|
|
543
|
+
// `blocked`, which each caller turns into its own line. `running` names which of `detach` are running (the caller's
|
|
544
|
+
// member read), so a stopped proxy is detached with nothing asked.
|
|
545
|
+
const { blocked, detached } = await detachEndpoints(docker, { network, endpoints: detach, running, gate });
|
|
546
|
+
if (blocked) return { removed: false, absent: false, detached: [], command: null, blocked };
|
|
547
|
+
// AGAIN, immediately before the verb that kills a launch. This is what removes the SCALING: with k
|
|
548
|
+
// endpoints the `rm` used to be k+1 commands after the only guard, and it is now always one. Asked only
|
|
549
|
+
// when there is something to re-ask ABOUT: with nothing detached, nothing has happened since the first ask
|
|
550
|
+
// and a second identical `docker ps -a` back to back would be a round trip that answers itself.
|
|
551
|
+
if (detached.length > 0 && !(await stillClear())) {
|
|
552
|
+
// PUT BACK WHAT WAS TAKEN. Reported separately from `detached`, which means "removed from this network
|
|
553
|
+
// by this pass": an endpoint that was detached and then restored was not, and a caller printing
|
|
554
|
+
// `detached` must not name it. A reconnect that fails is named too, because that endpoint IS now off a
|
|
555
|
+
// network someone may be using and nothing else will put it back.
|
|
556
|
+
const restored = [];
|
|
557
|
+
const lost = [];
|
|
558
|
+
for (const endpoint of detached) {
|
|
559
|
+
if ((await runWith(docker, ["network", "connect", network, endpoint]))?.code === 0) restored.push(endpoint);
|
|
560
|
+
else lost.push(endpoint);
|
|
561
|
+
}
|
|
562
|
+
return { removed: false, absent: false, detached: [], restored, lost, command: null, aborted: true };
|
|
563
|
+
}
|
|
564
|
+
if ((await runWith(docker, ["network", "rm", network]))?.code === 0) return { removed: true, absent: false, detached, command: null };
|
|
565
|
+
// Silent ONLY when the daemon says it is not there. Anything else -- a timeout, an unreachable daemon, a
|
|
566
|
+
// race that attached something between the inspect and the rm -- is said, with the command.
|
|
567
|
+
const inspected = await runWith(docker, ["network", "inspect", network]);
|
|
568
|
+
if (networkAbsentInDaemonWords(inspected)) return { removed: true, absent: true, detached, command: null };
|
|
569
|
+
// `bin` (issue #354) only spells the command an operator is told to type; the runner is the caller's, already bound.
|
|
570
|
+
return { removed: false, absent: false, detached, command: `${bin} network rm ${network}` };
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
/**
|
|
574
|
+
* The spawn-based runner the job path and the sandbox use: `runDocker` for the network verbs, as before, and a CAPTURING,
|
|
575
|
+
* BOUNDED spawn when the detach gate reads the runtime with its own `{ timeoutMs, maxBuffer }` (a `podman info` body is
|
|
576
|
+
* tens of KiB, past `runDocker`'s 4 KiB, and a teardown in a `finally` must not hang on a wedged daemon).
|
|
577
|
+
*/
|
|
578
|
+
function spawnRunner(spawnFn, bin) {
|
|
579
|
+
return (args, opts) => (opts ? runCapture(spawnFn, args, bin, opts) : runDocker(spawnFn, args, false, bin));
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
function runCapture(spawnFn, args, bin, { timeoutMs, maxBuffer }) {
|
|
583
|
+
return new Promise((resolve) => {
|
|
584
|
+
let child;
|
|
585
|
+
try {
|
|
586
|
+
child = spawnFn(bin, args, { stdio: ["ignore", "pipe", "ignore"] });
|
|
587
|
+
} catch {
|
|
588
|
+
resolve({ code: null, stdout: "" });
|
|
589
|
+
return;
|
|
590
|
+
}
|
|
591
|
+
let stdout = "";
|
|
592
|
+
let done = false;
|
|
593
|
+
const finish = (value) => {
|
|
594
|
+
if (done) return;
|
|
595
|
+
done = true;
|
|
596
|
+
clearTimeout(timer);
|
|
597
|
+
resolve(value);
|
|
598
|
+
};
|
|
599
|
+
const timer = setTimeout(() => {
|
|
600
|
+
try {
|
|
601
|
+
child.kill?.("SIGKILL");
|
|
602
|
+
} catch {
|
|
603
|
+
// already gone
|
|
604
|
+
}
|
|
605
|
+
finish({ code: null, stdout: "" });
|
|
606
|
+
}, timeoutMs);
|
|
607
|
+
child.stdout?.setEncoding?.("utf8");
|
|
608
|
+
child.stdout?.on?.("data", (chunk) => {
|
|
609
|
+
if (stdout.length < maxBuffer) stdout += chunk;
|
|
610
|
+
});
|
|
611
|
+
child.on?.("error", () => finish({ code: null, stdout: "" }));
|
|
612
|
+
child.on?.("close", (code) => finish({ code, stdout }));
|
|
613
|
+
});
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
/** One step through a caller's runner, as `{ code: null }` when it throws. */
|
|
617
|
+
async function runWith(docker, args) {
|
|
618
|
+
try {
|
|
619
|
+
return await docker(args);
|
|
620
|
+
} catch {
|
|
621
|
+
return { code: null, stdout: "" };
|
|
622
|
+
}
|
|
192
623
|
}
|
|
193
624
|
|
|
194
625
|
/**
|
|
@@ -197,11 +628,11 @@ export async function removeJobNetwork(spawnFn, { network, proxy = DEFAULT_EGRES
|
|
|
197
628
|
* binary is no answer. Same shape as image-preflight.mjs's own runDocker and doctor's runCmd, so all
|
|
198
629
|
* three agree on what "present" means.
|
|
199
630
|
*/
|
|
200
|
-
function runDocker(spawnFn, args, capture = false) {
|
|
631
|
+
function runDocker(spawnFn, args, capture = false, bin = "docker") {
|
|
201
632
|
return new Promise((resolve) => {
|
|
202
633
|
let child;
|
|
203
634
|
try {
|
|
204
|
-
child = spawnFn(
|
|
635
|
+
child = spawnFn(bin, args, { stdio: capture ? ["ignore", "pipe", "ignore"] : "ignore" });
|
|
205
636
|
} catch {
|
|
206
637
|
resolve({ code: null, stdout: "" });
|
|
207
638
|
return;
|
package/src/entry.mjs
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* "Is this module the program node was asked to run?" (issue #489), for every bin and entry file.
|
|
3
|
+
*
|
|
4
|
+
* The old guard compared `import.meta.url` with `file://${process.argv[1]}`, or tested that argv[1] ended in the
|
|
5
|
+
* file's own name. Neither holds through npm's bin link: npm installs `.bin/pi-dispatch` as a SYMLINK to
|
|
6
|
+
* `src/cli.mjs`, node puts the link's own path in argv[1] and the resolved path in `import.meta.url`, and the
|
|
7
|
+
* link's name ends in `pi-dispatch`, not `cli.mjs`. So `npx @edgehero/pi-dispatch init`, a local `.bin` and a
|
|
8
|
+
* global install all exited 0 having run nothing, while `node .../src/cli.mjs` (how `/dispatch setup` and the
|
|
9
|
+
* rendered service units call it) worked, which is why nothing that ships noticed.
|
|
10
|
+
*
|
|
11
|
+
* The rule is the one question underneath both old tests: do argv[1] and this module name the same FILE once
|
|
12
|
+
* links are resolved. `realpathSync` resolves the link, and both sides go through that same call, so a
|
|
13
|
+
* spelling it keeps (a case difference, say) is kept on both; `fileURLToPath` turns the module URL into a path on every
|
|
14
|
+
* platform, where string concatenation produced `file://C:\...`.
|
|
15
|
+
* A missing or unreadable argv[1] (node -e, the REPL, a test runner) is simply "not the entry".
|
|
16
|
+
*/
|
|
17
|
+
import { realpathSync } from "node:fs";
|
|
18
|
+
import { fileURLToPath } from "node:url";
|
|
19
|
+
|
|
20
|
+
export function isEntryModule(moduleUrl, { argv1 = process.argv[1], realpath = realpathSync } = {}) {
|
|
21
|
+
if (typeof argv1 !== "string" || argv1 === "") return false;
|
|
22
|
+
try {
|
|
23
|
+
return realpath(argv1) === realpath(fileURLToPath(moduleUrl));
|
|
24
|
+
} catch {
|
|
25
|
+
return false;
|
|
26
|
+
}
|
|
27
|
+
}
|