@edgehero/pi-dispatch 1.10.2 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +300 -148
- package/README.md +50 -0
- package/deploy/com.pi-dispatch.worker.plist +9 -3
- package/deploy/docker-compose.yml +49 -16
- package/deploy/egress-proxy.conf +32 -2
- package/deploy/nssm-install.cmd +12 -6
- package/deploy/pi-dispatch-egress-out.network +10 -0
- package/deploy/pi-dispatch-egress-proxy.container +50 -0
- package/deploy/pi-dispatch-netns-keeper.container +80 -0
- package/deploy/pi-dispatch-netns-keeper.network +18 -0
- package/deploy/pi-dispatch-valkey.container +51 -0
- package/deploy/pi-dispatch-valkey.network +16 -0
- package/deploy/receiver.service +6 -0
- package/deploy/worker-env-wrapper.cmd +11 -0
- package/deploy/worker-env-wrapper.sh +60 -34
- package/deploy/worker.service +18 -8
- package/package.json +14 -4
- package/src/azure-host.mjs +19 -0
- package/src/azure-identity.mjs +18 -2
- package/src/backend-conformance.mjs +71 -18
- package/src/backend-local.mjs +637 -21
- package/src/backend-podman.mjs +1168 -0
- package/src/backend-registry.mjs +86 -3
- package/src/backends.mjs +489 -37
- package/src/branch.mjs +7 -2
- package/src/cancel-cli.mjs +174 -0
- package/src/cancel-state.mjs +125 -0
- package/src/cli.mjs +188 -90
- package/src/config.mjs +503 -43
- package/src/connection.mjs +374 -8
- package/src/container-spec.mjs +102 -7
- package/src/daemon-facts.mjs +167 -0
- package/src/deployment-venue.mjs +158 -0
- package/src/docker-run.mjs +146 -15
- package/src/doctor.mjs +4701 -414
- package/src/egress-conf-copy.mjs +166 -0
- package/src/egress-proxy-state.mjs +151 -0
- package/src/egress.mjs +455 -25
- package/src/entry.mjs +27 -0
- package/src/env-allowlist.mjs +222 -40
- package/src/env-file.mjs +1869 -33
- package/src/exit-code.mjs +15 -0
- package/src/flow-gate.mjs +5 -3
- package/src/forgejo-host.mjs +19 -0
- package/src/forgejo-identity.mjs +21 -2
- package/src/get-token.mjs +67 -18
- package/src/git-dirty.mjs +9 -1
- package/src/git-hardening.mjs +33 -0
- package/src/github-app-setup.mjs +29 -12
- package/src/github-prompt.mjs +4 -1
- package/src/gitlab-host.mjs +19 -0
- package/src/gitlab-identity.mjs +19 -2
- package/src/host-registry.mjs +32 -5
- package/src/identity.mjs +29 -4
- package/src/image-preflight.mjs +46 -11
- package/src/image-ref.mjs +21 -0
- package/src/index.mjs +387 -17
- package/src/init.mjs +197 -38
- package/src/job-user.mjs +252 -0
- package/src/json-duplicates.mjs +204 -0
- package/src/live-probes.mjs +1020 -0
- package/src/materialize.mjs +4 -11
- package/src/netns-keeper.mjs +264 -0
- package/src/on-failure.mjs +119 -0
- package/src/outbox.mjs +7 -0
- package/src/podman-stack.mjs +1304 -0
- package/src/prepare-github.mjs +6 -6
- package/src/prepare-local.mjs +51 -17
- package/src/prepare.mjs +27 -6
- package/src/processor.mjs +505 -26
- package/src/provider-key.mjs +41 -0
- package/src/provider-steering.mjs +144 -0
- package/src/queue.mjs +35 -8
- package/src/redact.mjs +84 -0
- package/src/reserved-env.mjs +7 -3
- package/src/retention-sweep.mjs +178 -0
- package/src/run-container.mjs +181 -14
- package/src/run-history.mjs +105 -16
- package/src/runtime-observations.mjs +1152 -0
- package/src/runtime-settings.mjs +13 -8
- package/src/sandbox-cli.mjs +100 -95
- package/src/sandbox-store.mjs +612 -45
- package/src/sandbox.mjs +1459 -37
- package/src/schedules.mjs +16 -3
- package/src/secret-profiles.mjs +2 -1
- package/src/secrets.mjs +23 -6
- package/src/service-env.mjs +247 -0
- package/src/service.mjs +618 -28
- package/src/session-store.mjs +678 -53
- package/src/start.mjs +1395 -268
- package/src/transient.mjs +240 -0
- package/src/triggers-file.mjs +71 -15
- package/src/triggers.mjs +176 -19
- package/src/up.mjs +1399 -85
- package/src/valkey-auth.mjs +529 -0
- package/src/valkey-endpoint.mjs +367 -0
- package/src/watch-closer.mjs +158 -0
package/src/up.mjs
CHANGED
|
@@ -14,6 +14,15 @@
|
|
|
14
14
|
* file happens to name" (the same reasoning as jobs running with --pull=never).
|
|
15
15
|
* - no secrets printed: the generated WEBHOOK_SECRET is announced, never echoed.
|
|
16
16
|
*
|
|
17
|
+
* The native rootless `podman` venue (issue #430): when PI_BACKENDS lists podman, `up` also gates on the venue's own
|
|
18
|
+
* `podman info` rule, pulls the default image into this account's Podman store, and installs Valkey and the egress
|
|
19
|
+
* proxy as Quadlet units through the installer `service install` uses (podman-stack.mjs), after init. When the list
|
|
20
|
+
* does not name `local`, nothing here runs docker at all. The docker path of a deployment that blesses `local` is
|
|
21
|
+
* unchanged, with one correction: the egress step looks for the proxy PI_EGRESS_PROXY names rather than the shipped
|
|
22
|
+
* name, and starts nothing in place of an overriding one. The venue keys (PI_BACKENDS, PI_EGRESS, PI_EGRESS_PROXY)
|
|
23
|
+
* come from this shell when it sets them and otherwise from the deployment's `.env`, the file `service install`
|
|
24
|
+
* reads, so the two entry points decide the venue from the same place (`deploymentVenueEnv`).
|
|
25
|
+
*
|
|
17
26
|
* Converge-style, not transactional: a declined or failed step is reported and the pass continues, so
|
|
18
27
|
* one flaky pull does not hide the doctor report that says what else is missing. Exit code mirrors
|
|
19
28
|
* doctor's: 0 unless the docker daemon was unreachable up front (nothing else can be probed, so up
|
|
@@ -21,11 +30,26 @@
|
|
|
21
30
|
*/
|
|
22
31
|
import { randomBytes } from "node:crypto";
|
|
23
32
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
24
|
-
import { chmodSync, existsSync, readFileSync, renameSync, statSync, writeFileSync } from "node:fs";
|
|
33
|
+
import { chmodSync, existsSync, lstatSync, readFileSync, realpathSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs";
|
|
34
|
+
import { mkdirSync, readFileSync as readPackageFile } from "node:fs";
|
|
25
35
|
import { connect as netConnect } from "node:net";
|
|
26
|
-
import {
|
|
27
|
-
import {
|
|
28
|
-
import {
|
|
36
|
+
import { lookup as dnsLookup } from "node:dns/promises";
|
|
37
|
+
import { homedir, networkInterfaces, userInfo } from "node:os";
|
|
38
|
+
import { dirname, join, posix, resolve, win32 } from "node:path";
|
|
39
|
+
import { fileURLToPath } from "node:url";
|
|
40
|
+
import { PODMAN_BOOT_REFUSING_CAUSES, makePodmanInfoReader, decidePodmanJobUser, podmanJobUserRefusal } from "./backend-podman.mjs";
|
|
41
|
+
import { venuesOf } from "./backends.mjs";
|
|
42
|
+
import { logsDirPath, settingsFilePath } from "./config.mjs";
|
|
43
|
+
import { DEFAULT_EGRESS_PROXY, egressArmed as egressArmedFn, egressProxyName } from "./egress.mjs";
|
|
44
|
+
import { envKeyIsBlank, envValueShown, readEnvAssignments, updateEnvFile } from "./env-file.mjs";
|
|
45
|
+
import { COMPOSE_FILE, COMPOSE_VALKEY_OVERRIDE, OWNER_MARKER_KEY, VALKEY_PASSWORD_KEY, VALKEY_PORT_KEY, OWNER_CHECK_CONTAINER, OWNER_CHECK_EXEC, VALKEY_VOLUME_RECORD, adoptVolumeQuestion, composeArgs, ownerCheckAnswer, readVolumeRecord, valkeyOwnerCheckArgs, volumeRecordMatches, volumeRecordText, composeProjectName, foreignContainerSentence, foreignMarkerRefusal, foreignVolumeLabelRefusal, foreignVolumeRefusal, foreignVolumeUsers, isLoopbackHost, newValkeyPassword, unadoptedVolumeRefusal, valkeyContainerOwner, valkeyDockerRunArgs, valkeyPasswordDecision, valkeyPortEnvDecision, valkeyVolumeCreateArgs, valkeyVolumeOwner } from "./valkey-auth.mjs";
|
|
46
|
+
import { deploymentValkeyEnv, deploymentVenueEnv } from "./deployment-venue.mjs";
|
|
47
|
+
import { PACKAGED_EGRESS_PROXY_CONF, judgeProxyConfCopy, packageCopyName, readPackagedProxyConf, replaceProxyConfCopy } from "./egress-conf-copy.mjs";
|
|
48
|
+
import { EGRESS_PROXY_IMAGE, PROXY_STATE_FORMAT, jobNetworksOf, parseProxyState, shippedProxyDrift } from "./egress-proxy-state.mjs";
|
|
49
|
+
import { DEFAULT_VALKEY_PORT, NETNS_KEEPER, NETNS_KEEPER_FORMAT, QUADLET_FILES, STACK_KEYS, applyStack, decideValkey, describeRollBack, passwdNameFrom, readSubuidRanges, rollBackWrites, valkeySharedOn, VALKEY_SHARED_KEY, readValkeyKeys, valkeyTarget, judgeNetnsKeeper, keeperUnderRunningProxyHint, managerEnvRefusal, describeAction, foreignContainerRefusal, foreignContainers, lingerNote, planStack, readLinger, readStackKeys, stackComponents, unknownContainerRefusal, userBusRefusal } from "./podman-stack.mjs";
|
|
50
|
+
|
|
51
|
+
// The shipped Quadlet templates, module-relative like service.mjs's: worker/deploy in a checkout, <pkg>/deploy under npm.
|
|
52
|
+
const TEMPLATES_DIR = resolve(dirname(fileURLToPath(import.meta.url)), "..", "deploy");
|
|
29
53
|
|
|
30
54
|
// The one image up may ever fetch, and the local name jobs run under. Literal on purpose (not
|
|
31
55
|
// env.PI_JOB_IMAGE): an operator who pointed PI_JOB_IMAGE elsewhere has outgrown the quickstart, and
|
|
@@ -34,35 +58,20 @@ const UPSTREAM_IMAGE = "ghcr.io/edgehero/pi-job:latest";
|
|
|
34
58
|
const LOCAL_IMAGE = "pi-job:latest";
|
|
35
59
|
const PULL_ARGS = ["pull", UPSTREAM_IMAGE];
|
|
36
60
|
const TAG_ARGS = ["tag", UPSTREAM_IMAGE, LOCAL_IMAGE];
|
|
61
|
+
// The same two, into THIS account's rootless store (issue #430): a rootless Podman sees neither root's images nor
|
|
62
|
+
// docker's. Podman's CLI takes the same argv, and `podman tag` of the short name makes `localhost/pi-job:latest`, which
|
|
63
|
+
// `--pull=never` resolves `pi-job:latest` to with no registry lookup (measured, docs/podman.md step 5).
|
|
64
|
+
const PODMAN_EXISTS_ARGS = ["image", "exists", LOCAL_IMAGE];
|
|
37
65
|
|
|
38
66
|
// deploy/docker-compose.yml's Valkey service, reproduced as one docker run: same image, AOF on
|
|
39
67
|
// (REQ-QUEUE-BURST-NO-DROP), bound to localhost only (the queue is not a public surface), same
|
|
40
|
-
// healthcheck, restart unless-stopped, and a named volume standing in for the compose volume.
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
"--restart",
|
|
48
|
-
"unless-stopped",
|
|
49
|
-
"-p",
|
|
50
|
-
"127.0.0.1:6379:6379",
|
|
51
|
-
"-v",
|
|
52
|
-
"pi-dispatch-valkey-data:/data",
|
|
53
|
-
"--health-cmd",
|
|
54
|
-
"valkey-cli ping",
|
|
55
|
-
"--health-interval",
|
|
56
|
-
"10s",
|
|
57
|
-
"--health-timeout",
|
|
58
|
-
"3s",
|
|
59
|
-
"--health-retries",
|
|
60
|
-
"5",
|
|
61
|
-
"valkey/valkey:8",
|
|
62
|
-
"valkey-server",
|
|
63
|
-
"--appendonly",
|
|
64
|
-
"yes",
|
|
65
|
-
];
|
|
68
|
+
// healthcheck, restart unless-stopped, and a named volume standing in for the compose volume. Issue #468: its password
|
|
69
|
+
// travels as `-e VALKEY_PASSWORD` with the value in the docker CLI's own environment, never on an argv, and reaches
|
|
70
|
+
// valkey-server as configuration on stdin (valkey-auth.mjs holds the one copy of this argv; doctor's fix runs it too).
|
|
71
|
+
// The upgrade (issue #468): a pi-dispatch-valkey container that answers without a password is stopped (SIGTERM, so
|
|
72
|
+
// Valkey writes its AOF out), removed, and run again from the argv above. The volume, and the queue in it, is kept.
|
|
73
|
+
const VALKEY_STOP_ARGS = ["stop", "pi-dispatch-valkey"];
|
|
74
|
+
const VALKEY_RM_ARGS = ["rm", "pi-dispatch-valkey"];
|
|
66
75
|
|
|
67
76
|
// deploy/docker-compose.yml's `egress` profile, reproduced as one docker run (REQ-EGRESS-ALLOWLIST):
|
|
68
77
|
// same digest-pinned image, same two mounts, same explicit container name, same restart policy, on the
|
|
@@ -73,6 +82,13 @@ const VALKEY_RUN_ARGS = [
|
|
|
73
82
|
// The per-job networks are NOT here: the worker creates one per job and attaches this container to it for
|
|
74
83
|
// the life of that run. This is only the proxy and its way out.
|
|
75
84
|
const EGRESS_NETWORK_ARGS = ["network", "create", "pi-dispatch-egress-out"];
|
|
85
|
+
// A stopped CURRENT shipped proxy is started (a paused one unpaused) as it is; a STALE one is removed and run again from
|
|
86
|
+
// the argv below (issue #453; the reasons are at step e2).
|
|
87
|
+
const EGRESS_START_ARGS = ["start", "pi-dispatch-egress-proxy"];
|
|
88
|
+
const EGRESS_UNPAUSE_ARGS = ["unpause", "pi-dispatch-egress-proxy"];
|
|
89
|
+
const EGRESS_RM_ARGS = ["rm", "-f", "pi-dispatch-egress-proxy"];
|
|
90
|
+
// Issue #484: how a running proxy reads refreshed rules (see the refresh step in `runUp` for why a restart).
|
|
91
|
+
const EGRESS_RESTART_ARGS = ["restart", "pi-dispatch-egress-proxy"];
|
|
76
92
|
const EGRESS_RUN_ARGS = [
|
|
77
93
|
"run",
|
|
78
94
|
"-d",
|
|
@@ -82,32 +98,147 @@ const EGRESS_RUN_ARGS = [
|
|
|
82
98
|
"unless-stopped",
|
|
83
99
|
"--network",
|
|
84
100
|
"pi-dispatch-egress-out",
|
|
101
|
+
// `z` for the same measured reason as the compose file's mounts (issue #355): on an SELinux-enforcing host squid
|
|
102
|
+
// cannot read an unlabelled config and crash-loops; `z` (shared, never `Z`) relabels the one file so any container
|
|
103
|
+
// may read it, and is a no-op where SELinux is off.
|
|
85
104
|
"-v",
|
|
86
|
-
"./deploy/egress-proxy.conf:/etc/squid/squid.conf:ro",
|
|
105
|
+
"./deploy/egress-proxy.conf:/etc/squid/squid.conf:ro,z",
|
|
87
106
|
"-v",
|
|
88
|
-
"./egress-allowlist.conf:/etc/pi-dispatch/allowlist.conf:ro",
|
|
89
|
-
|
|
107
|
+
"./egress-allowlist.conf:/etc/pi-dispatch/allowlist.conf:ro,z",
|
|
108
|
+
EGRESS_PROXY_IMAGE,
|
|
90
109
|
];
|
|
91
110
|
|
|
111
|
+
/**
|
|
112
|
+
* Whether `p` is `folder` itself or sits inside it, compared on the separator rather than on a prefix so
|
|
113
|
+
* `/srv/deploy-old` is not read as being inside `/srv/deploy`.
|
|
114
|
+
*/
|
|
115
|
+
function underFolder(p, folder, platform) {
|
|
116
|
+
const norm = (x) => {
|
|
117
|
+
const slashed = String(x).replace(/\\/g, "/").replace(/\/+$/, "");
|
|
118
|
+
// Windows paths are case-insensitive and accept either separator, and `nssm-install.cmd` is a
|
|
119
|
+
// supported deployment: a byte compare there misses `c:\pi\deploy\logs` against `C:/pi/deploy`
|
|
120
|
+
// and writes exactly the value this guard exists to refuse. POSIX is case-SENSITIVE, so folding
|
|
121
|
+
// there would refuse a different directory that merely looks alike.
|
|
122
|
+
return platform === "win32" ? slashed.toLowerCase() : slashed;
|
|
123
|
+
};
|
|
124
|
+
const a = norm(p);
|
|
125
|
+
const b = norm(folder);
|
|
126
|
+
return a === b || a.startsWith(`${b}/`);
|
|
127
|
+
}
|
|
128
|
+
|
|
92
129
|
export async function runUp(argv = [], deps = {}) {
|
|
93
130
|
const {
|
|
94
131
|
env = process.env,
|
|
95
132
|
spawn = nodeSpawn,
|
|
96
133
|
out = (s) => process.stdout.write(s),
|
|
97
134
|
prompt = defaultPrompt,
|
|
98
|
-
|
|
135
|
+
// `realpathSync` is in this list for a reason worth keeping: `updateEnvFile` resolves the path so a
|
|
136
|
+
// `.env` symlinked at a shared env file is edited THROUGH the link rather than replaced by a
|
|
137
|
+
// regular file. It calls it optionally, so leaving it out here made that repair dead code in the
|
|
138
|
+
// only production caller, and the test that covered it attached the method to its own fake.
|
|
139
|
+
// `lstatSync` for the rules refresh (issue #484), which follows no symlink.
|
|
140
|
+
fs = { existsSync, lstatSync, readFileSync, writeFileSync, renameSync, statSync, chmodSync, realpathSync, unlinkSync },
|
|
99
141
|
probeTcp = defaultProbeTcp,
|
|
142
|
+
// Issue #468: whether the Valkey on 127.0.0.1:6379 answers a client that sends NO password ("ok" when it does),
|
|
143
|
+
// through the project's one connection function. Real only where the TCP probe is: a test that stands in for the
|
|
144
|
+
// host's listener answers this too, or gets "unknown".
|
|
145
|
+
probeValkeyAuth = deps.probeTcp === undefined ? (url) => defaultProbeValkeyAuth(url, { env, cwd: deps.cwd ?? process.cwd() }) : async () => "unknown",
|
|
146
|
+
// PR #475's review (the volume gap): record, then read back, the `pi-dispatch:owner` marker in a Valkey this pass
|
|
147
|
+
// started (connection.mjs' `claimValkeyOwner`). Real only where the TCP probe is, like the one above: a test that
|
|
148
|
+
// stands in for the host's listener answers this too, or gets "this folder".
|
|
149
|
+
claimValkeyOwner = deps.probeTcp === undefined ? (url, folder) => defaultClaimValkeyOwner(url, folder, { env, cwd: deps.cwd ?? process.cwd() }) : async (_url, folder) => ({ owner: folder, claimed: false }),
|
|
150
|
+
// Issue #464: VALKEY_URL's host resolved as the worker's client resolves it (every address), and which addresses
|
|
151
|
+
// are this host's, for the Valkey owner rule on the podman venue.
|
|
152
|
+
lookup = (host, opts) => dnsLookup(host, opts),
|
|
153
|
+
interfaces = networkInterfaces,
|
|
154
|
+
// Issue #464 (gate round 2): how `getsubids` is run for this account's subordinate ranges (null: not installed).
|
|
155
|
+
runSync = undefined,
|
|
156
|
+
// PR #463: how a path resolves through symlinks, for the manager-environment rule (a symlinked home).
|
|
157
|
+
realpath = (p) => realpathSync(p),
|
|
100
158
|
cwd = process.cwd(),
|
|
101
159
|
// Injected so tests can assert the secret never reaches output without fishing it back out of
|
|
102
160
|
// the written file. 32 bytes hex, matching doctor's `openssl rand -hex 32` fix line.
|
|
161
|
+
// The owner check's wait while its Valkey loads the queue (PR #475's round-cap re-review); a seam for its test.
|
|
162
|
+
sleep = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
103
163
|
randomHex = () => randomBytes(32).toString("hex"),
|
|
164
|
+
// Issue #468: the Valkey password, made the same way (valkey-auth.mjs), injected so a test can hold it never
|
|
165
|
+
// reaches the output.
|
|
166
|
+
newPassword = newValkeyPassword,
|
|
167
|
+
// Injected for the path compare below, which has to fold case on Windows and must not on POSIX.
|
|
168
|
+
platform = process.platform,
|
|
169
|
+
// The two resolved defaults `up` pins, injected for the same reason the clock is elsewhere: they
|
|
170
|
+
// read the real `homedir()`, so a test asserting the written `.env` byte for byte would otherwise
|
|
171
|
+
// assert whichever account ran it.
|
|
172
|
+
logsDirPathFn = logsDirPath,
|
|
173
|
+
settingsFilePathFn = settingsFilePath,
|
|
174
|
+
// The podman venue's seams (issue #430). The info read is the venue's own bounded reader, so `up` asks exactly
|
|
175
|
+
// the question the worker asks at boot; the ids are the ones the job-user rule judges.
|
|
176
|
+
readPodmanInfo = makePodmanInfoReader(),
|
|
177
|
+
euid = typeof process.geteuid === "function" ? process.geteuid() : null,
|
|
178
|
+
egid = typeof process.getegid === "function" ? process.getegid() : null,
|
|
179
|
+
home = homedir(),
|
|
180
|
+
// Whose linger `up` reads after starting the podman stack: the login already running it, as service.mjs reads
|
|
181
|
+
// it. Resolved LAZILY (`userName` below), never as a default here: `userInfo()` throws for a uid with no passwd
|
|
182
|
+
// entry, and the docker path, which never asks, must not die of it.
|
|
183
|
+
user,
|
|
184
|
+
userInfoFn = userInfo,
|
|
185
|
+
templatesDir = TEMPLATES_DIR,
|
|
186
|
+
// The shipped Quadlet templates are the package's files, read for real even where `fs` is a test's fake of the
|
|
187
|
+
// deployment folder; injectable so a test can hand in its own.
|
|
188
|
+
readTemplate = (name) => readPackageFile(join(templatesDir, name), "utf8"),
|
|
189
|
+
mkdir = mkdirSync,
|
|
190
|
+
// Issue #484: the installed package's rules, which a differing folder copy is compared with and refreshed from, and
|
|
191
|
+
// the clock the backup's name is stamped with.
|
|
192
|
+
readPackagedConf = readPackagedProxyConf,
|
|
193
|
+
now = Date.now,
|
|
104
194
|
} = deps;
|
|
105
195
|
let { runInitFn, runDoctorFn } = deps;
|
|
106
196
|
const yes = argv.includes("--yes");
|
|
107
197
|
const summary = [];
|
|
198
|
+
// env-internal USER: the login running `up`, only ever read on the podman path (see the `user` seam above).
|
|
199
|
+
const userName = () => user ?? (env.USER || userInfoFn().username);
|
|
108
200
|
|
|
201
|
+
// The venue keys, from the same place `service install` reads them (issue #430 review, D2). Nothing loads `.env`
|
|
202
|
+
// into a process, so `up` used to decide the venue from this shell alone: an operator who followed the docs and put
|
|
203
|
+
// PI_BACKENDS=podman in `.env` got the docker pass, and the service they installed next got podman. A key the shell
|
|
204
|
+
// sets and the file does not is taken from the shell (what the wizard relies on when it runs `up` with
|
|
205
|
+
// PI_BACKENDS=podman before init has written `.env`); `.env` fills the keys the shell leaves unset; a key BOTH set
|
|
206
|
+
// differently stops `up` (round 2, E5), because the service will run the file's value. A `.env` line touching one
|
|
207
|
+
// of these keys that the loaders read differently stops `up` before anything runs, as `service install` refuses.
|
|
208
|
+
const venue = deploymentVenueEnv({ env, fs, envPath: join(cwd, ".env"), platform, command: "up" });
|
|
209
|
+
if (venue.error) {
|
|
210
|
+
out(`✗ ${venue.error}\n\nup: cannot tell which venue this deployment runs: fix the above, then re-run \`pi-dispatch up\`.\n`);
|
|
211
|
+
return 1;
|
|
212
|
+
}
|
|
213
|
+
for (const line of venue.notes) out(`⚠ ${line}\n`);
|
|
214
|
+
const venueEnv = venue.env;
|
|
215
|
+
|
|
216
|
+
// Which runtimes this pass drives (issue #430). With `local` blessed (which the unset default is) everything below
|
|
217
|
+
// is exactly what it always was, and a list that also names podman adds the podman steps after the docker ones. A
|
|
218
|
+
// list WITHOUT `local` never touches docker at all: such a host may have no docker, and asking it would only fail.
|
|
219
|
+
const venues = venuesOf(venueEnv);
|
|
220
|
+
const dockerUsed = venues.localUsed;
|
|
221
|
+
// Issue #464: which Valkey the SERVICE's worker will use, and whether PI_VALKEY_SHARED lets it be one another uid
|
|
222
|
+
// holds, read from `.env` where `service install` reads them, with this shell's value only where the file sets none.
|
|
223
|
+
// Asked only where the podman stack decides its Valkey (docker's Valkey is the queue wherever `local` is blessed),
|
|
224
|
+
// and before anything runs, as the venue keys are.
|
|
225
|
+
let valkeyKeys = { env: {}, fromFile: {}, notes: [] };
|
|
226
|
+
if (venues.podmanUsed && !venues.localUsed) {
|
|
227
|
+
valkeyKeys = deploymentValkeyEnv({ env, fs, envPath: join(cwd, ".env"), platform });
|
|
228
|
+
if (valkeyKeys.error) {
|
|
229
|
+
out(`✗ ${valkeyKeys.error}\n\nup: cannot tell which Valkey this deployment's worker uses: fix the above, then re-run \`pi-dispatch up\`.\n`);
|
|
230
|
+
return 1;
|
|
231
|
+
}
|
|
232
|
+
for (const line of valkeyKeys.notes) out(`⚠ ${line}\n`);
|
|
233
|
+
}
|
|
234
|
+
//
|
|
235
|
+
// The docker lines below keep their original text and indentation, the `else` and the unbraced block included, so
|
|
236
|
+
// that the diff of issue #430 shows the docker path as the byte-identical thing its tests pin it to be.
|
|
237
|
+
if (!dockerUsed) out("pi-dispatch up: one pass over the quickstart for the podman venue; every podman and systemctl action asks first (--yes accepts)\n\n");
|
|
238
|
+
else
|
|
109
239
|
out("pi-dispatch up — one pass over the quickstart; every docker action asks first (--yes accepts)\n\n");
|
|
110
240
|
|
|
241
|
+
if (dockerUsed) {
|
|
111
242
|
// (a) docker binary + daemon, before anything is offered: every mutation below runs through the
|
|
112
243
|
// docker CLI, so with the daemon down the prompts would only collect consent for failures.
|
|
113
244
|
// Distinguishes not-on-PATH from daemon-down exactly as doctor does (spawn error vs nonzero exit).
|
|
@@ -144,33 +275,24 @@ export async function runUp(argv = [], deps = {}) {
|
|
|
144
275
|
}
|
|
145
276
|
}
|
|
146
277
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
//
|
|
150
|
-
//
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
if (!accepted) {
|
|
163
|
-
out("skipped — start it later with `docker compose -f deploy/docker-compose.yml up -d`\n");
|
|
164
|
-
summary.push(["valkey", "skipped (declined) — the queue needs it before `pi-dispatch worker` can drain"]);
|
|
165
|
-
} else if (await runStreamed(spawn, "docker", VALKEY_VOLUME_ARGS, out) !== 0) {
|
|
166
|
-
out("✗ docker volume create failed — continuing; doctor below will re-check Valkey\n");
|
|
167
|
-
summary.push(["valkey", "volume create FAILED — `docker compose -f deploy/docker-compose.yml up -d` is the fallback"]);
|
|
168
|
-
} else if (await runStreamed(spawn, "docker", VALKEY_RUN_ARGS, out) !== 0) {
|
|
169
|
-
out("✗ docker run failed — continuing; doctor below will re-check Valkey\n");
|
|
170
|
-
summary.push(["valkey", "container start FAILED — `docker compose -f deploy/docker-compose.yml up -d` is the fallback"]);
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// (a2)(b2) the podman venue: its gate and its job image, in this account's own store (issue #430). `podmanReady`
|
|
281
|
+
// is what lets the stack step below run at all.
|
|
282
|
+
let podmanReady = false;
|
|
283
|
+
let valkeyRefused = false;
|
|
284
|
+
if (venues.podmanUsed) {
|
|
285
|
+
const gate = await podmanGate({ readPodmanInfo, platform, euid, egid, out });
|
|
286
|
+
if (!gate.ok) {
|
|
287
|
+
summary.push(["podman", `NOT ready: ${gate.why}`]);
|
|
288
|
+
if (!dockerUsed) {
|
|
289
|
+
out("\nup: cannot continue without a usable rootless Podman: fix the above, then re-run `pi-dispatch up`.\n");
|
|
290
|
+
return 1;
|
|
291
|
+
}
|
|
292
|
+
out(" the docker steps above are unaffected; the podman steps below are skipped\n");
|
|
171
293
|
} else {
|
|
172
|
-
|
|
173
|
-
|
|
294
|
+
podmanReady = true;
|
|
295
|
+
await podmanImageStep({ spawn, out, yes, prompt, summary });
|
|
174
296
|
}
|
|
175
297
|
}
|
|
176
298
|
|
|
@@ -179,24 +301,453 @@ export async function runUp(argv = [], deps = {}) {
|
|
|
179
301
|
// nothing the operator wrote can be lost here.
|
|
180
302
|
out("\ninit (never overwrites — an existing file is reported and kept):\n");
|
|
181
303
|
runInitFn ??= (await import("./init.mjs")).runInit;
|
|
182
|
-
|
|
183
|
-
|
|
304
|
+
// The venues THIS pass decided (issue #453), so init decides nothing of its own here. And no "Next:" from init (issue
|
|
305
|
+
// #480): that ladder is this pass, so printing it here told a reader to pull the image and start Valkey by hand while
|
|
306
|
+
// `up` was doing both; the created and kept lines stay, and `up` closes with its own next steps below.
|
|
307
|
+
// A throw here (a folder init cannot write, PR #488's review) is said and the pass goes on: the steps below and doctor
|
|
308
|
+
// still say what else is missing, and an aborted `up` said nothing after the image pull.
|
|
309
|
+
let initCode;
|
|
310
|
+
try {
|
|
311
|
+
initCode = runInitFn(cwd, { out, venues, steps: false });
|
|
312
|
+
} catch (err) {
|
|
313
|
+
initCode = null;
|
|
314
|
+
out(`✗ init could not finish: ${err?.message ?? err}\n`);
|
|
315
|
+
}
|
|
316
|
+
if (initCode === 1) summary.push(["init", "ran, and REFUSED a file (said above); the others were kept or scaffolded"]);
|
|
317
|
+
else if (initCode === null) summary.push(["init", "FAILED (said above); the steps below still ran"]);
|
|
318
|
+
else summary.push(["init", "ran: existing files were kept untouched, missing ones scaffolded"]);
|
|
184
319
|
|
|
185
320
|
// (e) WEBHOOK_SECRET, only into a .env that exists (init just scaffolded one unless the operator
|
|
186
321
|
// keeps env elsewhere — a service-manager deployment gets no file invented for it). Same
|
|
187
322
|
// never-clobber contract at key granularity: a value the operator set survives. The value itself
|
|
188
323
|
// is NEVER printed — a webhook secret in a scrollback is a webhook secret in a pastebin.
|
|
189
324
|
const envPath = join(cwd, ".env");
|
|
325
|
+
// WHAT AN EMPTY VALUE COSTS IS PER KEY, and the first version of this said one thing for all five.
|
|
326
|
+
// `config.mjs` reads these two with `??`, so an empty string survives, and `start.mjs` calls
|
|
327
|
+
// `loadPauseWindows`/`loadScopedLimits` unconditionally at boot, which throw on a path that does not
|
|
328
|
+
// exist: the worker does not ignore the feature, it refuses to start. `PI_LOGS_DIR` and
|
|
329
|
+
// `PI_SETTINGS_FILE` use `||` and fall back to the account default; `WEBHOOK_SECRET` reads as absent.
|
|
330
|
+
const EMPTY_REFUSES_BOOT = new Set(["PI_PAUSE_WINDOWS_FILE", "PI_SCOPED_LIMITS_FILE"]);
|
|
331
|
+
const emptyNote = (key) =>
|
|
332
|
+
EMPTY_REFUSES_BOOT.has(key)
|
|
333
|
+
? `left untouched: the line is there and its value is EMPTY, which is not the same as no line -- a shell that sources this file exports it as "", the worker keeps it and REFUSES TO BOOT. up never clobbers a key an operator wrote, so fill it in or delete the line`
|
|
334
|
+
: `left untouched: the line is there and its value is empty, which reads as unset. up never clobbers a key an operator wrote, so fill it in or delete the line`;
|
|
335
|
+
// The blank test is `env-file.mjs`'s, and the reader underneath it is the one doctor uses -- which is
|
|
336
|
+
// what issue #365 was actually about: two callers answering "is this key set" with two hand-rolled
|
|
337
|
+
// regexes is how `up` and doctor came to print opposite sentences about one file in one run.
|
|
338
|
+
//
|
|
339
|
+
// NOT the same CALL, since issue #384, and the difference is the consequence rather than the question.
|
|
340
|
+
// `up` prints a sentence, so it asks the loose per-line answer and prefers "EMPTY" to "already set" on a
|
|
341
|
+
// file it cannot fully read. Doctor's answer becomes an exit code, so it asks the same reader for the
|
|
342
|
+
// platform's own loader and requires the vouch beside it: a refusal reached by inference is the one
|
|
343
|
+
// verdict this project will not print.
|
|
344
|
+
const writtenButEmpty = (key) => {
|
|
345
|
+
try {
|
|
346
|
+
return envKeyIsBlank(String(fs.readFileSync(envPath, "utf8")), key);
|
|
347
|
+
} catch {
|
|
348
|
+
return false;
|
|
349
|
+
}
|
|
350
|
+
};
|
|
351
|
+
// Windows takes forward slashes everywhere Node does, and writing them is what keeps these four values
|
|
352
|
+
// inside the BARE set: `deploy/worker-env-wrapper.cmd` says "Values MUST be UNQUOTED -- cmd's `set`
|
|
353
|
+
// keeps surrounding quotes as part of the value", so a quoted `C:\pi\deploy\logs` there is a
|
|
354
|
+
// directory that does not exist, behind a ✓. `config.mjs`'s own defaults are already spelled this way.
|
|
355
|
+
const forEnvFile = (value) => (platform === "win32" ? String(value).replace(/\\/g, "/") : value);
|
|
356
|
+
// What `up` wrote, layered over `env` for its OWN doctor step below. Writing a line into `.env`
|
|
357
|
+
// configures the SERVICE (through `EnvironmentFile=` and the wrappers) and configures nothing about the
|
|
358
|
+
// process running right now, because NOTHING in this project loads `.env` into an environment -- see
|
|
359
|
+
// `docs/secrets.md`, and `worker/test/service.test.mjs` pins that a `PI_ENV_SETUP` line in `./.env` is
|
|
360
|
+
// deliberately not honoured. Without this layer, `up` would write the two files' paths and then, three
|
|
361
|
+
// lines later, warn that they are unset (issue #357).
|
|
362
|
+
const wrote = {};
|
|
190
363
|
if (fs.existsSync(envPath)) {
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
364
|
+
// Wrapped like the four below it, and for the same reasons: a read-only deployment directory, a full
|
|
365
|
+
// disk, a `.env` this account does not own. Unwrapped, `up` died with a raw stack trace where a
|
|
366
|
+
// summary row was the whole point.
|
|
367
|
+
try {
|
|
368
|
+
if (updateEnvFile(envPath, "WEBHOOK_SECRET", randomHex(), { fs, platform }).changed) {
|
|
369
|
+
out("\n✓ generated WEBHOOK_SECRET into .env (32 random bytes, hex — value not shown)\n");
|
|
370
|
+
summary.push(["WEBHOOK_SECRET", "generated into .env (value not shown; the receiver verifies deliveries with it)"]);
|
|
371
|
+
} else {
|
|
372
|
+
const empty = writtenButEmpty("WEBHOOK_SECRET");
|
|
373
|
+
// ⚠ and not ✓ for the empty line: every other ✓ in this block means "this is fine", and an empty
|
|
374
|
+
// secret is a key the operator has to go and fill in.
|
|
375
|
+
out(empty ? `\n⚠ WEBHOOK_SECRET has a line in .env and its value is EMPTY — left untouched\n` : "\n✓ WEBHOOK_SECRET already set in .env — left untouched\n");
|
|
376
|
+
summary.push(["WEBHOOK_SECRET", empty ? emptyNote("WEBHOOK_SECRET") : "already set — left untouched"]);
|
|
377
|
+
}
|
|
378
|
+
} catch (err) {
|
|
379
|
+
out(`\n✗ WEBHOOK_SECRET could not be written: ${err?.message}\n`);
|
|
380
|
+
summary.push(["WEBHOOK_SECRET", `NOT written: ${err?.message}`]);
|
|
381
|
+
}
|
|
382
|
+
// (e1) the four paths four separate files already promise `up` writes, and it never did
|
|
383
|
+
// (`deploy/worker.service`, `deploy/com.pi-dispatch.worker.plist`, `deploy/nssm-install.cmd`,
|
|
384
|
+
// `.env.example`). Same never-clobber discipline as WEBHOOK_SECRET, at key granularity: a value the
|
|
385
|
+
// operator set survives untouched.
|
|
386
|
+
//
|
|
387
|
+
// TWO get the deployment folder and TWO get the RESOLVED ACCOUNT DEFAULT, and the split is the
|
|
388
|
+
// sharpest edge in this change rather than an inconsistency. `pause-windows.json` and
|
|
389
|
+
// `scoped-limits.json` are scaffolded by `init` into this folder, and the panel defaults to this
|
|
390
|
+
// folder, so pointing the worker here is what makes the three agree. `PI_LOGS_DIR` and
|
|
391
|
+
// `PI_SETTINGS_FILE` are different in kind: `makeLogReaper` unlinks EVERY `.log` and `.json` in
|
|
392
|
+
// `PI_LOGS_DIR` past the window with no name shape and no ownership check, so a deployment folder
|
|
393
|
+
// there would eat `triggers.json`, `pause-windows.json`, `scoped-limits.json` and
|
|
394
|
+
// `subscriptions.json` thirty days in, silently, and the worker would then run nothing while
|
|
395
|
+
// reporting success. `<deployment>/logs` is no better: `service.mjs` creates exactly that directory
|
|
396
|
+
// at install time and the plist puts `worker.out.log` in it. So these two get what
|
|
397
|
+
// `logsDirPath`/`settingsFilePath` would have resolved anyway: behaviour byte-unchanged, the value
|
|
398
|
+
// simply made explicit, which is the whole point -- a worker under another `User=` and the panel
|
|
399
|
+
// then cannot silently resolve two different directories.
|
|
400
|
+
for (const [key, raw, durable] of [
|
|
401
|
+
["PI_PAUSE_WINDOWS_FILE", join(cwd, "pause-windows.json"), false],
|
|
402
|
+
["PI_SCOPED_LIMITS_FILE", join(cwd, "scoped-limits.json"), false],
|
|
403
|
+
["PI_LOGS_DIR", logsDirPathFn(env), true],
|
|
404
|
+
["PI_SETTINGS_FILE", settingsFilePathFn(env), true],
|
|
405
|
+
]) {
|
|
406
|
+
const value = forEnvFile(raw);
|
|
407
|
+
// `logsDirPath` and `settingsFilePath` RESOLVE rather than default: `env.PI_LOGS_DIR || <default>`.
|
|
408
|
+
// So on a shell that already exports one of them at the deployment folder -- which three shipped
|
|
409
|
+
// files told operators to do for a year, before this change made the promise true -- `up` would
|
|
410
|
+
// persist exactly the value the rest of this block exists to prevent, and print a ✓ over it. The
|
|
411
|
+
// refusal is here rather than in the resolver because the resolver is right for every other
|
|
412
|
+
// caller: the worker SHOULD honour an exported path. What must never happen is `up` writing one
|
|
413
|
+
// into the deployment's permanent config without anyone deciding to.
|
|
414
|
+
// TWO refusals, and both are about the same resolver. `logsDirPath` is `env.PI_LOGS_DIR || <default>`,
|
|
415
|
+
// so whatever this shell exports passes straight through, unvalidated and un-absolutised.
|
|
416
|
+
//
|
|
417
|
+
// RELATIVE is the sharper of the two and it is not hypothetical: `PI_LOGS_DIR=.` resolves against
|
|
418
|
+
// the unit's `WorkingDirectory`, which IS the deployment folder, so the retention sweep then
|
|
419
|
+
// deletes `triggers.json`, `pause-windows.json`, `scoped-limits.json` and `subscriptions.json`
|
|
420
|
+
// thirty days in. All three deploy templates document this key as absolute for exactly that
|
|
421
|
+
// reason. INSIDE THIS FOLDER is the same harm reached with an absolute path.
|
|
422
|
+
//
|
|
423
|
+
// The refusals live here and not in the resolver, because the resolver is right for the worker:
|
|
424
|
+
// an exported path SHOULD be honoured at run time. What must never happen is `up` copying one
|
|
425
|
+
// into the deployment's permanent config, where it outlives the shell that set it.
|
|
426
|
+
// Only a value THIS SHELL supplied is checked, and that narrowing matters both ways. The hazard
|
|
427
|
+
// is the pass-through: `logsDirPath` is `env.PI_LOGS_DIR || <default>`, so an exported value
|
|
428
|
+
// lands in the deployment's permanent config where it outlives the shell that set it. The
|
|
429
|
+
// computed default is never a hazard even when it sits under `cwd` -- a deployment folder that
|
|
430
|
+
// IS the service account's home makes `<home>/.pi-dispatch/logs` "inside" it, and refusing
|
|
431
|
+
// there would reject the very path the worker resolves anyway, with a message blaming the
|
|
432
|
+
// operator for a layout that is fine.
|
|
433
|
+
const fromShell = typeof env[key] === "string" && env[key].trim() !== "";
|
|
434
|
+
// Both checks follow the INJECTED platform, not this process's. `node:path`'s default export is
|
|
435
|
+
// already the right one in production, but a drive-letter path is "relative" to the posix
|
|
436
|
+
// implementation, so a test running on POSIX would see the absolute check short-circuit and
|
|
437
|
+
// never reach the folder compare it meant to exercise. Choosing explicitly makes the Windows
|
|
438
|
+
// half of both rules reachable from a test on any host, which is the only way `nssm-install.cmd`
|
|
439
|
+
// gets covered at all.
|
|
440
|
+
const isAbsoluteOn = platform === "win32" ? win32.isAbsolute : posix.isAbsolute;
|
|
441
|
+
const why = !durable || !fromShell ? null : !isAbsoluteOn(value) ? "is not an absolute path, and a relative one resolves against the service's WorkingDirectory, which is this folder" : underFolder(value, cwd, platform) ? "is inside this deployment folder" : null;
|
|
442
|
+
if (why) {
|
|
443
|
+
out(`✗ ${key} is ${value} in this shell, which ${why} — not written into .env\n`);
|
|
444
|
+
summary.push([key, `NOT written: ${key} is set to ${value} in the environment you ran up from, and that ${why}. The log retention sweep deletes every .log and .json there past its window, so this would be persisted for the service. Unset it here, or set it by hand to a directory the worker owns`]);
|
|
445
|
+
continue;
|
|
446
|
+
}
|
|
447
|
+
// A path this file cannot represent so that BOTH consumers read it back is refused rather than
|
|
448
|
+
// mangled: `renderEnvValue` throws on a single quote, which the shells want escaped as `'\''`
|
|
449
|
+
// and systemd's parser does not understand.
|
|
450
|
+
let changed;
|
|
451
|
+
try {
|
|
452
|
+
({ changed } = updateEnvFile(envPath, key, value, { fs, platform }));
|
|
453
|
+
} catch (err) {
|
|
454
|
+
out(`✗ ${key} could not be written: ${err?.message}\n`);
|
|
455
|
+
summary.push([key, `NOT written: ${err?.message}. Set it by hand, or move the deployment somewhere without that character in its path`]);
|
|
456
|
+
continue;
|
|
457
|
+
}
|
|
458
|
+
if (changed) {
|
|
459
|
+
wrote[key] = value;
|
|
460
|
+
out(`✓ ${key}=${value} written into .env\n`);
|
|
461
|
+
summary.push([key, `written into .env (${value})`]);
|
|
462
|
+
} else {
|
|
463
|
+
summary.push([key, writtenButEmpty(key) ? emptyNote(key) : "already set — left untouched"]);
|
|
464
|
+
}
|
|
197
465
|
}
|
|
198
466
|
} else {
|
|
199
467
|
summary.push(["WEBHOOK_SECRET", "no .env here — skipped (set it wherever your env lives)"]);
|
|
468
|
+
for (const key of ["PI_PAUSE_WINDOWS_FILE", "PI_SCOPED_LIMITS_FILE", "PI_LOGS_DIR", "PI_SETTINGS_FILE"]) {
|
|
469
|
+
summary.push([key, "no .env here — skipped (set it wherever your env lives)"]);
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
// (e0) VALKEY_PASSWORD (issue #468), where docker's Valkey is the queue: generated into .env when the deployment has
|
|
474
|
+
// none and runs a Valkey of its own on this host (`valkeyPasswordDecision`), never over a value, the file narrowed to
|
|
475
|
+
// this account, the value never printed. The podman venue's Valkey gets its password in the stack step below.
|
|
476
|
+
const dockerValkeyPassword = dockerUsed ? dockerValkeyPasswordStep({ fs, envPath, platform, out, summary, newPassword }) : null;
|
|
477
|
+
|
|
478
|
+
if (dockerUsed) {
|
|
479
|
+
// (c) Valkey. A bare TCP probe of the default bind, not a redis PING: dependency-free, and the
|
|
480
|
+
// honest claim is only "something is listening". If our own compose-named container is up, docker
|
|
481
|
+
// can say so; if a listener exists that we cannot name, up must NOT offer a second Valkey; the
|
|
482
|
+
// port is taken, and `docker run` would only fail after consent.
|
|
483
|
+
//
|
|
484
|
+
// AFTER init and the .env keys since issue #468, where it used to run first: the Valkey this step starts takes its
|
|
485
|
+
// password from `.env`, which a fresh folder has only once init has scaffolded it. The prompts keep their order
|
|
486
|
+
// (the image, then Valkey, then the egress proxy), since init and the .env keys ask nothing.
|
|
487
|
+
const dockerEnv = { ...env };
|
|
488
|
+
// The deployment's password, and never a shell's: the container must start with the one the service's clients send.
|
|
489
|
+
if (dockerValkeyPassword) dockerEnv[VALKEY_PASSWORD_KEY] = dockerValkeyPassword;
|
|
490
|
+
else delete dockerEnv[VALKEY_PASSWORD_KEY];
|
|
491
|
+
// WHERE, from VALKEY_URL as the service reads it (`deploymentValkeyEnv`, the podman step's reader since #464; read
|
|
492
|
+
// here, after init, so a fresh folder's scaffolded .env counts): its loopback port is probed and published. It was
|
|
493
|
+
// always 6379, so a deployment whose VALKEY_URL named another port was told "nothing is listening" and given a
|
|
494
|
+
// Valkey its worker never dialled, or had its own listener on that port ignored. Another host: nothing is added
|
|
495
|
+
// here. An IPv6 literal cannot reach a Valkey published on 127.0.0.1, so it is refused with the fix.
|
|
496
|
+
const dockerValkey = dockerValkeyWhere({ env, fs, envPath, platform });
|
|
497
|
+
for (const note of dockerValkey.notes) out(`⚠ ${note}\n`);
|
|
498
|
+
const vport = dockerValkey.port;
|
|
499
|
+
// THE rule (PR #475's review, round 3): a Valkey container or volume is this deployment's only when it is PROVABLY
|
|
500
|
+
// this deployment's (valkey-auth.mjs, `valkeyContainerIsOurs`): up's label naming this folder, compose's working dir
|
|
501
|
+
// here, or, for a legacy unlabelled pi-dispatch-valkey, this deployment's VALKEY_URL port. Anything else is never
|
|
502
|
+
// stopped, removed or reused, and no Valkey is started on the volume while one that is not ours mounts it.
|
|
503
|
+
let deployment = cwd;
|
|
504
|
+
try {
|
|
505
|
+
deployment = realpath(cwd);
|
|
506
|
+
} catch {
|
|
507
|
+
// Unresolvable: the folder as given.
|
|
508
|
+
}
|
|
509
|
+
const dirs = [deployment, cwd];
|
|
510
|
+
const query = (cmd, args) => runCmdQuery(spawn, cmd, args);
|
|
511
|
+
const VALKEY_RUN_ARGS = valkeyDockerRunArgs({ port: vport, deployment });
|
|
512
|
+
// Labelled with this folder when up creates it (the volume gap): a label cannot be added later.
|
|
513
|
+
const VALKEY_VOLUME_ARGS = valkeyVolumeCreateArgs(deployment);
|
|
514
|
+
// PI_VALKEY_PORT into .env where VALKEY_URL's port is not 6379 (round 3), so a plain compose run in this folder
|
|
515
|
+
// publishes where the worker dials; never over a different value, which is named instead.
|
|
516
|
+
if (!dockerValkey.error && !dockerValkey.remote) dockerValkeyPortStep({ fs, envPath, platform, port: vport, out, summary });
|
|
517
|
+
// Before any start on pi-dispatch-valkey-data: refused, named, while a container that is not ours mounts it.
|
|
518
|
+
const volumeClear = async () => {
|
|
519
|
+
const check = await foreignVolumeUsers({ dirs, port: vport, query });
|
|
520
|
+
if (!check.unknown && check.foreign.length === 0) return true;
|
|
521
|
+
valkeyRefused = "volume";
|
|
522
|
+
out(`✗ ${foreignVolumeRefusal(check)}\n`);
|
|
523
|
+
summary.push(["valkey", `NOT started: ${foreignVolumeRefusal(check)}`]);
|
|
524
|
+
return false;
|
|
525
|
+
};
|
|
526
|
+
// The volume's own owner (PR #475's review, the volume gap: with no container on it, a second deployment's `up` started
|
|
527
|
+
// on the first one's queue). Another folder's label: never used. No label (made before the label, or by hand): used
|
|
528
|
+
// only after a question that --yes does not answer, since that queue cannot be attributed; `servedByOurs` where this
|
|
529
|
+
// deployment's own container serves it now (the upgrade restart), which attributes it. Sets `volumeState`.
|
|
530
|
+
let volumeState = null;
|
|
531
|
+
const volumeUsable = async ({ servedByOurs = false } = {}) => {
|
|
532
|
+
const v = await valkeyVolumeOwner({ dirs, query });
|
|
533
|
+
const refuse = (why) => {
|
|
534
|
+
valkeyRefused = "volume";
|
|
535
|
+
out(`✗ ${why}\n`);
|
|
536
|
+
summary.push(["valkey", `NOT started: ${why}`]);
|
|
537
|
+
return false;
|
|
538
|
+
};
|
|
539
|
+
if (v.unknown) return refuse(`whose pi-dispatch-valkey-data is could not be read (${v.unknown}), so no Valkey is started on it`);
|
|
540
|
+
if (v.ours === false) return refuse(foreignVolumeLabelRefusal(v.owner));
|
|
541
|
+
// Adopted before (PR #475's round-cap re-review): the record in this folder names this very volume by its CreatedAt.
|
|
542
|
+
if (v.unlabelled && volumeRecordMatches(readVolumeRecord(cwd, fs), v)) {
|
|
543
|
+
volumeState = "recorded";
|
|
544
|
+
return true;
|
|
545
|
+
}
|
|
546
|
+
if (v.unlabelled && !servedByOurs) {
|
|
547
|
+
out("\n");
|
|
548
|
+
if (!/^y(es)?$/i.test(String((await prompt(adoptVolumeQuestion(deployment))) ?? "").trim())) return refuse(unadoptedVolumeRefusal());
|
|
549
|
+
// Whose queue it holds, read BEFORE any Valkey on it is published (round-cap re-review: a wrong `y` ran this
|
|
550
|
+
// deployment's Valkey on another's data for a second): a Valkey with no network loads it, and its marker is read
|
|
551
|
+
// through `docker exec`.
|
|
552
|
+
const check = await ownerPreCheck();
|
|
553
|
+
if (check.error) return refuse(`whose queue pi-dispatch-valkey-data holds could not be read before starting (${check.error}), so no Valkey is started on it`);
|
|
554
|
+
if (check.owner && !dirs.includes(check.owner)) return refuse(`${foreignMarkerRefusal(check.owner)}: read by a Valkey with no network, so no Valkey was published on it`);
|
|
555
|
+
recordVolume(v);
|
|
556
|
+
volumeState = "adopted";
|
|
557
|
+
return true;
|
|
558
|
+
}
|
|
559
|
+
// Served by this deployment's own container now (the upgrade restart), which attributes it: recorded too.
|
|
560
|
+
if (v.unlabelled && servedByOurs) recordVolume(v);
|
|
561
|
+
volumeState = v.absent ? "absent" : v.unlabelled ? "served" : "ours";
|
|
562
|
+
return true;
|
|
563
|
+
};
|
|
564
|
+
// The owner check itself: `valkeyOwnerCheckArgs` (no network, no port, no password), the marker read through
|
|
565
|
+
// `docker exec`, retried while the queue loads, and the container stopped and removed whatever it answered.
|
|
566
|
+
const ownerPreCheck = async () => {
|
|
567
|
+
out(`checking whose queue it holds first, with a Valkey that has no network:\n docker ${valkeyOwnerCheckArgs(deployment).join(" ")}\n docker ${OWNER_CHECK_EXEC.join(" ")}\n`);
|
|
568
|
+
if ((await runStreamed(spawn, "docker", valkeyOwnerCheckArgs(deployment), out)) !== 0) return { error: `its docker run failed (a leftover ${OWNER_CHECK_CONTAINER}? \`docker rm -f ${OWNER_CHECK_CONTAINER}\`)` };
|
|
569
|
+
try {
|
|
570
|
+
for (let i = 0; i < 40; i++) {
|
|
571
|
+
const a = ownerCheckAnswer(await runCmdQuery(spawn, "docker", OWNER_CHECK_EXEC));
|
|
572
|
+
if (!a.retry) return a;
|
|
573
|
+
await sleep(500);
|
|
574
|
+
}
|
|
575
|
+
return { error: "its Valkey did not answer within 20 s" };
|
|
576
|
+
} finally {
|
|
577
|
+
await runStreamed(spawn, "docker", ["stop", OWNER_CHECK_CONTAINER], out);
|
|
578
|
+
await runStreamed(spawn, "docker", ["rm", OWNER_CHECK_CONTAINER], out);
|
|
579
|
+
}
|
|
580
|
+
};
|
|
581
|
+
// The adoption's record (round-cap re-review): create-only, 0600, never over a different one.
|
|
582
|
+
const recordVolume = (v) => {
|
|
583
|
+
const path = join(cwd, VALKEY_VOLUME_RECORD);
|
|
584
|
+
// Nothing to record where docker gave no CreatedAt: the next `up` asks again.
|
|
585
|
+
if (typeof v.createdAt !== "string" || v.createdAt === "") return;
|
|
586
|
+
try {
|
|
587
|
+
if (fs.existsSync(path)) {
|
|
588
|
+
if (!volumeRecordMatches(readVolumeRecord(cwd, fs), v)) out(`⚠ ${path} names another pi-dispatch-valkey-data and is left as it is: this adoption is not recorded, so the next \`up\` asks again\n`);
|
|
589
|
+
return;
|
|
590
|
+
}
|
|
591
|
+
fs.writeFileSync(path, volumeRecordText(v), { mode: 0o600, flag: "wx" });
|
|
592
|
+
out(`✓ recorded the adopted volume (its CreatedAt) in ${path}, so a later \`up\` uses it without asking\n`);
|
|
593
|
+
} catch (err) {
|
|
594
|
+
out(`⚠ the adoption could not be recorded in ${path} (${err?.message ?? err}), so the next \`up\` asks again\n`);
|
|
595
|
+
}
|
|
596
|
+
};
|
|
597
|
+
// After every start on the volume: the queue's own `pi-dispatch:owner`, recorded where it is missing and checked where
|
|
598
|
+
// it is not. Another folder's stops what this pass just started, at once (`stop`), and fails up.
|
|
599
|
+
const ownerMarkerOk = async (stop) => {
|
|
600
|
+
const r = await claimValkeyOwner(`redis://127.0.0.1:${vport}`, deployment);
|
|
601
|
+
if (r.error) {
|
|
602
|
+
valkeyRefused = "marker";
|
|
603
|
+
out(`✗ Valkey was started, but its ${OWNER_MARKER_KEY} could not be recorded or read (${r.error}), so whose queue it holds is unconfirmed\n`);
|
|
604
|
+
summary.push(["valkey", `started, owner marker UNREAD: ${r.error}`]);
|
|
605
|
+
return false;
|
|
606
|
+
}
|
|
607
|
+
if (!dirs.includes(r.owner)) {
|
|
608
|
+
valkeyRefused = "volume";
|
|
609
|
+
const stopped = await stop();
|
|
610
|
+
out(`✗ ${foreignMarkerRefusal(r.owner)}: the Valkey this pass started on it was ${stopped ? "stopped again at once" : "NOT stopped (the stop failed): stop it now"}\n`);
|
|
611
|
+
summary.push(["valkey", `STOPPED: ${foreignMarkerRefusal(r.owner)}`]);
|
|
612
|
+
return false;
|
|
613
|
+
}
|
|
614
|
+
if (r.claimed) out(`✓ recorded ${OWNER_MARKER_KEY}=${deployment} in its queue\n`);
|
|
615
|
+
return true;
|
|
616
|
+
};
|
|
617
|
+
const stopRun = async () => (await runStreamed(spawn, "docker", VALKEY_STOP_ARGS, out)) === 0 && (await runStreamed(spawn, "docker", VALKEY_RM_ARGS, out)) === 0;
|
|
618
|
+
// A folder the setup wizard handed to compose (PR #475's review): its Valkey is compose's valkey, named by the folder's
|
|
619
|
+
// project and the override, published on VALKEY_URL's port through PI_VALKEY_PORT.
|
|
620
|
+
const handedOver = fs.existsSync(join(cwd, COMPOSE_VALKEY_OVERRIDE));
|
|
621
|
+
const project = composeProjectName(cwd);
|
|
622
|
+
const composeBase = composeArgs(handedOver ? { project, override: true } : {});
|
|
623
|
+
const composeEnv = { ...dockerEnv, [VALKEY_PORT_KEY]: String(vport) };
|
|
624
|
+
const ourName = handedOver ? `${project}-valkey-1` : "pi-dispatch-valkey";
|
|
625
|
+
const shownCompose = (args) => `${vport !== 6379 ? `${VALKEY_PORT_KEY}=${vport} ` : ""}docker ${quoteArgs(args)}`;
|
|
626
|
+
const composeNote = handedOver ? "" : ` (in a folder /dispatch setup laid out, with -p ${project} after \`compose\`)`;
|
|
627
|
+
// Issue #480: a compose line is a way to start Valkey later only where the folder holds the compose file (a clone, or
|
|
628
|
+
// a folder /dispatch setup copied it into). A folder init made has none, and there the way is this command again.
|
|
629
|
+
const laterValkey = () => (composeHere(cwd, fs) ? `start it later with \`${shownCompose([...composeBase, "up", "-d"])}\`${composeNote}` : "start it later by running `pi-dispatch up` again and accepting");
|
|
630
|
+
let existing = null;
|
|
631
|
+
if (dockerValkey.error) {
|
|
632
|
+
valkeyRefused = "url";
|
|
633
|
+
out(`✗ ${dockerValkey.error}\n`);
|
|
634
|
+
summary.push(["valkey", `NOT added: ${dockerValkey.error}`]);
|
|
635
|
+
} else if (dockerValkey.remote) {
|
|
636
|
+
out(`✓ VALKEY_URL names ${dockerValkey.remote}, not this host, so no Valkey is added here\n`);
|
|
637
|
+
summary.push(["valkey", "VALKEY_URL names another host, nothing added here"]);
|
|
638
|
+
} else if (await probeTcp("127.0.0.1", vport)) {
|
|
639
|
+
const who = await valkeyContainerOwner(ourName, { dirs, port: vport, query });
|
|
640
|
+
const ours = who.ours === true;
|
|
641
|
+
out(`✓ something is listening on ${vport}, assuming your Valkey${ours ? ` (it is the ${ourName} container)` : ""}\n`);
|
|
642
|
+
if (who.ours === false) out(`⚠ ${foreignContainerSentence(ourName, who.owner)}\n`);
|
|
643
|
+
// Issue #468: a Valkey there that answers a client sending NO password, while this deployment has one, was started
|
|
644
|
+
// before the password existed (the upgrade of a deployment that had none). Ours is offered a restart with it; one
|
|
645
|
+
// we did not start is named, with the compose command that recreates the compose-started one.
|
|
646
|
+
const open = dockerValkeyPassword ? (await probeValkeyAuth(`redis://127.0.0.1:${vport}`)) === "ok" : false;
|
|
647
|
+
const cost = "A job running right now is interrupted, and the worker and the receiver get NOAUTH until they are restarted: pause first if a job runs (pi-dispatch pause, wait for active jobs, then pi-dispatch resume after the restart)";
|
|
648
|
+
if (open && ours && handedOver) {
|
|
649
|
+
// The wizard handed this folder's Valkey to compose (PR #475's review): compose recreates it with the password,
|
|
650
|
+
// on the same volume, where `docker run` would start a second one beside it.
|
|
651
|
+
const recreate = [...composeBase, "up", "-d", "--force-recreate", "valkey"];
|
|
652
|
+
const accepted = (await volumeClear()) && (await volumeUsable({ servedByOurs: true })) ? await consent(`${ourName} answers without a password, and ${VALKEY_PASSWORD_KEY} is set in .env. up would recreate it with the password (compose stops it first, so Valkey writes its AOF out; the pi-dispatch-valkey-data volume, and the queue in it, is kept). ${cost}:`, [shownCompose(recreate)], { yes, out, prompt }) : null;
|
|
653
|
+
if (accepted === null) {
|
|
654
|
+
// Refused above (`volumeClear` said why): nothing was asked, so nothing was declined.
|
|
655
|
+
} else if (!accepted) {
|
|
656
|
+
out(`skipped: until it restarts with ${VALKEY_PASSWORD_KEY}, any local account can reach this queue\n`);
|
|
657
|
+
summary.push(["valkey", `${ourName} runs WITHOUT a password (declined the restart with it); \`pi-dispatch up\` again offers it`]);
|
|
658
|
+
} else {
|
|
659
|
+
const code = await runStreamed(spawn, "docker", recreate, out, { env: composeEnv });
|
|
660
|
+
out(code === 0 ? `✓ recreated ${ourName} with ${VALKEY_PASSWORD_KEY} (same volume). Restart the worker and the receiver now so they send it\n` : `✗ compose could not recreate its valkey (exit ${code}): its own output above says why\n`);
|
|
661
|
+
if (code === 0) await ownerMarkerOk(async () => (await runStreamed(spawn, "docker", [...composeBase, "stop", "valkey"], out, { env: composeEnv })) === 0);
|
|
662
|
+
summary.push(["valkey", code === 0 ? `recreated ${ourName} with ${VALKEY_PASSWORD_KEY} (volume kept); restart the worker and the receiver` : `recreate with the password FAILED (exit ${code}); see compose's output`]);
|
|
663
|
+
}
|
|
664
|
+
} else if (open && ours) {
|
|
665
|
+
const accepted = (await volumeClear()) && (await volumeUsable({ servedByOurs: true })) ? await consent(
|
|
666
|
+
`pi-dispatch-valkey answers without a password, and ${VALKEY_PASSWORD_KEY} is set in .env. up would restart it with the password (stopped first, so Valkey writes its AOF out; the pi-dispatch-valkey-data volume, and the queue in it, is kept). ${cost}:`,
|
|
667
|
+
[`docker ${VALKEY_STOP_ARGS.join(" ")}`, `docker ${VALKEY_RM_ARGS.join(" ")}`, `docker ${quoteArgs(VALKEY_RUN_ARGS)}`],
|
|
668
|
+
{ yes, out, prompt },
|
|
669
|
+
) : null;
|
|
670
|
+
if (accepted === null) {
|
|
671
|
+
// Refused above (`volumeClear` said why).
|
|
672
|
+
} else if (!accepted) {
|
|
673
|
+
out(`skipped: until it restarts with ${VALKEY_PASSWORD_KEY}, any local account can reach this queue\n`);
|
|
674
|
+
summary.push(["valkey", `pi-dispatch-valkey runs WITHOUT a password (declined the restart with it); \`pi-dispatch up\` again offers it`]);
|
|
675
|
+
} else if (await runStreamed(spawn, "docker", VALKEY_STOP_ARGS, out) !== 0 || await runStreamed(spawn, "docker", VALKEY_RM_ARGS, out) !== 0) {
|
|
676
|
+
out("✗ stopping or removing pi-dispatch-valkey failed: it runs as it was, without a password; doctor below will re-check\n");
|
|
677
|
+
summary.push(["valkey", "restart with the password FAILED at stop/rm; pi-dispatch-valkey still runs without one"]);
|
|
678
|
+
} else if (await runStreamed(spawn, "docker", VALKEY_RUN_ARGS, out, { env: dockerEnv }) !== 0) {
|
|
679
|
+
out("✗ docker run failed after pi-dispatch-valkey was removed: no Valkey runs now; its volume is kept. Re-run `pi-dispatch up`\n");
|
|
680
|
+
summary.push(["valkey", "restart with the password FAILED at docker run: NO Valkey runs now (the volume is kept); re-run `pi-dispatch up`"]);
|
|
681
|
+
} else if (await ownerMarkerOk(stopRun)) {
|
|
682
|
+
out(`✓ restarted Valkey with ${VALKEY_PASSWORD_KEY} (container pi-dispatch-valkey, same volume). Restart the worker and the receiver now so they send it: \`pi-dispatch service restart\` (and \`pi-dispatch service restart --receiver\`), or restart them however you run them\n`);
|
|
683
|
+
summary.push(["valkey", `restarted pi-dispatch-valkey with ${VALKEY_PASSWORD_KEY} (volume kept); restart the worker and the receiver`]);
|
|
684
|
+
}
|
|
685
|
+
} else if (open) {
|
|
686
|
+
out(`⚠ the Valkey on ${vport} answers without a password, and it is not a container up started, so up leaves it: restart it with ${VALKEY_PASSWORD_KEY} from .env${composeHere(cwd, fs) ? ` (compose: \`${shownCompose([...composeBase, "up", "-d"])}\`${composeNote}, which recreates it and keeps its volume)` : ""}\n`);
|
|
687
|
+
summary.push(["valkey", `port ${vport} has a listener that answers WITHOUT a password, left alone: restart it with ${VALKEY_PASSWORD_KEY}`]);
|
|
688
|
+
} else {
|
|
689
|
+
summary.push(["valkey", ours ? `container ${ourName} already running` : `port ${vport} already has a listener, left alone`]);
|
|
690
|
+
}
|
|
691
|
+
} else if (handedOver) {
|
|
692
|
+
// The wizard handed this folder's Valkey to compose (PR #475's review): up starts THAT one, compose's valkey on
|
|
693
|
+
// up's old volume, with the folder's project and the override, never a `docker run` of its own beside it (measured:
|
|
694
|
+
// up re-created pi-dispatch-valkey on the same volume, and the wizard's next compose run then failed to bind).
|
|
695
|
+
const start = [...composeBase, "up", "-d", "valkey"];
|
|
696
|
+
const accepted = (await volumeClear()) && (await volumeUsable()) ? await consent(`Nothing is listening on 127.0.0.1:${vport}. This folder's Valkey is compose's (${COMPOSE_VALKEY_OVERRIDE}, written by /dispatch setup), so up would start it:`, [shownCompose(start)], { yes, out, prompt }) : null;
|
|
697
|
+
if (accepted === null) {
|
|
698
|
+
// Refused above (`volumeClear` said why): nothing was asked, so nothing was declined.
|
|
699
|
+
} else if (!accepted) {
|
|
700
|
+
out(`skipped: start it later with \`${shownCompose(start)}\`\n`);
|
|
701
|
+
summary.push(["valkey", "skipped (declined): the queue needs it before `pi-dispatch worker` can drain"]);
|
|
702
|
+
} else {
|
|
703
|
+
const code = await runStreamed(spawn, "docker", start, out, { env: composeEnv });
|
|
704
|
+
if (code === 0 && !(await ownerMarkerOk(async () => (await runStreamed(spawn, "docker", [...composeBase, "stop", "valkey"], out, { env: composeEnv })) === 0))) {
|
|
705
|
+
// Said above; nothing more to add.
|
|
706
|
+
} else if (code === 0) {
|
|
707
|
+
out(`✓ started compose's valkey (project ${project}, the pi-dispatch-valkey-data volume, 127.0.0.1:${vport})\n`);
|
|
708
|
+
summary.push(["valkey", `started compose's valkey (project ${project}, volume pi-dispatch-valkey-data)`]);
|
|
709
|
+
} else {
|
|
710
|
+
// Reported, never swallowed: compose prints why (a port another container holds, most often), and up exits
|
|
711
|
+
// non-zero on it.
|
|
712
|
+
valkeyRefused = "compose";
|
|
713
|
+
out(`✗ compose could not start its valkey (exit ${code}): its own output above says why (another container on 127.0.0.1:${vport} is the usual one: \`docker ps\` shows which)\n`);
|
|
714
|
+
summary.push(["valkey", `compose's valkey FAILED to start (exit ${code}); see compose's output`]);
|
|
715
|
+
}
|
|
716
|
+
}
|
|
717
|
+
} else if ((existing = await valkeyContainerOwner("pi-dispatch-valkey", { dirs, port: vport, query })) && !existing.absent) {
|
|
718
|
+
// A pi-dispatch-valkey that does not answer this deployment's port: this deployment's, stopped (started as it is,
|
|
719
|
+
// never replaced), or another's, never touched (PR #475's review, round 3), or one docker could not describe.
|
|
720
|
+
if (existing.ours) {
|
|
721
|
+
out(`⚠ pi-dispatch-valkey is this deployment's and is not listening on ${vport}: \`docker start pi-dispatch-valkey\` starts it as it is\n`);
|
|
722
|
+
summary.push(["valkey", "pi-dispatch-valkey exists (this deployment's) and is not running; `docker start pi-dispatch-valkey`"]);
|
|
723
|
+
} else {
|
|
724
|
+
valkeyRefused = "volume";
|
|
725
|
+
const why = existing.unknown ? `whether pi-dispatch-valkey is this deployment's could not be read (${existing.unknown})` : foreignContainerSentence("pi-dispatch-valkey", existing.owner);
|
|
726
|
+
out(`✗ ${why}, and a second container of that name cannot be started beside it. Give this deployment a Valkey of its own (compose in this folder, which names its container and volume after the folder's project)\n`);
|
|
727
|
+
summary.push(["valkey", `NOT started: ${why}`]);
|
|
728
|
+
}
|
|
729
|
+
} else if ((await volumeClear()) && (await volumeUsable())) {
|
|
730
|
+
// The volume is created (labelled) only where it does not exist; an adopted or labelled one is used as it is.
|
|
731
|
+
const creates = volumeState === "absent";
|
|
732
|
+
const accepted = await consent(
|
|
733
|
+
`Nothing is listening on 127.0.0.1:${vport}. up would start Valkey (same semantics as deploy/docker-compose.yml)${volumeState === "adopted" ? " on the adopted pi-dispatch-valkey-data" : ""}:`,
|
|
734
|
+
[...(creates ? [`docker ${quoteArgs(VALKEY_VOLUME_ARGS)}`] : []), `docker ${quoteArgs(VALKEY_RUN_ARGS)}`],
|
|
735
|
+
{ yes, out, prompt },
|
|
736
|
+
);
|
|
737
|
+
if (!accepted) {
|
|
738
|
+
out(`skipped: ${laterValkey()}\n`);
|
|
739
|
+
summary.push(["valkey", "skipped (declined): the queue needs it before `pi-dispatch worker` can drain"]);
|
|
740
|
+
} else if (creates && await runStreamed(spawn, "docker", VALKEY_VOLUME_ARGS, out) !== 0) {
|
|
741
|
+
out("✗ docker volume create failed; continuing, doctor below will re-check Valkey\n");
|
|
742
|
+
summary.push(["valkey", composeHere(cwd, fs) ? `volume create FAILED; \`${shownCompose([...composeBase, "up", "-d"])}\` is the fallback` : "volume create FAILED; re-run `pi-dispatch up`"]);
|
|
743
|
+
} else if (await runStreamed(spawn, "docker", VALKEY_RUN_ARGS, out, { env: dockerEnv }) !== 0) {
|
|
744
|
+
out("✗ docker run failed; continuing, doctor below will re-check Valkey\n");
|
|
745
|
+
summary.push(["valkey", composeHere(cwd, fs) ? `container start FAILED; \`${shownCompose([...composeBase, "up", "-d"])}\` is the fallback` : "container start FAILED; re-run `pi-dispatch up`"]);
|
|
746
|
+
} else if (await ownerMarkerOk(stopRun)) {
|
|
747
|
+
out(`✓ started Valkey (container pi-dispatch-valkey, AOF on, bound to 127.0.0.1, ${dockerValkeyPassword ? `with ${VALKEY_PASSWORD_KEY} from .env` : "WITHOUT a password: .env sets none"})\n`);
|
|
748
|
+
summary.push(["valkey", "started container pi-dispatch-valkey (durable: --appendonly yes, restart unless-stopped)"]);
|
|
749
|
+
}
|
|
750
|
+
}
|
|
200
751
|
}
|
|
201
752
|
|
|
202
753
|
// (e2) the egress policy's proxy, and ONLY when the operator has already armed it. up never invents
|
|
@@ -206,19 +757,183 @@ export async function runUp(argv = [], deps = {}) {
|
|
|
206
757
|
// AFTER init, deliberately: init has just scaffolded egress-allowlist.conf, and starting a proxy whose
|
|
207
758
|
// allowlist file does not exist gets a directory created by docker where a file belonged and a squid
|
|
208
759
|
// that fails confusingly. If the file is still missing, this step declines itself and says which file.
|
|
209
|
-
|
|
210
|
-
|
|
760
|
+
const proxyName = egressProxyName(venueEnv);
|
|
761
|
+
// (e1b) the folder's copy of the proxy's rules (issue #484), before the proxy step so that a proxy started, replaced
|
|
762
|
+
// or restarted below reads the refreshed file. Only where the shipped docker proxy mounts it: a proxy PI_EGRESS_PROXY
|
|
763
|
+
// names is the operator's, and the podman venue mounts an account-owned copy `service install` writes.
|
|
764
|
+
const confRefreshed = dockerUsed && egressArmedFn(venueEnv) && proxyName === DEFAULT_EGRESS_PROXY ? await proxyConfRefreshStep({ fs, cwd, out, prompt, summary, readPackagedConf, now, yes }) : false;
|
|
765
|
+
if (dockerUsed && egressArmedFn(venueEnv) && proxyName !== DEFAULT_EGRESS_PROXY) {
|
|
766
|
+
// PI_EGRESS_PROXY names the operator's own proxy (issue #430). This step used to look for, and offer to start, the
|
|
767
|
+
// shipped name whatever that key said, so it could report a proxy present that no job attaches to, or start one
|
|
768
|
+
// beside the one the worker actually uses. It now asks about the name the worker attaches by and starts nothing:
|
|
769
|
+
// that container is the operator's, and the shipped command below would not produce it.
|
|
770
|
+
// RUNNING, read off stdout alone, exactly as the podman path below (issue #453): `docker inspect` exits 0 for an
|
|
771
|
+
// exited container too, so the exit code called a stopped proxy "present" while every job was refused pre-spend.
|
|
772
|
+
// `.State.Status`, not `.State.Running` (issue #453): the status word is what names a paused or restarting container.
|
|
773
|
+
const inspect = await runCmdQuery(spawn, "docker", ["inspect", "--format={{.State.Status}}", proxyName]);
|
|
774
|
+
const status = inspect.code === 0 ? inspect.stdout.trim() : null;
|
|
775
|
+
if (status === "running") {
|
|
776
|
+
out(`\n✓ Egress proxy present (${proxyName}, named by PI_EGRESS_PROXY)\n`);
|
|
777
|
+
summary.push(["egress", `${proxyName} (PI_EGRESS_PROXY) present, left untouched`]);
|
|
778
|
+
} else if (inspect.code === 0) {
|
|
779
|
+
// It exists and is not running. Still the operator's container: reported, never started, for the reason above.
|
|
780
|
+
out(`\n✗ PI_EGRESS_PROXY names ${proxyName}, and that container exists but is not running (${status || "no status"}). up starts only the shipped ${DEFAULT_EGRESS_PROXY}, so start ${proxyName} yourself\n`);
|
|
781
|
+
summary.push(["egress", `${proxyName} (PI_EGRESS_PROXY) exists but is not running; every job is refused pre-spend until it runs`]);
|
|
782
|
+
} else {
|
|
783
|
+
out(`\n✗ PI_EGRESS_PROXY names ${proxyName}, and docker has no container of that name. up starts only the shipped ${DEFAULT_EGRESS_PROXY}, so start ${proxyName} yourself\n`);
|
|
784
|
+
summary.push(["egress", `${proxyName} (PI_EGRESS_PROXY) not found; every job is refused pre-spend until it runs`]);
|
|
785
|
+
}
|
|
786
|
+
} else if (dockerUsed && egressArmedFn(venueEnv)) {
|
|
787
|
+
// RUNNING and CURRENT, from one inspect (issue #453; `egress-proxy-state.mjs` says why each). Running is
|
|
788
|
+
// `.State.Status` "running". Current is the pinned image with its own entrypoint and command and THIS folder's two
|
|
789
|
+
// files mounted and nothing else: a container of the shipped name that differs is not this deployment's proxy,
|
|
790
|
+
// whether it is running or stopped. Mounts whose sources do not resolve on this host are UNKNOWN, never stale.
|
|
791
|
+
const inspect = await runCmdQuery(spawn, "docker", ["inspect", PROXY_STATE_FORMAT, DEFAULT_EGRESS_PROXY]);
|
|
792
|
+
const state = inspect.code === 0 ? parseProxyState(inspect.stdout) : null;
|
|
793
|
+
const judged = state ? shippedProxyDrift(state, { cwd, platform, realpath: (p) => (typeof fs.realpathSync === "function" ? fs.realpathSync(p) : p) }) : { drift: [], unknown: null };
|
|
794
|
+
const drift = judged.drift;
|
|
795
|
+
// The files a start or a recreate mounts. Both, not the allowlist alone: a missing egress-proxy.conf is bind-mounted
|
|
796
|
+
// as a directory the runtime creates in its place (measured, Docker Engine 29.8.1: the host path became a root-owned
|
|
797
|
+
// directory, and the create failed "not a directory: Are you trying to mount a directory onto a file", exit 127).
|
|
798
|
+
// A DIRECTORY at either path counts as missing too (PR #488's review): it is mounted where squid reads a file.
|
|
799
|
+
const fileProblem = (f) => (!fs.existsSync(join(cwd, f)) ? "is not here" : pathIsDirectory(fs, join(cwd, f)) ? "here is a directory, not a file" : null);
|
|
800
|
+
const missingFile = ["egress-allowlist.conf", "deploy/egress-proxy.conf"].find((f) => fileProblem(f) !== null);
|
|
801
|
+
const missingSaid = missingFile ? `${missingFile} ${fileProblem(missingFile)}` : "";
|
|
802
|
+
const noFile = missingFile ? (fileProblem(missingFile) === "is not here" ? `no ${missingFile} in this folder` : `${missingFile} in this folder is a directory`) : "";
|
|
803
|
+
// Whether the proxy's network is missing, filled by `egressNetworkLines` when an offer is about to be shown.
|
|
804
|
+
const network = { missing: false };
|
|
805
|
+
// STALE BY ITS IMAGE, ENTRYPOINT OR COMMAND (PR #456's final check): those are compared on every host, and a
|
|
806
|
+
// difference in any of them says the container is not the shipped proxy whatever its mounts hold, so an unknown
|
|
807
|
+
// mount is no reason to keep it. Docker Desktop reports every bind source as a path of its VM, so without this a
|
|
808
|
+
// proxy made from an older squid there was never replaced by `up`. HOW DESKTOP REPORTS `.Config.Image`, the
|
|
809
|
+
// entrypoint and the command is NOT MEASURED here (measured: Docker Engine 29.8.1, and Podman 4.9.3 and 5.8.1,
|
|
810
|
+
// the last with an image-id-created proxy reading as the pinned digest's spelling), so a Desktop spelling this
|
|
811
|
+
// does not expect would read stale and be offered, shown line for line and asked, never removed silently. An
|
|
812
|
+
// unknown mount still blocks a replace grounded on mounts alone.
|
|
813
|
+
const configStale = state ? shippedProxyDrift(state, { cwd, compareMounts: false }).drift.length > 0 : false;
|
|
814
|
+
const unjudged = Boolean(judged.unknown) && !configStale;
|
|
815
|
+
if (judged.unknown) out(`\n⚠ could not compare ${DEFAULT_EGRESS_PROXY}'s mounts on this host: ${judged.unknown}. Everything else about it is still judged, and up replaces it while one of its own two mounts is unknown only when its image, entrypoint or command shows it stale\n`);
|
|
816
|
+
if (inspect.code === 0 && !state) {
|
|
817
|
+
// It exists and its state could not be read: never started, never replaced, and never a `docker run` on a name
|
|
818
|
+
// that is taken.
|
|
819
|
+
out(`\n✗ ${DEFAULT_EGRESS_PROXY} exists, but its state could not be read from \`docker inspect\`, so up leaves it as it is; \`docker inspect ${DEFAULT_EGRESS_PROXY}\` shows it\n`);
|
|
820
|
+
summary.push(["egress", "proxy exists, state unreadable: left as it is"]);
|
|
821
|
+
} else if ((state?.status === "running" || state?.status === "paused") && drift.length === 0 && confRefreshed) {
|
|
822
|
+
// Issue #484: the rules were just refreshed under a running proxy, and squid reads them only at start. A RESTART,
|
|
823
|
+
// not `squid -k reconfigure`: the refresh renamed a new file over the old one, and a running container's bind
|
|
824
|
+
// mount still holds the old file (a file bind mount is of the inode; a start mounts the path again). Not the
|
|
825
|
+
// replace below either, which removes the container and with it every job network it is attached to; a restart
|
|
826
|
+
// keeps them. `--yes` never covers restarting a proxy jobs are using, as it never covers replacing one.
|
|
827
|
+
//
|
|
828
|
+
// A PAUSED current proxy takes this path too (PR #491's review): `unpause` alone resumes the same squid on the
|
|
829
|
+
// same mount, which still holds the old file, so it would run the old rules while `up` had just said they were
|
|
830
|
+
// replaced. It is unpaused and then restarted, two lines shown as one answer; unpausing first, rather than
|
|
831
|
+
// restarting a paused container, keeps the step off how each runtime stops a frozen one, which is not measured.
|
|
832
|
+
const paused = state.status === "paused";
|
|
833
|
+
const lines = paused ? [EGRESS_UNPAUSE_ARGS, EGRESS_RESTART_ARGS] : [EGRESS_RESTART_ARGS];
|
|
834
|
+
const shown = lines.map((a) => `docker ${a.join(" ")}`);
|
|
835
|
+
out(`\n${paused ? "⚠ Egress proxy is paused (pi-dispatch-egress-proxy), and holds" : "✓ Egress proxy present (pi-dispatch-egress-proxy), still running"} the rules it read at start\n`);
|
|
836
|
+
const attached = jobNetworksOf(state);
|
|
837
|
+
if (judged.unknown || missingFile) {
|
|
838
|
+
const why = judged.unknown ? "its mounts could not be compared on this host (above), so whether it mounts this folder's file is not known" : `${missingSaid}, and a restart would mount it as a directory`;
|
|
839
|
+
out(`not restarted: ${why}; \`${shown.join(" && ")}\` loads the refreshed rules once that is settled\n`);
|
|
840
|
+
summary.push(["egress", `proxy left ${state.status} on the rules it started with: not restarted, since ${why}`]);
|
|
841
|
+
} else {
|
|
842
|
+
if (yes && attached.length > 0) out("--yes does not cover restarting a proxy that jobs are using: answer below, or stop the worker and re-run\n");
|
|
843
|
+
if (
|
|
844
|
+
await consent(
|
|
845
|
+
`up would ${paused ? "unpause and then restart" : "restart"} it so squid reads the refreshed deploy/egress-proxy.conf${attached.length > 0 ? `. It is attached to ${attached.join(", ")}: those jobs lose their egress while it restarts` : ""}:`,
|
|
846
|
+
shown,
|
|
847
|
+
{ yes: yes && attached.length === 0, out, prompt },
|
|
848
|
+
)
|
|
849
|
+
) {
|
|
850
|
+
let failed = null;
|
|
851
|
+
for (const args of lines) {
|
|
852
|
+
if ((await runStreamed(spawn, "docker", args, out)) !== 0) {
|
|
853
|
+
failed = args[0];
|
|
854
|
+
break;
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
if (failed) {
|
|
858
|
+
out(`✗ could not ${failed} the egress proxy; continuing, doctor below will re-check it\n`);
|
|
859
|
+
summary.push(["egress", `${failed} FAILED after the rules refresh: see \`docker logs pi-dispatch-egress-proxy\``]);
|
|
860
|
+
} else {
|
|
861
|
+
summary.push(["egress", `${paused ? "unpaused and restarted" : "restarted"} pi-dispatch-egress-proxy on the refreshed rules`]);
|
|
862
|
+
}
|
|
863
|
+
} else {
|
|
864
|
+
out(`skipped: it ${paused ? "stays paused, and holds" : "runs"} the old rules until \`${shown.join(" && ")}\`\n`);
|
|
865
|
+
summary.push(["egress", `proxy left ${state.status} on the old rules (declined): \`${shown.join(" && ")}\` loads the refreshed ones`]);
|
|
866
|
+
}
|
|
867
|
+
}
|
|
868
|
+
} else if (state?.status === "running" && drift.length === 0) {
|
|
211
869
|
out("\n✓ Egress proxy already present (pi-dispatch-egress-proxy)\n");
|
|
212
|
-
summary.push(["egress", "proxy already present
|
|
213
|
-
} else if (
|
|
214
|
-
|
|
215
|
-
|
|
870
|
+
summary.push(["egress", judged.unknown ? "proxy already present, left untouched (its mounts could not be compared on this host)" : "proxy already present, left untouched"]);
|
|
871
|
+
} else if (state && drift.length > 0) {
|
|
872
|
+
// STALE, running or not: never started as it is. Its policy is not this deployment's, so the offer is to replace
|
|
873
|
+
// it with the shipped one, shown line for line. A running one may be carrying jobs: its job and sandbox networks
|
|
874
|
+
// are named, since removing it cuts each of them off mid-run.
|
|
875
|
+
out(`\n✗ ${DEFAULT_EGRESS_PROXY} exists (${state.status}) but is not this deployment's proxy: ${drift.join("; ")}\n`);
|
|
876
|
+
const attached = jobNetworksOf(state);
|
|
877
|
+
// Never removed on its mounts while one of its two own mounts is unknown (round-cap re-review): the other findings
|
|
878
|
+
// are said. Its image, entrypoint or command is ground enough (`configStale` above).
|
|
879
|
+
if (yes && attached.length > 0 && state.status === "running" && !unjudged && !missingFile) out("--yes does not cover replacing a proxy that jobs are using: answer below, or stop the worker and re-run\n");
|
|
880
|
+
if (unjudged) {
|
|
881
|
+
out(`not replaced: one of its own mounts could not be compared on this host (above), so up leaves it as it is; \`docker ${EGRESS_RM_ARGS.join(" ")}\` and \`pi-dispatch up\` replace it if you have checked it\n`);
|
|
882
|
+
summary.push(["egress", "stale proxy left as it is: one of its mounts could not be compared on this host"]);
|
|
883
|
+
} else if (missingFile) {
|
|
884
|
+
out(`✗ ${missingSaid}, so up cannot recreate it from this folder; run \`pi-dispatch init\` here (or \`up\` from the deployment folder), then \`up\` again\n`);
|
|
885
|
+
summary.push(["egress", `stale proxy left as it is: ${noFile} to recreate it from; its policy is not this deployment's`]);
|
|
886
|
+
} else if (
|
|
887
|
+
await consent(
|
|
888
|
+
`up would replace it with the shipped proxy (the same semantics as deploy/docker-compose.yml --profile egress)${attached.length > 0 ? `. It is attached to ${attached.join(", ")}: removing it cuts those jobs off from their egress mid-run, so stop the worker first (and let running jobs finish)` : ""}:`,
|
|
889
|
+
[`docker ${EGRESS_RM_ARGS.join(" ")}`, ...(await egressNetworkLines(spawn, network)), `docker ${quoteArgs(EGRESS_RUN_ARGS)}`],
|
|
890
|
+
// `--yes` does not cover cutting live jobs off (gate round 3): with a job network attached to a RUNNING
|
|
891
|
+
// proxy the lines are printed and a person must answer, so an unattended `up --yes` never does it.
|
|
892
|
+
{ yes: yes && !(attached.length > 0 && state.status === "running"), out, prompt },
|
|
893
|
+
)
|
|
894
|
+
) {
|
|
895
|
+
await runStreamed(spawn, "docker", EGRESS_RM_ARGS, out);
|
|
896
|
+
await ensureEgressNetwork(spawn, out, network);
|
|
897
|
+
if ((await runStreamed(spawn, "docker", EGRESS_RUN_ARGS, out)) !== 0) {
|
|
898
|
+
out("✗ could not start the egress proxy; continuing, doctor below will re-check it\n");
|
|
899
|
+
summary.push(["egress", "recreate failed: every job refuses pre-spend until it is up (costs no budget, runs nothing)"]);
|
|
900
|
+
} else {
|
|
901
|
+
summary.push(["egress", "replaced the stale pi-dispatch-egress-proxy with the shipped one"]);
|
|
902
|
+
}
|
|
903
|
+
} else {
|
|
904
|
+
out(`skipped: until it is replaced, jobs use a proxy whose policy is not this deployment's (\`docker ${EGRESS_RM_ARGS.join(" ")}\`, then \`pi-dispatch up\`)\n`);
|
|
905
|
+
summary.push(["egress", "stale proxy left as it is (declined): its policy is not this deployment's"]);
|
|
906
|
+
}
|
|
907
|
+
} else if (missingFile) {
|
|
908
|
+
out(`\n✗ the egress policy is on but ${missingSaid}, so no proxy is started without it\n`);
|
|
909
|
+
summary.push(["egress", `skipped: ${noFile}; run \`pi-dispatch init\` here (or \`up\` from the deployment folder), then \`up\` again`]);
|
|
910
|
+
} else if (state?.status === "restarting") {
|
|
911
|
+
// A crash loop: the restart policy is already starting it, and its squid keeps exiting. Nothing to start.
|
|
912
|
+
out(`\n✗ ${DEFAULT_EGRESS_PROXY} is restarting: its squid keeps exiting and its restart policy keeps bringing it back. \`docker logs ${DEFAULT_EGRESS_PROXY}\` says why; meanwhile each job is retried once, then failed\n`);
|
|
913
|
+
summary.push(["egress", "proxy crash-looping: see `docker logs pi-dispatch-egress-proxy`"]);
|
|
914
|
+
} else if (state) {
|
|
915
|
+
// Current, and stopped or paused. `docker run --name` would fail on the name conflict, so the SAME container is
|
|
916
|
+
// offered back: `unpause` for a paused one, else `start`, which is what `docker compose --profile egress up -d`
|
|
917
|
+
// does for a stopped container whose configuration is unchanged. It destroys nothing, and a compose-created
|
|
918
|
+
// container stays compose's.
|
|
919
|
+
const args = state.status === "paused" ? EGRESS_UNPAUSE_ARGS : EGRESS_START_ARGS;
|
|
920
|
+
if (await consent(`The egress policy is on (PI_EGRESS=0 opts out) and the allowlist proxy exists but is ${state.status}. up would ${args[0]} that same container (as deploy/docker-compose.yml --profile egress up -d does):`, [`docker ${args.join(" ")}`], { yes, out, prompt })) {
|
|
921
|
+
if ((await runStreamed(spawn, "docker", args, out)) !== 0) {
|
|
922
|
+
out(`✗ could not ${args[0]} the egress proxy; \`docker ${EGRESS_RM_ARGS.join(" ")}\` and re-run \`pi-dispatch up\` recreates it. Continuing; doctor below will re-check it\n`);
|
|
923
|
+
summary.push(["egress", `${args[0]} of the ${state.status} proxy failed; every job refuses pre-spend until it is up (costs no budget, runs nothing)`]);
|
|
924
|
+
} else {
|
|
925
|
+
summary.push(["egress", `${args[0] === "unpause" ? "unpaused" : "started"} the ${state.status} pi-dispatch-egress-proxy`]);
|
|
926
|
+
}
|
|
927
|
+
} else {
|
|
928
|
+
out(`skipped: ${args[0]} it later with \`docker ${args.join(" ")}\`\n`);
|
|
929
|
+
summary.push(["egress", "skipped (declined): every job is refused pre-spend until the proxy is up (PI_EGRESS=0 opts out)"]);
|
|
930
|
+
}
|
|
216
931
|
} else if (
|
|
217
|
-
await consent("The egress policy is on (PI_EGRESS=0 opts out) but the allowlist proxy is not running. up would start it (same semantics as deploy/docker-compose.yml --profile egress):", [
|
|
932
|
+
await consent("The egress policy is on (PI_EGRESS=0 opts out) but the allowlist proxy is not running. up would start it (same semantics as deploy/docker-compose.yml --profile egress):", [...(await egressNetworkLines(spawn, network)), `docker ${quoteArgs(EGRESS_RUN_ARGS)}`], { yes, out, prompt })
|
|
218
933
|
) {
|
|
219
|
-
// The network
|
|
220
|
-
//
|
|
221
|
-
await
|
|
934
|
+
// The network's create is not checked: one made between the ask and here is not a failure. The proxy is what
|
|
935
|
+
// matters and it is checked.
|
|
936
|
+
await ensureEgressNetwork(spawn, out, network);
|
|
222
937
|
if ((await runStreamed(spawn, "docker", EGRESS_RUN_ARGS, out)) !== 0) {
|
|
223
938
|
out("✗ could not start the egress proxy — continuing; doctor below will re-check it\n");
|
|
224
939
|
summary.push(["egress", "start failed — every job refuses pre-spend until it is up (costs no budget, runs nothing)"]);
|
|
@@ -226,22 +941,57 @@ export async function runUp(argv = [], deps = {}) {
|
|
|
226
941
|
summary.push(["egress", "started pi-dispatch-egress-proxy on pi-dispatch-egress-out"]);
|
|
227
942
|
}
|
|
228
943
|
} else {
|
|
229
|
-
out(
|
|
944
|
+
out(`skipped: ${composeHere(cwd, fs) ? `start it later with \`${composeCommandFor(cwd, fs, ["--profile", "egress", "up", "-d"])}\`` : "start it later by running `pi-dispatch up` again and accepting"}\n`);
|
|
230
945
|
summary.push(["egress", "skipped (declined) — every job is refused pre-spend until the proxy is up (PI_EGRESS=0 opts out)"]);
|
|
231
946
|
}
|
|
232
947
|
}
|
|
233
948
|
|
|
949
|
+
// (e3) the podman venue's stack: Valkey and, while the policy is armed, the proxy, as Quadlet units through the SAME
|
|
950
|
+
// installer `pi-dispatch service install` uses (issue #430). After init for the reason (e2) gives: the proxy mounts
|
|
951
|
+
// the allowlist init scaffolds.
|
|
952
|
+
if (podmanReady) {
|
|
953
|
+
const stackState = { valkeyRefused: false };
|
|
954
|
+
await podmanStackStep({ env: venueEnv, valkeyKeys: valkeyKeys.env, state: stackState, venues, spawn, out, yes, prompt, summary, fs, probeTcp, lookup, interfaces, runSync, cwd, home, user: userName(), euid, newPassword, platform, templatesDir, readTemplate, mkdir, realpath });
|
|
955
|
+
valkeyRefused = stackState.valkeyRefused;
|
|
956
|
+
}
|
|
957
|
+
|
|
234
958
|
// (f) doctor — always, verbatim: up converges what it can, doctor is the judge of what remains
|
|
235
959
|
// (provider key, forge env, overlay …), and its verdict is up's exit code.
|
|
236
960
|
out("\ndoctor:\n");
|
|
237
961
|
runDoctorFn ??= (await import("./doctor.mjs")).runDoctor;
|
|
238
|
-
|
|
962
|
+
// Layered, per the note at step (e1): the lines just written configure the service and not this
|
|
963
|
+
// process, so an unlayered call would warn about exactly what `up` had converged a moment earlier.
|
|
964
|
+
//
|
|
965
|
+
// `env` WINS, and the filter is what makes that true rather than the spread order. `wrote` holds keys
|
|
966
|
+
// that were empty in the FILE, which is a different question from whether this shell sets them: an
|
|
967
|
+
// operator exporting `PI_SCOPED_LIMITS_FILE=/etc/pi/limits.json` with the key still commented in `.env`
|
|
968
|
+
// would otherwise have doctor read back the file `up` chose instead of the one their worker loads.
|
|
969
|
+
// SETS, not "sets to something usable", and the `.trim()` this dropped was hiding a refused boot from
|
|
970
|
+
// up's own verdict (issue #384). A shell that exports `PI_PAUSE_WINDOWS_FILE= ` is a shell whose
|
|
971
|
+
// foreground worker reads three spaces, keeps them (`config.mjs` uses `??`, not `||`) and throws at
|
|
972
|
+
// `start.mjs` before it takes a job. Filling that key from `wrote` handed doctor a path the operator
|
|
973
|
+
// does not have set, so doctor judged a deployment nobody is running and `up` exited 0 on one that
|
|
974
|
+
// cannot start. Doctor now names the blank itself, which is the only line that tells the operator what
|
|
975
|
+
// to do about it. For the other two keys `up` writes, this hands doctor the truth rather than a default:
|
|
976
|
+
// `PI_LOGS_DIR=""` does resolve through `||` to what `up` wrote, but `PI_LOGS_DIR=" "` does NOT -- three
|
|
977
|
+
// spaces are truthy, so `config.mjs` keeps them and the worker really does use a directory named three
|
|
978
|
+
// spaces. Filling that key in from `wrote` made doctor judge a deployment the operator is not running,
|
|
979
|
+
// which is the same defect one key over.
|
|
980
|
+
const layered = { ...env };
|
|
981
|
+
for (const [key, value] of Object.entries(wrote)) {
|
|
982
|
+
if (typeof env[key] !== "string") layered[key] = value;
|
|
983
|
+
}
|
|
984
|
+
// The venue keys `.env` supplied (D2), by the same env-wins rule, so doctor judges the venue this pass drove.
|
|
985
|
+
for (const [key, value] of Object.entries(venue.fromFile)) {
|
|
986
|
+
if (typeof env[key] !== "string") layered[key] = value;
|
|
987
|
+
}
|
|
988
|
+
const doctorCode = await runDoctorFn(layered, { out });
|
|
239
989
|
|
|
240
990
|
// (g) the summary: what ran, what was skipped, what was already there — then the two commands
|
|
241
991
|
// that actually start work, so "up is green" flows straight into the first job.
|
|
242
992
|
out("\nup: summary\n");
|
|
243
993
|
for (const [name, note] of summary) {
|
|
244
|
-
out(` ${name.padEnd(
|
|
994
|
+
out(` ${name.padEnd(21)} ${note}\n`);
|
|
245
995
|
}
|
|
246
996
|
out(`
|
|
247
997
|
Next:
|
|
@@ -252,9 +1002,386 @@ Next:
|
|
|
252
1002
|
pi-dispatch run ./my-project --task "add type hints to utils.py"
|
|
253
1003
|
queue your first job from another terminal
|
|
254
1004
|
`);
|
|
1005
|
+
// A worker run BY HAND reads only its shell, so on the podman venue it needs the list exported; the service reads
|
|
1006
|
+
// `.env` itself. Printed only there, so every other deployment's closing text is unchanged.
|
|
1007
|
+
if (venues.podmanUsed) {
|
|
1008
|
+
out(" (podman venue: a worker started by hand reads only its shell, so export the same PI_BACKENDS there first, or run it as a service with `pi-dispatch service install`, which reads .env)\n");
|
|
1009
|
+
}
|
|
1010
|
+
// Issue #464: a Valkey this pass refused (another account's, or one no one could name) is a failed `up` whatever doctor
|
|
1011
|
+
// says: a worker started now would take that queue's jobs, and an exit 0 read by a script or the wizard says ready.
|
|
1012
|
+
if (valkeyRefused) {
|
|
1013
|
+
// Worded per reason (gate round 2): an owner refusal and a URL the Quadlet Valkey cannot serve are different faults.
|
|
1014
|
+
out(
|
|
1015
|
+
valkeyRefused === "owner"
|
|
1016
|
+
? "\nup: the Valkey VALKEY_URL reaches is not this account's (above); give this account its own port, or opt in with PI_VALKEY_SHARED=1 in .env, then re-run `pi-dispatch up`.\n"
|
|
1017
|
+
: valkeyRefused === "password"
|
|
1018
|
+
? `\nup: ${VALKEY_PASSWORD_KEY} in .env cannot be handed to Valkey (above); fix it, then re-run \`pi-dispatch up\`.\n`
|
|
1019
|
+
: valkeyRefused === "compose"
|
|
1020
|
+
? "\nup: compose could not start this folder's Valkey (above); free the port it names, then re-run `pi-dispatch up`.\n"
|
|
1021
|
+
: valkeyRefused === "marker"
|
|
1022
|
+
? `\nup: the Valkey up started could not be asked whose queue it holds (${OWNER_MARKER_KEY}, above); check it answers, then re-run \`pi-dispatch up\`.\n`
|
|
1023
|
+
: valkeyRefused === "volume"
|
|
1024
|
+
? "\nup: the Valkey container or volume named above is not this deployment's (or not attributed to it), and up never stops, removes, reuses or starts beside one; take the way out it names, then re-run `pi-dispatch up`.\n"
|
|
1025
|
+
: "\nup: VALKEY_URL cannot be served as it is written (above); fix it in .env, then re-run `pi-dispatch up`.\n",
|
|
1026
|
+
);
|
|
1027
|
+
return doctorCode !== 0 ? doctorCode : 1;
|
|
1028
|
+
}
|
|
255
1029
|
return doctorCode;
|
|
256
1030
|
}
|
|
257
1031
|
|
|
1032
|
+
/**
|
|
1033
|
+
* The podman venue's gate: the same `podman info` read and the same ordered job-user rule the worker boots with
|
|
1034
|
+
* (`decidePodmanJobUser`), so `up` cannot pass a host the worker then refuses, or refuse one it would run. Printed in
|
|
1035
|
+
* the worker's own refusal words.
|
|
1036
|
+
*/
|
|
1037
|
+
async function podmanGate({ readPodmanInfo, platform, euid, egid, out }) {
|
|
1038
|
+
let read;
|
|
1039
|
+
try {
|
|
1040
|
+
read = await readPodmanInfo();
|
|
1041
|
+
} catch {
|
|
1042
|
+
read = { answered: false, reason: "spawn-failed", transient: true };
|
|
1043
|
+
}
|
|
1044
|
+
const decision = decidePodmanJobUser({ platform, euid, egid, read });
|
|
1045
|
+
if (decision.mode === "worker") {
|
|
1046
|
+
out(`✓ rootless Podman answering (podman jobs run as ${decision.user}, with --userns=keep-id)\n`);
|
|
1047
|
+
return { ok: true };
|
|
1048
|
+
}
|
|
1049
|
+
// The worker's boot-refusing causes stop `up` in the worker's own words. `up` is DELIBERATELY STRICTER than the
|
|
1050
|
+
// worker on the rest (round 2, E7, which reverses round 1's D7): a podman info that timed out or that nothing could
|
|
1051
|
+
// read is a per-job retry at the worker, but `up` is about to install units and run podman commands that have no
|
|
1052
|
+
// timeout of their own, for a Podman it could not see. A wedged podman would hang the pass, and a REMOTE podman
|
|
1053
|
+
// whose info merely timed out would get local Quadlet units. So only an answered, usable info continues.
|
|
1054
|
+
if (decision.mode === "unmappable" && PODMAN_BOOT_REFUSING_CAUSES.has(decision.cause)) {
|
|
1055
|
+
const why = podmanJobUserRefusal(decision);
|
|
1056
|
+
out(`✗ podman venue not usable here\n → ${why}\n`);
|
|
1057
|
+
return { ok: false, why };
|
|
1058
|
+
}
|
|
1059
|
+
const why = `${decision.mode === "unmappable" ? podmanJobUserRefusal(decision) : `podman info did not answer (${decision.reason}).`} up stops here although the worker would only retry this per job: it will not install units for a Podman it could not see. Check that \`podman info --format json\` answers as this account, then re-run.`;
|
|
1060
|
+
out(`✗ podman venue not answering\n → ${why}\n`);
|
|
1061
|
+
return { ok: false, why };
|
|
1062
|
+
}
|
|
1063
|
+
|
|
1064
|
+
/** The default job image into this account's Podman store, mirroring the docker step (b) line for line. */
|
|
1065
|
+
async function podmanImageStep({ spawn, out, yes, prompt, summary }) {
|
|
1066
|
+
if ((await runCmd(spawn, "podman", PODMAN_EXISTS_ARGS)) === 0) {
|
|
1067
|
+
out(`✓ Job image present in this account's Podman store (${LOCAL_IMAGE})\n`);
|
|
1068
|
+
summary.push(["podman job image", `already present (${LOCAL_IMAGE})`]);
|
|
1069
|
+
return;
|
|
1070
|
+
}
|
|
1071
|
+
const accepted = await consent(
|
|
1072
|
+
`The default job image (${LOCAL_IMAGE}) is not in this account's Podman store. up would run:`,
|
|
1073
|
+
[`podman ${PULL_ARGS.join(" ")}`, `podman ${TAG_ARGS.join(" ")}`],
|
|
1074
|
+
{ yes, out, prompt },
|
|
1075
|
+
);
|
|
1076
|
+
if (!accepted) {
|
|
1077
|
+
out("skipped: pull it later with the two commands above\n");
|
|
1078
|
+
summary.push(["podman job image", "skipped (declined): jobs run with --pull=never, so nothing fetches it later"]);
|
|
1079
|
+
} else if ((await runStreamed(spawn, "podman", PULL_ARGS, out)) !== 0) {
|
|
1080
|
+
out("✗ podman pull failed: continuing; doctor below will re-check the image\n");
|
|
1081
|
+
summary.push(["podman job image", "pull FAILED: re-run `pi-dispatch up`, or pull by hand"]);
|
|
1082
|
+
} else if ((await runStreamed(spawn, "podman", TAG_ARGS, out)) !== 0) {
|
|
1083
|
+
out("✗ podman tag failed: continuing; doctor below will re-check the image\n");
|
|
1084
|
+
summary.push(["podman job image", `pulled, but tagging as ${LOCAL_IMAGE} FAILED: re-run the tag command by hand`]);
|
|
1085
|
+
} else {
|
|
1086
|
+
out(`✓ pulled and tagged ${LOCAL_IMAGE} in this account's Podman store\n`);
|
|
1087
|
+
summary.push(["podman job image", `pulled ${UPSTREAM_IMAGE} and tagged it ${LOCAL_IMAGE}`]);
|
|
1088
|
+
}
|
|
1089
|
+
}
|
|
1090
|
+
|
|
1091
|
+
/**
|
|
1092
|
+
* The podman venue's stack step. Decides what is missing the way the docker steps do (a listener on 6379 is left
|
|
1093
|
+
* alone, a proxy that exists is left alone), then plans the Quadlet files with `planStack`, shows `plan.actions`, and on
|
|
1094
|
+
* consent hands those same objects to `applyStack`: what runs is literally what was shown.
|
|
1095
|
+
*/
|
|
1096
|
+
async function podmanStackStep({ env, valkeyKeys = {}, state = {}, venues, spawn, out, yes, prompt, summary, fs, probeTcp, lookup, interfaces, runSync, cwd, home, user, euid, templatesDir, readTemplate, mkdir, realpath, newPassword = newValkeyPassword, platform = "linux" }) {
|
|
1097
|
+
const armed = egressArmedFn(env);
|
|
1098
|
+
let includeValkey = false;
|
|
1099
|
+
let valkeyPort = DEFAULT_VALKEY_PORT;
|
|
1100
|
+
if (!venues.localUsed) {
|
|
1101
|
+
// Issue #464: the rule `service install` applies (`decideValkey`), on VALKEY_URL and PI_VALKEY_SHARED as the
|
|
1102
|
+
// service reads them (`deploymentValkeyEnv`): every address the URL reaches is judged, and a listener there is taken
|
|
1103
|
+
// to be the Valkey only when it is this account's (or shared on purpose, opted in). Anything else is refused,
|
|
1104
|
+
// nothing is added for it, and `up` exits non-zero.
|
|
1105
|
+
const decided = await decideValkey({
|
|
1106
|
+
venues,
|
|
1107
|
+
url: valkeyKeys.VALKEY_URL,
|
|
1108
|
+
installed: false,
|
|
1109
|
+
probeTcp,
|
|
1110
|
+
lookup,
|
|
1111
|
+
interfaces,
|
|
1112
|
+
euid,
|
|
1113
|
+
user,
|
|
1114
|
+
shared: valkeySharedOn(valkeyKeys[VALKEY_SHARED_KEY]),
|
|
1115
|
+
subuids: readSubuidRanges({ user, euid, fs: { readFileSync: (p, enc) => (fs.readFileSync ?? readFileSync)(p, enc) }, run: runSync }),
|
|
1116
|
+
fs: { readFileSync: (p, enc) => (fs.readFileSync ?? readFileSync)(p, enc) },
|
|
1117
|
+
ownerName: (uid) => passwdNameFrom({ readFileSync: (p, enc) => (fs.readFileSync ?? readFileSync)(p, enc) }, uid),
|
|
1118
|
+
envPath: join(cwd, ".env"),
|
|
1119
|
+
});
|
|
1120
|
+
valkeyPort = decided.port;
|
|
1121
|
+
if (decided.error) {
|
|
1122
|
+
state.valkeyRefused = "url";
|
|
1123
|
+
out(`\n✗ ${decided.error}\n`);
|
|
1124
|
+
summary.push(["valkey", `NOT added: ${decided.error}`]);
|
|
1125
|
+
} else if (decided.refusal) {
|
|
1126
|
+
state.valkeyRefused = "owner";
|
|
1127
|
+
out(`\n✗ ${decided.refusal.text}\n`);
|
|
1128
|
+
summary.push(["valkey", `NOT added and NOT adopted: ${decided.refusal.short}`]);
|
|
1129
|
+
} else if (decided.include) {
|
|
1130
|
+
includeValkey = true;
|
|
1131
|
+
} else if (decided.remote) {
|
|
1132
|
+
out(`\n✓ ${decided.notes[0]}\n`);
|
|
1133
|
+
summary.push(["valkey", "VALKEY_URL names another host, nothing added here"]);
|
|
1134
|
+
} else {
|
|
1135
|
+
// "It is the pi-dispatch-valkey container" only when that container PUBLISHES this port: a container of that name
|
|
1136
|
+
// on another port (a second deployment's) says nothing about who answers here.
|
|
1137
|
+
const ps = await runCmdQuery(spawn, "podman", ["ps", "--filter", "name=^pi-dispatch-valkey$", "--format", "{{.Names}}|{{.Ports}}"]);
|
|
1138
|
+
const ours = ps.code === 0 && ps.stdout.split("\n").some((l) => valkeyContainerPublishes(l, decided.port));
|
|
1139
|
+
out(`\n✓ something is listening on ${decided.port}, assuming your Valkey${ours ? " (it is the pi-dispatch-valkey container, published there)" : ""}: ${decided.notes[0] ?? ""}\n`);
|
|
1140
|
+
summary.push(["valkey", ours ? `container pi-dispatch-valkey already running on ${decided.port}` : `port ${decided.port} already has a listener, left alone`]);
|
|
1141
|
+
}
|
|
1142
|
+
}
|
|
1143
|
+
// Issue #468: the password the Quadlet Valkey starts with, read from .env NOW (init, earlier in this pass, may have just
|
|
1144
|
+
// written the file and a password into it): kept when set, generated when the Valkey this step adds is the
|
|
1145
|
+
// deployment's own and it has none, written into .env only once the plan is accepted.
|
|
1146
|
+
const envPath = join(cwd, ".env");
|
|
1147
|
+
let valkeyPassword = null;
|
|
1148
|
+
let pendingPassword = null;
|
|
1149
|
+
if (includeValkey) {
|
|
1150
|
+
let fileKeys = {};
|
|
1151
|
+
let readError = null;
|
|
1152
|
+
if (fs.existsSync(envPath)) {
|
|
1153
|
+
try {
|
|
1154
|
+
const read = readValkeyKeys(fs.readFileSync(envPath), { loader: "systemd", path: envPath });
|
|
1155
|
+
if (read.error) readError = read.error;
|
|
1156
|
+
else fileKeys = read.keys;
|
|
1157
|
+
} catch (err) {
|
|
1158
|
+
readError = `${envPath} could not be read (${err?.message ?? err})`;
|
|
1159
|
+
}
|
|
1160
|
+
}
|
|
1161
|
+
const decided = readError ? { error: readError } : valkeyPasswordDecision({ ...valkeyKeys, ...fileKeys }, { envPath });
|
|
1162
|
+
if (decided.error) {
|
|
1163
|
+
state.valkeyRefused = "password";
|
|
1164
|
+
out(`\n✗ ${decided.error}\n`);
|
|
1165
|
+
summary.push(["valkey", `NOT added: ${decided.error}`]);
|
|
1166
|
+
includeValkey = false;
|
|
1167
|
+
} else {
|
|
1168
|
+
if (decided.note) out(`\n⚠ ${decided.note}\n`);
|
|
1169
|
+
if (decided.generate && fs.existsSync(envPath)) pendingPassword = newPassword();
|
|
1170
|
+
else if (decided.generate) out(`\n⚠ no .env here, so the Valkey up adds has no password: run \`pi-dispatch init\` here, then \`pi-dispatch service install --force\`, which gives it one\n`);
|
|
1171
|
+
valkeyPassword = decided.password ?? pendingPassword;
|
|
1172
|
+
}
|
|
1173
|
+
}
|
|
1174
|
+
const components = stackComponents({ venues, env, includeValkey, armed, valkeyPort, valkeyPassword });
|
|
1175
|
+
for (const note of components.notes) out(`\n⚠ ${note}\n`);
|
|
1176
|
+
let keeperRestart = false;
|
|
1177
|
+
if (components.proxy) {
|
|
1178
|
+
// RUNNING, read off the output: `podman inspect` exits 0 for an EXITED container too and prints `false`
|
|
1179
|
+
// (measured), so the exit code alone called a stopped hand-started proxy "present" and offered nothing, which is
|
|
1180
|
+
// exactly the upgrade this step exists for. A stopped one is offered the unit, under the foreign-container rule.
|
|
1181
|
+
// stdout ALONE (round 2, E2): podman's stderr warnings on an ordinary account would otherwise make `true` unequal.
|
|
1182
|
+
// `.State.Status` (issue #453): Podman reports a paused container `paused` and one between restarts `stopped`
|
|
1183
|
+
// (measured on 4.9.3 and 5.8.1), and only `running` carries a job's traffic.
|
|
1184
|
+
const inspect = await runCmdQuery(spawn, "podman", ["inspect", "--format={{.State.Status}}", DEFAULT_EGRESS_PROXY]);
|
|
1185
|
+
if (inspect.code === 0 && inspect.stdout.trim() === "running") {
|
|
1186
|
+
out(`\n✓ Egress proxy already present under this account's Podman (${DEFAULT_EGRESS_PROXY})\n`);
|
|
1187
|
+
summary.push(["egress (podman)", "proxy already present, left untouched"]);
|
|
1188
|
+
components.proxy = false;
|
|
1189
|
+
} else if (!fs.existsSync(join(cwd, "egress-allowlist.conf")) || pathIsDirectory(fs, join(cwd, "egress-allowlist.conf"))) {
|
|
1190
|
+
out(`\n✗ the egress policy is on but egress-allowlist.conf ${pathIsDirectory(fs, join(cwd, "egress-allowlist.conf")) ? "here is a directory, not a file" : "is not here"}, not starting a proxy with no allowlist\n`);
|
|
1191
|
+
summary.push(["egress (podman)", "skipped: no egress-allowlist.conf in this folder; run `pi-dispatch init` here, then `up` again"]);
|
|
1192
|
+
components.proxy = false;
|
|
1193
|
+
}
|
|
1194
|
+
}
|
|
1195
|
+
if (components.keeper) {
|
|
1196
|
+
// The keeper (issue #458) under the proxy's rule: one that HOLDS is left alone, read off stdout as above, by the
|
|
1197
|
+
// shared rule (`judgeNetnsKeeper`: running, bridge mode, on its own network; issue #463's gate found a keeper on
|
|
1198
|
+
// `--network none` called "already running" here). Anything else is offered the unit, under the foreign-container
|
|
1199
|
+
// rule, which is what stops a `--replace` over a container of that name that is not ours. It is offered with the
|
|
1200
|
+
// allowlist missing too, since it mounts nothing and a proxy started later needs it just the same.
|
|
1201
|
+
const inspect = await runCmdQuery(spawn, "podman", ["inspect", NETNS_KEEPER_FORMAT, NETNS_KEEPER]);
|
|
1202
|
+
const keeper = judgeNetnsKeeper({ code: inspect.code, stdout: inspect.stdout });
|
|
1203
|
+
if (keeper.holds) {
|
|
1204
|
+
out(`\n✓ Rootless network keeper already running under this account's Podman on its own bridge network (${NETNS_KEEPER})\n`);
|
|
1205
|
+
summary.push(["netns keeper (podman)", "already running, left untouched"]);
|
|
1206
|
+
components.keeper = false;
|
|
1207
|
+
} else if (keeper.exists) {
|
|
1208
|
+
// There but not holding (paused, running off its bridge, exited): its unit, if the files are ours and unchanged,
|
|
1209
|
+
// is RESTARTED, since `start` on a unit that is still active does nothing and "installed and started" would
|
|
1210
|
+
// then be said over a keeper still not holding (PR #463 round 2). A container that is not ours stops at the
|
|
1211
|
+
// foreign-container rule below, as before.
|
|
1212
|
+
out(`\n⚠ ${NETNS_KEEPER} ${keeper.problem}: offering its unit, restarted\n`);
|
|
1213
|
+
keeperRestart = true;
|
|
1214
|
+
}
|
|
1215
|
+
}
|
|
1216
|
+
if (!components.valkey && !components.proxy && !components.keeper) return;
|
|
1217
|
+
const plan = planStack({ components, templatesDir, deployDir: cwd, home, fs, readTemplate, restartUnits: keeperRestart ? [QUADLET_FILES.keeper.unit] : [] });
|
|
1218
|
+
if (plan.error) {
|
|
1219
|
+
out(`\n✗ ${plan.error}\n`);
|
|
1220
|
+
summary.push(["podman stack", `NOT installed: ${plan.error}`]);
|
|
1221
|
+
return;
|
|
1222
|
+
}
|
|
1223
|
+
const changed = plan.files.filter((f) => f.state === "changed");
|
|
1224
|
+
if (changed.length > 0) {
|
|
1225
|
+
// up has no --force, and a file someone edited is theirs until they say otherwise (init's contract).
|
|
1226
|
+
out(`\n✗ ${changed.map((f) => f.path).join(", ")} differs from what this version renders, left untouched\n`);
|
|
1227
|
+
summary.push(["podman stack", "NOT installed: a Quadlet file differs; `pi-dispatch service install --force` replaces it"]);
|
|
1228
|
+
return;
|
|
1229
|
+
}
|
|
1230
|
+
// The shared installer's foreign-container rule (D3): a container of a unit's name that the unit does not own would
|
|
1231
|
+
// be removed by the unit's `podman run --replace`. up has no --force, so it names both ways out and installs nothing.
|
|
1232
|
+
// No user manager to talk to (`sudo -iu`, measured): said before anything is written (round 2, E8).
|
|
1233
|
+
const bus = plan.actions.length > 0 ? userBusRefusal({ env, user, euid }) : null;
|
|
1234
|
+
if (bus) {
|
|
1235
|
+
out(`\n✗ ${bus}\n`);
|
|
1236
|
+
summary.push(["podman stack", "NOT installed: no user manager reachable from this shell"]);
|
|
1237
|
+
return;
|
|
1238
|
+
}
|
|
1239
|
+
// The manager's own XDG_RUNTIME_DIR and XDG_CONFIG_HOME (PR #463 round 3): service install's refusal, the same words.
|
|
1240
|
+
const managerEnv = plan.actions.length > 0 ? managerEnvRefusal(await runCmdQuery(spawn, "systemctl", ["--user", "show-environment"]), { home, euid, user, realpath }) : null;
|
|
1241
|
+
if (managerEnv) {
|
|
1242
|
+
out(`\n✗ ${managerEnv}\n`);
|
|
1243
|
+
summary.push(["podman stack", "NOT installed: the user manager's environment names another account's directories"]);
|
|
1244
|
+
return;
|
|
1245
|
+
}
|
|
1246
|
+
const containers = await foreignContainers(plan, (cmd, args) => runCmdQuery(spawn, cmd, args));
|
|
1247
|
+
if (containers.unknown.length > 0) {
|
|
1248
|
+
const why = unknownContainerRefusal(containers.unknown);
|
|
1249
|
+
out(`\n✗ ${why}\n`);
|
|
1250
|
+
summary.push(["podman stack", "NOT installed: podman did not say whether the containers exist"]);
|
|
1251
|
+
return;
|
|
1252
|
+
}
|
|
1253
|
+
const foreign = containers.found;
|
|
1254
|
+
if (foreign.length > 0) {
|
|
1255
|
+
const why = foreignContainerRefusal(foreign, { forceHint: "run `pi-dispatch service install --force`, which replaces it and says so" });
|
|
1256
|
+
out(`\n✗ ${why}\n`);
|
|
1257
|
+
summary.push(["podman stack", `NOT installed: ${foreign.map((f) => f.container).join(", ")} exists and is not the Quadlet unit's`]);
|
|
1258
|
+
return;
|
|
1259
|
+
}
|
|
1260
|
+
const named = [components.valkey ? "Valkey" : null, components.proxy ? "the egress proxy" : null, components.keeper ? `the rootless network keeper (${NETNS_KEEPER}, which keeps the proxy's route out on Podman 4.x)` : null].filter(Boolean);
|
|
1261
|
+
const parts = named.length > 1 ? `${named.slice(0, -1).join(", ")} and ${named.at(-1)}` : named[0];
|
|
1262
|
+
const several = named.length > 1;
|
|
1263
|
+
const accepted = await consent(
|
|
1264
|
+
`${parts} ${several ? "are" : "is"} not running under this account's Podman. up would install ${several ? "them" : "it"} as Quadlet units in your user manager (the same installer \`pi-dispatch service install\` uses; systemd brings them back at boot while linger is on):`,
|
|
1265
|
+
[...(pendingPassword ? [`generate ${VALKEY_PASSWORD_KEY} into ${envPath} (32 random bytes, hex; the value is not shown)`] : []), ...plan.actions.map(describeAction)],
|
|
1266
|
+
{ yes, out, prompt },
|
|
1267
|
+
);
|
|
1268
|
+
const row = several ? "podman stack" : components.valkey ? "valkey" : components.proxy ? "egress (podman)" : "netns keeper (podman)";
|
|
1269
|
+
if (!accepted) {
|
|
1270
|
+
out("skipped: `pi-dispatch service install` installs the same units with the worker\n");
|
|
1271
|
+
summary.push([row, `skipped (declined): ${components.valkey ? "the queue needs Valkey before `pi-dispatch worker` can drain" : components.proxy ? "every podman job is refused pre-spend until the proxy is up (PI_EGRESS=0 opts out)" : "on Podman 4.x the first egress job's teardown cuts the proxy's route out, and every later one gets 503"}`]);
|
|
1272
|
+
return;
|
|
1273
|
+
}
|
|
1274
|
+
// The fs seam gains mkdir here only: up's own writes (`.env`) never needed one, and the Quadlet directory usually
|
|
1275
|
+
// does not exist on a fresh account.
|
|
1276
|
+
// Issue #464: every write journalled, as `service install` does. The plan writes every file before its first command
|
|
1277
|
+
// (planStack), so a failed WRITE puts back every file this run wrote and names any it could not; a failed COMMAND
|
|
1278
|
+
// leaves the files its units run from, named.
|
|
1279
|
+
const journal = [];
|
|
1280
|
+
const stackFs = { ...fs, mkdirSync: fs.mkdirSync ?? mkdir, unlinkSync: fs.unlinkSync ?? unlinkSync };
|
|
1281
|
+
if (pendingPassword) {
|
|
1282
|
+
// Issue #468: the password into .env before any stack file, journalled with what was there, so a failed write
|
|
1283
|
+
// below puts .env back with the rest. Never over a value; the file narrowed to this account; the value unshown.
|
|
1284
|
+
let previous = null;
|
|
1285
|
+
try {
|
|
1286
|
+
previous = fs.readFileSync(envPath);
|
|
1287
|
+
if (!updateEnvFile(envPath, VALKEY_PASSWORD_KEY, pendingPassword, { fs, platform, narrow: true }).changed) throw new Error("the file already assigns it");
|
|
1288
|
+
} catch (err) {
|
|
1289
|
+
out(`✗ ${VALKEY_PASSWORD_KEY} could not be written into ${envPath} (${err?.message ?? err}), so nothing was installed. Continuing; doctor below will re-check\n`);
|
|
1290
|
+
summary.push([row, `NOT installed: ${VALKEY_PASSWORD_KEY} could not be written into .env`]);
|
|
1291
|
+
return;
|
|
1292
|
+
}
|
|
1293
|
+
journal.push({ path: envPath, existed: true, previous });
|
|
1294
|
+
out(`✓ generated ${VALKEY_PASSWORD_KEY} into .env (value not shown; .env is now readable by this account only)\n`);
|
|
1295
|
+
}
|
|
1296
|
+
const applied = await applyStack(plan, { fs: stackFs, run: (cmd, args) => runStreamed(spawn, cmd, args, out), journal });
|
|
1297
|
+
if (!applied.ok) {
|
|
1298
|
+
if (!applied.ran) {
|
|
1299
|
+
const rolled = describeRollBack(rollBackWrites(stackFs, journal));
|
|
1300
|
+
out(`✗ ${applied.failed} failed (${applied.message ?? "write error"}), so nothing was installed: ${rolled}. Continuing; doctor below will re-check\n`);
|
|
1301
|
+
summary.push([row, `install FAILED at: ${applied.failed}; ${rolled}`]);
|
|
1302
|
+
return;
|
|
1303
|
+
}
|
|
1304
|
+
const remain = [...new Set(journal.map((e) => e.path))];
|
|
1305
|
+
out(`✗ ${applied.failed} failed: continuing; doctor below will re-check. \`journalctl --user -u ${plan.start.join(" -u ")}\` has the details. Files this run wrote, which remain: ${remain.join(", ") || "none"}\n`);
|
|
1306
|
+
summary.push([row, `install FAILED at: ${applied.failed}`]);
|
|
1307
|
+
return;
|
|
1308
|
+
}
|
|
1309
|
+
const restarted = plan.restart ?? [];
|
|
1310
|
+
const fresh = plan.start.filter((u) => !restarted.includes(u));
|
|
1311
|
+
if (fresh.length > 0) out(`✓ started ${fresh.join(" ")}\n`);
|
|
1312
|
+
if (restarted.length > 0) out(`✓ restarted ${restarted.join(" ")}\n`);
|
|
1313
|
+
// `up` refuses a changed file above, so the one restart it makes is a keeper that was there and not holding.
|
|
1314
|
+
summary.push([row, `installed as Quadlet units and ${restarted.length > 0 ? `${fresh.length > 0 ? `started (${fresh.join(", ")}) and ` : ""}restarted (${restarted.join(", ")})` : `started (${plan.start.join(", ")})`}`]);
|
|
1315
|
+
// A keeper (re)started under a proxy that stays up, ours left running or the operator's own PI_EGRESS_PROXY, leaves
|
|
1316
|
+
// that proxy for the operator to restart, named: the worker and doctor both ask for exactly that (PR #463 round 3).
|
|
1317
|
+
const hint = keeperUnderRunningProxyHint(plan, egressProxyName(env), { keeperStarting: plan.start.includes(QUADLET_FILES.keeper.unit) });
|
|
1318
|
+
if (hint) out(`⚠ ${hint}\n`);
|
|
1319
|
+
out(lingerNote(await readLinger(user, (cmd, args) => runCmdCapture(spawn, cmd, args)), user));
|
|
1320
|
+
}
|
|
1321
|
+
|
|
1322
|
+
/**
|
|
1323
|
+
* Whether a `podman ps --format {{.Names}}|{{.Ports}}` line is the pi-dispatch-valkey container publishing host port
|
|
1324
|
+
* `port` to its 6379 (issue #464). Ports read like `127.0.0.1:16468->6379/tcp` (measured, Podman 5.8.1 and 4.9.3), a
|
|
1325
|
+
* list comma-separated.
|
|
1326
|
+
*/
|
|
1327
|
+
export function valkeyContainerPublishes(line, port) {
|
|
1328
|
+
const [name, ports = ""] = String(line).trim().split("|");
|
|
1329
|
+
if (name !== "pi-dispatch-valkey") return false;
|
|
1330
|
+
return ports.split(",").some((p) => {
|
|
1331
|
+
const m = /:(\d+)->6379\/tcp$/.exec(p.trim());
|
|
1332
|
+
return m !== null && Number(m[1]) === port;
|
|
1333
|
+
});
|
|
1334
|
+
}
|
|
1335
|
+
|
|
1336
|
+
/**
|
|
1337
|
+
* Issue #484: the folder's `deploy/egress-proxy.conf` against the installed package's copy. Nothing is said for an
|
|
1338
|
+
* identical or an absent copy (the proxy step says what a missing file costs); a differing one is offered for
|
|
1339
|
+
* replacement, shown and asked. Returns true when it was replaced.
|
|
1340
|
+
*
|
|
1341
|
+
* `--yes` DOES NOT COVER IT, and that is the one departure from up's consent rule ("--yes accepts them all"): every other
|
|
1342
|
+
* action --yes accepts creates or starts something, or replaces a container this project made, while this replaces a
|
|
1343
|
+
* FILE that reads the same whether an upgrade left it behind or the operator edited it on purpose, and nothing here can
|
|
1344
|
+
* tell the two apart. So the lines are printed and a person answers, as for removing a proxy jobs are using. The old
|
|
1345
|
+
* bytes are kept beside it either way (`replaceProxyConfCopy`), since an edit lost to a wrong "y" is still the
|
|
1346
|
+
* operator's.
|
|
1347
|
+
*/
|
|
1348
|
+
async function proxyConfRefreshStep({ fs, cwd, out, prompt, summary, readPackagedConf, now, yes }) {
|
|
1349
|
+
const rel = "deploy/egress-proxy.conf";
|
|
1350
|
+
const path = join(cwd, rel);
|
|
1351
|
+
// A directory there is the proxy step's to say (it starts nothing on one), so it is not read as a copy here.
|
|
1352
|
+
if (pathIsDirectory(fs, path)) return false;
|
|
1353
|
+
const judged = judgeProxyConfCopy({ path, read: (p) => fs.readFileSync(p, "utf8"), readPackaged: readPackagedConf });
|
|
1354
|
+
if (judged.state === "absent" || judged.state === "same") return false;
|
|
1355
|
+
if (judged.state !== "differs") {
|
|
1356
|
+
const why = judged.state === "no-package" ? `the package's own copy (${PACKAGED_EGRESS_PROXY_CONF}) could not be read (${judged.error})` : `it could not be read (${judged.error})`;
|
|
1357
|
+
out(`\n⚠ ${rel} was not compared with ${packageCopyName()}: ${why}\n`);
|
|
1358
|
+
summary.push(["egress rules", `not compared: ${why}`]);
|
|
1359
|
+
return false;
|
|
1360
|
+
}
|
|
1361
|
+
out(`\n⚠ ${rel} differs from ${packageCopyName()}: ${judged.summary}. An upgrade does not rewrite it, so it is either an older version's rules or an edit of your own (\`diff ${rel} ${PACKAGED_EGRESS_PROXY_CONF}\` shows which)\n`);
|
|
1362
|
+
if (yes) out("--yes does not cover replacing a file that may hold your own edits: answer below\n");
|
|
1363
|
+
const accepted = await consent(
|
|
1364
|
+
`up would replace it with ${packageCopyName()}, keeping yours beside it:`,
|
|
1365
|
+
[`cp ${rel} ${rel}.bak-<timestamp>`, `write ${PACKAGED_EGRESS_PROXY_CONF}'s content beside ${rel}, then rename it over ${rel}`],
|
|
1366
|
+
{ yes: false, out, prompt },
|
|
1367
|
+
);
|
|
1368
|
+
if (!accepted) {
|
|
1369
|
+
out(`skipped: ${rel} is left as it is; \`pi-dispatch up\` offers this again\n`);
|
|
1370
|
+
summary.push(["egress rules", `${rel} differs from ${packageCopyName()}, left as it is (declined)`]);
|
|
1371
|
+
return false;
|
|
1372
|
+
}
|
|
1373
|
+
const done = replaceProxyConfCopy({ path, text: judged.packaged, fs, now });
|
|
1374
|
+
if (!done.ok) {
|
|
1375
|
+
out(`✗ not replaced: ${done.reason}\n`);
|
|
1376
|
+
summary.push(["egress rules", `NOT replaced: ${done.reason}`]);
|
|
1377
|
+
return false;
|
|
1378
|
+
}
|
|
1379
|
+
const backup = `${rel}${done.backup.slice(path.length)}`;
|
|
1380
|
+
out(`✓ replaced ${rel} with ${packageCopyName()}; yours is kept as ${backup}. squid reads it only at start, so a proxy already running or paused is offered a restart below\n`);
|
|
1381
|
+
summary.push(["egress rules", `${rel} replaced with ${packageCopyName()} (the old one is ${backup})`]);
|
|
1382
|
+
return true;
|
|
1383
|
+
}
|
|
1384
|
+
|
|
258
1385
|
/**
|
|
259
1386
|
* Show the exact commands, then ask. Printing happens with or without `--yes`: consent is what the
|
|
260
1387
|
* flag waives, never visibility — every host mutation is on screen before it runs. The prompt
|
|
@@ -271,9 +1398,146 @@ async function consent(intro, commands, { yes, out, prompt }) {
|
|
|
271
1398
|
return /^y(es)?$/i.test(String(answer ?? "").trim());
|
|
272
1399
|
}
|
|
273
1400
|
|
|
274
|
-
/**
|
|
1401
|
+
/**
|
|
1402
|
+
* Re-join an argv array for display, quoting the args that contain spaces (e.g. "valkey-cli ping"). An arg with a
|
|
1403
|
+
* double quote or a `$` in it (issue #468: the Valkey start script) is shown in single quotes, which is how a shell
|
|
1404
|
+
* would take it unchanged; none of those args has a single quote of its own.
|
|
1405
|
+
*/
|
|
275
1406
|
function quoteArgs(args) {
|
|
276
|
-
return args.map((a) => (a.includes(" ") ? `"${a}"` : a)).join(" ");
|
|
1407
|
+
return args.map((a) => (/["$]/.test(a) ? `'${a}'` : a.includes(" ") ? `"${a}"` : a)).join(" ");
|
|
1408
|
+
}
|
|
1409
|
+
|
|
1410
|
+
/**
|
|
1411
|
+
* Issue #468: the password docker's Valkey starts with, from the deployment's `.env` (`dockerValkeyPasswordStep`): kept
|
|
1412
|
+
* when set, generated into the file when the deployment runs its own Valkey on this host and has none, null otherwise.
|
|
1413
|
+
* Said on screen and in the summary without the value.
|
|
1414
|
+
*/
|
|
1415
|
+
function dockerValkeyPasswordStep({ fs, envPath, platform, out, summary, newPassword }) {
|
|
1416
|
+
if (!fs.existsSync(envPath)) {
|
|
1417
|
+
summary.push([VALKEY_PASSWORD_KEY, "no .env here, skipped (set it wherever your env lives); the Valkey up starts then has no password"]);
|
|
1418
|
+
return null;
|
|
1419
|
+
}
|
|
1420
|
+
const loader = platform === "linux" ? "systemd" : platform === "darwin" ? "shell" : "cmd";
|
|
1421
|
+
let keys;
|
|
1422
|
+
try {
|
|
1423
|
+
const read = readValkeyKeys(fs.readFileSync(envPath), { loader, path: envPath });
|
|
1424
|
+
if (read.error) throw new Error(read.error);
|
|
1425
|
+
keys = read.keys;
|
|
1426
|
+
} catch (err) {
|
|
1427
|
+
out(`\n✗ ${VALKEY_PASSWORD_KEY}: ${err?.message ?? err}\n`);
|
|
1428
|
+
summary.push([VALKEY_PASSWORD_KEY, `NOT decided: ${err?.message ?? err}`]);
|
|
1429
|
+
return null;
|
|
1430
|
+
}
|
|
1431
|
+
const decided = valkeyPasswordDecision(keys, { envPath });
|
|
1432
|
+
if (decided.error) {
|
|
1433
|
+
out(`\n✗ ${decided.error}\n`);
|
|
1434
|
+
summary.push([VALKEY_PASSWORD_KEY, `NOT usable: ${decided.error}`]);
|
|
1435
|
+
return null;
|
|
1436
|
+
}
|
|
1437
|
+
if (decided.note) {
|
|
1438
|
+
out(`\n⚠ ${decided.note}\n`);
|
|
1439
|
+
summary.push([VALKEY_PASSWORD_KEY, decided.note]);
|
|
1440
|
+
return null;
|
|
1441
|
+
}
|
|
1442
|
+
if (!decided.generate) {
|
|
1443
|
+
summary.push([VALKEY_PASSWORD_KEY, "already set, left untouched (value not shown)"]);
|
|
1444
|
+
return decided.password;
|
|
1445
|
+
}
|
|
1446
|
+
const password = newPassword();
|
|
1447
|
+
try {
|
|
1448
|
+
if (!updateEnvFile(envPath, VALKEY_PASSWORD_KEY, password, { fs, platform, narrow: true }).changed) throw new Error("the file already assigns it");
|
|
1449
|
+
} catch (err) {
|
|
1450
|
+
out(`\n✗ ${VALKEY_PASSWORD_KEY} could not be written: ${err?.message ?? err}\n`);
|
|
1451
|
+
summary.push([VALKEY_PASSWORD_KEY, `NOT written: ${err?.message ?? err}. The Valkey up starts then has no password`]);
|
|
1452
|
+
return null;
|
|
1453
|
+
}
|
|
1454
|
+
out(`\n✓ generated ${VALKEY_PASSWORD_KEY} into .env (32 random bytes, hex; value not shown; .env is now readable by this account only)\n`);
|
|
1455
|
+
summary.push([VALKEY_PASSWORD_KEY, "generated into .env (value not shown; Valkey requires it, and every client sends it)"]);
|
|
1456
|
+
return password;
|
|
1457
|
+
}
|
|
1458
|
+
|
|
1459
|
+
/**
|
|
1460
|
+
* PI_VALKEY_PORT in .env (PR #475's review, round 3): written where VALKEY_URL's port is not 6379 and .env names none,
|
|
1461
|
+
* a differing value named and left alone (`valkeyPortEnvDecision`).
|
|
1462
|
+
*/
|
|
1463
|
+
function dockerValkeyPortStep({ fs, envPath, platform, port, out, summary }) {
|
|
1464
|
+
if (!fs.existsSync(envPath)) return;
|
|
1465
|
+
const loader = platform === "linux" ? "systemd" : platform === "darwin" ? "shell" : "cmd";
|
|
1466
|
+
let decided;
|
|
1467
|
+
try {
|
|
1468
|
+
const found = readEnvAssignments(fs.readFileSync(envPath), [VALKEY_PORT_KEY], { loader })[VALKEY_PORT_KEY];
|
|
1469
|
+
decided = valkeyPortEnvDecision(found === undefined ? undefined : (found.value ?? "?"), port, { envPath });
|
|
1470
|
+
if (decided.write && !updateEnvFile(envPath, VALKEY_PORT_KEY, decided.write, { fs, platform }).changed) throw new Error("the file already assigns it");
|
|
1471
|
+
} catch (err) {
|
|
1472
|
+
out(`\n✗ ${VALKEY_PORT_KEY} could not be written: ${err?.message ?? err}\n`);
|
|
1473
|
+
summary.push([VALKEY_PORT_KEY, `NOT written: ${err?.message ?? err}`]);
|
|
1474
|
+
return;
|
|
1475
|
+
}
|
|
1476
|
+
if (decided.conflict) {
|
|
1477
|
+
out(`\n⚠ ${decided.conflict}\n`);
|
|
1478
|
+
summary.push([VALKEY_PORT_KEY, decided.conflict]);
|
|
1479
|
+
} else if (decided.write) {
|
|
1480
|
+
out(`\n✓ wrote ${VALKEY_PORT_KEY}=${port} into .env (VALKEY_URL's port), so compose in this folder publishes its Valkey there\n`);
|
|
1481
|
+
summary.push([VALKEY_PORT_KEY, `written: ${port} (VALKEY_URL's port)`]);
|
|
1482
|
+
}
|
|
1483
|
+
}
|
|
1484
|
+
|
|
1485
|
+
/** Whether `path` is a directory, followed through a link; false where the seam cannot say (a test's fake fs). */
|
|
1486
|
+
function pathIsDirectory(fs, path) {
|
|
1487
|
+
try {
|
|
1488
|
+
return typeof fs.statSync === "function" && fs.statSync(path).isDirectory() === true;
|
|
1489
|
+
} catch {
|
|
1490
|
+
return false;
|
|
1491
|
+
}
|
|
1492
|
+
}
|
|
1493
|
+
|
|
1494
|
+
/**
|
|
1495
|
+
* Whether this folder holds the compose file its compose lines name (issue #480: a folder init made does not). Every
|
|
1496
|
+
* compose line `up` prints as a later or fallback step asks this first; the hand-over's own consented compose runs are
|
|
1497
|
+
* reached only in a folder the wizard wrote the override into beside that file.
|
|
1498
|
+
*/
|
|
1499
|
+
function composeHere(cwd, fs) {
|
|
1500
|
+
return fs.existsSync(join(cwd, COMPOSE_FILE));
|
|
1501
|
+
}
|
|
1502
|
+
|
|
1503
|
+
/**
|
|
1504
|
+
* The compose command a message names for this folder (PR #475's review): the folder's project and the hand-over
|
|
1505
|
+
* override where the wizard wrote one, else the docs' plain command with the note a wizard folder needs.
|
|
1506
|
+
*/
|
|
1507
|
+
function composeCommandFor(cwd, fs, tail) {
|
|
1508
|
+
const handedOver = fs.existsSync(join(cwd, COMPOSE_VALKEY_OVERRIDE));
|
|
1509
|
+
const project = composeProjectName(cwd);
|
|
1510
|
+
const args = [...composeArgs(handedOver ? { project, override: true } : {}), ...tail];
|
|
1511
|
+
return `docker ${args.join(" ")}${handedOver ? "" : `\` (in a folder /dispatch setup laid out, with -p ${project} after \`compose`}`;
|
|
1512
|
+
}
|
|
1513
|
+
|
|
1514
|
+
/**
|
|
1515
|
+
* Where docker's Valkey is, from VALKEY_URL as the service reads it (`deploymentValkeyEnv`): `{ port, notes }`, with
|
|
1516
|
+
* `remote` for another host (nothing is added for it) or `error` for a URL the Valkey `up` starts could not serve.
|
|
1517
|
+
*/
|
|
1518
|
+
function dockerValkeyWhere({ env, fs, envPath, platform }) {
|
|
1519
|
+
const read = deploymentValkeyEnv({ env, fs, envPath, platform });
|
|
1520
|
+
if (read.error) return { port: 6379, notes: [], error: `${read.error}. up judges the Valkey the service will use, so it adds none` };
|
|
1521
|
+
const url = read.env.VALKEY_URL;
|
|
1522
|
+
const target = valkeyTarget(url);
|
|
1523
|
+
if (target.error) return { port: 6379, notes: read.notes, error: `VALKEY_URL: ${target.error}` };
|
|
1524
|
+
if (!isLoopbackHost(target.host)) return { port: target.port, notes: read.notes, remote: target.host };
|
|
1525
|
+
if (target.host.includes(":")) {
|
|
1526
|
+
return { port: target.port, notes: read.notes, error: `VALKEY_URL's host is ${target.host}, and the Valkey up starts is published on 127.0.0.1 only, so the worker would not reach it. Write VALKEY_URL=redis://127.0.0.1:${target.port}` };
|
|
1527
|
+
}
|
|
1528
|
+
return { port: target.port, notes: read.notes };
|
|
1529
|
+
}
|
|
1530
|
+
|
|
1531
|
+
/** `claimValkeyOwner` (connection.mjs) with the deployment's credential, as every client of the project gets it. */
|
|
1532
|
+
async function defaultClaimValkeyOwner(url, folder, { env, cwd }) {
|
|
1533
|
+
const { claimValkeyOwner, valkeyClientContext } = await import("./connection.mjs");
|
|
1534
|
+
return claimValkeyOwner(url, folder, { context: valkeyClientContext({ env, cwd }) });
|
|
1535
|
+
}
|
|
1536
|
+
|
|
1537
|
+
/** Whether the Valkey at `url` answers a client that sends no password: its `valkeyAuthState`, through connection.mjs. */
|
|
1538
|
+
async function defaultProbeValkeyAuth(url, { env, cwd }) {
|
|
1539
|
+
const { valkeyAuthState, valkeyClientContext } = await import("./connection.mjs");
|
|
1540
|
+
return (await valkeyAuthState(url, { withoutPassword: true, context: valkeyClientContext({ env, cwd }) })).state;
|
|
277
1541
|
}
|
|
278
1542
|
|
|
279
1543
|
/** Exit code of a spawned command; null when it could not launch (not on PATH) — mirrors doctor. */
|
|
@@ -295,11 +1559,13 @@ function runCmd(spawn, cmd, args) {
|
|
|
295
1559
|
* Run a CONSENTED command with its stdout+stderr streamed to `out` as it happens — a docker pull's
|
|
296
1560
|
* progress is the operator's confirmation that the thing they approved is the thing running.
|
|
297
1561
|
*/
|
|
298
|
-
function runStreamed(spawn, cmd, args, out) {
|
|
1562
|
+
function runStreamed(spawn, cmd, args, out, { env } = {}) {
|
|
299
1563
|
return new Promise((resolve) => {
|
|
300
1564
|
let child;
|
|
301
1565
|
try {
|
|
302
|
-
|
|
1566
|
+
// `env` only where a value must reach the command without being on its argv (issue #468: `docker run -e
|
|
1567
|
+
// VALKEY_PASSWORD` reads it from here); otherwise the child inherits this process's, as before.
|
|
1568
|
+
child = spawn(cmd, args, env ? { stdio: ["ignore", "pipe", "pipe"], env } : { stdio: ["ignore", "pipe", "pipe"] });
|
|
303
1569
|
} catch {
|
|
304
1570
|
resolve(null);
|
|
305
1571
|
return;
|
|
@@ -311,6 +1577,54 @@ function runStreamed(spawn, cmd, args, out) {
|
|
|
311
1577
|
});
|
|
312
1578
|
}
|
|
313
1579
|
|
|
1580
|
+
/**
|
|
1581
|
+
* Like runCmdCapture with the streams SEPARATE, for a podman query whose answer is stdout alone (round 2, E2): podman
|
|
1582
|
+
* prints warnings on stderr on ordinary accounts, and a merged capture made `true` and our own label unequal.
|
|
1583
|
+
*/
|
|
1584
|
+
function runCmdQuery(spawn, cmd, args) {
|
|
1585
|
+
return new Promise((resolve) => {
|
|
1586
|
+
let child;
|
|
1587
|
+
try {
|
|
1588
|
+
child = spawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
|
|
1589
|
+
} catch {
|
|
1590
|
+
resolve({ code: null, stdout: "", stderr: "" });
|
|
1591
|
+
return;
|
|
1592
|
+
}
|
|
1593
|
+
let stdout = "";
|
|
1594
|
+
let stderr = "";
|
|
1595
|
+
child.stdout?.on("data", (d) => (stdout += d));
|
|
1596
|
+
child.stderr?.on("data", (d) => (stderr += d));
|
|
1597
|
+
child.on("error", () => resolve({ code: null, stdout, stderr }));
|
|
1598
|
+
child.on("close", (code) => resolve({ code, stdout, stderr }));
|
|
1599
|
+
});
|
|
1600
|
+
}
|
|
1601
|
+
|
|
1602
|
+
/**
|
|
1603
|
+
* `docker network create pi-dispatch-egress-out` only where it does not exist (exec round 3): a recreate otherwise printed
|
|
1604
|
+
* the daemon's "network name ... already used" every time. A quiet inspect first; its answer is never a failure, since the
|
|
1605
|
+
* run that follows is what is checked.
|
|
1606
|
+
*
|
|
1607
|
+
* ASKED BEFORE THE CONSENT, and the create line shown only when the network is missing (PR #456's final check): `--yes`
|
|
1608
|
+
* "runs exactly the lines shown", and the line it used to show, `... create pi-dispatch-egress-out (only if it does
|
|
1609
|
+
* not exist yet)`, was not a command anyone could run. The inspect is read-only, so it runs before the question; the
|
|
1610
|
+
* create after it runs exactly when its line was shown.
|
|
1611
|
+
*/
|
|
1612
|
+
async function egressNetworkLines(spawn, network) {
|
|
1613
|
+
network.missing = (await runCmdQuery(spawn, "docker", ["network", "inspect", EGRESS_NETWORK_ARGS[2]])).code !== 0;
|
|
1614
|
+
return network.missing ? [`docker ${EGRESS_NETWORK_ARGS.join(" ")}`] : [];
|
|
1615
|
+
}
|
|
1616
|
+
/**
|
|
1617
|
+
* The network the run needs, made at RUN time (PR #466 gate round 1): asked again after the answer, since it can be
|
|
1618
|
+
* removed while the prompt waits, and then `docker run --network` fails "network not found" (measured). Created when it
|
|
1619
|
+
* is missing then; when its create line was not shown, because it existed at the question, that is said before it runs.
|
|
1620
|
+
* One that appeared meanwhile is left as it is.
|
|
1621
|
+
*/
|
|
1622
|
+
async function ensureEgressNetwork(spawn, out, network) {
|
|
1623
|
+
if ((await runCmdQuery(spawn, "docker", ["network", "inspect", EGRESS_NETWORK_ARGS[2]])).code === 0) return;
|
|
1624
|
+
if (!network.missing) out(`${EGRESS_NETWORK_ARGS[2]} was removed while the question waited, and the proxy's run needs it: \`docker ${EGRESS_NETWORK_ARGS.join(" ")}\`\n`);
|
|
1625
|
+
await runStreamed(spawn, "docker", EGRESS_NETWORK_ARGS, out);
|
|
1626
|
+
}
|
|
1627
|
+
|
|
314
1628
|
/** Like runCmd but with stdout+stderr captured, for read-only lookups (docker ps). */
|
|
315
1629
|
function runCmdCapture(spawn, cmd, args) {
|
|
316
1630
|
return new Promise((resolve) => {
|