@edgehero/pi-dispatch 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +160 -0
- package/deploy/com.pi-dispatch.worker.plist +66 -0
- package/deploy/nssm-install.cmd +59 -0
- package/deploy/receiver.service +36 -0
- package/deploy/worker-env-wrapper.cmd +50 -0
- package/deploy/worker-env-wrapper.sh +63 -0
- package/deploy/worker.service +55 -0
- package/package.json +83 -0
- package/src/azure-auth.mjs +61 -0
- package/src/azure-host.mjs +236 -0
- package/src/azure-identity.mjs +63 -0
- package/src/azure-prompt.mjs +118 -0
- package/src/branch.mjs +80 -0
- package/src/budget.mjs +179 -0
- package/src/cli.mjs +208 -0
- package/src/config.mjs +329 -0
- package/src/connection.mjs +40 -0
- package/src/cron.mjs +94 -0
- package/src/docker-run.mjs +119 -0
- package/src/doctor.mjs +1127 -0
- package/src/env-allowlist.mjs +198 -0
- package/src/env-file.mjs +153 -0
- package/src/exit-code.mjs +32 -0
- package/src/flow-gate.mjs +82 -0
- package/src/forgejo-auth.mjs +77 -0
- package/src/forgejo-host.mjs +172 -0
- package/src/forgejo-identity.mjs +74 -0
- package/src/forgejo-prompt.mjs +123 -0
- package/src/forges.mjs +148 -0
- package/src/get-token.mjs +226 -0
- package/src/git-dirty.mjs +16 -0
- package/src/github-app-setup.mjs +517 -0
- package/src/github-host.mjs +159 -0
- package/src/github-prompt.mjs +286 -0
- package/src/gitlab-auth.mjs +72 -0
- package/src/gitlab-host.mjs +200 -0
- package/src/gitlab-identity.mjs +61 -0
- package/src/gitlab-prompt.mjs +123 -0
- package/src/identity.mjs +57 -0
- package/src/image-preflight.mjs +180 -0
- package/src/import-pi.mjs +451 -0
- package/src/index.mjs +177 -0
- package/src/init.mjs +77 -0
- package/src/job-id.mjs +100 -0
- package/src/materialize.mjs +138 -0
- package/src/outbox.mjs +179 -0
- package/src/packages.mjs +188 -0
- package/src/pause-windows.mjs +218 -0
- package/src/prepare-github.mjs +260 -0
- package/src/prepare-local.mjs +76 -0
- package/src/prepare.mjs +199 -0
- package/src/pricing.mjs +168 -0
- package/src/processor.mjs +360 -0
- package/src/queue.mjs +152 -0
- package/src/run-container.mjs +133 -0
- package/src/run-history.mjs +534 -0
- package/src/runtime-settings.mjs +188 -0
- package/src/sandbox-cli.mjs +156 -0
- package/src/sandbox-store.mjs +269 -0
- package/src/sandbox.mjs +171 -0
- package/src/scheduler-stall-guard.mjs +67 -0
- package/src/schedules.mjs +62 -0
- package/src/service.mjs +677 -0
- package/src/session-key.mjs +108 -0
- package/src/session-store.mjs +249 -0
- package/src/start.mjs +502 -0
- package/src/subscriptions.mjs +208 -0
- package/src/triggers.mjs +491 -0
- package/src/up.mjs +315 -0
package/src/doctor.mjs
ADDED
|
@@ -0,0 +1,1127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `pi-dispatch doctor` — preflight the host before the first job. Prints a ✓/⚠/✗ line per prerequisite
|
|
3
|
+
* with a one-line fix, and exits non-zero if any hard check fails, so it is usable in a setup script.
|
|
4
|
+
*
|
|
5
|
+
* Reads the handful of values it needs with config.mjs's own defaults rather than loadConfig, so it
|
|
6
|
+
* runs even when GitHub auth is unset — a local-folder deployment needs none of it. Mirrors the kill
|
|
7
|
+
* switch in cli.mjs, which reads only VALKEY_URL for the same reason (it must work when GitHub is
|
|
8
|
+
* misconfigured). The provider key is checked for presence only and never printed (secrets-and-pii).
|
|
9
|
+
*
|
|
10
|
+
* GitHub auth gets two advisory (never failing) checks: the default GITHUB_AUTH_SOURCE=gh mints from the
|
|
11
|
+
* operator's FULL-scope gh login, which then reaches every token-carrying job container — the opposite of
|
|
12
|
+
* the App path's per-repo short-lived tokens (CONST-TOKEN-SCOPED-PER-JOB) — so doctor names the scopes it
|
|
13
|
+
* carries; and gh is preflighted inside the job image, since a token that works host-side but not
|
|
14
|
+
* in-container fails jobs mid-run, not at submit. Token values travel via the spawn env only, never argv.
|
|
15
|
+
*
|
|
16
|
+
* Issue #80 adds the RECEIVER's half of the preflight. Doctor runs on the worker host, but the triggers
|
|
17
|
+
* file names forges whose deliveries only ever arrive if the receiver can boot -- and the receiver is
|
|
18
|
+
* deliberately fail-loud (receiver/src/config.mjs), so a missing WEBHOOK_SECRET or a half-set forge env
|
|
19
|
+
* block is a refusal the operator otherwise meets at deploy time with no forewarning. Doctor mirrors
|
|
20
|
+
* exactly the variables each forge loader hard-requires and WARNS about what boot will refuse -- never
|
|
21
|
+
* fails, because the worker host may legitimately not be the receiver host, and a deployment can be
|
|
22
|
+
* mid-setup. Secrets are checked for presence only and never printed, same rule as the provider key. The
|
|
23
|
+
* github repos the triggers file names also get a READ-ONLY branch-protection preflight, so
|
|
24
|
+
* REQ-BRANCH-PROTECTION-PRECONDITION surfaces at setup time instead of as a refusal comment on the first
|
|
25
|
+
* paid trigger.
|
|
26
|
+
*
|
|
27
|
+
* The overlay checks (REQ-GLOBAL-PI-OVERLAY, INT-TRIGGERS-FILE-CONTRACT) exist because nothing about the
|
|
28
|
+
* overlay is visible from the worker host once jobs are running. BOTH halves of it -- `extensions/` and the
|
|
29
|
+
* staged `packages/` -- now load by default, so the state worth surfacing is no longer "armed": an armed
|
|
30
|
+
* thing is one the operator just switched on and remembers. The dangerous state now is STAGED AND FORGOTTEN,
|
|
31
|
+
* so doctor's overlay lines answer "what will actually load into my job containers", and the ⚠ marks the
|
|
32
|
+
* live third-party code rather than the switch.
|
|
33
|
+
*
|
|
34
|
+
* The silent-failure checks that outlive the flip are unchanged, because they never depended on the default:
|
|
35
|
+
* a manifest naming a staged dir that is gone, and a trigger that explicitly requires packages nobody staged.
|
|
36
|
+
* Both end the same way -- pi skips an absent local source with no error, and the flow exits 0 without the
|
|
37
|
+
* tools it was written for.
|
|
38
|
+
*
|
|
39
|
+
* `doctor --fix` (issue #80, REQ-DEPLOYMENT-BOOTSTRAP) turns SOME fix lines into offers, per failing check.
|
|
40
|
+
* The tier ladder is deliberate: a silent tier for the two fixes whose decision the operator already made
|
|
41
|
+
* (init's create-only scaffolds; mkdir of a directory an env var already names), a prompt tier (y/N,
|
|
42
|
+
* default No, the exact command shown first) for the rest, and a never tier for everything doctor could
|
|
43
|
+
* only fix by guessing -- see the fixAction comment at its first use below. Offering fixes changes NOTHING
|
|
44
|
+
* about severity: a --fix run still exits by the same failed/ok logic, warns stay warns, and the fix pass
|
|
45
|
+
* happens at most once (check, fix, re-check -- never a loop).
|
|
46
|
+
*/
|
|
47
|
+
import { chmodSync, closeSync, existsSync, mkdirSync, openSync, readdirSync, readFileSync, readSync, rmSync, statSync } from "node:fs";
|
|
48
|
+
import { homedir } from "node:os";
|
|
49
|
+
import { join } from "node:path";
|
|
50
|
+
import { fileURLToPath } from "node:url";
|
|
51
|
+
import { spawn as nodeSpawn } from "node:child_process";
|
|
52
|
+
import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
|
|
53
|
+
import { isForgeKind } from "./forges.mjs";
|
|
54
|
+
import { findLiteralSecret, ADMIN_RE } from "./import-pi.mjs";
|
|
55
|
+
import { PACKAGES_SUBDIR, readStageManifest } from "./packages.mjs";
|
|
56
|
+
import { parseTriggers } from "./triggers.mjs";
|
|
57
|
+
|
|
58
|
+
const NODE_FLOOR = [22, 19]; // pi's engine floor (22.19.0)
|
|
59
|
+
|
|
60
|
+
// Which env var holds the credential for each provider. Presence is checked, value never read out.
|
|
61
|
+
// anthropic lists the OAuth token too because it silently takes precedence over the API key upstream.
|
|
62
|
+
const PROVIDER_KEYS = {
|
|
63
|
+
anthropic: ["ANTHROPIC_API_KEY", "ANTHROPIC_OAUTH_TOKEN"],
|
|
64
|
+
openai: ["OPENAI_API_KEY"],
|
|
65
|
+
google: ["GEMINI_API_KEY", "GOOGLE_API_KEY"],
|
|
66
|
+
gemini: ["GEMINI_API_KEY", "GOOGLE_API_KEY"],
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
// gh login scopes that reach well past what a job should ever hold — called out by name in the fix line.
|
|
70
|
+
const BROAD_SCOPES = ["admin:org", "delete_repo", "workflow"];
|
|
71
|
+
|
|
72
|
+
// The loopback Valkey, as one docker argv. Mirrors deploy/docker-compose.yml exactly: AOF on (the
|
|
73
|
+
// wait-list must survive a reboot, REQ-QUEUE-BURST-NO-DROP), bound to 127.0.0.1 only (the queue is not a
|
|
74
|
+
// public surface), restart unless-stopped, data on a named volume. Container and volume names are
|
|
75
|
+
// pi-dispatch-prefixed so compose's own `valkey`/`valkey-data` never collide with these.
|
|
76
|
+
const VALKEY_RUN = ["run", "-d", "--name", "pi-dispatch-valkey", "--restart", "unless-stopped", "-p", "127.0.0.1:6379:6379", "-v", "pi-dispatch-valkey-data:/data", "valkey/valkey:8", "valkey-server", "--appendonly", "yes"];
|
|
77
|
+
|
|
78
|
+
export async function runDoctor(env = process.env, deps = {}) {
|
|
79
|
+
const {
|
|
80
|
+
cwd = process.cwd(),
|
|
81
|
+
out = (s) => process.stdout.write(s),
|
|
82
|
+
spawn = nodeSpawn,
|
|
83
|
+
probeValkey = defaultProbeValkey,
|
|
84
|
+
fileExists = existsSync,
|
|
85
|
+
nodeVersion = process.versions.node,
|
|
86
|
+
// --fix (REQ-DEPLOYMENT-BOOTSTRAP): offer to run the exact fixes doctor already prints. The prompt
|
|
87
|
+
// is injectable so tests drive consent hermetically; the default is a readline y/N that answers No
|
|
88
|
+
// on empty input AND on non-TTY stdin -- a piped or CI `doctor --fix` runs nothing from the prompt
|
|
89
|
+
// tier, because nobody was at the keyboard to consent.
|
|
90
|
+
fix = false,
|
|
91
|
+
promptFn = defaultPromptFn,
|
|
92
|
+
// fs seams for the fixActions, injectable for the same hermetic-test reason as fileExists.
|
|
93
|
+
mkdir = mkdirSync,
|
|
94
|
+
chmod = chmodSync,
|
|
95
|
+
rm = rmSync,
|
|
96
|
+
} = deps;
|
|
97
|
+
const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm };
|
|
98
|
+
|
|
99
|
+
let checks = await collectChecks(env, seams);
|
|
100
|
+
let failed = render(checks, out);
|
|
101
|
+
|
|
102
|
+
if (fix) {
|
|
103
|
+
const ran = await applyFixes(checks, seams, promptFn);
|
|
104
|
+
if (ran > 0) {
|
|
105
|
+
// Converge-to-green: the probes are idempotent and cheap, so ONE full re-collect answers "did
|
|
106
|
+
// the fixes take" without bookkeeping about which probe fed which check. At most once,
|
|
107
|
+
// structurally -- the re-check never re-enters the fix pass, so a fix that did not take is
|
|
108
|
+
// reported still-failing rather than retried forever.
|
|
109
|
+
checks = await collectChecks(env, seams);
|
|
110
|
+
const passing = checks.filter((c) => c.ok).length;
|
|
111
|
+
out(`\nre-check after fixes: ${passing} of ${checks.length} checks pass\n`);
|
|
112
|
+
for (const c of checks.filter((c) => !c.ok)) {
|
|
113
|
+
out(`${c.warn ? "⚠" : "✗"} ${c.label}\n → ${c.fix}\n`);
|
|
114
|
+
}
|
|
115
|
+
// Recomputed with the SAME failed/ok logic as the first pass (warn-not-fail): --fix changes
|
|
116
|
+
// what doctor does, never how it judges. A converged run exits 0 because the checks pass now,
|
|
117
|
+
// not because attempting fixes earns credit.
|
|
118
|
+
failed = checks.some((c) => !c.ok && !c.warn);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
out(failed ? "\ndoctor: some checks failed — fix the above, then re-run.\n" : "\ndoctor: ready. Start the worker with `pi-dispatch worker`.\n");
|
|
123
|
+
return failed ? 1 : 0;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** The ✓/⚠/✗ lines plus each failure's fix, exactly as doctor has always printed them. Returns whether
|
|
127
|
+
* any HARD check failed (a ⚠ never fails doctor). */
|
|
128
|
+
function render(checks, out) {
|
|
129
|
+
let failed = false;
|
|
130
|
+
for (const c of checks) {
|
|
131
|
+
out(`${c.ok ? "✓" : c.warn ? "⚠" : "✗"} ${c.label}\n`);
|
|
132
|
+
if (!c.ok) {
|
|
133
|
+
out(` → ${c.fix}\n`);
|
|
134
|
+
if (!c.warn) failed = true;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
return failed;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* The --fix pass (REQ-DEPLOYMENT-BOOTSTRAP): walk the rendered checks IN ORDER and act on each failing one
|
|
142
|
+
* that carries a fixAction. Returns how many fixes actually RAN -- a declined offer counts for nothing, so
|
|
143
|
+
* a decline-everything run re-checks nothing and ends exactly like a fix-less one.
|
|
144
|
+
*/
|
|
145
|
+
async function applyFixes(checks, seams, promptFn) {
|
|
146
|
+
let ran = 0;
|
|
147
|
+
for (const c of checks) {
|
|
148
|
+
if (c.ok || !c.fixAction) continue;
|
|
149
|
+
const fa = c.fixAction;
|
|
150
|
+
if (fa.tier === "prompt") {
|
|
151
|
+
// The exact command first, then consent, default No. The same philosophy that runs jobs with
|
|
152
|
+
// --pull=never holds here: nothing is fetched or started implicitly -- the y keypress IS the
|
|
153
|
+
// operator running the command themselves, and doctor only saves the retyping after it.
|
|
154
|
+
seams.out(`\nfix available: ${c.label}\n $ ${fa.describe}\n`);
|
|
155
|
+
if (!(await promptFn("run this? [y/N] "))) {
|
|
156
|
+
seams.out(`skipped: ${c.label}\n`);
|
|
157
|
+
continue;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
ran++;
|
|
161
|
+
let res;
|
|
162
|
+
try {
|
|
163
|
+
res = await fa.run(seams);
|
|
164
|
+
} catch (e) {
|
|
165
|
+
res = { ok: false, note: e?.message ?? String(e) };
|
|
166
|
+
}
|
|
167
|
+
seams.out(`${res.ok ? "fixed" : "fix failed"}: ${c.label}${res.note ? ` — ${res.note}` : ""}\n`);
|
|
168
|
+
}
|
|
169
|
+
return ran;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* The default --fix consent prompt: y/N over readline, No unless the operator typed y/yes. Two refusals
|
|
174
|
+
* are load-bearing: EMPTY input is No (plain enter must never consent), and NON-TTY stdin is No without
|
|
175
|
+
* reading at all -- a piped or CI `doctor --fix` has nobody at the keyboard, so the prompt tier must
|
|
176
|
+
* execute nothing there. Streams are injectable and the function exported so tests exercise both refusals
|
|
177
|
+
* without owning the process's real stdin.
|
|
178
|
+
*/
|
|
179
|
+
export async function defaultPromptFn(question, { input = process.stdin, output = process.stdout } = {}) {
|
|
180
|
+
if (!input.isTTY) return false;
|
|
181
|
+
const { createInterface } = await import("node:readline/promises");
|
|
182
|
+
const rl = createInterface({ input, output });
|
|
183
|
+
try {
|
|
184
|
+
const answer = (await rl.question(question)).trim().toLowerCase();
|
|
185
|
+
return answer === "y" || answer === "yes";
|
|
186
|
+
} finally {
|
|
187
|
+
rl.close();
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* Run every probe and return the check list without rendering -- runDoctor renders it, and under --fix
|
|
193
|
+
* collects it a second time for the converge re-check. Exported for the never-tier doctrine pin in the
|
|
194
|
+
* tests (githubProtectionPreflight's precedent): the test walks the returned array and fails on any check
|
|
195
|
+
* that grows a `fixAction` outside the allowed set, so the never tier stays a tested contract rather than
|
|
196
|
+
* a comment.
|
|
197
|
+
*/
|
|
198
|
+
export async function collectChecks(env, seams) {
|
|
199
|
+
const { cwd, spawn, probeValkey, fileExists, nodeVersion } = seams;
|
|
200
|
+
|
|
201
|
+
const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
|
|
202
|
+
const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
|
|
203
|
+
const provider = env.PI_PROVIDER ?? "anthropic";
|
|
204
|
+
|
|
205
|
+
const checks = [];
|
|
206
|
+
checks.push(nodeCheck(nodeVersion));
|
|
207
|
+
|
|
208
|
+
checks.push({
|
|
209
|
+
ok: fileExists(join(cwd, ".env")),
|
|
210
|
+
warn: true, // advisory: env may be supplied by a service manager instead of a file
|
|
211
|
+
label: ".env present",
|
|
212
|
+
fix: "run `pi-dispatch init` to scaffold one (or supply env via your service manager)",
|
|
213
|
+
// `fixAction` -- what `doctor --fix` may offer for a failing check (REQ-DEPLOYMENT-BOOTSTRAP):
|
|
214
|
+
// { tier: "silent"|"prompt", describe: <the exact command>, run(seams) }. Silent runs unprompted
|
|
215
|
+
// and is reported after; ONLY two fixes qualify, because in both the operator already made the
|
|
216
|
+
// decision and only the mechanical remainder is left: (1) THIS one, delegating absent config files
|
|
217
|
+
// to init, which is create-only by contract (init.mjs header) and so can overwrite nothing; and
|
|
218
|
+
// (2) mkdir -p + chmod 700 of a directory an env var already names (the session store, below).
|
|
219
|
+
// Prompt shows the exact command and defaults to No. EVERY other check deliberately carries no
|
|
220
|
+
// fixAction -- the never tier: doctor never rewrites malformed JSON, never touches triggers or
|
|
221
|
+
// pause-windows CONTENT, never guesses a semantic env value (PI_GLOBAL_ALLOW_EXTENSIONS and kin),
|
|
222
|
+
// never touches branch protection, and never pulls a trigger-named run.image -- each custom image
|
|
223
|
+
// is a per-flow trust posture the operator chose, so only the deployment's OWN default image ever
|
|
224
|
+
// gains an offer. Fail loud and let the operator decide; the plain fix line still prints as before.
|
|
225
|
+
fixAction: {
|
|
226
|
+
tier: "silent",
|
|
227
|
+
describe: "pi-dispatch init",
|
|
228
|
+
run: async ({ out }) => {
|
|
229
|
+
// Lazy import: a doctor run without --fix (or without this failure) never loads init.
|
|
230
|
+
const { runInit } = await import("./init.mjs");
|
|
231
|
+
const code = runInit(cwd, { out });
|
|
232
|
+
return { ok: code === 0, note: "scaffolded by `pi-dispatch init` (create-only: existing files were kept)" };
|
|
233
|
+
},
|
|
234
|
+
},
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
const dockerCode = await runCmd(spawn, "docker", ["info"]);
|
|
238
|
+
checks.push({
|
|
239
|
+
ok: dockerCode === 0,
|
|
240
|
+
label: "Docker daemon reachable",
|
|
241
|
+
fix: dockerCode === null ? "install Docker — `docker` was not found on PATH" : "start Docker (the daemon is not responding)",
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
// Read once, used twice, so the triggers file is parsed a single time: `images` drives the per-trigger
|
|
245
|
+
// image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
|
|
246
|
+
// `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
|
|
247
|
+
// run.packages: true, which arms nothing any more but is still an operator statement of intent.
|
|
248
|
+
const { requiring, optingOut, resuming, replicating, images, forges, repositories } = readTriggerFacts(env, fileExists, cwd);
|
|
249
|
+
|
|
250
|
+
// Only meaningful if docker itself responds; otherwise the image check is noise on top of a down daemon.
|
|
251
|
+
const imageCode = dockerCode === 0 ? await runCmd(spawn, "docker", ["image", "inspect", jobImage]) : null;
|
|
252
|
+
checks.push({
|
|
253
|
+
ok: imageCode === 0,
|
|
254
|
+
label: `Job image present (${jobImage})`,
|
|
255
|
+
fix: "docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest (or build image/Dockerfile)",
|
|
256
|
+
// Prompt tier, and ONLY for the deployment default: a PI_JOB_IMAGE the operator overrode is a trust
|
|
257
|
+
// choice this command cannot honestly satisfy (pulling ghcr's pi-job would not make THEIR image
|
|
258
|
+
// exist), so an overridden name keeps the plain fix line -- the same never-tier reasoning as the
|
|
259
|
+
// trigger-named run.image checks below. Jobs run with --pull=never and that stays true: the y
|
|
260
|
+
// keypress IS the operator pulling the repo's own image themselves.
|
|
261
|
+
...(jobImage === "pi-job:latest"
|
|
262
|
+
? {
|
|
263
|
+
fixAction: {
|
|
264
|
+
tier: "prompt",
|
|
265
|
+
describe: "docker pull ghcr.io/edgehero/pi-job:latest && docker tag ghcr.io/edgehero/pi-job:latest pi-job:latest",
|
|
266
|
+
run: async ({ spawn }) => {
|
|
267
|
+
if ((await runCmd(spawn, "docker", ["pull", "ghcr.io/edgehero/pi-job:latest"])) !== 0) return { ok: false, note: "docker pull failed" };
|
|
268
|
+
if ((await runCmd(spawn, "docker", ["tag", "ghcr.io/edgehero/pi-job:latest", "pi-job:latest"])) !== 0) return { ok: false, note: "docker tag failed" };
|
|
269
|
+
return { ok: true };
|
|
270
|
+
},
|
|
271
|
+
},
|
|
272
|
+
}
|
|
273
|
+
: {}),
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
// Issue #41: every DISTINCT image a trigger names in run.image, minus the deployment default already
|
|
277
|
+
// checked above. Two silent-failure modes, and both used to be impossible because there was one image.
|
|
278
|
+
// 1. the image was never built -- a job that refuses pre-spend at 03:00 in a log nobody is reading, and
|
|
279
|
+
// with --pull=never nothing will fetch it either, so this line is the only warning that arrives first.
|
|
280
|
+
// 2. the image is present but is not a pi-job image. An entrypoint that is not the runner either exits
|
|
281
|
+
// 126/127 or, worse, runs whatever it does have and exits 0 -- a job the queue records as COMPLETED
|
|
282
|
+
// that never started the agent. Warn, never fail: an operator MAY legitimately ship a wrapper
|
|
283
|
+
// entrypoint that execs the runner, and a ✗ here is reserved for certainties.
|
|
284
|
+
// A deployment with no run.image anywhere adds no lines at all, so its output is byte-identical.
|
|
285
|
+
for (const img of dockerCode === 0 ? images.filter((i) => i !== jobImage) : []) {
|
|
286
|
+
const code = await runCmd(spawn, "docker", ["image", "inspect", img]);
|
|
287
|
+
checks.push({
|
|
288
|
+
ok: code === 0,
|
|
289
|
+
label: `Trigger job image present (${img})`,
|
|
290
|
+
fix: `docker pull ${img} (or build it) -- a trigger names it in run.image, and jobs run with --pull=never, so the worker never fetches it at job time`,
|
|
291
|
+
});
|
|
292
|
+
if (code !== 0) continue;
|
|
293
|
+
const entry = await runCmdCapture(spawn, "docker", ["image", "inspect", "--format={{json .Config.Entrypoint}}", img]);
|
|
294
|
+
if (entry.code === 0 && !entry.output.includes("entrypoint.sh")) {
|
|
295
|
+
checks.push({
|
|
296
|
+
ok: false,
|
|
297
|
+
warn: true,
|
|
298
|
+
label: `${img} does not appear to carry the pi-dispatch runner entrypoint`,
|
|
299
|
+
fix: "build your job image FROM this repo's image/Dockerfile so it keeps /entrypoint.sh -- an image without the runner can exit 0 without ever starting the agent, and the queue records that as success (docs/job-image.md)",
|
|
300
|
+
});
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
// The receiver itself, when the triggers file names ANY forge (issue #80). Only forge deliveries need
|
|
305
|
+
// the receiver at all, so a cron/local-only deployment gets no receiver noise here. WARNS rather than
|
|
306
|
+
// fails, same doctrine as the gitlab block below: a deployment can legitimately be mid-setup (or run
|
|
307
|
+
// the receiver on another host with its own env), and doctor's job is to say what will not work, not
|
|
308
|
+
// to refuse.
|
|
309
|
+
if (forges.length > 0) {
|
|
310
|
+
// Presence only, value never read out (secrets-and-pii) -- without it the receiver refuses to boot,
|
|
311
|
+
// because a webhook it cannot verify is a forgeable paid-agent trigger (CONST-HMAC-OVER-RAW-BODY).
|
|
312
|
+
const webhookSecret = env.WEBHOOK_SECRET;
|
|
313
|
+
if (typeof webhookSecret !== "string" || webhookSecret.trim() === "") {
|
|
314
|
+
checks.push({
|
|
315
|
+
ok: false,
|
|
316
|
+
warn: true,
|
|
317
|
+
label: `triggers.json has ${forges.join("/")} triggers but WEBHOOK_SECRET is unset -- the receiver will refuse to start`,
|
|
318
|
+
fix: "generate one (`openssl rand -hex 32`) and set WEBHOOK_SECRET in .env -- `pi-dispatch-receiver` verifies every delivery's signature against it and refuses to boot without it",
|
|
319
|
+
});
|
|
320
|
+
} else {
|
|
321
|
+
checks.push({ ok: true, label: "WEBHOOK_SECRET set -- the receiver (`pi-dispatch-receiver`) can verify deliveries" });
|
|
322
|
+
}
|
|
323
|
+
// Validated only WHEN SET: unset (or empty) means the loader's own default of 3000, which needs no
|
|
324
|
+
// line. The malformed value IS echoed -- a port is not a secret, and naming the shape it actually
|
|
325
|
+
// has is what makes the warn actionable. Mirrors `positiveInt` (worker config.mjs) exactly, because
|
|
326
|
+
// that is the parse the receiver refuses to boot on.
|
|
327
|
+
const port = env.RECEIVER_PORT;
|
|
328
|
+
if (port !== undefined && port !== "") {
|
|
329
|
+
const n = Number.parseInt(port, 10);
|
|
330
|
+
if (!Number.isInteger(n) || n < 1 || String(n) !== String(port).trim()) {
|
|
331
|
+
checks.push({
|
|
332
|
+
ok: false,
|
|
333
|
+
warn: true,
|
|
334
|
+
label: `RECEIVER_PORT is ${JSON.stringify(port)}, which is not a positive integer -- the receiver will refuse to start`,
|
|
335
|
+
fix: "set RECEIVER_PORT to a TCP port number, or drop it for the default (3000)",
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
// GitLab, when the triggers file names it. WARNS rather than fails, matching the github auth checks
|
|
342
|
+
// below and for the same reason: a deployment can legitimately be mid-setup, and doctor's job is to say
|
|
343
|
+
// what will not work, not to refuse.
|
|
344
|
+
if (forges.includes("gitlab")) {
|
|
345
|
+
const token = env.GITLAB_TOKEN;
|
|
346
|
+
if (typeof token !== "string" || token.trim() === "") {
|
|
347
|
+
checks.push({
|
|
348
|
+
ok: false,
|
|
349
|
+
warn: true,
|
|
350
|
+
label: "triggers.json has gitlab triggers but GITLAB_TOKEN is unset",
|
|
351
|
+
fix: "set GITLAB_TOKEN to a project or group access token with the `api` scope -- gitlab jobs cannot clone, comment, or resolve the actor's access level without it",
|
|
352
|
+
});
|
|
353
|
+
} else {
|
|
354
|
+
checks.push({ ok: true, label: `gitlab triggers configured (${env.GITLAB_URL ?? "https://gitlab.com"})` });
|
|
355
|
+
// The scope an operator cannot narrow. Said out loud because it is the one place GitLab is
|
|
356
|
+
// weaker than the github App path and an operator should know which trade they made
|
|
357
|
+
// (CONST-TOKEN-SCOPED-PER-JOB).
|
|
358
|
+
checks.push({
|
|
359
|
+
ok: true,
|
|
360
|
+
warn: true,
|
|
361
|
+
label: "a GitLab project access token needs the `api` scope to post notes, which grants full project API read/write",
|
|
362
|
+
fix: "scope the token to ONE project and rotate it on a schedule -- GitLab offers no contents-vs-issues split, and no short-expiry equivalent of a GitHub App token",
|
|
363
|
+
});
|
|
364
|
+
}
|
|
365
|
+
// The receiver-boot half (issue #80), mirrored from receiver/src/config.mjs loadGitLabConfig: once
|
|
366
|
+
// ANY GITLAB_* variable is set, boot refuses without a chosen mode and a secret -- and with NONE
|
|
367
|
+
// set there is no /gitlab route at all, so these triggers can never fire either way. The mode value
|
|
368
|
+
// is echoed (it is a choice, not a secret); the secret is presence-only.
|
|
369
|
+
const glMode = env.GITLAB_WEBHOOK_MODE;
|
|
370
|
+
if (glMode !== "signature" && glMode !== "token") {
|
|
371
|
+
checks.push({
|
|
372
|
+
ok: false,
|
|
373
|
+
warn: true,
|
|
374
|
+
label: `triggers.json has gitlab triggers but GITLAB_WEBHOOK_MODE is ${glMode === undefined ? "unset" : JSON.stringify(glMode)} -- the receiver will refuse to start`,
|
|
375
|
+
fix: 'set it to "signature" (HMAC, GitLab 19.0+) or "token" (X-Gitlab-Token, any version) -- deliberately undefaulted, which verification a deployment runs must be a thing somebody chose (.env.example, docs/gitlab.md)',
|
|
376
|
+
});
|
|
377
|
+
}
|
|
378
|
+
if (typeof env.GITLAB_WEBHOOK_SECRET !== "string" || env.GITLAB_WEBHOOK_SECRET.trim() === "") {
|
|
379
|
+
checks.push({
|
|
380
|
+
ok: false,
|
|
381
|
+
warn: true,
|
|
382
|
+
label: "triggers.json has gitlab triggers but GITLAB_WEBHOOK_SECRET is unset -- the receiver cannot verify deliveries and will refuse to start",
|
|
383
|
+
fix: "set GITLAB_WEBHOOK_SECRET in .env to the secret configured on the project webhook (.env.example, docs/gitlab.md)",
|
|
384
|
+
});
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
// Forgejo, when the triggers file names it (issue #80) -- the gitlab block's twin, and previously the
|
|
389
|
+
// gap: a forgejo misconfiguration hard-failed at receiver boot with no preflight warning. The variable
|
|
390
|
+
// set mirrors receiver/src/config.mjs loadForgejoConfig exactly (those three are what boot
|
|
391
|
+
// hard-requires), so this warns about precisely what the receiver will refuse. Presence-only for all
|
|
392
|
+
// three: FORGEJO_URL is no secret, but one rule for the set is one rule to audit.
|
|
393
|
+
if (forges.includes("forgejo")) {
|
|
394
|
+
const missing = ["FORGEJO_URL", "FORGEJO_WEBHOOK_SECRET", "FORGEJO_TOKEN"].filter((k) => typeof env[k] !== "string" || env[k].trim() === "");
|
|
395
|
+
if (missing.length > 0) {
|
|
396
|
+
checks.push({
|
|
397
|
+
ok: false,
|
|
398
|
+
warn: true,
|
|
399
|
+
label: `triggers.json has forgejo triggers but ${missing.join(", ")} ${missing.length === 1 ? "is" : "are"} unset -- the receiver will refuse to start (or serve no /forgejo endpoint at all)`,
|
|
400
|
+
fix: "set them in .env (.env.example documents each; docs/forgejo.md walks the webhook setup); FORGEJO_BOT_ID is also needed when FORGEJO_TOKEN is repository-scoped -- a scoped token cannot call GET /user to identify itself",
|
|
401
|
+
});
|
|
402
|
+
} else {
|
|
403
|
+
checks.push({ ok: true, label: `forgejo triggers configured (${env.FORGEJO_URL})` });
|
|
404
|
+
}
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
// Azure DevOps, when the triggers file names it (issue #80) -- same shape, mirrored from
|
|
408
|
+
// receiver/src/config.mjs loadAzureConfig. AZURE_WEBHOOK_MODE gets its own line because it is
|
|
409
|
+
// required-UNDEFAULTED: Azure offers no HMAC at all, so both modes are shared-secret compares, and
|
|
410
|
+
// which header carries the secret must be a thing somebody decided. AZURE_WEBHOOK_HEADER joins the
|
|
411
|
+
// required set only under mode=header, exactly as boot requires it.
|
|
412
|
+
if (forges.includes("azure")) {
|
|
413
|
+
const azMode = env.AZURE_WEBHOOK_MODE;
|
|
414
|
+
const azModeOk = azMode === "basic" || azMode === "header";
|
|
415
|
+
if (!azModeOk) {
|
|
416
|
+
checks.push({
|
|
417
|
+
ok: false,
|
|
418
|
+
warn: true,
|
|
419
|
+
label: `triggers.json has azure triggers but AZURE_WEBHOOK_MODE is ${azMode === undefined ? "unset" : JSON.stringify(azMode)} -- the receiver will refuse to start`,
|
|
420
|
+
fix: 'set it to "basic" (HTTP Basic on the service hook) or "header" (a custom header) -- deliberately undefaulted, Azure offers no HMAC, so which shared-secret compare gates the endpoint must be a chosen thing (.env.example, docs/azure-devops.md)',
|
|
421
|
+
});
|
|
422
|
+
}
|
|
423
|
+
const azRequired = ["AZURE_WEBHOOK_SECRET", "AZURE_TOKEN", "AZURE_ORG_URL", ...(azMode === "header" ? ["AZURE_WEBHOOK_HEADER"] : [])];
|
|
424
|
+
const azMissing = azRequired.filter((k) => typeof env[k] !== "string" || env[k].trim() === "");
|
|
425
|
+
if (azMissing.length > 0) {
|
|
426
|
+
checks.push({
|
|
427
|
+
ok: false,
|
|
428
|
+
warn: true,
|
|
429
|
+
label: `triggers.json has azure triggers but ${azMissing.join(", ")} ${azMissing.length === 1 ? "is" : "are"} unset -- the receiver will refuse to start (or serve no /azure endpoint at all)`,
|
|
430
|
+
fix: "set them in .env (.env.example documents each; docs/azure-devops.md walks the service-hook setup)",
|
|
431
|
+
});
|
|
432
|
+
} else if (azModeOk) {
|
|
433
|
+
checks.push({ ok: true, label: `azure triggers configured (${env.AZURE_ORG_URL})` });
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
// REQ-BRANCH-PROTECTION-PRECONDITION, preflighted (issue #80). The worker refuses a forge-backed job
|
|
438
|
+
// on an unprotected default branch BEFORE any spend -- correct, but that answer arrives at the first
|
|
439
|
+
// paid trigger, as a refusal comment a requester is already waiting on. Doctor asks the same question
|
|
440
|
+
// READ-ONLY at setup time, for each github repo the triggers file NAMES. Warn, never fail: the
|
|
441
|
+
// worker's own gate stays the enforcement, this is the early copy of its answer.
|
|
442
|
+
if (forges.includes("github")) {
|
|
443
|
+
if (repositories.length === 0) {
|
|
444
|
+
// A github label/comment trigger takes its repository from each delivery's payload, and the
|
|
445
|
+
// shared schema admits `run.repository` only on azure triggers today (triggers.mjs,
|
|
446
|
+
// validateRepository) -- so there is nothing here to ask GitHub about. Said in ONE line so a
|
|
447
|
+
// green doctor cannot read as "this REQ was preflighted": it is enforced per job, just not
|
|
448
|
+
// checkable from here.
|
|
449
|
+
checks.push({
|
|
450
|
+
ok: true,
|
|
451
|
+
label: "github triggers take their repository from each delivery -- branch protection cannot be preflighted per repo here, and is enforced per job before any spend (REQ-BRANCH-PROTECTION-PRECONDITION)",
|
|
452
|
+
});
|
|
453
|
+
} else {
|
|
454
|
+
checks.push(...(await githubProtectionPreflight(spawn, repositories)));
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
// The default source `gh` mints job tokens from the operator's gh login, so the FULL-scope login token
|
|
459
|
+
// reaches every token-carrying job container — the opposite of the App path's per-repo short-lived
|
|
460
|
+
// tokens (CONST-TOKEN-SCOPED-PER-JOB). Both checks below warn, never fail: a local-only deployment with
|
|
461
|
+
// the default source is valid, for the same reason the worker's own auth at start is best-effort.
|
|
462
|
+
const ghSource = env.GITHUB_AUTH_SOURCE ?? "gh"; // config.mjs's own default, read directly — no loadConfig
|
|
463
|
+
if (ghSource === "gh") {
|
|
464
|
+
// gh writes `auth status` to stdout or stderr depending on version — capture both combined.
|
|
465
|
+
const status = await runCmdCapture(spawn, "gh", ["auth", "status"]);
|
|
466
|
+
if (status.code === 0) {
|
|
467
|
+
const scopes = parseGhTokenScopes(status.output);
|
|
468
|
+
const broad = (scopes ?? []).filter((s) => BROAD_SCOPES.includes(s));
|
|
469
|
+
checks.push({
|
|
470
|
+
ok: false,
|
|
471
|
+
warn: true,
|
|
472
|
+
label: `GITHUB_AUTH_SOURCE=gh forwards your full gh login into every token-carrying job container (${
|
|
473
|
+
scopes ? `scopes: ${scopes.join(", ")}` : "scopes not reported (fine-grained token)"
|
|
474
|
+
})`,
|
|
475
|
+
fix:
|
|
476
|
+
(broad.length > 0 ? `this token carries broad scopes (${broad.join(", ")}) -- ` : "") +
|
|
477
|
+
"use a fine-grained PAT (GITHUB_AUTH_SOURCE=pat) or a GitHub App for per-job scoping -- see SECURITY.md",
|
|
478
|
+
});
|
|
479
|
+
} else {
|
|
480
|
+
checks.push({
|
|
481
|
+
ok: false,
|
|
482
|
+
warn: true,
|
|
483
|
+
label: "GITHUB_AUTH_SOURCE is gh but `gh auth status` failed",
|
|
484
|
+
fix: "run `gh auth login` (or switch GITHUB_AUTH_SOURCE) -- github jobs and run.github cron triggers will refuse to run",
|
|
485
|
+
});
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
// GITHUB_AUTH_SOURCE=app: completeness of the credential triple loadGitHubAuth hard-requires
|
|
490
|
+
// (config.mjs), preflighted here so a half-finished App setup surfaces as doctor lines instead of a
|
|
491
|
+
// boot refusal. WARN, never fail, same doctrine as the rest of the github block: a deployment can
|
|
492
|
+
// legitimately be mid-setup. The private key gets a hygiene pass on top — presence, POSIX mode, and
|
|
493
|
+
// a first-bytes PEM sniff — but its CONTENTS never reach output: only the leading bytes are read
|
|
494
|
+
// (never the whole key into memory), and nothing from the file is ever echoed. Every fix line points
|
|
495
|
+
// at `pi-dispatch setup github`, which mints all three values and writes the PEM 0600 in one pass.
|
|
496
|
+
if (ghSource === "app") {
|
|
497
|
+
const setupFix = "run `pi-dispatch setup github` -- it mints the App, writes these .env lines, and lands the key mode 0600";
|
|
498
|
+
const numeric = (v) => typeof v === "string" && /^\d+$/.test(v.trim());
|
|
499
|
+
for (const name of ["GITHUB_APP_ID", "GITHUB_APP_INSTALLATION_ID"]) {
|
|
500
|
+
checks.push({
|
|
501
|
+
ok: numeric(env[name]),
|
|
502
|
+
warn: true,
|
|
503
|
+
label: numeric(env[name])
|
|
504
|
+
? `${name} set (${env[name].trim()})`
|
|
505
|
+
: `GITHUB_AUTH_SOURCE=app but ${name} is ${env[name] ? `not numeric (${JSON.stringify(env[name])})` : "unset"} -- github jobs cannot mint tokens`,
|
|
506
|
+
fix: setupFix,
|
|
507
|
+
});
|
|
508
|
+
}
|
|
509
|
+
const keyPath = env.GITHUB_APP_PRIVATE_KEY_PATH;
|
|
510
|
+
if (!keyPath) {
|
|
511
|
+
checks.push({ ok: false, warn: true, label: "GITHUB_AUTH_SOURCE=app but GITHUB_APP_PRIVATE_KEY_PATH is unset -- the worker will refuse to boot", fix: setupFix });
|
|
512
|
+
} else if (!fileExists(keyPath)) {
|
|
513
|
+
checks.push({ ok: false, warn: true, label: `GITHUB_APP_PRIVATE_KEY_PATH does not exist (${keyPath})`, fix: setupFix });
|
|
514
|
+
} else {
|
|
515
|
+
checks.push({ ok: true, label: `GitHub App private key present (${keyPath})` });
|
|
516
|
+
// POSIX mode only -- on win32 stat modes are synthetic (0666-ish for everything), so a warn
|
|
517
|
+
// there would fire on every healthy deployment and teach operators to ignore it.
|
|
518
|
+
if (process.platform !== "win32") {
|
|
519
|
+
try {
|
|
520
|
+
const loose = statSync(keyPath).mode & 0o077;
|
|
521
|
+
if (loose !== 0) {
|
|
522
|
+
checks.push({
|
|
523
|
+
ok: false,
|
|
524
|
+
warn: true,
|
|
525
|
+
label: `the App private key at ${keyPath} is group/world-readable`,
|
|
526
|
+
fix: `chmod 600 ${keyPath} -- any local user can read the App's signing key right now (\`pi-dispatch setup github\` writes it 0600)`,
|
|
527
|
+
});
|
|
528
|
+
}
|
|
529
|
+
} catch {
|
|
530
|
+
// stat raced a deletion or an exotic fs: the presence line above already covered existence.
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
// First bytes only: enough to see "-----BEGIN", never the key material, and never echoed.
|
|
534
|
+
try {
|
|
535
|
+
const fd = openSync(keyPath, "r");
|
|
536
|
+
const head = Buffer.alloc(16);
|
|
537
|
+
let read = 0;
|
|
538
|
+
try {
|
|
539
|
+
read = readSync(fd, head, 0, head.length, 0);
|
|
540
|
+
} finally {
|
|
541
|
+
closeSync(fd);
|
|
542
|
+
}
|
|
543
|
+
if (!head.toString("utf8", 0, read).startsWith("-----BEGIN")) {
|
|
544
|
+
checks.push({
|
|
545
|
+
ok: false,
|
|
546
|
+
warn: true,
|
|
547
|
+
label: `the file at GITHUB_APP_PRIVATE_KEY_PATH does not look like a PEM (first line is not "-----BEGIN ...") -- contents not shown`,
|
|
548
|
+
fix: setupFix,
|
|
549
|
+
});
|
|
550
|
+
}
|
|
551
|
+
} catch {
|
|
552
|
+
checks.push({ ok: false, warn: true, label: `the App private key at ${keyPath} exists but is not readable by this user`, fix: setupFix });
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
// Preflight gh INSIDE the job image: a token that works host-side but not in-container (no egress from
|
|
558
|
+
// containers, stale image) fails jobs mid-run, not at submit. Only meaningful when docker and the image
|
|
559
|
+
// are green; otherwise it is noise on top of the failures already reported above.
|
|
560
|
+
if (dockerCode === 0 && imageCode === 0) {
|
|
561
|
+
if (ghSource === "app") {
|
|
562
|
+
checks.push({ ok: true, label: "in-image gh auth: skipped (GITHUB_AUTH_SOURCE=app mints per-job)" });
|
|
563
|
+
} else {
|
|
564
|
+
let token = "";
|
|
565
|
+
if (ghSource === "gh") {
|
|
566
|
+
const minted = await runCmdCapture(spawn, "gh", ["auth", "token"]);
|
|
567
|
+
if (minted.code === 0) token = minted.output.trim();
|
|
568
|
+
// mint failed → skip: the status check above already warned that gh auth is broken
|
|
569
|
+
} else if (ghSource === "pat") {
|
|
570
|
+
const patVar = env.GITHUB_PAT_VAR ?? "GITHUB_PAT"; // config.mjs's patVar default, read directly
|
|
571
|
+
token = (env[patVar] ?? "").trim(); // absent → skip; loadConfig fails loud at worker boot anyway
|
|
572
|
+
}
|
|
573
|
+
if (token) {
|
|
574
|
+
// Value-less `-e` flags: docker forwards GH_TOKEN/GITHUB_TOKEN from the spawn env, so the
|
|
575
|
+
// token value never enters argv (visible in `ps`) and never reaches doctor's output.
|
|
576
|
+
const probe = await runCmdCapture(
|
|
577
|
+
spawn,
|
|
578
|
+
"docker",
|
|
579
|
+
["run", "--rm", "--pull=never", "-e", "GH_TOKEN", "-e", "GITHUB_TOKEN", "--entrypoint", "gh", jobImage, "auth", "status"],
|
|
580
|
+
{ env: { ...env, GH_TOKEN: token, GITHUB_TOKEN: token } },
|
|
581
|
+
);
|
|
582
|
+
checks.push({
|
|
583
|
+
ok: probe.code === 0,
|
|
584
|
+
warn: true,
|
|
585
|
+
label:
|
|
586
|
+
probe.code === 0
|
|
587
|
+
? `gh authenticates inside the job image (${jobImage})`
|
|
588
|
+
: `gh cannot authenticate inside the job image (${jobImage})`,
|
|
589
|
+
fix: "check network egress from containers or rebuild/pull the job image -- jobs that use gh will fail mid-run",
|
|
590
|
+
});
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
}
|
|
594
|
+
|
|
595
|
+
checks.push({
|
|
596
|
+
ok: await probeValkey(valkeyUrl),
|
|
597
|
+
label: `Valkey reachable (${valkeyUrl})`,
|
|
598
|
+
fix: "docker compose -f deploy/docker-compose.yml up -d",
|
|
599
|
+
// Prompt tier, and only for a LOOPBACK url (the shipped default): starting a local container cannot
|
|
600
|
+
// make a remote VALKEY_URL reachable, so a pointed-elsewhere deployment keeps the plain fix line
|
|
601
|
+
// rather than an offer that would mask the real problem. The argv mirrors the compose file's
|
|
602
|
+
// semantics exactly (VALKEY_RUN above).
|
|
603
|
+
...(/^redis:\/\/(127\.0\.0\.1|localhost)(:6379)?\/?$/.test(valkeyUrl)
|
|
604
|
+
? {
|
|
605
|
+
fixAction: {
|
|
606
|
+
tier: "prompt",
|
|
607
|
+
describe: `docker ${VALKEY_RUN.join(" ")}`,
|
|
608
|
+
run: async ({ spawn }) =>
|
|
609
|
+
(await runCmd(spawn, "docker", VALKEY_RUN)) === 0
|
|
610
|
+
? { ok: true }
|
|
611
|
+
: { ok: false, note: "docker run failed (is a container named pi-dispatch-valkey already present? `docker start pi-dispatch-valkey`)" },
|
|
612
|
+
},
|
|
613
|
+
}
|
|
614
|
+
: {}),
|
|
615
|
+
});
|
|
616
|
+
|
|
617
|
+
const keys = PROVIDER_KEYS[provider] ?? [`${provider.toUpperCase()}_API_KEY`];
|
|
618
|
+
let keyOk = keys.some((k) => (env[k] ?? "").trim().length > 0);
|
|
619
|
+
let keyNote = "";
|
|
620
|
+
// The key may come from pi's auth.json when the env has none (ON by default; PI_AUTH_FROM_PI=0 forces
|
|
621
|
+
// env-only) — so don't falsely report it missing.
|
|
622
|
+
const authFromPi = env.PI_AUTH_FROM_PI !== "0";
|
|
623
|
+
if (!keyOk && authFromPi) {
|
|
624
|
+
const agentDir = env.PI_CODING_AGENT_DIR || join(homedir(), ".pi", "agent");
|
|
625
|
+
try {
|
|
626
|
+
const cred = JSON.parse(readFileSync(join(agentDir, "auth.json"), "utf8"))?.[provider];
|
|
627
|
+
if (cred?.type === "api_key" && cred.key) {
|
|
628
|
+
keyOk = true;
|
|
629
|
+
keyNote = " — from pi auth.json";
|
|
630
|
+
} else if (cred?.type === "oauth") {
|
|
631
|
+
keyNote = " — pi login is OAuth/subscription: not usable for an unattended service, configure an API key";
|
|
632
|
+
}
|
|
633
|
+
} catch {}
|
|
634
|
+
}
|
|
635
|
+
checks.push({
|
|
636
|
+
ok: keyOk,
|
|
637
|
+
label: `Provider key set (${provider}: ${keys.join(" or ")})${keyNote}`,
|
|
638
|
+
fix: authFromPi ? `run \`pi login\` with an API key for ${provider}, or set ${keys[0]} in .env` : `set ${keys[0]} in .env`,
|
|
639
|
+
});
|
|
640
|
+
|
|
641
|
+
|
|
642
|
+
// REQ-GLOBAL-PI-OVERLAY: read the extensions opt-out through the WORKER's own parser, so doctor reports
|
|
643
|
+
// the exact posture the worker will boot with and refuses the exact values it refuses. Checked with or
|
|
644
|
+
// without an overlay configured, because a malformed knob stops boot either way -- and a `false` an
|
|
645
|
+
// operator wrote believing it disabled their extensions is precisely the value they need told about.
|
|
646
|
+
let extensionsEnabled = true;
|
|
647
|
+
let extensionsInvalid = false;
|
|
648
|
+
try {
|
|
649
|
+
extensionsEnabled = globalExtensionsEnabled(env);
|
|
650
|
+
} catch {
|
|
651
|
+
extensionsInvalid = true;
|
|
652
|
+
checks.push({
|
|
653
|
+
ok: false,
|
|
654
|
+
label: `PI_GLOBAL_ALLOW_EXTENSIONS is ${JSON.stringify(env.PI_GLOBAL_ALLOW_EXTENSIONS)}, which is neither on nor off`,
|
|
655
|
+
fix: 'set it to exactly "0" to disable the overlay\'s extensions, or leave it unset to load them -- the worker refuses to boot on any other value',
|
|
656
|
+
});
|
|
657
|
+
}
|
|
658
|
+
|
|
659
|
+
// Global pi overlay (REQ-GLOBAL-PI-OVERLAY), only when configured. The overlay is mounted :ro into an
|
|
660
|
+
// adversarial-input container, so the load-bearing checks are that it holds NO credential.
|
|
661
|
+
const overlay = env.PI_GLOBAL_PI_DIR;
|
|
662
|
+
if (overlay) {
|
|
663
|
+
const dirOk = fileExists(overlay);
|
|
664
|
+
checks.push({ ok: dirOk, label: `Global overlay dir exists (${overlay})`, fix: "run `pi-dispatch import-pi`, or fix PI_GLOBAL_PI_DIR" });
|
|
665
|
+
if (dirOk) {
|
|
666
|
+
const overlayAuth = join(overlay, "auth.json");
|
|
667
|
+
checks.push({
|
|
668
|
+
ok: !fileExists(overlayAuth),
|
|
669
|
+
label: "Overlay is credential-free (no auth.json)",
|
|
670
|
+
fix: "delete auth.json from the overlay — the provider key belongs in env, never a mounted file",
|
|
671
|
+
// Prompt, not silent, even though deleting it is always right for the OVERLAY: the file may
|
|
672
|
+
// be the operator's only copy of a credential they meant to keep elsewhere, and doctor
|
|
673
|
+
// deleting an operator's file unasked is a line not worth crossing for one saved keypress.
|
|
674
|
+
fixAction: {
|
|
675
|
+
tier: "prompt",
|
|
676
|
+
describe: `rm ${overlayAuth}`,
|
|
677
|
+
run: async ({ rm }) => {
|
|
678
|
+
rm(overlayAuth);
|
|
679
|
+
return { ok: true };
|
|
680
|
+
},
|
|
681
|
+
},
|
|
682
|
+
});
|
|
683
|
+
const modelsPath = join(overlay, "models.json");
|
|
684
|
+
let modelsOk = true;
|
|
685
|
+
let modelsFix = "";
|
|
686
|
+
if (fileExists(modelsPath)) {
|
|
687
|
+
try {
|
|
688
|
+
const leak = findLiteralSecret(JSON.parse(readFileSync(modelsPath, "utf8")));
|
|
689
|
+
if (leak) {
|
|
690
|
+
modelsOk = false;
|
|
691
|
+
modelsFix = `literal secret at ${leak} — move it to env/auth.json or a "$VAR" reference`;
|
|
692
|
+
}
|
|
693
|
+
} catch {
|
|
694
|
+
modelsOk = false;
|
|
695
|
+
modelsFix = "overlay models.json is not valid JSON";
|
|
696
|
+
}
|
|
697
|
+
}
|
|
698
|
+
checks.push({ ok: modelsOk, label: "Overlay models.json is credential-free", fix: modelsFix });
|
|
699
|
+
// Staged extensions load unless the operator opted out, so this pair reports what WILL run, not
|
|
700
|
+
// what is switched on. The ⚠ sits on the loading case: it is the one where code the operator may
|
|
701
|
+
// have staged months ago is executing against adversarial input right now. It stays a warning and
|
|
702
|
+
// never a failure -- a vetted overlay that loads is the intended deployment, not a fault.
|
|
703
|
+
// Suppressed when the knob is malformed: the ✗ above already says the worker will not boot, and a
|
|
704
|
+
// second line guessing which way it would have resolved would be worse than silence.
|
|
705
|
+
if (fileExists(join(overlay, "extensions")) && !extensionsInvalid) {
|
|
706
|
+
if (extensionsEnabled) {
|
|
707
|
+
checks.push({
|
|
708
|
+
ok: false,
|
|
709
|
+
warn: true,
|
|
710
|
+
label: "Overlay extensions LOAD in every job (PI_GLOBAL_ALLOW_EXTENSIONS is not 0)",
|
|
711
|
+
fix: "they run code against adversarial input with open egress — vet each; set PI_GLOBAL_ALLOW_EXTENSIONS=0 in .env to disable them",
|
|
712
|
+
});
|
|
713
|
+
} else {
|
|
714
|
+
checks.push({ ok: true, label: "Overlay extensions present but disabled (PI_GLOBAL_ALLOW_EXTENSIONS=0)" });
|
|
715
|
+
}
|
|
716
|
+
}
|
|
717
|
+
|
|
718
|
+
// Staged pi packages (REQ-GLOBAL-PI-OVERLAY): pinned third-party code the operator staged with
|
|
719
|
+
// `import-pi --with-packages`, loaded by every job whose trigger did not set `run.packages: false`.
|
|
720
|
+
// Keyed on the dir the same way the extensions pair above is, so a deployment that stages none
|
|
721
|
+
// prints nothing here.
|
|
722
|
+
const packagesDir = join(overlay, PACKAGES_SUBDIR);
|
|
723
|
+
if (fileExists(packagesDir)) {
|
|
724
|
+
// The restage offer shared by the two staleness checks below (prompt tier: it fetches and
|
|
725
|
+
// runs npm on this host). A child process through the injected spawn rather than an
|
|
726
|
+
// in-process call, so import-pi's own gates run unmodified -- the literal-secret abort, the
|
|
727
|
+
// admin-extension block, the printed-names vetting -- and its output is forwarded so the
|
|
728
|
+
// operator still reads the names of exactly what will load into their job containers.
|
|
729
|
+
const restageFixAction = {
|
|
730
|
+
tier: "prompt",
|
|
731
|
+
describe: `pi-dispatch import-pi --with-packages --to ${overlay}`,
|
|
732
|
+
run: async ({ spawn, out }) => {
|
|
733
|
+
const cli = fileURLToPath(new URL("./cli.mjs", import.meta.url));
|
|
734
|
+
// npm staging can be slow, so 10 minutes rather than runCmdCapture's default 30s.
|
|
735
|
+
const res = await runCmdCapture(spawn, process.execPath, [cli, "import-pi", "--with-packages", "--to", overlay], { env, cwd, timeoutMs: 600000 });
|
|
736
|
+
if (res.output) out(res.output);
|
|
737
|
+
return { ok: res.code === 0 };
|
|
738
|
+
},
|
|
739
|
+
};
|
|
740
|
+
const manifest = readStageManifest({ globalPiDir: overlay, readFile: (p) => readFileSync(p, "utf8"), fileExists });
|
|
741
|
+
if (!manifest) {
|
|
742
|
+
checks.push({
|
|
743
|
+
ok: false,
|
|
744
|
+
label: `Staged packages manifest readable (${PACKAGES_SUBDIR}/packages.json)`,
|
|
745
|
+
fix: "re-run `pi-dispatch import-pi --with-packages` -- without the manifest nothing knows what is staged, so no package is ever loaded",
|
|
746
|
+
fixAction: restageFixAction,
|
|
747
|
+
});
|
|
748
|
+
} else {
|
|
749
|
+
// A manifest entry whose dir is gone loads nothing, and pi reports no error for a package
|
|
750
|
+
// it was never told about -- the stage is only as real as the dirs behind the names.
|
|
751
|
+
const missing = manifest.packages.filter((p) => !fileExists(join(packagesDir, p.dir))).map((p) => p.name);
|
|
752
|
+
checks.push({
|
|
753
|
+
ok: missing.length === 0,
|
|
754
|
+
label: `Staged packages present (${manifest.packages.map((p) => `${p.name}@${p.version}`).join(", ")})`,
|
|
755
|
+
fix: `staged dir missing for ${missing.join(", ")} -- re-run \`pi-dispatch import-pi --with-packages\` to restage`,
|
|
756
|
+
fixAction: restageFixAction,
|
|
757
|
+
});
|
|
758
|
+
// The admin extension's twin, and blocked for the same reason import-pi blocks that one.
|
|
759
|
+
const admin = manifest.packages.filter((p) => ADMIN_RE.test(p.name) || ADMIN_RE.test(p.dir)).map((p) => p.name);
|
|
760
|
+
if (admin.length > 0) {
|
|
761
|
+
checks.push({
|
|
762
|
+
ok: false,
|
|
763
|
+
label: `Staged package looks like the dispatch admin (${admin.join(", ")})`,
|
|
764
|
+
fix: "remove it from the overlay -- a package that can enqueue paid jobs from INSIDE a job container is a recursion vector",
|
|
765
|
+
});
|
|
766
|
+
}
|
|
767
|
+
// There is no dormant state left to report: a staged, manifested package loads into every
|
|
768
|
+
// job whose trigger did not opt out -- INCLUDING jobs no trigger file describes at all
|
|
769
|
+
// (a matched webhook, `dispatch_run`, the CLI). So "staged" IS "loading", and the honest
|
|
770
|
+
// line says so and names how much of the trigger file withholds it. Warn, never fail: this
|
|
771
|
+
// is the intended posture, and it is stated so a forgotten stage cannot read as inert.
|
|
772
|
+
// Sits inside the manifest branch because an unreadable manifest loads NOTHING -- the ✗
|
|
773
|
+
// above is that case, and claiming these load there would be the opposite of the truth.
|
|
774
|
+
checks.push({
|
|
775
|
+
ok: false,
|
|
776
|
+
warn: true,
|
|
777
|
+
label: `Staged packages LOAD in every job (${optingOut} trigger(s) opt out with run.packages: false)`,
|
|
778
|
+
fix: "they run third-party code against adversarial input with open egress -- vet each, keep every version exactly pinned, and set run.packages: false on any trigger that must not load them",
|
|
779
|
+
});
|
|
780
|
+
}
|
|
781
|
+
} else if (requiring > 0) {
|
|
782
|
+
// The silently-package-less job, and the one check the flip does NOT touch: `run.packages:
|
|
783
|
+
// true` is no longer an arming switch, but it is still an operator asserting "this flow needs
|
|
784
|
+
// the staged packages". Nothing staged means PI_PACKAGES is never emitted and the flow runs
|
|
785
|
+
// WITHOUT the tools it was written for -- on a clean exit 0.
|
|
786
|
+
checks.push({
|
|
787
|
+
ok: false,
|
|
788
|
+
label: `${requiring} trigger(s) require staged packages (run.packages: true) but nothing is staged in ${packagesDir}`,
|
|
789
|
+
fix: "declare them in pi-packages.json and run `pi-dispatch import-pi --with-packages`, or drop run.packages from the trigger -- otherwise the flow runs without its tools and still exits 0",
|
|
790
|
+
});
|
|
791
|
+
}
|
|
792
|
+
}
|
|
793
|
+
} else if (requiring > 0) {
|
|
794
|
+
// Same silent failure one level up: the staged set lives INSIDE the overlay, so no overlay means the
|
|
795
|
+
// packages are not mounted at all, however carefully they were staged.
|
|
796
|
+
checks.push({
|
|
797
|
+
ok: false,
|
|
798
|
+
label: `${requiring} trigger(s) require staged packages (run.packages: true) but PI_GLOBAL_PI_DIR is unset`,
|
|
799
|
+
fix: "set PI_GLOBAL_PI_DIR -- staged packages live inside the overlay and are mounted with it, so with no overlay there is nothing to load",
|
|
800
|
+
});
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
// REQ-RESUMABLE-SESSION. Only reported when a trigger actually asked for it: a deployment that does
|
|
804
|
+
// not use resume should not be told about a directory it has no reason to create.
|
|
805
|
+
if (resuming > 0) {
|
|
806
|
+
const sessionsDir = env.PI_SESSIONS_DIR;
|
|
807
|
+
if (!sessionsDir) {
|
|
808
|
+
checks.push({
|
|
809
|
+
ok: false,
|
|
810
|
+
label: `${resuming} trigger(s) set run.resume but PI_SESSIONS_DIR is unset`,
|
|
811
|
+
fix: "set PI_SESSIONS_DIR to a private directory (mode 0700, OUTSIDE any git repo) -- these jobs refuse pre-spend until you do, deliberately, rather than running unpersisted and looking like they worked",
|
|
812
|
+
});
|
|
813
|
+
} else {
|
|
814
|
+
const exists = fileExists(sessionsDir);
|
|
815
|
+
checks.push({
|
|
816
|
+
ok: exists,
|
|
817
|
+
label: `Session store ${exists ? "exists" : "does not exist"} (${sessionsDir})`,
|
|
818
|
+
fix: `create it: mkdir -p ${sessionsDir} && chmod 700 ${sessionsDir}`,
|
|
819
|
+
// Silent tier: setting PI_SESSIONS_DIR WAS the decision, and it has already been made -- the
|
|
820
|
+
// mkdir is the mechanical remainder, creates only the path the env var names, and 0700 is
|
|
821
|
+
// the mode the fix line already prescribes (transcripts are PII-bearing, host-only).
|
|
822
|
+
fixAction: {
|
|
823
|
+
tier: "silent",
|
|
824
|
+
describe: `mkdir -p ${sessionsDir} && chmod 700 ${sessionsDir}`,
|
|
825
|
+
run: async ({ mkdir, chmod }) => {
|
|
826
|
+
mkdir(sessionsDir, { recursive: true });
|
|
827
|
+
chmod(sessionsDir, 0o700);
|
|
828
|
+
return { ok: true, note: "mode 0700" };
|
|
829
|
+
},
|
|
830
|
+
},
|
|
831
|
+
});
|
|
832
|
+
// Not a failure -- a warning, because it is a disclosure the operator may have accepted
|
|
833
|
+
// knowingly. A transcript holds tool output, file contents and the agent's own reasoning, which
|
|
834
|
+
// is strictly more than logs/<jobId>.log holds, and that one is opt-in for this reason.
|
|
835
|
+
checks.push({
|
|
836
|
+
ok: true,
|
|
837
|
+
warn: true,
|
|
838
|
+
label: `${resuming} trigger(s) persist agent transcripts to ${sessionsDir} -- PII-bearing, host-only, never committed`,
|
|
839
|
+
fix: "confirm it is outside every git repo and on a disk you would put issue text on; PI_SESSIONS_TTL_DAYS bounds how long a transcript stays resumable",
|
|
840
|
+
});
|
|
841
|
+
}
|
|
842
|
+
}
|
|
843
|
+
|
|
844
|
+
// REQ-REPLICA-RUNS. A warning, never a failure -- replicas are an opt-in an operator chose in a reviewed
|
|
845
|
+
// file, and the harness is doing exactly what was asked. What is worth saying is the arithmetic: each
|
|
846
|
+
// replica reserves its OWN budget slot before its own tokens (CONST-BUDGET-BEFORE-TOKENS), so a delivery
|
|
847
|
+
// on a `replicas: 2` trigger consumes two, and the daily cap divides accordingly. Only reported when a
|
|
848
|
+
// trigger actually asked for it.
|
|
849
|
+
if (replicating > 0) {
|
|
850
|
+
checks.push({
|
|
851
|
+
ok: true,
|
|
852
|
+
warn: true,
|
|
853
|
+
// Both facts live in the LABEL rather than the fix, because an `ok: true` check never prints its
|
|
854
|
+
// fix line (the loop below) -- and the concurrency half is the one an operator most often has
|
|
855
|
+
// wrong: replicas above PI_CONCURRENCY queue instead of racing, which looks like the feature
|
|
856
|
+
// silently not working.
|
|
857
|
+
label: `${replicating} trigger(s) set run.replicas -- one delivery reserves one budget slot PER replica; PI_CONCURRENCY bounds how many actually race`,
|
|
858
|
+
fix: "confirm the daily/weekly/monthly caps account for the multiplier, and that PI_CONCURRENCY is at least the largest run.replicas",
|
|
859
|
+
});
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
// REQ-RESURRECTABLE-SANDBOX. A warning, never a failure: retention is a convenience, and the only thing
|
|
863
|
+
// worth surfacing is that finished runs' directories -- a repository clone plus the run's prompt.md and
|
|
864
|
+
// event.json, so issue text -- are sitting on disk, and how many. An operator who never opens a sandbox
|
|
865
|
+
// should still know they are being kept.
|
|
866
|
+
{
|
|
867
|
+
const retentionHours = nonNegativeEnvInt(env.PI_SANDBOX_RETENTION_HOURS, 24);
|
|
868
|
+
const sandboxDir = env.PI_SANDBOX_DIR || defaultSandboxDir(env);
|
|
869
|
+
if (retentionHours === 0) {
|
|
870
|
+
checks.push({ ok: true, label: "Workspace retention off (PI_SANDBOX_RETENTION_HOURS=0) — finished runs are deleted, none are re-openable" });
|
|
871
|
+
} else {
|
|
872
|
+
const kept = countRetained(sandboxDir, fileExists);
|
|
873
|
+
checks.push({
|
|
874
|
+
ok: true,
|
|
875
|
+
warn: kept.count > 0,
|
|
876
|
+
label: `${kept.count} retained workspace(s) in ${sandboxDir}, swept after ${retentionHours}h — re-open one with \`pi-dispatch sandbox <jobId>\``,
|
|
877
|
+
fix: "each holds the run's clone plus its prompt.md/event.json (issue text); PI_SANDBOX_RETENTION_HOURS=0 turns retention off entirely",
|
|
878
|
+
});
|
|
879
|
+
}
|
|
880
|
+
}
|
|
881
|
+
|
|
882
|
+
return checks;
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
/**
|
|
886
|
+
* How many retained workspaces are sitting in the retention root.
|
|
887
|
+
*
|
|
888
|
+
* Reads its OWN env rather than loadConfig, like every other doctor check (`doctor.mjs` header): a broken
|
|
889
|
+
* GitHub auth must not stop the operator finding out how much disk this is using. Never throws -- an
|
|
890
|
+
* unreadable or absent root reports zero, which is the honest answer to "how many can I open".
|
|
891
|
+
*/
|
|
892
|
+
function countRetained(sandboxDir, fileExists) {
|
|
893
|
+
if (!fileExists(sandboxDir)) return { count: 0 };
|
|
894
|
+
try {
|
|
895
|
+
return { count: readdirSync(sandboxDir).length };
|
|
896
|
+
} catch {
|
|
897
|
+
return { count: 0 };
|
|
898
|
+
}
|
|
899
|
+
}
|
|
900
|
+
|
|
901
|
+
/** PI_SANDBOX_RETENTION_HOURS, parsed the same permissive way the admin's own env reads are. */
|
|
902
|
+
function nonNegativeEnvInt(raw, fallback) {
|
|
903
|
+
if (raw === undefined || raw === "") return fallback;
|
|
904
|
+
const n = Number.parseInt(raw, 10);
|
|
905
|
+
return Number.isInteger(n) && n >= 0 && String(n) === String(raw).trim() ? n : fallback;
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
function nodeCheck(version) {
|
|
909
|
+
const [maj, min] = version.split(".").map((n) => Number.parseInt(n, 10));
|
|
910
|
+
const ok = maj > NODE_FLOOR[0] || (maj === NODE_FLOOR[0] && min >= NODE_FLOOR[1]);
|
|
911
|
+
return {
|
|
912
|
+
ok,
|
|
913
|
+
label: `Node ≥ ${NODE_FLOOR[0]}.${NODE_FLOOR[1]} (have ${version})`,
|
|
914
|
+
fix: `upgrade Node to ${NODE_FLOOR[0]}.${NODE_FLOOR[1]} or newer`,
|
|
915
|
+
};
|
|
916
|
+
}
|
|
917
|
+
|
|
918
|
+
/**
|
|
919
|
+
* What does the trigger file say about the staged packages (INT-TRIGGERS-FILE-CONTRACT)?
|
|
920
|
+
*
|
|
921
|
+
* `images` is the sorted set of distinct `run.image` values across the file (issue #41), so doctor can check
|
|
922
|
+
* that every image a trigger names is actually on this host -- with `--pull=never` nothing will fetch one at
|
|
923
|
+
* job time, so this is the only warning that arrives BEFORE the trigger fires at 03:00.
|
|
924
|
+
*
|
|
925
|
+
* `optingOut` counts `run.packages: false` -- the only thing that now withholds the staged set from a job.
|
|
926
|
+
* `requiring` counts explicit `run.packages: true`, which arms nothing any more but is still an operator
|
|
927
|
+
* asserting "this flow needs those packages"; that assertion is what makes an empty stage a hard failure.
|
|
928
|
+
*
|
|
929
|
+
* `repositories` is the sorted set of distinct `run.repository` values on github-kind triggers, feeding the
|
|
930
|
+
* branch-protection preflight (issue #80). Note the shared schema currently ADMITS `run.repository` only on
|
|
931
|
+
* azure label/comment triggers (triggers.mjs, validateRepository), so this set is empty today for every
|
|
932
|
+
* valid file -- collected here anyway, rather than hard-coded empty, so the preflight lights up the day the
|
|
933
|
+
* schema grows the field for github instead of silently never running.
|
|
934
|
+
*
|
|
935
|
+
* Parsed with the SHARED `parseTriggers`, so doctor counts exactly the entries the worker and receiver will
|
|
936
|
+
* act on -- a truthy `"true"` string is rejected there and therefore never counted here.
|
|
937
|
+
*
|
|
938
|
+
* Swallows ANY error to zeroes -- a missing, unreadable, or malformed triggers file already fails LOUD at
|
|
939
|
+
* worker boot (config.mjs, schedules.mjs), so re-reporting the parse failure here would only bury doctor's
|
|
940
|
+
* own findings under a second copy of a diagnosis the operator already gets.
|
|
941
|
+
*/
|
|
942
|
+
function readTriggerFacts(env, fileExists, cwd) {
|
|
943
|
+
const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, images: [], forges: [], repositories: [] };
|
|
944
|
+
try {
|
|
945
|
+
// Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
|
|
946
|
+
// (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
|
|
947
|
+
// the receiver will not boot. An absent file still means "no triggers at all", exactly as before.
|
|
948
|
+
const path = env.PI_TRIGGERS_FILE ?? join(cwd, "triggers.json");
|
|
949
|
+
if (!fileExists(path)) return none;
|
|
950
|
+
const triggers = parseTriggers(readFileSync(path, "utf8"), path);
|
|
951
|
+
return {
|
|
952
|
+
requiring: triggers.filter((t) => t.run.packages === true).length,
|
|
953
|
+
resuming: triggers.filter((t) => t.run.resume === true).length,
|
|
954
|
+
// REQ-REPLICA-RUNS. `> 1` rather than `!== undefined` because the loader already refuses anything
|
|
955
|
+
// else -- this counts triggers that will actually multiply spend, which is the only reason to say so.
|
|
956
|
+
replicating: triggers.filter((t) => t.run.replicas > 1).length,
|
|
957
|
+
optingOut: triggers.filter((t) => t.run.packages === false).length,
|
|
958
|
+
images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
|
|
959
|
+
// The forges this file actually needs credentials for. Read from the triggers rather than from
|
|
960
|
+
// the env, so the check answers "is what you configured enough for what you wrote" instead of
|
|
961
|
+
// "did you set some variables".
|
|
962
|
+
//
|
|
963
|
+
// `isForgeKind` rather than a written-out pair: this whole function is wrapped in `catch { return
|
|
964
|
+
// none }`, so a forge missing from a hand-written filter would not merely be unchecked -- doctor
|
|
965
|
+
// would report all-green and never mention that the credential it needs was never looked for.
|
|
966
|
+
forges: [...new Set(triggers.map((t) => t.run.kind).filter(isForgeKind))].sort(),
|
|
967
|
+
repositories: [...new Set(triggers.filter((t) => t.run.kind === "github" && typeof t.run.repository === "string").map((t) => t.run.repository))].sort(),
|
|
968
|
+
};
|
|
969
|
+
} catch {
|
|
970
|
+
return none;
|
|
971
|
+
}
|
|
972
|
+
}
|
|
973
|
+
|
|
974
|
+
/**
|
|
975
|
+
* READ-ONLY branch-protection preflight for the github repos the triggers file names (issue #80,
|
|
976
|
+
* REQ-BRANCH-PROTECTION-PRECONDITION). Two `gh api` GETs per repo -- resolve the default branch, then ask
|
|
977
|
+
* the protection endpoint -- and never anything else: doctor reports repo settings, it does not change
|
|
978
|
+
* them, so the fix line SHOWS the settings page rather than running a PUT.
|
|
979
|
+
*
|
|
980
|
+
* A non-zero exit on the protection endpoint deliberately conflates GitHub's determinate 404 ("no
|
|
981
|
+
* protection") with transient errors. The worker's own gate does the 404-vs-retryable split, because there
|
|
982
|
+
* a false "unprotected" would disarm the never-merge backstop (github-host.mjs, issue #61) -- here every
|
|
983
|
+
* answer is an advisory warn, and a warn that occasionally fires on a flaky API is acceptable where a
|
|
984
|
+
* false ✓ would not be.
|
|
985
|
+
*
|
|
986
|
+
* Exported rather than folded into runDoctor: the shared schema admits `run.repository` only on azure
|
|
987
|
+
* triggers today (see readTriggerFacts), so no valid triggers file can reach this loop through runDoctor
|
|
988
|
+
* yet -- tests exercise it directly, and the runDoctor wiring is already live for the day the schema
|
|
989
|
+
* grows the field for github. Returns check objects in runDoctor's `{ok, warn, label, fix}` shape.
|
|
990
|
+
*/
|
|
991
|
+
export async function githubProtectionPreflight(spawn, repositories) {
|
|
992
|
+
const checks = [];
|
|
993
|
+
// gh availability first, mirroring the GITHUB_AUTH_SOURCE=gh handling in runDoctor: one warn covers
|
|
994
|
+
// every repo, and the loop is skipped rather than producing one confusing failure line per repo.
|
|
995
|
+
const status = await runCmdCapture(spawn, "gh", ["auth", "status"]);
|
|
996
|
+
if (status.code !== 0) {
|
|
997
|
+
checks.push({
|
|
998
|
+
ok: false,
|
|
999
|
+
warn: true,
|
|
1000
|
+
label: `branch-protection preflight skipped: gh is unavailable or not logged in (${repositories.length} github repo(s) named in triggers.json)`,
|
|
1001
|
+
fix: "install gh and run `gh auth login` -- the preflight is a read-only `gh api` per repo; the worker still enforces REQ-BRANCH-PROTECTION-PRECONDITION at job time either way",
|
|
1002
|
+
});
|
|
1003
|
+
return checks;
|
|
1004
|
+
}
|
|
1005
|
+
// Bounded so a large trigger file cannot turn doctor into a network crawl: two API round-trips per
|
|
1006
|
+
// repo, five repos. The rest are not silently dropped -- the cap line says so, and job time enforces.
|
|
1007
|
+
const capped = repositories.slice(0, 5);
|
|
1008
|
+
if (repositories.length > capped.length) {
|
|
1009
|
+
checks.push({
|
|
1010
|
+
ok: true,
|
|
1011
|
+
label: `branch-protection preflight capped at ${capped.length} of ${repositories.length} repos -- the rest are still enforced per job before any spend`,
|
|
1012
|
+
});
|
|
1013
|
+
}
|
|
1014
|
+
for (const repo of capped) {
|
|
1015
|
+
const branch = await runCmdCapture(spawn, "gh", ["api", `repos/${repo}`, "--jq", ".default_branch"]);
|
|
1016
|
+
const name = branch.code === 0 ? branch.output.trim() : "";
|
|
1017
|
+
if (!name) {
|
|
1018
|
+
checks.push({
|
|
1019
|
+
ok: false,
|
|
1020
|
+
warn: true,
|
|
1021
|
+
label: `could not resolve the default branch of ${repo} -- branch protection not preflighted`,
|
|
1022
|
+
fix: "check the run.repository value and this gh login's access to it; the worker still refuses an unprotected repo at job time",
|
|
1023
|
+
});
|
|
1024
|
+
continue;
|
|
1025
|
+
}
|
|
1026
|
+
const code = await runCmd(spawn, "gh", ["api", `repos/${repo}/branches/${name}/protection`]);
|
|
1027
|
+
checks.push(
|
|
1028
|
+
code === 0
|
|
1029
|
+
? { ok: true, label: `default branch of ${repo} is protected (${name})` }
|
|
1030
|
+
: {
|
|
1031
|
+
ok: false,
|
|
1032
|
+
warn: true,
|
|
1033
|
+
label: `default branch of ${repo} is not protected -- the worker refuses forge jobs on unprotected repos before any spend (REQ-BRANCH-PROTECTION-PRECONDITION)`,
|
|
1034
|
+
fix: `protect ${name} at https://github.com/${repo}/settings/branches (see SECURITY.md) -- a read-only preflight, doctor never changes repo settings`,
|
|
1035
|
+
},
|
|
1036
|
+
);
|
|
1037
|
+
}
|
|
1038
|
+
return checks;
|
|
1039
|
+
}
|
|
1040
|
+
|
|
1041
|
+
/** Resolve a spawned command's exit code; null means it could not be launched (e.g. not on PATH). */
|
|
1042
|
+
function runCmd(spawn, cmd, args) {
|
|
1043
|
+
return new Promise((resolve) => {
|
|
1044
|
+
let child;
|
|
1045
|
+
try {
|
|
1046
|
+
child = spawn(cmd, args, { stdio: "ignore" });
|
|
1047
|
+
} catch {
|
|
1048
|
+
resolve(null);
|
|
1049
|
+
return;
|
|
1050
|
+
}
|
|
1051
|
+
child.on("error", () => resolve(null)); // ENOENT etc. — the binary is not available
|
|
1052
|
+
child.on("close", (code) => resolve(code));
|
|
1053
|
+
});
|
|
1054
|
+
}
|
|
1055
|
+
|
|
1056
|
+
/**
|
|
1057
|
+
* Like runCmd but collects stdout+stderr into one combined string — gh moves its human output between
|
|
1058
|
+
* the two across versions, so callers get both. Resolves `{code, output}`; `code: null` when the command
|
|
1059
|
+
* could not be launched or overran the timeout (default 30s, so a hung docker daemon cannot stall doctor).
|
|
1060
|
+
* `opts.env` is passed through to the spawn so secrets can travel via env instead of argv; `opts.cwd`
|
|
1061
|
+
* likewise, for the child-process fixActions that must run where doctor's own cwd seam points.
|
|
1062
|
+
*/
|
|
1063
|
+
function runCmdCapture(spawn, cmd, args, opts = {}) {
|
|
1064
|
+
const { timeoutMs = 30000 } = opts;
|
|
1065
|
+
return new Promise((resolve) => {
|
|
1066
|
+
let child;
|
|
1067
|
+
try {
|
|
1068
|
+
child = spawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"], ...(opts.env ? { env: opts.env } : {}), ...(opts.cwd ? { cwd: opts.cwd } : {}) });
|
|
1069
|
+
} catch {
|
|
1070
|
+
resolve({ code: null, output: "" });
|
|
1071
|
+
return;
|
|
1072
|
+
}
|
|
1073
|
+
let output = "";
|
|
1074
|
+
let done = false;
|
|
1075
|
+
const finish = (code) => {
|
|
1076
|
+
if (done) return;
|
|
1077
|
+
done = true;
|
|
1078
|
+
clearTimeout(timer);
|
|
1079
|
+
resolve({ code, output });
|
|
1080
|
+
};
|
|
1081
|
+
const timer = setTimeout(() => {
|
|
1082
|
+
try {
|
|
1083
|
+
child.kill();
|
|
1084
|
+
} catch {}
|
|
1085
|
+
finish(null);
|
|
1086
|
+
}, timeoutMs);
|
|
1087
|
+
child.stdout?.on("data", (d) => (output += d));
|
|
1088
|
+
child.stderr?.on("data", (d) => (output += d));
|
|
1089
|
+
child.on("error", () => finish(null)); // ENOENT etc. — the binary is not available
|
|
1090
|
+
child.on("close", (code) => finish(code));
|
|
1091
|
+
});
|
|
1092
|
+
}
|
|
1093
|
+
|
|
1094
|
+
/**
|
|
1095
|
+
* Pull the scope list out of `gh auth status` output. The line reads like
|
|
1096
|
+
* ` - Token scopes: 'gist', 'read:org', 'repo', 'workflow'` (older gh omits the quotes). Returns null
|
|
1097
|
+
* when the line is absent — fine-grained tokens report no classic scopes at all.
|
|
1098
|
+
*/
|
|
1099
|
+
function parseGhTokenScopes(output) {
|
|
1100
|
+
const m = output.match(/Token scopes:\s*(.+)/);
|
|
1101
|
+
if (!m) return null;
|
|
1102
|
+
return m[1]
|
|
1103
|
+
.split(",")
|
|
1104
|
+
.map((s) => s.trim().replace(/^'(.*)'$/, "$1"))
|
|
1105
|
+
.filter((s) => s.length > 0);
|
|
1106
|
+
}
|
|
1107
|
+
|
|
1108
|
+
/**
|
|
1109
|
+
* Reachability probe with a raw, fail-fast ioredis client. `lazyConnect` holds the connect until the
|
|
1110
|
+
* error handler is attached, so a down Valkey is reported as one ✗ line — not the ioredis stack traces
|
|
1111
|
+
* a BullMQ Queue's internal client would dump. Reuses `parseConnection`'s fail-fast options (cli.mjs:88).
|
|
1112
|
+
*/
|
|
1113
|
+
async function defaultProbeValkey(url) {
|
|
1114
|
+
const { Redis } = await import("ioredis");
|
|
1115
|
+
const { parseConnection } = await import("./connection.mjs");
|
|
1116
|
+
const client = new Redis({ ...parseConnection(url, { failFast: true }), lazyConnect: true });
|
|
1117
|
+
client.on("error", () => {}); // swallow connect errors + retries; reachability is the ✓/✗, not a trace
|
|
1118
|
+
try {
|
|
1119
|
+
await client.connect();
|
|
1120
|
+
await client.ping();
|
|
1121
|
+
return true;
|
|
1122
|
+
} catch {
|
|
1123
|
+
return false;
|
|
1124
|
+
} finally {
|
|
1125
|
+
client.disconnect();
|
|
1126
|
+
}
|
|
1127
|
+
}
|