@edgehero/pi-dispatch 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/doctor.mjs CHANGED
@@ -46,7 +46,7 @@
46
46
  */
47
47
  import { chmodSync, closeSync, existsSync, lstatSync, mkdirSync, mkdtempSync, openSync, readdirSync, readFileSync, readSync, rmSync, statSync } from "node:fs";
48
48
  import { homedir, tmpdir } from "node:os";
49
- import { join } from "node:path";
49
+ import { dirname, join } from "node:path";
50
50
  import { fileURLToPath } from "node:url";
51
51
  import { spawn as nodeSpawn } from "node:child_process";
52
52
  import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
@@ -56,6 +56,8 @@ import { agentDirFrom, readHostPi } from "./host-pi.mjs";
56
56
  import { PACKAGES_SUBDIR, readStagedSkills, readStageManifest } from "./packages.mjs";
57
57
  import { copySkillTree } from "./copy-tree.mjs";
58
58
  import { SKILL_NAME_RE } from "./flow-gate.mjs";
59
+ import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
60
+ import { installedUnitPaths, readUnitSeam } from "./service.mjs";
59
61
  import { parseTriggers } from "./triggers.mjs";
60
62
 
61
63
  const NODE_FLOOR = [22, 19]; // pi's engine floor (22.19.0)
@@ -100,8 +102,13 @@ export async function runDoctor(env = process.env, deps = {}) {
100
102
  // default is a real path in the developer's home directory and the host comparison may spawn their
101
103
  // package manager -- neither belongs in a unit test, and "no network, no Docker" is the same rule.
102
104
  agentDir = agentDirFrom(env),
105
+ // Where the service manager's units live, and which formats to read them in (issue #216). Seams
106
+ // rather than bare process.platform/homedir() because the --env-setup check has to be exercised
107
+ // for all three unit formats, and only one of them exists on whichever host runs the suite.
108
+ platform = process.platform,
109
+ home = homedir(),
103
110
  } = deps;
104
- const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir };
111
+ const seams = { cwd, out, spawn, probeValkey, fileExists, nodeVersion, mkdir, chmod, rm, agentDir, platform, home };
105
112
 
106
113
  let checks = await collectChecks(env, seams);
107
114
  let failed = render(checks, out);
@@ -203,7 +210,7 @@ export async function defaultPromptFn(question, { input = process.stdin, output
203
210
  * a comment.
204
211
  */
205
212
  export async function collectChecks(env, seams) {
206
- const { cwd, spawn, probeValkey, fileExists, nodeVersion, agentDir = agentDirFrom(env) } = seams;
213
+ const { cwd, spawn, probeValkey, fileExists, nodeVersion, platform, agentDir = agentDirFrom(env) } = seams;
207
214
 
208
215
  const jobImage = env.PI_JOB_IMAGE ?? "pi-job:latest";
209
216
  const valkeyUrl = env.VALKEY_URL ?? "redis://127.0.0.1:6379";
@@ -241,6 +248,10 @@ export async function collectChecks(env, seams) {
241
248
  },
242
249
  });
243
250
 
251
+ // Right after `.env present`, because it answers the same question that check raises: where DOES this
252
+ // deployment's environment come from. [] unless a seam is configured (issue #216).
253
+ checks.push(...(await envSetupChecks(env, seams)));
254
+
244
255
  const dockerCode = await runCmd(spawn, "docker", ["info"]);
245
256
  checks.push({
246
257
  ok: dockerCode === 0,
@@ -252,7 +263,17 @@ export async function collectChecks(env, seams) {
252
263
  // image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
253
264
  // `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
254
265
  // run.packages: true, which arms nothing any more but is still an operator statement of intent.
255
- const { requiring, optingOut, resuming, replicating, instructing, commands, images, skillsDirs, forges, repositories, flows } = readTriggerFacts(env, fileExists, cwd);
266
+ const { requiring, optingOut, resuming, replicating, instructing, commands, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
267
+ // FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
268
+ // zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
269
+ // file unconditionally and refuses to start without it, which is the consequence worth naming.
270
+ if (parseError) {
271
+ checks.push({
272
+ ok: false,
273
+ label: `triggers file does not parse -- the receiver will refuse to start: ${parseError}`,
274
+ fix: `fix ${triggersFilePath} so it loads (the message above names the entry and the reason), then re-run doctor -- every trigger-derived check below is skipped until it parses`,
275
+ });
276
+ }
256
277
 
257
278
  // Only meaningful if docker itself responds; otherwise the image check is noise on top of a down daemon.
258
279
  const imageCode = dockerCode === 0 ? await runCmd(spawn, "docker", ["image", "inspect", jobImage]) : null;
@@ -439,6 +460,11 @@ export async function collectChecks(env, seams) {
439
460
  }
440
461
  }
441
462
 
463
+ // REQ-EGRESS-ALLOWLIST (issue #202). [] when PI_EGRESS=0, so a deployment that declined it gets
464
+ // byte-identical output. Gated on docker and the image, because two of these checks run a container and
465
+ // the rest are noise on top of a down daemon.
466
+ checks.push(...(await egressChecks(env, seams, { dockerCode, imageCode, jobImage })));
467
+
442
468
  // The receiver itself, when the triggers file names ANY forge (issue #80). Only forge deliveries need
443
469
  // the receiver at all, so a cron/local-only deployment gets no receiver noise here. WARNS rather than
444
470
  // fails, same doctrine as the gitlab block below: a deployment can legitimately be mid-setup (or run
@@ -644,16 +670,43 @@ export async function collectChecks(env, seams) {
644
670
  fix: setupFix,
645
671
  });
646
672
  }
673
+ // Two ways to supply the key, exactly one of them at a time (issue #208): a path to a file, or the
674
+ // PEM itself in GITHUB_APP_PRIVATE_KEY for a deployment whose environment comes from a secrets
675
+ // manager. The hygiene pass below belongs to the PATH variant -- there is no mode to check and no
676
+ // file to stat on a value that only ever exists in this process's environment, and whoever supplies
677
+ // that environment owns its hygiene. What survives for both is the shape sniff, and the rule that
678
+ // nothing from the key reaches output.
647
679
  const keyPath = env.GITHUB_APP_PRIVATE_KEY_PATH;
648
- if (!keyPath) {
649
- checks.push({ ok: false, warn: true, label: "GITHUB_AUTH_SOURCE=app but GITHUB_APP_PRIVATE_KEY_PATH is unset -- the worker will refuse to boot", fix: setupFix });
680
+ const inlineKey = (env.GITHUB_APP_PRIVATE_KEY ?? "").trim();
681
+ if (inlineKey !== "" && keyPath) {
682
+ checks.push({
683
+ ok: false,
684
+ warn: true,
685
+ label: "GITHUB_APP_PRIVATE_KEY and GITHUB_APP_PRIVATE_KEY_PATH are both set -- the worker will refuse to boot",
686
+ fix: "unset one of them: the inline value for a deployment fed by a secrets manager, the path for a key on disk (docs/secrets.md)",
687
+ });
688
+ } else if (inlineKey !== "") {
689
+ // A flattened key still starts with the header -- the `\n` escapes come after it -- so one sniff
690
+ // covers both accepted forms.
691
+ if (inlineKey.startsWith("-----BEGIN")) {
692
+ checks.push({ ok: true, label: "GitHub App private key supplied inline (GITHUB_APP_PRIVATE_KEY)" });
693
+ } else {
694
+ checks.push({
695
+ ok: false,
696
+ warn: true,
697
+ label: `GITHUB_APP_PRIVATE_KEY does not look like a PEM (does not begin "-----BEGIN ...") -- the worker will refuse to boot; contents not shown`,
698
+ fix: "check for a truncated paste, or point GITHUB_APP_PRIVATE_KEY_PATH at the key file instead (docs/secrets.md)",
699
+ });
700
+ }
701
+ } else if (!keyPath) {
702
+ checks.push({ ok: false, warn: true, label: "GITHUB_AUTH_SOURCE=app but neither GITHUB_APP_PRIVATE_KEY_PATH nor GITHUB_APP_PRIVATE_KEY is set -- the worker will refuse to boot", fix: setupFix });
650
703
  } else if (!fileExists(keyPath)) {
651
704
  checks.push({ ok: false, warn: true, label: `GITHUB_APP_PRIVATE_KEY_PATH does not exist (${keyPath})`, fix: setupFix });
652
705
  } else {
653
706
  checks.push({ ok: true, label: `GitHub App private key present (${keyPath})` });
654
707
  // POSIX mode only -- on win32 stat modes are synthetic (0666-ish for everything), so a warn
655
708
  // there would fire on every healthy deployment and teach operators to ignore it.
656
- if (process.platform !== "win32") {
709
+ if (platform !== "win32") {
657
710
  try {
658
711
  const loose = statSync(keyPath).mode & 0o077;
659
712
  if (loose !== 0) {
@@ -689,6 +742,26 @@ export async function collectChecks(env, seams) {
689
742
  } catch {
690
743
  checks.push({ ok: false, warn: true, label: `the App private key at ${keyPath} exists but is not readable by this user`, fix: setupFix });
691
744
  }
745
+ // Mode 0600 protects the key from other users on this host; it does nothing once the file is in
746
+ // a commit. `setup github` writes the key into the DEPLOYMENT FOLDER, and a deployment folder is
747
+ // very often a checkout -- so the last thing between an App signing key and a public repository
748
+ // can be one `git add -A`. This repo's own .gitignore covers *.pem; the operator's may not, and
749
+ // a key they renamed or brought themselves is the same accident.
750
+ //
751
+ // Exit 1 is the ONLY case that warns: git says "this is a work tree, and that path is not
752
+ // ignored". 0 means covered, 128 means no work tree at all, and null means git could not be
753
+ // launched. Every one of those is silence, because a check nobody can silence must never cry
754
+ // wolf -- the cost of a missed warning here is one operator reading the doc, and the cost of a
755
+ // false one is every operator learning to scroll past doctor.
756
+ const ignoreCode = await runCmd(spawn, "git", [...GIT_READ_FLAGS, "-C", dirname(keyPath), "check-ignore", "-q", keyPath]);
757
+ if (ignoreCode === 1) {
758
+ checks.push({
759
+ ok: false,
760
+ warn: true,
761
+ label: `the App private key at ${keyPath} is inside a git work tree that does not ignore it`,
762
+ fix: `move it outside that repo, or ignore it there (\`*.pem\`) -- one \`git add -A\` commits the App's signing key, which can mint a token for every repository the App is installed on, and mode 0600 does not survive a commit`,
763
+ });
764
+ }
692
765
  }
693
766
  }
694
767
 
@@ -1210,9 +1283,11 @@ function nodeCheck(version) {
1210
1283
  * Parsed with the SHARED `parseTriggers`, so doctor counts exactly the entries the worker and receiver will
1211
1284
  * act on -- a truthy `"true"` string is rejected there and therefore never counted here.
1212
1285
  *
1213
- * Swallows ANY error to zeroes -- a missing, unreadable, or malformed triggers file already fails LOUD at
1214
- * worker boot (config.mjs, schedules.mjs), so re-reporting the parse failure here would only bury doctor's
1215
- * own findings under a second copy of a diagnosis the operator already gets.
1286
+ * A missing file still reads as zeroes and says nothing: that is an ordinary cron-less deployment. A file
1287
+ * that EXISTS and does not parse is reported instead, with the reason. The old justification here -- that
1288
+ * such a file "already fails LOUD at worker boot" -- was false for the deployment that needs doctor most:
1289
+ * the worker reads this file only when PI_TRIGGERS_FILE is set, so on a receiver-only host nothing else
1290
+ * says a word, while the zeroes quietly disarm every forge, image and flow check below.
1216
1291
  */
1217
1292
  /** lstat, so a symlinked skillsDir is judged on its own inode -- copy-tree.mjs's rule, restated. */
1218
1293
  function dirExists(dir) {
@@ -1307,7 +1382,7 @@ async function repoFlowAtHead(spawn, folder, flow) {
1307
1382
  }
1308
1383
 
1309
1384
  function readTriggerFacts(env, fileExists, cwd) {
1310
- const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, images: [], skillsDirs: [], forges: [], repositories: [], flows: [] };
1385
+ const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
1311
1386
  try {
1312
1387
  // Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
1313
1388
  // (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
@@ -1359,9 +1434,290 @@ function readTriggerFacts(env, fileExists, cwd) {
1359
1434
  }))
1360
1435
  .filter((f) => typeof f.flow === "string"),
1361
1436
  };
1437
+ } catch (e) {
1438
+ // REPORTED, not swallowed. This catch used to justify itself with "a malformed triggers file already
1439
+ // fails LOUD at worker boot", and that premise does not hold: the worker reads the file only when
1440
+ // PI_TRIGGERS_FILE is set, so a receiver-only deployment gets no loud failure anywhere. Worse, the
1441
+ // zeroes below silently disarm the WEBHOOK_SECRET check, every per-forge credential check, the
1442
+ // per-image checks and the flow-tier probes -- so doctor came back GREENER than a healthy
1443
+ // deployment, which is the one direction a preflight must never fail in.
1444
+ //
1445
+ // The counts stay zero, because every downstream check reads them and a half-parsed file has no
1446
+ // honest counts to give. What changes is that the reason travels with them.
1447
+ // Only a TAGGED config refusal is reported. parseTriggers throws `piDispatchConfig` errors; anything
1448
+ // else here is an fs failure on a path the guard above already said existed (a race, a permission,
1449
+ // a directory), which is not a statement about the file's CONTENT and has no fix an operator can act
1450
+ // on from this line. Those keep the old silent zeroes.
1451
+ if (e?.piDispatchConfig !== true) return none;
1452
+ return { ...none, parseError: e.message, path: triggersPath(env, cwd) };
1453
+ }
1454
+ }
1455
+
1456
+ /** The triggers path doctor would have read, so a parse failure can name it. */
1457
+ function triggersPath(env, cwd) {
1458
+ return env.PI_TRIGGERS_FILE ?? join(cwd, "triggers.json");
1459
+ }
1460
+
1461
+ /**
1462
+ * The `--env-setup` script (issue #216). `pi-dispatch service render|install --env-setup <path>` names a
1463
+ * script the service manager SOURCES at every boot, as the service user, with the deployment's
1464
+ * environment -- and after that nothing ever looks at it again. resolveEnvSetup checked it existed once,
1465
+ * at render time, on a host that may not be this one.
1466
+ *
1467
+ * doctor has to DISCOVER the path before it can check it, because --env-setup is a render-time flag and
1468
+ * the rendered unit is the only place it lives. Two sources, in this order:
1469
+ *
1470
+ * 1. The installed units for THIS deployment -- the file that actually boots, and so the honest
1471
+ * answer. A unit whose WorkingDirectory names some other folder belongs to some other deployment on
1472
+ * the same host and is deliberately skipped: doctor is this deployment's preflight, and warning
1473
+ * about a neighbour's unit would fire forever on a host that runs two.
1474
+ * 2. PI_ENV_SETUP in doctor's OWN environment, and only when (1) found nothing. That is what launchd
1475
+ * and nssm put in front of the wrapper, so it is the right answer for a doctor run through the same
1476
+ * environment the service gets. It is a different question from (1), which is why every line below
1477
+ * names the source it came from rather than blurring the two.
1478
+ *
1479
+ * Everything here is warn-tier and nothing carries a `fixAction` -- the never tier
1480
+ * (REQ-DEPLOYMENT-BOOTSTRAP): doctor does not chmod an operator's file and does not move it. Nor does it
1481
+ * ever OPEN the script. The script holds no secret by design, but what it holds is the commands that
1482
+ * fetch them, and a preflight that echoed those would be publishing the map instead of the treasure.
1483
+ *
1484
+ * Returns [] when no seam is configured, so a deployment that does not use one gets byte-identical
1485
+ * output.
1486
+ */
1487
+ /**
1488
+ * REQ-EGRESS-ALLOWLIST. What the shipped egress policy actually is on this host, read back from docker
1489
+ * rather than assumed from the compose file that was supposed to create it.
1490
+ *
1491
+ * Returns [] when `PI_EGRESS=0`, so a deployment that declined the policy gets byte-identical output --
1492
+ * the same convention envSetupChecks follows one feature over. Armed is the DEFAULT, so most deployments
1493
+ * see these lines.
1494
+ *
1495
+ * TIERING, and it is the whole editorial judgement here. The proxy's PRESENCE is a hard failure when the
1496
+ * policy is armed: the worker refuses every job pre-spend without it, so a ✓ would be a lie and a ⚠ would
1497
+ * under-report a deployment that cannot run anything. Everything that needs the NETWORK to answer is
1498
+ * warn-tier, on doctor's own rule that a ✗ is reserved for certainties: a custom provider base URL, a
1499
+ * corporate egress path or a transient provider blip each make a red here a false alarm, and an operator
1500
+ * who learns to scroll past doctor costs more than a missed warning does.
1501
+ *
1502
+ * NOTHING here carries a `fixAction` -- the never tier (REQ-DEPLOYMENT-BOOTSTRAP). One candidate was
1503
+ * considered and refused: a prompt-tier offer to start the proxy, on the Valkey precedent. That offer
1504
+ * starts a QUEUE, whose failure mode is that nothing runs. This one would stand up a SECURITY CONTROL
1505
+ * whose allowlist the operator has not written yet, turning "no policy" into "a policy that fails every
1506
+ * job inside a paid container". It is also not one argv but a compose profile and a file that must already
1507
+ * exist, and doctor "never guesses a semantic env value".
1508
+ */
1509
+ async function egressChecks(env, seams, { dockerCode, imageCode, jobImage }) {
1510
+ const { spawn } = seams;
1511
+ // The SAME parse the worker boots with (egress.mjs), never a second `=== "1"`: doctor reporting a
1512
+ // policy that is off, or nothing about one that is on, is worse than doctor not checking at all.
1513
+ // A malformed value is the worker's boot failure to report, not doctor's to guess at, so it reads as
1514
+ // armed here and the `.env` check above is what fails.
1515
+ let armed;
1516
+ try {
1517
+ armed = egressArmed(env);
1362
1518
  } catch {
1363
- return none;
1519
+ armed = true;
1520
+ }
1521
+ if (!armed) return [];
1522
+ const proxy = env.PI_EGRESS_PROXY || DEFAULT_EGRESS_PROXY;
1523
+ const checks = [];
1524
+
1525
+ if (dockerCode !== 0) {
1526
+ checks.push({
1527
+ ok: false,
1528
+ warn: true,
1529
+ label: "Egress policy: not checked (the Docker daemon did not answer)",
1530
+ fix: "start Docker, then re-run doctor -- the policy lives in docker's own networks and containers, so none of it can be read from here",
1531
+ });
1532
+ return checks;
1533
+ }
1534
+
1535
+ // `docker inspect` on the container, not `ps`: it answers present-vs-absent and running-vs-stopped in
1536
+ // one call, and those are two different fixes. The FIELD_SEP habit is image-preflight.mjs's -- neither
1537
+ // a boolean nor a health word can contain "|".
1538
+ const state = await runCmdCapture(spawn, "docker", ["inspect", `--format={{.State.Running}}|{{if .State.Health}}{{.State.Health.Status}}{{else}}none{{end}}`, proxy]);
1539
+ const [running, health] = state.code === 0 ? state.output.trim().split("|") : [];
1540
+ const up = running === "true";
1541
+ checks.push({
1542
+ ok: up,
1543
+ label: up ? `Egress proxy running (${proxy})` : state.code === 0 ? `Egress proxy is stopped (${proxy})` : `Egress proxy is not on this host (${proxy})`,
1544
+ fix: "docker compose -f deploy/docker-compose.yml --profile egress up -d -- the egress policy refuses every job pre-spend while this is down, which costs no budget but runs nothing (PI_EGRESS=0 opts out)",
1545
+ });
1546
+ if (!up) return checks;
1547
+
1548
+ // Advisory on purpose, and deliberately NOT what the money gate reads. A healthcheck can flap, and a
1549
+ // pre-spend gate that refuses on a flapping signal drops real work while one that retries on it burns
1550
+ // the second budget slot this whole requirement exists to save. Here a human is reading, so it is worth
1551
+ // saying: a squid that parsed its config and then wedged looks identical to a healthy one from outside.
1552
+ if (health && health !== "none") {
1553
+ checks.push({
1554
+ ok: health === "healthy",
1555
+ warn: true,
1556
+ label: `Egress proxy health: ${health}`,
1557
+ fix: `docker logs ${proxy} -- the container is up but its listener is not answering, so jobs will start and then fail to reach anything`,
1558
+ });
1364
1559
  }
1560
+
1561
+ // The end-to-end probe, and the only place in this codebase that proves the policy rather than
1562
+ // inspecting it. Two containers, on a throwaway network built exactly like a job's, gated on the image
1563
+ // being present because it uses the job image's own node -- which is the point: it proves the operator's
1564
+ // OWN image honours NODE_USE_ENV_PROXY, the property a stale image would silently lack and the one
1565
+ // whose absence turns the whole policy into an outage.
1566
+ //
1567
+ // Credential-free by construction: `api.anthropic.com` answers 401 to an unauthenticated request, so
1568
+ // reaching the provider and being refused for the key proves the entire path and costs nothing. That is
1569
+ // docs/egress.md's own method, promoted from prose to a check.
1570
+ if (imageCode !== 0) return checks;
1571
+ const net = `pi-dispatch-egress-doctor-${process.pid}`;
1572
+ if ((await runCmd(spawn, "docker", ["network", "create", "--internal", net])) !== 0) return checks;
1573
+ try {
1574
+ if ((await runCmd(spawn, "docker", ["network", "connect", net, proxy])) !== 0) return checks;
1575
+ for (const [slug, host, url, want] of [
1576
+ ["provider", "the provider", "https://api.anthropic.com/v1/messages", true],
1577
+ ["unlisted", "an unlisted host", "https://pi-dispatch-not-on-your-allowlist.example/", false],
1578
+ ]) {
1579
+ const probe = await runCmdCapture(spawn, "docker", [
1580
+ "run",
1581
+ "--rm",
1582
+ // Named, and outside the boot reaper's `pi-job-` filter by construction. `--rm` disposes of it,
1583
+ // so the name exists for the operator watching `docker ps` during a doctor run and for the one
1584
+ // reading `ps` afterwards to find out what a wedged probe was doing.
1585
+ "--name",
1586
+ `pi-dispatch-egress-probe-${slug}`,
1587
+ "--pull=never",
1588
+ `--network=${net}`,
1589
+ "-e",
1590
+ `HTTPS_PROXY=http://${proxy}:3128`,
1591
+ "-e",
1592
+ "NODE_USE_ENV_PROXY=1",
1593
+ "--entrypoint",
1594
+ "node",
1595
+ jobImage,
1596
+ "-e",
1597
+ // The URL rides ARGV, not the spawn env, and the difference from the in-image `gh` probe is
1598
+ // deliberate: that one carries a TOKEN, which must never be visible in `ps`. This carries a
1599
+ // public hostname, so argv is the honest place for it -- an operator reading `ps` during a
1600
+ // doctor run can see exactly which host is being probed.
1601
+ `fetch(${JSON.stringify(url)},{method:"POST"}).then(r=>{console.log("reached",r.status);process.exit(0)},e=>{console.log("blocked",e.cause?.code??e.message);process.exit(3)})`,
1602
+ ]);
1603
+ const reached = probe.code === 0;
1604
+ checks.push({
1605
+ ok: reached === want,
1606
+ warn: true,
1607
+ label: reached === want
1608
+ ? want
1609
+ ? `Egress policy reaches the provider (api.anthropic.com answered, so the whole path works and no key was spent)`
1610
+ : `Egress policy denies ${host} (the deny direction is the half an allowlist can silently lose)`
1611
+ : want
1612
+ ? `Egress policy does NOT reach the provider (api.anthropic.com)`
1613
+ : `Egress policy ALLOWS ${host} that is not on your allowlist`,
1614
+ fix: want
1615
+ ? `add api.anthropic.com to egress-allowlist.conf and restart the proxy -- until then every job starts, fails at its first turn, and spends two budget slots proving it (docs/egress.md)`
1616
+ : `check egress-allowlist.conf: a rule wider than you meant (a bare domain where you wanted a subdomain) lets a job reach hosts you did not list`,
1617
+ });
1618
+ }
1619
+ } finally {
1620
+ await runCmd(spawn, "docker", ["network", "disconnect", "-f", net, proxy]);
1621
+ await runCmd(spawn, "docker", ["network", "rm", net]);
1622
+ }
1623
+ return checks;
1624
+ }
1625
+
1626
+ async function envSetupChecks(env, seams) {
1627
+ const { cwd, spawn, fileExists, platform, home } = seams;
1628
+ const sources = new Map(); // setup path -> how doctor learned it; the first source to name it wins
1629
+
1630
+ if (platform === "win32") {
1631
+ for (const which of ["worker", "receiver"]) {
1632
+ const service = `pi-dispatch-${which}`;
1633
+ const got = await runCmdCapture(spawn, "nssm", ["get", service, "AppEnvironmentExtra"]);
1634
+ // Not installed, or nssm not on PATH: silence. Same doctrine as check-ignore below -- a check
1635
+ // nobody can silence must never cry wolf, and "could not ask" is not "misconfigured".
1636
+ if (got.code !== 0) continue;
1637
+ // No deployment match here: nssm keeps the folder in a SEPARATE AppDirectory property, and there
1638
+ // is exactly one machine-scoped service per name for it to be confused with.
1639
+ const { setup } = readUnitSeam(got.output, "win32");
1640
+ if (setup && !sources.has(setup)) sources.set(setup, `${service}'s AppEnvironmentExtra`);
1641
+ }
1642
+ } else {
1643
+ for (const { path } of installedUnitPaths(platform, home)) {
1644
+ if (!fileExists(path)) continue;
1645
+ let seam;
1646
+ try {
1647
+ seam = readUnitSeam(readFileSync(path, "utf8"), platform);
1648
+ } catch {
1649
+ continue; // a system-scope unit this user may not read: which deployment it serves is unknowable
1650
+ }
1651
+ if (!seam.setup || seam.deployDir !== cwd) continue;
1652
+ if (!sources.has(seam.setup)) sources.set(seam.setup, path);
1653
+ }
1654
+ }
1655
+
1656
+ const fromEnv = (env.PI_ENV_SETUP ?? "").trim();
1657
+ if (sources.size === 0 && fromEnv) sources.set(fromEnv, "PI_ENV_SETUP in this environment");
1658
+
1659
+ const checks = [];
1660
+ for (const [setup, source] of sources) {
1661
+ if (!fileExists(setup)) {
1662
+ checks.push({
1663
+ ok: false,
1664
+ warn: true,
1665
+ label: `the env-setup script at ${setup} does not exist (named by ${source})`,
1666
+ fix: "restore it, or re-render without --env-setup -- the service manager sources it at every boot, so until it is back the unit exits 1 in a restart loop and the worker never starts (docs/secrets.md)",
1667
+ });
1668
+ continue;
1669
+ }
1670
+ checks.push({ ok: true, label: `env-setup script present (${setup}, named by ${source})` });
1671
+
1672
+ // WRITABILITY, not readability -- deliberately `& 0o022` and not the App key's `& 0o077`. This file
1673
+ // is EXECUTED (sourced) by the account that holds the provider key and the forge token, so anyone
1674
+ // who can edit it owns the worker. That it is READABLE is fine: it holds no secret by design.
1675
+ // POSIX only, for the same reason the App key's mode check skips win32 -- stat modes are synthetic
1676
+ // there, so this would warn on every healthy Windows deployment and teach operators to scroll past.
1677
+ if (platform !== "win32") {
1678
+ try {
1679
+ if ((statSync(setup).mode & 0o022) !== 0) {
1680
+ checks.push({
1681
+ ok: false,
1682
+ warn: true,
1683
+ label: `the env-setup script at ${setup} is group/world-writable`,
1684
+ fix: `chmod go-w ${setup} -- the service manager sources it at every boot as the account that holds the provider key and the forge token, so whoever can edit it owns the worker`,
1685
+ });
1686
+ }
1687
+ } catch {
1688
+ // stat raced a deletion or an exotic fs: the presence line above already covered existence.
1689
+ }
1690
+ const dir = dirname(setup);
1691
+ try {
1692
+ const mode = statSync(dir).mode;
1693
+ // Sticky (0o1000) is exempt and must stay exempt: in a sticky directory a non-owner cannot
1694
+ // rename or delete someone else's file, so "anyone can replace it" would simply be false there.
1695
+ if ((mode & 0o022) !== 0 && (mode & 0o1000) === 0) {
1696
+ checks.push({
1697
+ ok: false,
1698
+ warn: true,
1699
+ label: `the directory holding the env-setup script (${dir}) is group/world-writable`,
1700
+ fix: `chmod go-w ${dir} -- the script's own mode does not help when anyone can replace the file, and the manager sources whatever is there at the next boot`,
1701
+ });
1702
+ }
1703
+ } catch {
1704
+ // an unreadable parent directory: nothing to claim either way.
1705
+ }
1706
+ }
1707
+
1708
+ // The #211 question, asked of a different file. Exit 1 is again the ONLY case that speaks: 0 means
1709
+ // ignored, 128 means no work tree, null means git could not be launched, and all three are silence.
1710
+ const ignoreCode = await runCmd(spawn, "git", [...GIT_READ_FLAGS, "-C", dirname(setup), "check-ignore", "-q", setup]);
1711
+ if (ignoreCode === 1) {
1712
+ checks.push({
1713
+ ok: false,
1714
+ warn: true,
1715
+ label: `the env-setup script at ${setup} is inside a git work tree that does not ignore it`,
1716
+ fix: "move it outside that repo, or ignore it there -- it holds no secret by design, but it holds the commands that FETCH them (client and project ids, a manager address, sometimes a path to a credential file), which is a map to every secret this deployment uses",
1717
+ });
1718
+ }
1719
+ }
1720
+ return checks;
1365
1721
  }
1366
1722
 
1367
1723
  /**