@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/LICENSE +661 -0
  2. package/README.md +94 -2
  3. package/dist/capture-client.d.mts +49 -0
  4. package/dist/capture-client.d.mts.map +1 -0
  5. package/dist/capture-client.mjs +352 -0
  6. package/dist/capture-client.mjs.map +1 -0
  7. package/dist/check-worker-code.d.mts +34 -0
  8. package/dist/check-worker-code.d.mts.map +1 -0
  9. package/dist/check-worker-code.mjs +173 -0
  10. package/dist/check-worker-code.mjs.map +1 -0
  11. package/dist/cli-flags.d.mts +71 -0
  12. package/dist/cli-flags.d.mts.map +1 -0
  13. package/dist/cli-flags.mjs +207 -0
  14. package/dist/cli-flags.mjs.map +1 -0
  15. package/dist/code-drift.d.mts +140 -0
  16. package/dist/code-drift.d.mts.map +1 -0
  17. package/dist/code-drift.mjs +284 -0
  18. package/dist/code-drift.mjs.map +1 -0
  19. package/dist/command-line-census.d.mts +33 -0
  20. package/dist/command-line-census.d.mts.map +1 -0
  21. package/dist/command-line-census.mjs +96 -0
  22. package/dist/command-line-census.mjs.map +1 -0
  23. package/dist/compare-workers.d.mts +3 -0
  24. package/dist/compare-workers.d.mts.map +1 -0
  25. package/dist/compare-workers.mjs +332 -0
  26. package/dist/compare-workers.mjs.map +1 -0
  27. package/dist/control-plane-isolation.d.mts +45 -0
  28. package/dist/control-plane-isolation.d.mts.map +1 -0
  29. package/dist/control-plane-isolation.mjs +67 -0
  30. package/dist/control-plane-isolation.mjs.map +1 -0
  31. package/dist/deploy-worker.d.mts +3 -0
  32. package/dist/deploy-worker.d.mts.map +1 -0
  33. package/dist/deploy-worker.mjs +333 -0
  34. package/dist/deploy-worker.mjs.map +1 -0
  35. package/dist/doctor.d.mts +216 -0
  36. package/dist/doctor.d.mts.map +1 -0
  37. package/dist/doctor.mjs +962 -0
  38. package/dist/doctor.mjs.map +1 -0
  39. package/dist/fleet-consistency.d.mts +235 -0
  40. package/dist/fleet-consistency.d.mts.map +1 -0
  41. package/dist/fleet-consistency.mjs +436 -0
  42. package/dist/fleet-consistency.mjs.map +1 -0
  43. package/dist/fleet-env.d.mts +228 -0
  44. package/dist/fleet-env.d.mts.map +1 -0
  45. package/dist/fleet-env.mjs +509 -0
  46. package/dist/fleet-env.mjs.map +1 -0
  47. package/dist/fleet-scripts.d.mts +11 -0
  48. package/dist/fleet-scripts.d.mts.map +1 -0
  49. package/dist/fleet-scripts.mjs +41 -0
  50. package/dist/fleet-scripts.mjs.map +1 -0
  51. package/dist/git-safe-env.d.mts +10 -0
  52. package/dist/git-safe-env.d.mts.map +1 -0
  53. package/dist/git-safe-env.mjs +44 -0
  54. package/dist/git-safe-env.mjs.map +1 -0
  55. package/dist/guest-run.d.mts +26 -0
  56. package/dist/guest-run.d.mts.map +1 -0
  57. package/dist/guest-run.mjs +164 -0
  58. package/dist/guest-run.mjs.map +1 -0
  59. package/dist/host-address.d.mts +33 -0
  60. package/dist/host-address.d.mts.map +1 -0
  61. package/dist/host-address.mjs +105 -0
  62. package/dist/host-address.mjs.map +1 -0
  63. package/dist/host-capacity.d.mts +64 -0
  64. package/dist/host-capacity.d.mts.map +1 -0
  65. package/dist/host-capacity.mjs +152 -0
  66. package/dist/host-capacity.mjs.map +1 -0
  67. package/dist/host-metrics.d.mts +116 -0
  68. package/dist/host-metrics.d.mts.map +1 -0
  69. package/dist/host-metrics.mjs +201 -0
  70. package/dist/host-metrics.mjs.map +1 -0
  71. package/dist/index.d.ts +23 -0
  72. package/dist/index.d.ts.map +1 -0
  73. package/dist/index.js +25 -0
  74. package/dist/index.js.map +1 -0
  75. package/dist/local-vm.d.ts +125 -0
  76. package/dist/local-vm.d.ts.map +1 -0
  77. package/dist/local-vm.js +360 -0
  78. package/dist/local-vm.js.map +1 -0
  79. package/dist/measure-guard.d.mts +34 -0
  80. package/dist/measure-guard.d.mts.map +1 -0
  81. package/dist/measure-guard.mjs +73 -0
  82. package/dist/measure-guard.mjs.map +1 -0
  83. package/dist/normalise-fleet.d.mts +2 -0
  84. package/dist/normalise-fleet.d.mts.map +1 -0
  85. package/dist/normalise-fleet.mjs +76 -0
  86. package/dist/normalise-fleet.mjs.map +1 -0
  87. package/dist/npm-cli-executable.d.mts +42 -0
  88. package/dist/npm-cli-executable.d.mts.map +1 -0
  89. package/dist/npm-cli-executable.mjs +159 -0
  90. package/dist/npm-cli-executable.mjs.map +1 -0
  91. package/dist/probe-outcome.d.mts +89 -0
  92. package/dist/probe-outcome.d.mts.map +1 -0
  93. package/dist/probe-outcome.mjs +104 -0
  94. package/dist/probe-outcome.mjs.map +1 -0
  95. package/dist/protocol-guard.d.mts +34 -0
  96. package/dist/protocol-guard.d.mts.map +1 -0
  97. package/dist/protocol-guard.mjs +121 -0
  98. package/dist/protocol-guard.mjs.map +1 -0
  99. package/dist/source-walk.d.mts +12 -0
  100. package/dist/source-walk.d.mts.map +1 -0
  101. package/dist/source-walk.mjs +56 -0
  102. package/dist/source-walk.mjs.map +1 -0
  103. package/dist/transient-fault.d.mts +6 -0
  104. package/dist/transient-fault.d.mts.map +1 -0
  105. package/dist/transient-fault.mjs +86 -0
  106. package/dist/transient-fault.mjs.map +1 -0
  107. package/dist/utm-deprecated.d.mts +6 -0
  108. package/dist/utm-deprecated.d.mts.map +1 -0
  109. package/dist/utm-deprecated.mjs +23 -0
  110. package/dist/utm-deprecated.mjs.map +1 -0
  111. package/dist/worker-code-check.d.mts +29 -0
  112. package/dist/worker-code-check.d.mts.map +1 -0
  113. package/dist/worker-code-check.mjs +78 -0
  114. package/dist/worker-code-check.mjs.map +1 -0
  115. package/dist/worker-health.d.mts +56 -0
  116. package/dist/worker-health.d.mts.map +1 -0
  117. package/dist/worker-health.mjs +73 -0
  118. package/dist/worker-health.mjs.map +1 -0
  119. package/dist/worker-http.d.mts +103 -0
  120. package/dist/worker-http.d.mts.map +1 -0
  121. package/dist/worker-http.mjs +277 -0
  122. package/dist/worker-http.mjs.map +1 -0
  123. package/dist/worker-stats.d.mts +66 -0
  124. package/dist/worker-stats.d.mts.map +1 -0
  125. package/dist/worker-stats.mjs +143 -0
  126. package/dist/worker-stats.mjs.map +1 -0
  127. package/package.json +96 -4
  128. package/src/local-worker/autounattend.xml +280 -0
  129. package/src/local-worker/build-vm.sh +218 -0
  130. package/src/local-worker/clone-worker.sh +141 -0
  131. package/src/local-worker/create-utm-vm.sh +202 -0
  132. package/src/local-worker/fetch-windows-iso.sh +238 -0
  133. package/src/local-worker/first-boot.cmd +58 -0
  134. package/src/local-worker/worker-ctl.sh +442 -0
  135. package/src/provisioning/README.md +28 -0
  136. package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
  137. package/src/provisioning/bare-metal/README.md +213 -0
  138. package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
  139. package/src/provisioning/bare-metal/autounattend.xml +428 -0
  140. package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
  141. package/src/provisioning/bootstrap-control-plane.sh +463 -0
  142. package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
  143. package/src/provisioning/build-lean-worker-image.ps1 +275 -0
  144. package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
  145. package/src/provisioning/provision-nvda-worker.ps1 +827 -0
  146. package/src/provisioning/set-display-mode.ps1 +411 -0
  147. package/src/provisioning/stamp-provision-revision.ps1 +184 -0
@@ -0,0 +1,962 @@
1
+ #!/usr/bin/env node
2
+ // @ts-check
3
+ // Can I run right now? One command, one answer.
4
+ //
5
+ // pnpm run doctor human-readable, with the fix for anything broken
6
+ // pnpm run doctor -- --json machine-readable, for an agent
7
+ //
8
+ // This exists because "is the environment ready" took five commands and some inference, and
9
+ // the inference went wrong: a healthy VM was reported as a corrupted one because `utmctl`
10
+ // answers `unknown` when the UTM app is closed. Every check below therefore reports what it
11
+ // observed AND the exact command that fixes it, so nothing has to be deduced.
12
+ //
13
+ // Exit codes: 0 ready, 1 something is broken (details in the report).
14
+ import { execFile, execFileSync } from "node:child_process";
15
+ import { promisify } from "node:util";
16
+ import { existsSync, readFileSync, realpathSync, statSync } from "node:fs";
17
+ import { resolve } from "node:path";
18
+ import { fileURLToPath, pathToFileURL } from "node:url";
19
+ import { homedir } from "node:os";
20
+ import { createRequire } from "node:module";
21
+ // The canonical scrubber for this package -- `packages/guards/src/git-env.mjs` re-exports the same function for the
22
+ // repo-root scripts. One helper, two entry points, so a git spawn cannot inherit GIT_DIR by either door.
23
+ import { sandboxGitEnv } from "./git-safe-env.mjs";
24
+ import { availableHostMemoryMb, workersHostCanRun } from "./host-capacity.mjs";
25
+ import { fleetConsistency, describeMismatches } from "./fleet-consistency.mjs";
26
+ import { assessWorker } from "./worker-health.mjs";
27
+ import { controlPlaneIsolation } from "./control-plane-isolation.mjs";
28
+ import { fleetScriptPaths } from "./fleet-scripts.mjs";
29
+ import { configuredWorkers, namedInventoryWorkers } from "./fleet-env.mjs";
30
+ import { refuseUnknownFlags } from "./cli-flags.mjs";
31
+ import { pnpmCliInvocation } from "./npm-cli-executable.mjs";
32
+ import { requestJson } from "./worker-http.mjs";
33
+ /**
34
+ * a mistyped `--json` prints for a human where a script expected a machine-readable answer, and the
35
+ * caller parses the prose.
36
+ *
37
+ * An unrecognised flag is otherwise IGNORED, so it runs the default and reports success.
38
+ */
39
+ refuseUnknownFlags(["--json"], { entry: import.meta.url, command: "pnpm run doctor" });
40
+ const run = promisify(execFile);
41
+ const JSON_OUT = process.argv.includes("--json");
42
+ // A11Y_WORKERS (plural) is how a bare-metal fleet is configured -- bootstrap-control-plane.sh tells you
43
+ // to set exactly that -- and this read only A11Y_WORKER. So doctor reported "no A11Y_WORKER set and no
44
+ // local VM tooling here" against a healthy fleet of ten machines, and doctor is the FIRST command
45
+ // CLAUDE.md tells an agent to run. The environment was fine and the entry point said it was broken.
46
+ //
47
+ // Parsed by fleet-env.mjs, which is the ONE place that answers "what is the fleet" -- see there for why
48
+ // the precedence had to be settled: doctor and worker:code disagreed, so they could report on two
49
+ // different sets of machines with nothing to say so.
50
+ const WORKERS_ENV = configuredWorkers();
51
+ const PAGES_PORT = Number(process.env.DATASET_PAGES_PORT || 5050);
52
+ // The lifecycle script ships beside this one in the fleet package. It was `../scripts/local-worker/...`,
53
+ // resolved from this module — correct while doctor lived in `scripts/`, and silently wrong the moment it
54
+ // moved: doctor then reported "no local VM tooling here" on a host with three registered VMs, which reads as
55
+ // a broken environment rather than a broken path.
56
+ const CTL = fleetScriptPaths().workerCtl;
57
+ // Resolved from THIS module, never the cwd. The scorer being resolved against `process.cwd()` is the
58
+ // defect that made a fresh clone unable to run its own default judge (see packages/scorer/src/index.ts),
59
+ // so nothing here may repeat it.
60
+ const SCORER_MODEL_DIR = fileURLToPath(new URL("../../scorer/models/screenreader-scorer/", import.meta.url));
61
+ // Same rule as SCORER_MODEL_DIR above: resolved from THIS module, never the cwd. `@a11ign/lab`
62
+ // owns the canonical `runs/` resolution (`packages/lab/src/dataset-paths.mjs`), but `lab` depends on
63
+ // `worker-fleet`, so this package cannot import it without a cycle — this is the same computation,
64
+ // duplicated for that reason rather than left cwd-anchored.
65
+ const DATASET = resolve(fileURLToPath(new URL("../../../", import.meta.url)), "runs/screenreader-dataset");
66
+ const PROBE_TIMEOUT_MS = 8000;
67
+ /** @type {{name: string, id: string, ok: boolean, detail: string, fix: string|null, note?: string|null,
68
+ * advisory?: boolean}[]} */
69
+ const checks = [];
70
+ /**
71
+ * DOES THIS CHECK DECIDE `ready`? -- #1073, product-manager's ruling: **the dataset check must REPORT and
72
+ * not gate.**
73
+ *
74
+ * `const ready = checks.every((c) => c.ok)` made every check a gate, so a freshly cloned checkout with a
75
+ * worker configured read **NOT READY** — because the stranger had not generated a TRAINING CORPUS they
76
+ * have no reason to want. `doctor` is the first command the README names and CLAUDE.md tells an agent to
77
+ * obey its `next_command`; a verdict of NOT READY on a machine that is ready for the documented purpose
78
+ * **teaches the reader that the verdict is not about them**, which is #1059's defect one level up.
79
+ *
80
+ * DECLARED HERE AND ENFORCED BY `add`, WHICH THROWS ON AN UNDECLARED NAME. A list that merely sat beside
81
+ * the checks would drift from them silently; a list a new check cannot bypass cannot. **The author of the
82
+ * next check has to say which kind it is**, which is the property the row asked for — not the location.
83
+ *
84
+ * THE GATE NARROWS, IT DOES NOT VANISH. Everything a capture run actually needs still decides: a
85
+ * readiness command that is always ready is worse than one that is never ready, because it is believed.
86
+ * `dataset` is the single entry that reports without deciding, and its absence is real information on the
87
+ * lab path -- which is why it is still PRINTED with its fix.
88
+ * @type {Readonly<Record<string, boolean>>}
89
+ */
90
+ const GATES = Object.freeze({
91
+ worker: true, // no worker, no capture
92
+ fleet: true, // the fleet is the worker, on a configured machine
93
+ judge: true, // a run that cannot score is not a run
94
+ pages: true, // the dataset page server a capture fetches from
95
+ run: true, // a run left mid-flight is a real obstruction
96
+ contention: true, // another session holds the pool
97
+ isolation: true, // the control-plane boundary
98
+ "dist-freshness": true,
99
+ "dist-resolution": true,
100
+ dataset: false, // #1073: lab work. Reported, never a reason a capture cannot happen.
101
+ // THE THREE BELOW ALWAYS PASS `ok: true` TODAY, so their gating value is not currently observable --
102
+ // stated so nobody reads these as decisions that were tested. They are declared because the throw
103
+ // demands it, and the reasons are what make the table a decision rather than a list.
104
+ "primary checkout": false, // a capture runs fine in a linked worktree; this guards the SHARED tree's
105
+ // read-only rule, which is a different kind of cannot-proceed and not one
106
+ // that stops a run. It reports, loudly, via `advise` on the unmarked case.
107
+ "fleet reach": false, // it fires only when SOME box is unreachable and says "the run will dispatch
108
+ // to the rest" -- a partial fleet is a smaller fleet, not no fleet. `fleet`
109
+ // above is the check that decides whether there is a fleet at all.
110
+ "host memory": false, // reports how many workers the host has room for; the run starts fewer
111
+ // rather than failing, which is the whole point of the number.
112
+ });
113
+ /**
114
+ * One check's verdict, and its REMEDY. Every parameter is typed here rather than inferred, because
115
+ * `remedy` defaulting to `null` infers as exactly `null` -- so the argument that matters most, what a
116
+ * reader is to do about a failed check, was the one the compiler refused.
117
+ *
118
+ * `id` is always `name` -- #256 is the first check queried individually by `--json`
119
+ * (`.checks[] | select(.id=="dist-freshness")`), and every check already has a unique `name`, so a
120
+ * separate parameter here would be a second spelling of the same fact rather than a new one.
121
+ *
122
+ * #1059: `fix` IS A COMMAND OR IT IS `null`, and human advice goes in `note`.
123
+ *
124
+ * CLAUDE.md tells an agent to read `next_command` and do that, and `next_command` is `fix`. A sentence
125
+ * there ("unlock the Mac if it is locked, then re-run …") is not something anything can run, so an
126
+ * automated reader loops. The advice is not lost -- it moves to the field nothing executes.
127
+ * ONE `remedy` ARGUMENT, not two: a bare string stays a command (which is what every existing call site
128
+ * passes), and the object form carries the advice beside it. Four parameters is the house limit and
129
+ * bundling cohesive arguments is the house answer to it.
130
+ * @param {string} name
131
+ * @param {boolean} ok
132
+ * @param {string} detail
133
+ * @param {string|null|{fix: string|null, note?: string|null}} [remedy] a runnable command, `null` when
134
+ * none can be constructed, or `{ fix, note }` when there is advice a shell cannot run
135
+ */
136
+ export const addCheck = (name, ok, detail, remedy = null) => {
137
+ // #1073: an undeclared check cannot inherit a default. The throw is the forcing function.
138
+ if (!Object.hasOwn(GATES, name)) {
139
+ throw new Error(`doctor: check "${name}" declares no entry in GATES -- say whether it decides `
140
+ + "`ready` (a capture cannot happen without it) or only reports (true/false in GATES).");
141
+ }
142
+ const { fix, note } = typeof remedy === "string" || remedy === null
143
+ ? { fix: remedy, note: null }
144
+ : { note: null, ...remedy };
145
+ checks.push({ name, id: name, ok, detail, fix, note });
146
+ };
147
+ /** The name every call site uses; `addCheck` is the same function, exported so a test can drive the
148
+ * GATES refusal through the real thing rather than through a copy of its rule. */
149
+ const add = addCheck;
150
+ /**
151
+ * A finding that is REAL and does not stop a run — reported every time, never blocking.
152
+ *
153
+ * `doctor` exits 0 when a run can PROCEED, and that contract is load-bearing: agents are told to read
154
+ * `next_command` and act on it. An architectural debt is not a reason a capture cannot happen, so filing
155
+ * one as a failure would make `doctor` say NOT READY on a machine that is fine — and a readiness command
156
+ * that cries wolf is one people stop running, which is how the checks it does own stop being read.
157
+ *
158
+ * Distinct from `ok: true` for the opposite reason: silence is how ADR 0012 went years describing a system
159
+ * that did not exist. It is stated on every run and excluded from the verdict.
160
+ */
161
+ const advise = (/** @type {string} */ name, /** @type {string} */ detail, /** @type {string|null} */ fix = null) => checks.push({ name, id: name, ok: true, advisory: true, detail, fix });
162
+ function commandError(/** @type {any} */ error) {
163
+ const observed = [error?.stderr, error?.stdout, /** @type {any} */ (error)?.message]
164
+ .map((value) => String(value ?? "").trim())
165
+ .find(Boolean) || "unknown command failure";
166
+ return observed.replace(/\s+/g, " ").slice(0, 400);
167
+ }
168
+ /**
169
+ * The remedy for a failed local-pool query, as a runnable command plus the advice that is not one.
170
+ *
171
+ * #1059: THE DEPRECATION REFUSAL GETS THE FLEET, NOT THE SCRIPT THAT JUST REFUSED. `worker-ctl.sh` refuses
172
+ * on a machine the deprecation is aimed at and says so -- *"Capture on the bare-metal fleet instead:
173
+ * npm run fleet:status"* -- and the `fix:` line beside it said to re-run the script. **Following it
174
+ * exactly reproduces the refusal**, which is the one thing a refusal must never do, and it contradicted
175
+ * the message printed directly above it.
176
+ * @param {string} observed
177
+ * @returns {{ fix: string|null, note: string|null }}
178
+ */
179
+ export function workerControlFix(observed) {
180
+ // The deprecation names itself; matching the REFUSAL rather than the script's name, because the same
181
+ // script succeeds under `A11Y_LOCAL_VM=1` and that is a different situation with a different remedy.
182
+ if (/DEPRECATED|refusing: set A11Y_LOCAL_VM/i.test(observed)) {
183
+ return {
184
+ fix: "pnpm run fleet:status",
185
+ note: "the local UTM VM path is deprecated and refused to run; capture on the bare-metal fleet. "
186
+ + `To use the deprecated local VM anyway: A11Y_LOCAL_VM=1 ${CTL} pool`,
187
+ };
188
+ }
189
+ if (/no VM named|no worker VM registered/i.test(observed)) {
190
+ return { fix: `${CTL} pool`,
191
+ note: "UTM has no registered worker VM; re-register the existing a11y-worker*.utm bundles in UTM first" };
192
+ }
193
+ return existsSync("/Applications/UTM.app")
194
+ ? { fix: `${CTL} pool`, note: "unlock the Mac first if it is locked" }
195
+ : { fix: `${CTL} pool`, note: "this launches UTM if it is installed" };
196
+ }
197
+ async function shell(/** @type {any} */ cmd, /** @type {any} */ args, timeout = 30000) {
198
+ const { stdout } = await run(cmd, args, { timeout, encoding: "utf8" });
199
+ return stdout.trim();
200
+ }
201
+ // `requestJson`, not `fetch` -- audit §9's "the HTTP client" row: a second hand-rolled probe of the same
202
+ // worker JSON API, with its own timeout mechanism, buys nothing and cost this project a silent-truncation
203
+ // bug once already (worker-http.mjs's own header). `requestJson` also carries the failure's CODE
204
+ // (`ECONNREFUSED`, `EHOSTUNREACH`) rather than fetch's undifferentiated `TypeError: fetch failed`.
205
+ async function httpJson(/** @type {any} */ url) {
206
+ const response = await requestJson(url, { timeoutMs: PROBE_TIMEOUT_MS });
207
+ if (!response.ok)
208
+ throw new Error(`HTTP ${response.status}`);
209
+ // `requestJson` returns `undefined` for unparseable JSON rather than throwing (its own docstring: a
210
+ // cache miss is a normal outcome for its usual callers) -- `fetch`'s `.json()` throws, and this function's
211
+ // callers rely on that to distinguish "answered with garbage" from "answered correctly". Restored here.
212
+ if (response.json === undefined)
213
+ throw new Error(`invalid JSON from ${url}`);
214
+ return response.json;
215
+ }
216
+ // --- the checks -----------------------------------------------------------
217
+ /**
218
+ * IS THE FLEET KEY SITTING NEXT TO 100 MB OF PACKAGES NOBODY AUDITED? — ADR 0012, checked rather than
219
+ * asserted.
220
+ *
221
+ * Found on 2026-08-29 to be violated on BOTH machines the ADR is about: the control plane carried 56 MB
222
+ * and 121 packages beside the key, and this laptop carries 103 MB beside the same key plus the lab key.
223
+ * The document was accurate about the intent and described a system that did not exist — which is worse
224
+ * than no document, because it is read as a guarantee.
225
+ *
226
+ * Reported by `doctor` because that is the command whose whole promise is that every check names its own
227
+ * fix, and because a check nobody runs is one this repo has learned not to write.
228
+ */
229
+ /**
230
+ * IS THIS CHECKOUT MARKED AS THE PRIMARY, and does that match what it looks like?
231
+ *
232
+ * The primary-checkout guards (`pre-commit`, `post-checkout`) are OPT-IN as of #198: they fire only where
233
+ * `git config --local a11y.primaryCheckout` is `true`. That is the correct default — inferring it from
234
+ * `.git` being a directory made the hook fire on the lab, which is an ordinary clone, and broke every
235
+ * `lab:job -e ref=<branch>`.
236
+ *
237
+ * But an opt-in guard nobody can find the switch for is an OFF guard, and "unmarked" must not read the
238
+ * same as "safe". So this reports the state on every run rather than only when something is wrong — the
239
+ * `isolation` check above takes the same shape for the same reason: a debt that is reported every run is
240
+ * a known one, and a debt reported never is a forgotten one.
241
+ *
242
+ * ADVISORY, never a hard failure. `doctor` exits 0 when a RUN can proceed, and an unmarked checkout can
243
+ * run perfectly well — it is the fleet-driving machine's protection that is missing, not its capability.
244
+ * A doctor that refused READY over this would be ignored, which is how a guard gets switched off.
245
+ *
246
+ * It does not GUESS which machine deserves the mark. `doctor` runs on laptops, worktrees, the lab and CI,
247
+ * and telling four of those five to mark themselves would be the #198 defect wearing an advisory's
248
+ * clothes. It states what is true and names the command; the operator decides.
249
+ */
250
+ function checkPrimaryCheckoutMark() {
251
+ const root = resolve(fileURLToPath(new URL(".", import.meta.url)), "..", "..", "..");
252
+ // A linked worktree's `.git` is a FILE (`gitdir: ...`) and it SHARES `.git/config` with the repository
253
+ // it was created from -- so the mark alone reads true inside every worktree off a marked primary. Both
254
+ // conditions, exactly as `scripts/git-hooks/lib/is-primary-checkout.sh` requires them; a doctor that
255
+ // disagreed with the hook it reports on would be worse than silent.
256
+ /** A linked worktree's `.git` is a FILE; an absent one is neither, and neither is the primary. */
257
+ let linked;
258
+ try {
259
+ linked = !statSync(resolve(root, ".git")).isDirectory();
260
+ }
261
+ catch {
262
+ linked = false;
263
+ }
264
+ /** `git config --get` exits 1 on an absent key: not marked, not an error. */
265
+ let marked;
266
+ try {
267
+ marked = execFileSync("git", ["config", "--local", "--get", "a11y.primaryCheckout"], { cwd: root, encoding: "utf8", env: sandboxGitEnv() }).trim() === "true";
268
+ }
269
+ catch {
270
+ marked = false;
271
+ }
272
+ if (linked) {
273
+ return add("primary checkout", true, "a linked worktree — never the primary, whatever the shared .git/config says");
274
+ }
275
+ if (marked) {
276
+ return add("primary checkout", true, "MARKED — pre-commit refuses commits here and post-checkout keeps it detached at origin/main");
277
+ }
278
+ advise("primary checkout", "not marked, so the primary-checkout guards are INERT here. Correct for the lab, a worker or a "
279
+ + "colleague's clone; wrong for the machine that drives the fleet.", "pnpm run primary:mark -- --set (only on the fleet-driving checkout — see docs/primary-checkout.md)");
280
+ }
281
+ function checkControlPlaneIsolation() {
282
+ // `~` is a SHELL expansion, not a filesystem one: `existsSync("~/.ssh/...")` is always false, which
283
+ // would make this guard report every machine as compliant. The silent-pass failure mode, in the guard
284
+ // written because a document silently passed.
285
+ const raw = process.env.A11Y_SSH_KEY || "~/.ssh/a11y-witness_ed25519";
286
+ const keyPath = raw.startsWith("~/") ? resolve(homedir(), raw.slice(2)) : raw;
287
+ const hasFleetKey = existsSync(keyPath);
288
+ const root = resolve(fileURLToPath(new URL(".", import.meta.url)), "..", "..", "..");
289
+ const hasNodeModules = existsSync(resolve(root, "node_modules"));
290
+ // A workspace is a checkout with sources, which is what makes "delete node_modules" the wrong advice
291
+ // here and the right advice on a control plane.
292
+ const isWorkspace = existsSync(resolve(root, "packages")) && existsSync(resolve(root, "package.json"));
293
+ const verdict = controlPlaneIsolation({ hasNodeModules, hasFleetKey, isWorkspace });
294
+ // NOT a hard failure: this machine cannot do its job without the key today, and a doctor that refuses
295
+ // to say READY over an architectural debt would simply be ignored. It is reported every run so it stays
296
+ // visible, which is the difference between a known debt and a forgotten one.
297
+ if (!verdict.violated)
298
+ return add("isolation", true, verdict.why);
299
+ advise("isolation", verdict.why, "docs/control-plane-plan.md L3 — drive the control plane rather than holding its keys");
300
+ }
301
+ /**
302
+ * Pure: does `resolvedRealPath` (already realpath'd) live under `thisCheckoutRoot` (also realpath'd)?
303
+ * Both must be realpath'd BEFORE calling this, never inside it -- comparing a symlinked path against a
304
+ * realpath'd root would report every worktree as foreign to itself, since a worktree's own files are
305
+ * reached through no symlink while its `node_modules` is one.
306
+ *
307
+ * @param {string} resolvedRealPath
308
+ * @param {string} thisCheckoutRootReal
309
+ * @returns {boolean}
310
+ */
311
+ export function resolvesToThisCheckout(resolvedRealPath, thisCheckoutRootReal) {
312
+ return resolvedRealPath === thisCheckoutRootReal
313
+ || resolvedRealPath.startsWith(thisCheckoutRootReal.endsWith("/") ? thisCheckoutRootReal : `${thisCheckoutRootReal}/`);
314
+ }
315
+ /**
316
+ * Pure: which checkout root does a resolved `packages/<name>/dist/...` path belong to? Everything before
317
+ * the first `/packages/` -- this repo's own, fixed layout, not a guess. `null` when the path does not look
318
+ * like it, which a caller must treat as "could not tell", never as "this checkout".
319
+ *
320
+ * @param {string} resolvedRealPath
321
+ * @returns {string | null}
322
+ */
323
+ export function checkoutRootFor(resolvedRealPath) {
324
+ const idx = resolvedRealPath.indexOf("/packages/");
325
+ return idx === -1 ? null : resolvedRealPath.slice(0, idx);
326
+ }
327
+ /** @type {(cmd: string, args: string[]) => string} */
328
+ const defaultTscRun = (cmd, args) => execFileSync(cmd, args, { encoding: "utf8" });
329
+ /**
330
+ * Whether `tsc --build --dry` -- the SAME authority the real build uses, deferred to rather than
331
+ * reimplemented -- considers `tsconfigPath`'s own outputs up to date.
332
+ *
333
+ * NOT a timestamp comparison, and that correction cost a wrong first version of this check (#256, live-
334
+ * measured): a directory's mtime does not move on rewrite; a file's mtime moves on `git checkout` with no
335
+ * content change at all, which is the ordinary case of switching branches; and `tsc --build` itself is
336
+ * content-addressed for the files it actually recompiles, so a real build can be legitimately up to date
337
+ * with source files whose mtimes are newer than its outputs. Measured on two live worktrees straight
338
+ * after a branch switch: several source files newer than `dist/index.js`, and `tsc --build --dry` still
339
+ * correctly reported "is up to date". A raw mtime comparison would have flagged both as stale --
340
+ * permanently, on every worktree, the moment `git checkout` runs -- which is exactly the "readiness
341
+ * command that cries wolf" this file's own `advise` doc warns against.
342
+ *
343
+ * `null`, not `false`, when the run failed outright or its report never mentioned this project at all --
344
+ * "could not tell" and "not up to date" need opposite responses (investigate vs. rebuild), and this repo's
345
+ * own rule is that a lookup failure is never silently read as a clean answer.
346
+ *
347
+ * @param {string} tsconfigPath
348
+ * @param {{ run?: (cmd: string, args: string[]) => string }} [deps]
349
+ * @returns {boolean | null}
350
+ */
351
+ export function tscProjectUpToDate(tsconfigPath, { run = defaultTscRun } = {}) {
352
+ /** @type {string} */
353
+ let output;
354
+ try {
355
+ const tsc = pnpmCliInvocation(["exec", "tsc", "--build", "--dry", tsconfigPath]);
356
+ output = run(tsc.command, tsc.args);
357
+ }
358
+ catch (error) {
359
+ // `--dry` still exits 0 for a stale project (measured); a thrown error here is a REAL failure --
360
+ // a missing tsconfig, a syntax error blocking even the dry check -- and its stdout, if any, is still
361
+ // worth reading rather than discarded.
362
+ output = /** @type {{stdout?: string}} */ (error)?.stdout ?? "";
363
+ if (!output)
364
+ return null;
365
+ }
366
+ const line = output.split("\n").find((l) => l.includes(tsconfigPath));
367
+ if (!line)
368
+ return null;
369
+ return /is up to date/.test(line);
370
+ }
371
+ /**
372
+ * ", N commit(s) behind origin/main" or "" -- best-effort, and silently empty on any failure (not a git
373
+ * checkout at all, no `origin/main`, `otherRoot` unknown). This is a NOTE on an already-advisory finding,
374
+ * not itself a fact `doctor` promises; the resolution mismatch is reported either way.
375
+ *
376
+ * @param {string | null} otherRoot
377
+ * @returns {string}
378
+ */
379
+ function behindOriginMainNote(otherRoot) {
380
+ if (!otherRoot)
381
+ return "";
382
+ try {
383
+ const behindBy = execFileSync("git", ["rev-list", "--count", "HEAD..origin/main"], { cwd: otherRoot, encoding: "utf8", env: sandboxGitEnv() }).trim();
384
+ return behindBy === "0" ? "" : `, ${behindBy} commit(s) behind origin/main`;
385
+ }
386
+ catch {
387
+ return "";
388
+ }
389
+ }
390
+ /**
391
+ * WHOSE dist a cross-package import actually resolves to, and is IT stale (#256) -- both computed from
392
+ * the exact SPECIFIER a real import site in this repo uses, never the bare package name. CLAUDE.md's own
393
+ * recorded lesson: "resolving @a11ign/judge does not prove @a11ign/judge/rules came from your
394
+ * tree" -- a package can export subpaths from elsewhere, so resolving the root proves nothing about a
395
+ * subpath. `@a11ign/judge/rules` is a real specifier this repo imports
396
+ * (`packages/lab/scripts/score-rules.ts` and others), not a synthetic probe.
397
+ *
398
+ * ADVISORY, never a hard failure -- same reasoning as `isolation` above: a worktree resolving to the
399
+ * primary's dist can still run every command correctly today, and a doctor that refused READY over an
400
+ * environmental fact would be ignored, which is how a guard gets switched off. It is reported every run
401
+ * so a stale answer is a known condition, not a silent one.
402
+ */
403
+ function checkCrossPackageDist() {
404
+ const specifier = "@a11ign/judge/rules";
405
+ const thisCheckoutRoot = resolve(fileURLToPath(new URL(".", import.meta.url)), "..", "..", "..");
406
+ /** @type {string} */
407
+ let resolvedRealPath;
408
+ try {
409
+ resolvedRealPath = realpathSync(createRequire(import.meta.url).resolve(specifier));
410
+ }
411
+ catch (error) {
412
+ return advise("dist-resolution", `could not resolve ${specifier} to check whose dist it comes from -- `
413
+ + `${ /** @type {Error} */(error).message}`, "pnpm run build");
414
+ }
415
+ if (resolvesToThisCheckout(resolvedRealPath, realpathSync(thisCheckoutRoot))) {
416
+ add("dist-resolution", true, `${specifier} resolves to this checkout's own dist`);
417
+ }
418
+ else {
419
+ const behindNote = behindOriginMainNote(checkoutRootFor(resolvedRealPath));
420
+ advise("dist-resolution", `${specifier} resolves to ${resolvedRealPath} (NOT this checkout${behindNote})`, "pnpm run primary:update && pnpm run build # if that is the primary checkout");
421
+ }
422
+ // THE HALF A RESOLUTION CHECK ALONE MISSES: resolving to your OWN tree is no protection if your own
423
+ // dist is stale -- so this checks freshness of whichever checkout the specifier ACTUALLY resolved to,
424
+ // not always this one. `tsc --build --dry`, never a raw mtime comparison -- see `tscProjectUpToDate`'s
425
+ // own doc for the live-measured reason.
426
+ const distRoot = checkoutRootFor(resolvedRealPath) ?? thisCheckoutRoot;
427
+ const tsconfigPath = resolve(distRoot, "packages/judge/tsconfig.json");
428
+ const upToDate = tscProjectUpToDate(tsconfigPath);
429
+ if (upToDate === null) {
430
+ return advise("dist-freshness", `could not ask tsc whether ${tsconfigPath} is up to date`, "pnpm run build");
431
+ }
432
+ if (upToDate) {
433
+ return add("dist-freshness", true, `packages/judge under ${distRoot} is up to date (tsc --build --dry)`);
434
+ }
435
+ advise("dist-freshness", `packages/judge under ${distRoot} is NOT up to date (tsc --build --dry) -- a `
436
+ + "build compiled before the source it now reflects", "pnpm run build # in that checkout");
437
+ }
438
+ // The DEFAULT here was "codex", and every part of that was wrong. `judge.ts` has no codex case at all —
439
+ // it offers local, anthropic and openai — so with JUDGE_BACKEND unset (the normal case) this told the
440
+ // operator to "install Codex and run: codex login" for a backend the product cannot use. And setting
441
+ // JUDGE_BACKEND=local fell into the other branch and checked JUDGE_BASE_URL, which local does not need.
442
+ // Both answers were wrong, in a command whose whole promise is that every check names its own fix.
443
+ //
444
+ // Mirrors judge.ts's default deliberately: a doctor that disagrees with the thing it inspects is worse
445
+ // than no doctor.
446
+ async function checkJudge() {
447
+ // `||`, not `??`: an env var set to the EMPTY string is how CI passes "unset", and `??` only defaults
448
+ // on nullish — so an empty JUDGE_BACKEND matched no backend and reported a typo that nobody made.
449
+ const backend = (process.env.JUDGE_BACKEND || "local").toLowerCase();
450
+ if (backend === "local") {
451
+ const weights = resolve(SCORER_MODEL_DIR, "model.safetensors");
452
+ return add("judge", existsSync(weights), existsSync(weights) ? "backend=local, trained scorer present" : "backend=local, but the trained scorer is missing", `expected weights at ${weights} — they ship in the repo, so this means an incomplete checkout`);
453
+ }
454
+ if (backend === "anthropic" || backend === "openai") {
455
+ const key = backend === "anthropic" ? "ANTHROPIC_API_KEY" : "JUDGE_BASE_URL";
456
+ return add("judge", !!process.env[key], `backend=${backend}`, `export ${key}=...`);
457
+ }
458
+ // Refuse an unknown backend rather than reporting on one that will not run — the same rule action.yml
459
+ // applies, because a typo must not quietly change which judge assessed the page.
460
+ add("judge", false, `backend=${backend} is not one of local, anthropic, openai`, "unset JUDGE_BACKEND to use the default local scorer");
461
+ }
462
+ // Workers, as a POOL, and with the right idea of what "ready" means.
463
+ //
464
+ // This check used to look at one VM and fail if it was not started. That was true before runs
465
+ // managed the VM themselves; now a stopped worker is the correct resting state -- a run starts
466
+ // what it needs and puts it back afterwards -- so reporting it as FAIL told an agent the
467
+ // environment was broken when it was idle. One did exactly that: it went hunting for a
468
+ // decommissioned worker on another host and then reached for the UTM GUI.
469
+ //
470
+ // Ready means "a run can proceed", not "everything is already running".
471
+ async function checkWorker() {
472
+ if (WORKERS_ENV.length)
473
+ return checkConfiguredFleet(WORKERS_ENV);
474
+ // THE INVENTORY IS A FLEET, and `doctor` could not see one.
475
+ //
476
+ // It resolved A11Y_WORKERS, then the local UTM pool, then gave up — so on a Mac with any registered VM
477
+ // it reported the DEPRECATED local guests and never `inventory.yml`, which every other fleet command
478
+ // treats as the source of truth. Measured 2026-08-29 on one machine, at one moment: `doctor` said
479
+ // "2 worker(s), all stopped — READY" while `worker:code` said "checking 5 worker(s) from inventory.yml"
480
+ // and `fleet:status` showed those five busy with a corpus run.
481
+ //
482
+ // Three commands describing three different fleets, and `doctor` is the one CLAUDE.md tells an agent to
483
+ // run FIRST. Its `next_command` said `training:capture`, which would have captured on the wrong machines.
484
+ //
485
+ // THE INVENTORY WINS OUTRIGHT. The local UTM pool is deprecated — it was a testing arrangement — so it
486
+ // is a fallback for a machine with no inventory, never a contender with one. Anything else reproduces
487
+ // the divergence above on any developer Mac that still has a bundle registered.
488
+ const inventory = namedInventoryWorkers();
489
+ if (inventory.length)
490
+ return checkConfiguredFleet(inventory);
491
+ if (process.platform !== "darwin" || !existsSync(CTL)) {
492
+ return add("worker", false, "no A11Y_WORKERS set, no inventory.yml fleet, and no local VM tooling here", "set A11Y_WORKERS=http://host:8765[,http://host2:8765], or see docs/getting-started.md");
493
+ }
494
+ let pool;
495
+ try {
496
+ pool = JSON.parse(await shell(CTL, ["pool"], 90000));
497
+ }
498
+ catch (e) {
499
+ return add("worker", false, `could not query the local pool (${commandError(e)})`, workerControlFix(commandError(e)));
500
+ }
501
+ if (!pool.length) {
502
+ return add("worker", false, "no worker VM registered", "UTM has no registered worker VM; re-register an existing a11y-worker*.utm bundle, or build one from docs/getting-started.md");
503
+ }
504
+ const running = pool.filter((/** @type {any} */ vm) => vm.state === "started");
505
+ const healthy = pool.filter((/** @type {any} */ vm) => vm.healthy);
506
+ const brokenlyRunning = running.filter((/** @type {any} */ vm) => !vm.healthy);
507
+ const summary = pool.map((/** @type {any} */ vm) => `${vm.name}=${vm.healthy ? vm.ip : vm.state}`).join(" ");
508
+ // A VM that is RUNNING but not answering is a genuine fault. One that is stopped is not.
509
+ if (brokenlyRunning.length) {
510
+ add("worker", false, `${summary} — ${brokenlyRunning.map((/** @type {any} */ v) => v.name).join(", ")} running but not answering`, "Start-ScheduledTask -TaskName a11ysrv on that guest, or " + `${CTL} stop && ${CTL} up`);
511
+ }
512
+ else if (healthy.length) {
513
+ add("worker", true, `${healthy.length}/${pool.length} ready — ${summary}`);
514
+ }
515
+ else {
516
+ add("worker", true, `${pool.length} worker(s), all stopped — a run starts them automatically (${summary})`);
517
+ }
518
+ // Same shape as a configured fleet, so the two diagnostics below have ONE implementation. They were
519
+ // pure functions over /health JSON that only the UTM branch could reach, which meant a bare-metal
520
+ // fleet -- the direction this project is going -- got neither.
521
+ const reachable = pool.filter((/** @type {any} */ v) => v.healthy && v.ip)
522
+ .map((/** @type {any} */ v) => ({ name: v.name, url: `http://${v.ip}:${v.port}` }));
523
+ const probed = await probeAll(reachable);
524
+ await checkDegradedWorkers(probed);
525
+ checkFleetConsistency(probed, pool.length);
526
+ checkHostCapacity(pool);
527
+ const busy = pool.filter((/** @type {any} */ vm) => vm.busy);
528
+ if (busy.length) {
529
+ add("contention", false, `${busy.map((/** @type {any} */ v) => v.name).join(", ")} busy with a capture — another shell or agent is using the pool`,
530
+ // #1059: no `fix`, because waiting is not a command. The advice is a note and `next_command` is null.
531
+ { fix: null, note: "wait for it, or you will both see the other's restarts as breakage" });
532
+ }
533
+ }
534
+ /**
535
+ * The fleet named by A11Y_WORKERS: probe every one, report per worker, never fail on a single loss.
536
+ *
537
+ * "Ready means a run can proceed" is this file's own rule, and for a fleet that means AT LEAST ONE
538
+ * worker answering -- not all of them. The dispatcher already evicts a worker after three consecutive
539
+ * failures and requeues its cases (capture-decisions.mjs), so one dead machine costs throughput, not
540
+ * the run. Failing the whole check for it would tell an agent the environment is broken when nine
541
+ * workers are sitting idle and ready, which is the exact mistake the comment above checkWorker
542
+ * describes for a stopped VM.
543
+ */
544
+ /**
545
+ * Ask every worker once, and keep the failures as data.
546
+ *
547
+ * Shared because both fleet branches feed the same two diagnostics, and because `doctor` used to probe
548
+ * `/health` THREE times per worker — once here, once for degradation, once for consistency — so the three
549
+ * sections could describe three different moments. A box that went busy between them was reported ready by
550
+ * one and silently skipped by the next.
551
+ *
552
+ * @param {{name: string, url: string}[]} workers
553
+ */
554
+ async function probeAll(workers) {
555
+ const probed = [];
556
+ for (const w of workers) {
557
+ try {
558
+ probed.push({ ...w, health: await httpJson(`${w.url}/health`) });
559
+ }
560
+ catch (e) {
561
+ probed.push({ ...w, health: null, error: /** @type {any} */ (e).message });
562
+ }
563
+ }
564
+ return probed;
565
+ }
566
+ async function checkConfiguredFleet(/** @type {any} */ workers) {
567
+ const probed = await probeAll(workers);
568
+ const reachable = probed.filter((p) => p.health);
569
+ const ready = reachable.filter((p) => p.health.ready);
570
+ const state = (/** @type {any} */ p) => {
571
+ if (!p.health)
572
+ return "unreachable";
573
+ if (p.health.busy)
574
+ return "busy";
575
+ return p.health.ready ? "ready" : "not-ready";
576
+ };
577
+ const summary = probed.map((p) => `${p.name}=${state(p)}`).join(" ");
578
+ if (!reachable.length) {
579
+ return add("worker", false, `${workers.length} configured, none answering — ${summary}`, "check those machines are up and their a11ysrv task is running; "
580
+ + `curl ${workers[0].url}/health from this host`);
581
+ }
582
+ // A worker that answers but is not ready is normal right after a boot and clears on its own --
583
+ // /health's own note says so -- which is why this reports the count rather than failing on it.
584
+ add("worker", true, `${ready.length}/${workers.length} ready — ${summary}`);
585
+ const unreachable = probed.filter((p) => !p.health);
586
+ if (unreachable.length) {
587
+ add("fleet reach", true, `${unreachable.length} not answering: ${unreachable.map((p) => p.name).join(", ")}`
588
+ + " — the run will dispatch to the rest");
589
+ }
590
+ // ONE PROBE, PASSED DOWN. These re-probed `/health` themselves, so `doctor` made three requests per
591
+ // worker and the three sections could describe three different moments — a box that went busy or
592
+ // unreachable between them was reported ready by one and silently skipped by the next.
593
+ await checkDegradedWorkers(probed);
594
+ await checkFleetConsistency(probed, workers.length);
595
+ // Contention only matters when there is nowhere left to dispatch. One busy worker in a fleet of ten
596
+ // is a run in progress, not a conflict -- flagging it would make doctor fail during normal use.
597
+ if (reachable.length && reachable.every((p) => p.health.busy)) {
598
+ add("contention", false, `all ${reachable.length} reachable worker(s) busy — another run or agent has the fleet`,
599
+ // #1059: no `fix`, because waiting is not a command. The advice is a note and `next_command` is null.
600
+ { fix: null, note: "wait for it, or you will both see the other's restarts as breakage" });
601
+ }
602
+ }
603
+ // A guest whose NVDA fails on every capture keeps serving and keeps passing — it just costs 4x.
604
+ //
605
+ // The run's eviction rule needs three consecutive FAILURES, and the worker's own retry means there are
606
+ // none, so this never surfaced anywhere. Measured on this pool: one worker needed a recovery on 4 of 4
607
+ // captures (nvdaStart 19.1s each, WALL 122.9s) beside one that needed none (WALL 40.6s). Reported, not
608
+ // failed: a degraded worker is slow, not broken, and pulling it costs more throughput than it saves.
609
+ async function checkDegradedWorkers(/** @type {any} */ probed) {
610
+ for (const w of probed) {
611
+ if (!w.health)
612
+ continue; // unreachable is already the worker check's business
613
+ const { degraded, reason } = assessWorker(w.health.vitals);
614
+ if (degraded) {
615
+ add(`worker ${w.name}`, true, `DEGRADED — ${reason}`, `re-provision ${w.name}: packages/worker-fleet/src/provisioning/provision-nvda-worker.ps1, elevated,`
616
+ + " in the interactive session");
617
+ }
618
+ }
619
+ }
620
+ /**
621
+ * Are the guests interchangeable?
622
+ *
623
+ * The pool assumes so: cases go to whichever worker is free, the cache lets any guest reuse another's
624
+ * evidence, and a good/bad pair is only comparable because both halves came from equivalent machines.
625
+ * Two real divergences happened in one day -- Edge auto-updated on one guest while the others stayed
626
+ * behind, and StartupBoostEnabled read 1 on two guests and 0 on a third -- and BOTH were caught by a
627
+ * human reading a console by eye. That is not a detection mechanism.
628
+ *
629
+ * Never a FAIL. A run on slightly mismatched guests is worse than one on matched guests and far better
630
+ * than no run, and a diagnostic must not be the thing that takes the pool offline.
631
+ */
632
+ function checkFleetConsistency(/** @type {any} */ probed, /** @type {number} */ configured) {
633
+ const guests = probed.filter((/** @type {any} */ w) => w.health)
634
+ .map((/** @type {any} */ w) => ({ worker: w.url, environment: w.health.environment, policy: undefined }));
635
+ const { consistent, mismatches, fields } = fleetConsistency(guests);
636
+ if (guests.length < 2)
637
+ return;
638
+ if (consistent) {
639
+ add("fleet", true, fleetAgreementLine({ agreeing: guests.length, configured, fields }));
640
+ return;
641
+ }
642
+ // The remedy NAMES THE FIELDS THAT DIFFER, derived from the mismatches, and that is the same fix as
643
+ // the agreement line below: it used to retype "browser, screen reader, OS and protocol" too, so a
644
+ // fleet split on `displayMode` was told to go and align four fields that already matched.
645
+ add("fleet", true, `INCONSISTENT — ${describeMismatches(mismatches).join("; ")}`, "re-provision the odd one out so every worker reports the same "
646
+ + `${mismatches.map((/** @type {any} */ m) => m.field).join(", ")}`);
647
+ }
648
+ /**
649
+ * The agreement sentence, with WHICH FIELDS AGREED DERIVED rather than retyped — #1997.
650
+ *
651
+ * "OF N CONFIGURED", because agreement among a SUBSET is not agreement. Unreachable guests are skipped
652
+ * by the caller — correctly, that check is not their business — so without the denominator "3 guests
653
+ * agree" reads as a whole fleet on a fleet of five, which is the examined-nothing shape one step in from
654
+ * zero.
655
+ *
656
+ * AND THE FIELD NAMES ARE NOT TYPED HERE. This line used to read "agree on browser, screen reader, OS
657
+ * and protocol": four names, by hand, beside a `MUST_MATCH` that has ten. `guidepupVersion`,
658
+ * `architecture`, `browserProfile`, `screenReaderSettings`, `provisionRevision` and `displayMode` were
659
+ * all compared and none of them was mentioned, and the remediation string repeated the same four — so
660
+ * the sentence had been making a POSITIVE, false claim about its own scope since the fifth field was
661
+ * added, and no test could notice because nothing tied the words to the list. A sentence enumerating
662
+ * what a machine compared is a second copy of that machine's predicate; derived, it cannot go stale.
663
+ *
664
+ * `fleet:status` says nothing about field coverage and this said something untrue about it, which is why
665
+ * #1997's fix has to reach both. Never a FAIL either way: a mismatched pool is worse than a matched one
666
+ * and far better than no pool, and a diagnostic must not be the thing that takes the fleet offline.
667
+ *
668
+ * AND THE LIST IS WHAT EVERY COMPARED GUEST REPORTED, NOT WHAT ANY ONE OF THEM DID — #2034, carrying
669
+ * #2019's ruling into the second of the two commands #1997 named. `fields.compared` is TRUE-IF-ANYBODY,
670
+ * so a field ONE guest of three reported was named inside a list introduced by the words "guests agree
671
+ * on", and the reporter count contradicting it sat in the same return value. Measured 2026-09-22 at
672
+ * #2033's head: `coverage: {"field":"displayMode","reported":1,"asked":3}` beside `3 of 3 guests agree
673
+ * on 10 compared field(s) (..., displayMode)`. One guest's display was read. Naming it is worse than
674
+ * counting it, because the naming is what #1997 added to make the sentence actionable.
675
+ *
676
+ * THREE FACTS, THREE SENTENCES, and that split is #2019's ruling rather than a style choice: `N of N`
677
+ * is agreement, `k of N` is *some boxes did not report it* and sends a reader to the BOXES, `0 of N` is
678
+ * *nobody could be asked* and sends them to the FIELD. Collapsing the middle one into either of the
679
+ * outer two is the defect this fixes in one direction and #1997's in the other.
680
+ *
681
+ * DERIVED FROM `coverage`, AND NOT FROM `compared`/`unchecked` — the same call `fleet:status` makes
682
+ * (`fieldCoverageGap`, #2019). The two lists are that same measurement thresholded at "anybody", so
683
+ * reading the whole case off one and the partial case off the other gives one fact two sources that can
684
+ * disagree. The lists stay in the return value, where a caller greps them for the remedy.
685
+ *
686
+ * @param {{ agreeing: number, configured: number,
687
+ * fields: { compared: string[], unchecked: string[],
688
+ * coverage?: { field: string, reported: number, asked: number }[] } }} input
689
+ * @returns {string}
690
+ */
691
+ export function fleetAgreementLine({ agreeing, configured, fields }) {
692
+ const rest = agreeing < configured ? " — the rest could not be asked" : "";
693
+ const coverage = fields.coverage;
694
+ // NO COUNTS SUPPLIED IS A CANNOT-ASK, NOT A PASS — `fleet:status` takes the same line on the same
695
+ // shape. A caller carrying the pre-#2019 `fields` has answered "did anybody report each field" and
696
+ // not "how many", so it cannot rule out the 1-of-3 case; falling back to `compared` here would
697
+ // restore the sentence this function exists to stop making, and nothing would say so.
698
+ if (coverage === undefined) {
699
+ return `${agreeing} of ${configured} guests agree, and no field coverage was supplied, so WHICH `
700
+ + `fields were compared, and by HOW MANY guests, was never asked${rest}`;
701
+ }
702
+ const whole = coverage.filter(({ reported, asked }) => reported === asked).map(({ field }) => field);
703
+ const named = whole.length === 0 ? "" : ` (${whole.join(", ")})`;
704
+ const gaps = [partialClause(coverage), uncheckedClause(coverage)].filter((clause) => clause !== "");
705
+ return `${agreeing} of ${configured} guests agree on ${whole.length} of ${coverage.length} `
706
+ + `field(s)${named}${rest}${gaps.join("")}`;
707
+ }
708
+ /** "it"/"them" for a clause that names a list, so one gap does not read as a plural. */
709
+ const itOrThem = (/** @type {number} */ count) => (count === 1 ? "it" : "them");
710
+ /**
711
+ * #2034's clause: the fields SOME compared guests reported and others did not, each with its `k of N`.
712
+ *
713
+ * NAMED WITH ITS COUNT, never counted — the same choice `fleet:status`' `partialClause` makes, and for
714
+ * the same reason #1997 gave for naming the zero case. "1 field was partly reported" sends a reader back
715
+ * to `doctor`; "displayMode (1 of 3 reported it)" tells them two boxes owe an answer, which is the
716
+ * finding — either the converge did not reach them, or their probe for the field failed (#1953).
717
+ *
718
+ * @param {{ field: string, reported: number, asked: number }[]} coverage
719
+ * @returns {string} empty when no field is partly reported
720
+ */
721
+ function partialClause(coverage) {
722
+ const partial = coverage.filter(({ reported, asked }) => reported > 0 && reported < asked);
723
+ if (partial.length === 0)
724
+ return "";
725
+ const named = partial.map(({ field, reported, asked }) => `${field} (${reported} of ${asked} reported it)`);
726
+ return `; reported by only SOME of the compared guests, so agreement says nothing about `
727
+ + `${itOrThem(partial.length)}: ${named.join(", ")}`;
728
+ }
729
+ /**
730
+ * #1997's clause: the fields asked of everybody and answered by nobody.
731
+ *
732
+ * NAMED, not counted: "one field was not compared" sends a reader back to doctor, and "displayMode was
733
+ * not compared" sends them to the deploy that would report it.
734
+ *
735
+ * @param {{ field: string, reported: number, asked: number }[]} coverage
736
+ * @returns {string} empty when every asked field drew at least one value
737
+ */
738
+ function uncheckedClause(coverage) {
739
+ const unchecked = coverage.filter(({ reported }) => reported === 0).map(({ field }) => field);
740
+ if (unchecked.length === 0)
741
+ return "";
742
+ return `; NOT compared on any guest, so agreement says nothing about ${itOrThem(unchecked.length)}: `
743
+ + unchecked.join(", ");
744
+ }
745
+ // Can this host actually hold the pool it has registered?
746
+ //
747
+ // Not a fault, and never a FAIL: a capped pool runs fine, just narrower. It is reported because the
748
+ // alternative is invisible. Three guests on this 36 GB Mac made every capture 1.6x slower than one
749
+ // and produced mute-NVDA failures, and from outside that reads as "the workers are degrading" rather
750
+ // than "the host is out of memory" — which is exactly how it was misread for a day.
751
+ function checkHostCapacity(/** @type {any} */ pool) {
752
+ const availableMb = availableHostMemoryMb();
753
+ if (availableMb === null || !pool.length)
754
+ return;
755
+ // Guests already up have paid for their memory and are not counted in `availableMb`, so they are
756
+ // added back — otherwise a running worker makes the host look smaller than it is.
757
+ const running = pool.filter((/** @type {any} */ vm) => vm.state === "started").length;
758
+ const poolSize = pool.length;
759
+ const limit = workersHostCanRun({ availableMb, alreadyRunning: running });
760
+ const detail = `~${availableMb} MB available — room for ${Math.min(limit, poolSize)} of ${poolSize} worker(s)`;
761
+ add("host memory", true, limit >= poolSize
762
+ ? detail
763
+ : `${detail}; the rest stay stopped so the run does not swap (override: A11Y_MAX_WORKERS)`);
764
+ }
765
+ // Dataset capture needs the pages served, and the guest must be able to reach them — the
766
+ // host's localhost is not reachable from inside the VM. The capture command leases the page
767
+ // server for the run, so an idle host with no listener on this port is ready, not broken.
768
+ async function checkDatasetPages() {
769
+ const manifestPath = resolve(DATASET, "manifest.json");
770
+ if (!existsSync(manifestPath)) {
771
+ return add("dataset", false, "no manifest — the dataset has not been generated", "pnpm run training:generate");
772
+ }
773
+ // Ask for a REAL page, not `/`.
774
+ //
775
+ // "Something answers on :5050" is not the same as "our pages are being served", and the difference
776
+ // has already cost a dataset: a stray server on that port reported "Capture complete: 3/3 cases"
777
+ // while every transcript read "Error code: 404". A leftover `npx serve` from another directory
778
+ // answers the root happily and 404s every case. Four of them were running on this host today, which
779
+ // is how likely that is.
780
+ const sample = JSON.parse(readFileSync(manifestPath, "utf8")).cases?.[0]?.id;
781
+ const probe = sample ? `${sample}/good.html` : "";
782
+ try {
783
+ const response = await fetch(`http://localhost:${PAGES_PORT}/${probe}`, { signal: AbortSignal.timeout(PROBE_TIMEOUT_MS) });
784
+ if (!response.ok) {
785
+ return add("pages", false, `:${PAGES_PORT} answers but returns HTTP ${response.status} for ${probe} — wrong directory`, `something else holds the port. Stop it; training:capture will lease the dataset server automatically`);
786
+ }
787
+ add("pages", true, `serving the dataset on :${PAGES_PORT} (verified ${probe || "/"})`);
788
+ }
789
+ catch {
790
+ add("pages", true, `nothing serving on :${PAGES_PORT} — training:capture leases it automatically`);
791
+ }
792
+ }
793
+ // A run left mid-flight is the difference between "start" and "--resume", and getting it
794
+ // wrong either re-captures for hours or silently skips work.
795
+ function checkRunState() {
796
+ const progress = resolve(DATASET, "capture-progress.json");
797
+ if (!existsSync(progress))
798
+ return add("run", true, "no capture run recorded");
799
+ const p = JSON.parse(readFileSync(progress, "utf8"));
800
+ if (!p.startedAt)
801
+ return add("run", true, "no capture run recorded");
802
+ if (!p.finishedAt) {
803
+ return add("run", false, `a run is UNFINISHED (started ${p.startedAt})`, "pnpm run training:wait, or pnpm run training:capture -- --resume --no-cache");
804
+ }
805
+ const failed = Object.values(p.cases ?? {}).filter((c) => c.status === "failed").length;
806
+ add("run", failed === 0, `last run ${p.outcome ?? "finished"}`, failed ? "pnpm run training:capture -- --resume --no-cache" : null);
807
+ }
808
+ // --- report ---------------------------------------------------------------
809
+ // The single most useful line for anything automated: what to run next. A list of green ticks
810
+ // still leaves a caller deciding, and deciding is where they go wrong.
811
+ /**
812
+ * #1059: A COMMAND, OR `null`. Never a sentence.
813
+ *
814
+ * `--json`'s `next_command` is the field CLAUDE.md tells an agent to read and obey, so anything in it that
815
+ * is not runnable makes an automated reader loop -- which is exactly what happened when a failing `worker`
816
+ * check put *"unlock the Mac if it is locked, then re-run …"* here.
817
+ *
818
+ * **`null` and an unrunnable string are different reports**, and a JSON consumer can act on the first: it
819
+ * means read the checks. `contention` has no command because waiting is not one, and it says so with
820
+ * `null` and a `note` rather than with an imperative nobody can execute.
821
+ * THE PARAMETER TYPE IS WHAT THIS FUNCTION READS, not the whole check shape: `ok` and `fix`, and nothing
822
+ * else. A test injecting a two-field object is stating exactly the inputs the verdict depends on, and a
823
+ * wider type would have made it carry an `id` and a `detail` the answer cannot possibly turn on.
824
+ * @param {{ok: boolean, fix: string|null}[]} [checkList]
825
+ * @returns {string|null}
826
+ */
827
+ export function nextCommand(checkList = checks) {
828
+ const broken = checkList.find((c) => !c.ok);
829
+ if (!broken)
830
+ return "pnpm run training:capture";
831
+ return broken.fix ?? null;
832
+ }
833
+ /**
834
+ * Is `line` something a shell can run? The shape `next_command` must have, and prose must not.
835
+ *
836
+ * #1059: the field carried *"unlock the Mac if it is locked, then re-run …"*, and CLAUDE.md tells an agent
837
+ * to read it and do that. A shape check is the only thing that can tell a command from an imperative
838
+ * sentence without running it: **a command begins with an executable token** — a pnpm/npm/node/git invocation,
839
+ * a path, or a `VAR=value` prefix — **and an English sentence begins with a verb or an article.**
840
+ * @param {string|null} line
841
+ * @returns {boolean}
842
+ */
843
+ export function isRunnableCommand(line) {
844
+ if (line === null)
845
+ return false; // absent is not unrunnable; the caller distinguishes them
846
+ const first = line.trim().split(/\s+/)[0] ?? "";
847
+ return /^(?:[A-Z][A-Z0-9_]*=\S*|pnpm|npm|npx|node|git|\.?\.?\/\S+|[a-z0-9_-]+\.(?:sh|mjs|js|ts|py))$/.test(first);
848
+ }
849
+ /**
850
+ * Only when RUN, never on import.
851
+ *
852
+ * Every check here probes something real -- it spawns the Python scorer, polls each worker's `/health`,
853
+ * looks for strays on the pages port and reads the run's progress file -- and then calls `process.exit`.
854
+ * So importing this file ran the whole diagnostic against the fleet and then terminated the IMPORTING
855
+ * process with doctor's verdict.
856
+ *
857
+ * A brace-depth scan for dangerous calls at module scope reports this file CLEAN, because the work is one
858
+ * call deeper inside `checkJudge`/`checkWorker`/`checkDatasetPages`. Indirection is that check's blind
859
+ * spot, which is why these guards were placed by reading each file rather than by running a tool over them.
860
+ */
861
+ /**
862
+ * `ready` over the checks that GATE. Exported so the case nobody runs -- a clean clone whose only failing
863
+ * check is `dataset` -- is drivable without a clean clone.
864
+ * @param {{name: string, ok: boolean}[]} checkList
865
+ * @returns {boolean}
866
+ */
867
+ export function readyFrom(checkList) {
868
+ return checkList.every((c) => c.ok || GATES[c.name] === false);
869
+ }
870
+ /** Every DECLARED check name, gating or not -- so a test can compare the two sets without a literal. */
871
+ export const allChecks = () => Object.keys(GATES);
872
+ /** Which checks decide `ready`, for a test that must not retype the list. */
873
+ export const gatingChecks = () => Object.entries(GATES).filter(([, gates]) => gates).map(([name]) => name);
874
+ /** The checks `doctor` runs, in order. Injectable so a throwing one can be driven without breaking a tree. */
875
+ const DEFAULT_STEPS = [checkPrimaryCheckoutMark, checkControlPlaneIsolation, checkCrossPackageDist,
876
+ checkJudge, checkWorker, checkDatasetPages, checkRunState];
877
+ /**
878
+ * #1082: A `--json` RUN THAT CANNOT PRODUCE JSON STILL PRODUCES JSON.
879
+ *
880
+ * Measured on `1e74e3d0`: a check threw, `doctor --json` exited 1 with **zero bytes on stdout** and 1,155
881
+ * bytes of stack on stderr — where a `--json` consumer never looks. **"Could not ask" and "no output" are
882
+ * different for a caller**, and only the first is actionable; the second is indistinguishable from a
883
+ * command that was never run.
884
+ *
885
+ * `2>&1` IS NOT THE FIX HERE, which is what makes this different from #1068's watch job. That job's stdout
886
+ * is prose, so merging stderr in was free. This stdout is a PARSED format, and redirecting into it
887
+ * produces invalid JSON — **worse than nothing, because a consumer that parses gets a syntax error rather
888
+ * than a document.** The fix has to be in the tool.
889
+ *
890
+ * `checks` is present and EMPTY rather than absent OR PARTIAL — and the partial list is the one actually
891
+ * worth refusing. A consumer reading `.checks[]` off an absent key crashes; off a partial one it reads a
892
+ * list that looks exactly like a complete verdict, with **no way to tell a check that is missing because
893
+ * it passed from one that is missing because the run died under it.** Empty says "no verdict" and cannot
894
+ * be mistaken for a short one.
895
+ * @param {unknown} error
896
+ * @returns {{ready: false, error: string, checks: never[]}}
897
+ */
898
+ export function errorDocument(error) {
899
+ const message = error instanceof Error ? error.message : String(error);
900
+ return { ready: false, error: message, checks: [] };
901
+ }
902
+ /**
903
+ * The whole run, as a function of its steps and its streams, returning an exit code.
904
+ *
905
+ * @param {{ steps?: (() => unknown)[], json?: boolean, out?: (line: string) => void,
906
+ * err?: (line: string) => void }} [deps]
907
+ * @returns {Promise<number>}
908
+ */
909
+ export async function doctorRun(deps = {}) {
910
+ const { steps = DEFAULT_STEPS, json = JSON_OUT, out = console.log, err = console.error } = deps;
911
+ try {
912
+ for (const step of steps)
913
+ await step();
914
+ }
915
+ catch (error) {
916
+ // NOTHING BUT JSON REACHES STDOUT IN `--json` MODE. A `console.log` here would corrupt the document
917
+ // for every consumer, which is the failure this exists to prevent rather than to introduce.
918
+ if (json)
919
+ out(JSON.stringify(errorDocument(error), null, 2));
920
+ // The HUMAN path keeps the stack. Only the parsed format has to give it up, and it gives it up for a
921
+ // document a consumer can read -- not to make the failure quieter.
922
+ else
923
+ err(error instanceof Error && error.stack ? error.stack : `doctor: ${errorDocument(error).error}`);
924
+ return 1;
925
+ }
926
+ return renderDoctor({ json, out });
927
+ }
928
+ /**
929
+ * @param {{ json: boolean, out: (line: string) => void }} deps
930
+ * @returns {number}
931
+ */
932
+ function renderDoctor({ json, out }) {
933
+ const ready = readyFrom(checks);
934
+ if (json) {
935
+ out(JSON.stringify({ ready, next_command: nextCommand(), checks }, null, 2));
936
+ }
937
+ else {
938
+ for (const c of checks) {
939
+ out(`${c.advisory ? "DEBT" : c.ok ? "OK " : "FAIL"} ${c.name.padEnd(11)} ${c.detail}`);
940
+ if ((!c.ok || c.advisory) && c.fix)
941
+ out(` fix: ${c.fix}`);
942
+ // #1059: the advice a shell cannot run is PRINTED, just not in the field something executes.
943
+ if ((!c.ok || c.advisory) && c.note)
944
+ out(` note: ${c.note}`);
945
+ }
946
+ out(`\n${ready ? "READY" : "NOT READY — see the fixes above"}`);
947
+ const next = nextCommand();
948
+ // NO COMMAND IS SAID AS SUCH. "next: null" would read as a bug; "no single command" is the answer.
949
+ out(next === null ? "next: no single command — read the fixes and notes above" : `next: ${next}`);
950
+ }
951
+ return ready ? 0 : 1;
952
+ }
953
+ async function main() {
954
+ process.exit(await doctorRun());
955
+ }
956
+ // REALPATH'D: `import.meta.url` is resolved through symlinks by Node's ESM loader and `process.argv[1]`
957
+ // is not, so a bin reached via its `.bin` symlink (which is how npm always installs one) mismatched here
958
+ // and this guard silently read false — the tool loaded, did nothing, and exited 0. `/var` and `/tmp` are
959
+ // themselves symlinks on macOS, so this fired every time. Same defect, same fix, as `cli.ts`'s `isProgram`.
960
+ if (import.meta.url === pathToFileURL(process.argv[1] ? realpathSync(process.argv[1]) : "").href)
961
+ await main();
962
+ //# sourceMappingURL=doctor.mjs.map