@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/LICENSE +661 -0
  2. package/README.md +94 -2
  3. package/dist/capture-client.d.mts +49 -0
  4. package/dist/capture-client.d.mts.map +1 -0
  5. package/dist/capture-client.mjs +352 -0
  6. package/dist/capture-client.mjs.map +1 -0
  7. package/dist/check-worker-code.d.mts +34 -0
  8. package/dist/check-worker-code.d.mts.map +1 -0
  9. package/dist/check-worker-code.mjs +142 -0
  10. package/dist/check-worker-code.mjs.map +1 -0
  11. package/dist/cli-flags.d.mts +71 -0
  12. package/dist/cli-flags.d.mts.map +1 -0
  13. package/dist/cli-flags.mjs +207 -0
  14. package/dist/cli-flags.mjs.map +1 -0
  15. package/dist/code-drift.d.mts +140 -0
  16. package/dist/code-drift.d.mts.map +1 -0
  17. package/dist/code-drift.mjs +284 -0
  18. package/dist/code-drift.mjs.map +1 -0
  19. package/dist/command-line-census.d.mts +33 -0
  20. package/dist/command-line-census.d.mts.map +1 -0
  21. package/dist/command-line-census.mjs +96 -0
  22. package/dist/command-line-census.mjs.map +1 -0
  23. package/dist/compare-workers.d.mts +3 -0
  24. package/dist/compare-workers.d.mts.map +1 -0
  25. package/dist/compare-workers.mjs +332 -0
  26. package/dist/compare-workers.mjs.map +1 -0
  27. package/dist/control-plane-isolation.d.mts +45 -0
  28. package/dist/control-plane-isolation.d.mts.map +1 -0
  29. package/dist/control-plane-isolation.mjs +67 -0
  30. package/dist/control-plane-isolation.mjs.map +1 -0
  31. package/dist/deploy-worker.d.mts +3 -0
  32. package/dist/deploy-worker.d.mts.map +1 -0
  33. package/dist/deploy-worker.mjs +333 -0
  34. package/dist/deploy-worker.mjs.map +1 -0
  35. package/dist/doctor.d.mts +216 -0
  36. package/dist/doctor.d.mts.map +1 -0
  37. package/dist/doctor.mjs +962 -0
  38. package/dist/doctor.mjs.map +1 -0
  39. package/dist/fleet-consistency.d.mts +235 -0
  40. package/dist/fleet-consistency.d.mts.map +1 -0
  41. package/dist/fleet-consistency.mjs +436 -0
  42. package/dist/fleet-consistency.mjs.map +1 -0
  43. package/dist/fleet-env.d.mts +228 -0
  44. package/dist/fleet-env.d.mts.map +1 -0
  45. package/dist/fleet-env.mjs +509 -0
  46. package/dist/fleet-env.mjs.map +1 -0
  47. package/dist/fleet-scripts.d.mts +11 -0
  48. package/dist/fleet-scripts.d.mts.map +1 -0
  49. package/dist/fleet-scripts.mjs +41 -0
  50. package/dist/fleet-scripts.mjs.map +1 -0
  51. package/dist/git-safe-env.d.mts +10 -0
  52. package/dist/git-safe-env.d.mts.map +1 -0
  53. package/dist/git-safe-env.mjs +44 -0
  54. package/dist/git-safe-env.mjs.map +1 -0
  55. package/dist/guest-run.d.mts +26 -0
  56. package/dist/guest-run.d.mts.map +1 -0
  57. package/dist/guest-run.mjs +164 -0
  58. package/dist/guest-run.mjs.map +1 -0
  59. package/dist/host-address.d.mts +33 -0
  60. package/dist/host-address.d.mts.map +1 -0
  61. package/dist/host-address.mjs +105 -0
  62. package/dist/host-address.mjs.map +1 -0
  63. package/dist/host-capacity.d.mts +64 -0
  64. package/dist/host-capacity.d.mts.map +1 -0
  65. package/dist/host-capacity.mjs +152 -0
  66. package/dist/host-capacity.mjs.map +1 -0
  67. package/dist/host-metrics.d.mts +116 -0
  68. package/dist/host-metrics.d.mts.map +1 -0
  69. package/dist/host-metrics.mjs +201 -0
  70. package/dist/host-metrics.mjs.map +1 -0
  71. package/dist/index.d.ts +23 -0
  72. package/dist/index.d.ts.map +1 -0
  73. package/dist/index.js +25 -0
  74. package/dist/index.js.map +1 -0
  75. package/dist/local-vm.d.ts +125 -0
  76. package/dist/local-vm.d.ts.map +1 -0
  77. package/dist/local-vm.js +360 -0
  78. package/dist/local-vm.js.map +1 -0
  79. package/dist/measure-guard.d.mts +34 -0
  80. package/dist/measure-guard.d.mts.map +1 -0
  81. package/dist/measure-guard.mjs +73 -0
  82. package/dist/measure-guard.mjs.map +1 -0
  83. package/dist/normalise-fleet.d.mts +2 -0
  84. package/dist/normalise-fleet.d.mts.map +1 -0
  85. package/dist/normalise-fleet.mjs +76 -0
  86. package/dist/normalise-fleet.mjs.map +1 -0
  87. package/dist/npm-cli-executable.d.mts +42 -0
  88. package/dist/npm-cli-executable.d.mts.map +1 -0
  89. package/dist/npm-cli-executable.mjs +159 -0
  90. package/dist/npm-cli-executable.mjs.map +1 -0
  91. package/dist/probe-outcome.d.mts +89 -0
  92. package/dist/probe-outcome.d.mts.map +1 -0
  93. package/dist/probe-outcome.mjs +104 -0
  94. package/dist/probe-outcome.mjs.map +1 -0
  95. package/dist/protocol-guard.d.mts +34 -0
  96. package/dist/protocol-guard.d.mts.map +1 -0
  97. package/dist/protocol-guard.mjs +121 -0
  98. package/dist/protocol-guard.mjs.map +1 -0
  99. package/dist/source-walk.d.mts +12 -0
  100. package/dist/source-walk.d.mts.map +1 -0
  101. package/dist/source-walk.mjs +56 -0
  102. package/dist/source-walk.mjs.map +1 -0
  103. package/dist/transient-fault.d.mts +6 -0
  104. package/dist/transient-fault.d.mts.map +1 -0
  105. package/dist/transient-fault.mjs +86 -0
  106. package/dist/transient-fault.mjs.map +1 -0
  107. package/dist/utm-deprecated.d.mts +6 -0
  108. package/dist/utm-deprecated.d.mts.map +1 -0
  109. package/dist/utm-deprecated.mjs +23 -0
  110. package/dist/utm-deprecated.mjs.map +1 -0
  111. package/dist/worker-code-check.d.mts +29 -0
  112. package/dist/worker-code-check.d.mts.map +1 -0
  113. package/dist/worker-code-check.mjs +85 -0
  114. package/dist/worker-code-check.mjs.map +1 -0
  115. package/dist/worker-health.d.mts +56 -0
  116. package/dist/worker-health.d.mts.map +1 -0
  117. package/dist/worker-health.mjs +73 -0
  118. package/dist/worker-health.mjs.map +1 -0
  119. package/dist/worker-http.d.mts +103 -0
  120. package/dist/worker-http.d.mts.map +1 -0
  121. package/dist/worker-http.mjs +277 -0
  122. package/dist/worker-http.mjs.map +1 -0
  123. package/dist/worker-stats.d.mts +66 -0
  124. package/dist/worker-stats.d.mts.map +1 -0
  125. package/dist/worker-stats.mjs +143 -0
  126. package/dist/worker-stats.mjs.map +1 -0
  127. package/package.json +96 -4
  128. package/src/local-worker/autounattend.xml +280 -0
  129. package/src/local-worker/build-vm.sh +218 -0
  130. package/src/local-worker/clone-worker.sh +141 -0
  131. package/src/local-worker/create-utm-vm.sh +202 -0
  132. package/src/local-worker/fetch-windows-iso.sh +238 -0
  133. package/src/local-worker/first-boot.cmd +58 -0
  134. package/src/local-worker/worker-ctl.sh +442 -0
  135. package/src/provisioning/README.md +28 -0
  136. package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
  137. package/src/provisioning/bare-metal/README.md +213 -0
  138. package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
  139. package/src/provisioning/bare-metal/autounattend.xml +428 -0
  140. package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
  141. package/src/provisioning/bootstrap-control-plane.sh +463 -0
  142. package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
  143. package/src/provisioning/build-lean-worker-image.ps1 +275 -0
  144. package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
  145. package/src/provisioning/provision-nvda-worker.ps1 +827 -0
  146. package/src/provisioning/set-display-mode.ps1 +411 -0
  147. package/src/provisioning/stamp-provision-revision.ps1 +184 -0
@@ -0,0 +1,463 @@
1
+ #!/usr/bin/env bash
2
+ # Stand up the CONTROL PLANE on Linux, so no run depends on a particular laptop.
3
+ #
4
+ # curl -fsSL <raw-url>/bootstrap-control-plane.sh | bash
5
+ #
6
+ # The pair to bootstrap-windows-worker.ps1: that one turns a Windows box into a capture worker,
7
+ # this one turns a Debian/Ubuntu box (an LXC on Proxmox, say) into the thing that drives them.
8
+ #
9
+ # ## Why this exists
10
+ #
11
+ # ADR 0001 says it already: "The control plane is portable; only capture workers are OS-bound."
12
+ # Portable meant *could*, not *does* — it ran on one Mac, and that Mac was in the path of every
13
+ # corpus run. Three separate ways that bit, all in one day:
14
+ #
15
+ # - macOS 26 blocks node from the local network by default, so every worker call failed with
16
+ # EHOSTUNREACH while curl and python worked fine. A privacy toggle stopped the fleet.
17
+ # - the dataset page server ran there, so the pages a capture reads lived on a laptop.
18
+ # - a four-hour corpus run needed that laptop awake, on the same network, unslept.
19
+ #
20
+ # ## What it does NOT need
21
+ #
22
+ # No utmctl, no VM management, none of the macOS-bound half of worker-fleet. Verified rather than
23
+ # hoped: `leaseWorker` returns at its first line when a worker is named explicitly, and
24
+ # `capture-screenreader-dataset.mjs` returns the explicit pool before `leaseWorkerPool` is called.
25
+ # So with A11Y_WORKERS set, the managed-VM path is never entered and nothing macOS-only runs.
26
+ #
27
+ # That is why this is a deployment rather than a port: `packages/lab` — the orchestrator — contains
28
+ # no macOS-only command at all.
29
+ #
30
+ # A11Y_REPO_URL default the public GitHub repo
31
+ # A11Y_REPO_PATH default ~/a11y-witness
32
+ # A11Y_WORKERS comma-separated worker URLs, e.g. http://192.0.2.10:8765
33
+ # A11Y_CORPUS_URL optional tar.gz of runs/ to seed the baseline corpus (69 MB at time of writing)
34
+ set -euo pipefail
35
+
36
+ REPO_URL="${A11Y_REPO_URL:-https://github.com/a11ign/a11ign.git}"
37
+ REPO_PATH="${A11Y_REPO_PATH:-$HOME/a11y-witness}"
38
+
39
+ # WHICH HALF OF THE CONTROL PLANE IS THIS? (A11Y_ROLE=control|lab, default both)
40
+ #
41
+ # ADR 0012 splits them, and the reason is credentials rather than tidiness: the SSH key that can
42
+ # reconfigure twelve Windows machines should not sit next to 100 MB of transitive dependencies and a
43
+ # Python venv, which are the largest supply-chain surface in the system.
44
+ #
45
+ # control ansible + the fleet key. No node_modules, no venv, no corpus. Rebuildable in a minute.
46
+ # lab pnpm install + venv + the corpus. Talks to workers over HTTP only. Holds NO key.
47
+ #
48
+ # `both` remains the default so a single-box setup still works and nobody is forced into two containers
49
+ # on day one -- but it is the thing to grow out of, not the target.
50
+ ROLE="${A11Y_ROLE:-both}"
51
+ case "$ROLE" in
52
+ control|lab|both) ;;
53
+ *) echo "A11Y_ROLE must be control, lab or both (got '$ROLE')" >&2; exit 1 ;;
54
+ esac
55
+ is_control() { [ "$ROLE" = control ] || [ "$ROLE" = both ]; }
56
+ is_lab() { [ "$ROLE" = lab ] || [ "$ROLE" = both ]; }
57
+
58
+ step() { printf '\n\033[36m[%s] %s\033[0m\n' "$1" "$2"; }
59
+ ok() { printf ' \033[32mOK %s\033[0m\n' "$1"; }
60
+ warn() { printf ' \033[33mWARN %s\033[0m\n' "$1"; }
61
+
62
+ step 1 'Preconditions'
63
+ [ "$(uname -s)" = "Linux" ] || { echo "This is the Linux control plane; run it on the host that will drive the workers." >&2; exit 1; }
64
+ # shellcheck disable=SC1091 # /etc/os-release is a runtime file, not an input to lint
65
+ if . /etc/os-release 2>/dev/null && [ -n "${PRETTY_NAME:-}" ]; then ok "$PRETTY_NAME"; else ok "$(uname -sr)"; fi
66
+
67
+ # Root or sudo, but do not assume either: an LXC console is usually already root, and demanding
68
+ # sudo there fails on a box that does not have it installed.
69
+ SUDO=""
70
+ # A SEPARATE variable for the environment-preserving form, not `$SUDO -E`. When SUDO is empty --
71
+ # which is the normal case, because an LXC console is root -- `$SUDO -E cmd` leaves `-E` as the
72
+ # command word, and the shell reports `-E: command not found`. That killed this script at the
73
+ # NodeSource step on both containers: `set -euo pipefail` aborted correctly, so it failed loudly
74
+ # rather than half-installing, but the cause reads as a missing binary rather than a quoting bug.
75
+ SUDO_E=""
76
+ if [ "$(id -u)" -ne 0 ]; then
77
+ command -v sudo >/dev/null || { echo "Not root and no sudo. Run as root." >&2; exit 1; }
78
+ SUDO="sudo"
79
+ SUDO_E="sudo -E"
80
+ fi
81
+ if [ -n "$SUDO" ]; then ok 'using sudo'; else ok 'running as root'; fi
82
+ ok "role: $ROLE"
83
+
84
+
85
+ step 2 'Node.js and git'
86
+ if command -v node >/dev/null && node -e 'process.exit(process.versions.node.split(".")[0] >= 20 ? 0 : 1)'; then
87
+ ok "node already present ($(node --version))"
88
+ else
89
+ # NodeSource rather than the distro package: Debian ships a node far older than this repo needs,
90
+ # and a version skew here surfaces as syntax errors in the orchestrator rather than as a version
91
+ # complaint.
92
+ $SUDO apt-get update -qq
93
+ $SUDO apt-get install -y -qq curl ca-certificates gnupg >/dev/null
94
+ curl -fsSL https://deb.nodesource.com/setup_lts.x | $SUDO_E bash - >/dev/null
95
+ $SUDO apt-get install -y -qq nodejs >/dev/null
96
+ ok "node installed ($(node --version))"
97
+ fi
98
+ command -v git >/dev/null || { $SUDO apt-get install -y -qq git >/dev/null; }
99
+ ok "git $(git --version | awk '{print $3}')"
100
+
101
+ step 3 'Repository'
102
+ if [ -d "$REPO_PATH/.git" ]; then
103
+ git -C "$REPO_PATH" pull --ff-only
104
+ ok "pulled $REPO_PATH ($(git -C "$REPO_PATH" rev-parse --short HEAD))"
105
+ else
106
+ git clone --quiet "$REPO_URL" "$REPO_PATH"
107
+ ok "cloned to $REPO_PATH ($(git -C "$REPO_PATH" rev-parse --short HEAD))"
108
+ fi
109
+ # A LAYER IN ITS OWN REPOSITORY IS A SECOND CHECKOUT BESIDE THE CORE'S (ADR 0039 item 6, row 6c, #3396). The core
110
+ # checkout above is one repository; a layer that `packages/control/layers.json` gives a `remote` lives in another,
111
+ # at its declared path inside this one, and the control plane and the lab both read it there. This clones it when
112
+ # it is absent and FETCHES it when it is there, and nothing more: where it STANDS is the pair's second half, which
113
+ # `fleet:deploy --layer-ref=` and a lab job's `layer_refs` set and read back, so a pull here would be a third
114
+ # thing moving it. With no layer that declares a `remote` the loop below has nothing to do, which is today.
115
+ #
116
+ # It REFUSES to clone over a directory that is not a git checkout: the monorepo's own copy of the layer is one,
117
+ # and a clone on top of it would be the core's tree answering for the layer. The path is excluded from the core's
118
+ # `git status` (`.git/info/exclude`, which is local and never committed), because a nested clone is otherwise
119
+ # `??` there, which reads as "somebody is working in the lab checkout" and stops every job from pulling.
120
+ LAYER_ROWS="$(node -e '
121
+ const layers = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).layers;
122
+ for (const [name, l] of Object.entries(layers)) {
123
+ if (!l.remote) continue;
124
+ if (!/^[A-Za-z0-9._\/-]+$/.test(l.path) || l.path.includes("..")) { console.error(`layer ${name}: path ${l.path} is not a plain relative path`); process.exit(1); }
125
+ if (!/^https:\/\/[A-Za-z0-9._\/-]+\.git$/.test(l.remote)) { console.error(`layer ${name}: remote ${l.remote} is not an https .git URL`); process.exit(1); }
126
+ console.log([name, l.path, l.remote, l.branch || ""].join("\t"));
127
+ }' "$REPO_PATH/packages/control/layers.json")"
128
+ while IFS=$'\t' read -r LAYER_NAME LAYER_DIR LAYER_REMOTE LAYER_BRANCH; do
129
+ [ -n "$LAYER_NAME" ] || continue
130
+ LAYER_PATH="$REPO_PATH/$LAYER_DIR"
131
+ if [ -d "$LAYER_PATH/.git" ]; then
132
+ git -C "$LAYER_PATH" fetch --quiet origin
133
+ ok "layer $LAYER_NAME fetched at $LAYER_DIR (on $(git -C "$LAYER_PATH" rev-parse --short HEAD); the pin moves it, not this)"
134
+ elif [ -e "$LAYER_PATH" ] && [ -n "$(ls -A "$LAYER_PATH" 2>/dev/null)" ]; then
135
+ echo "layer $LAYER_NAME is declared at $LAYER_DIR and $LAYER_PATH exists and is not a git checkout; refusing to clone over it." >&2
136
+ exit 1
137
+ else
138
+ git clone --quiet ${LAYER_BRANCH:+--branch "$LAYER_BRANCH"} "$LAYER_REMOTE" "$LAYER_PATH"
139
+ ok "layer $LAYER_NAME cloned to $LAYER_DIR ($(git -C "$LAYER_PATH" rev-parse --short HEAD))"
140
+ fi
141
+ mkdir -p "$REPO_PATH/.git/info"
142
+ grep -qxF "/$LAYER_DIR/" "$REPO_PATH/.git/info/exclude" 2>/dev/null || printf '/%s/\n' "$LAYER_DIR" >> "$REPO_PATH/.git/info/exclude"
143
+ done <<< "$LAYER_ROWS"
144
+ cd "$REPO_PATH"
145
+ if is_lab; then
146
+ # #2890: `corepack pnpm install --frozen-lockfile`, the SAME spelling `roles/worker/tasks/nvda.yml` uses
147
+ # (pinned by `provisioning-installs-with-pnpm.test.ts`). `packageManager` in package.json names the pnpm
148
+ # version and `pnpm-lock.yaml` is the specification: a plain install here resolved versions the lockfile
149
+ # never named, and a lab built that way captured evidence no other machine could reproduce.
150
+ command -v corepack >/dev/null || {
151
+ echo "corepack is not on PATH (Node 25+ no longer ships it). Use a Node that does (24 LTS), or install corepack from its own distribution" >&2
152
+ exit 1
153
+ }
154
+ export COREPACK_ENABLE_DOWNLOAD_PROMPT=0
155
+ # ONE-TIME MIGRATION: a tree npm made carries `node_modules/.package-lock.json`, and pnpm installed over it
156
+ # leaves every npm-hoisted package in place -- the hoisting that lets an undeclared import resolve.
157
+ if [ -f node_modules/.package-lock.json ]; then
158
+ ok 'node_modules was made by npm: removing it once, so nothing npm hoisted survives beside pnpm'
159
+ rm -rf "${REPO_PATH:?}/node_modules"
160
+ fi
161
+ corepack pnpm install --frozen-lockfile --silent
162
+ # The remedies this script prints below say `pnpm run ...`, so make `pnpm` a command an operator can type.
163
+ # `corepack enable` writes beside node, which needs root where node came from a package.
164
+ $SUDO corepack enable pnpm
165
+ ok 'dependencies installed'
166
+
167
+ # The LOCAL scorer is the default judge and the only one that ships (JUDGE_BACKEND defaults to
168
+ # `local`), so a lab without Python is a lab that cannot score. This step was missing entirely, and
169
+ # nothing here failed: `witness`, `eval` and `training:score` all name `.venv/bin/python` explicitly,
170
+ # so the absence surfaces as a missing interpreter partway through a run rather than at setup.
171
+ $SUDO apt-get install -y -qq python3-venv >/dev/null
172
+ [ -d "$REPO_PATH/.venv" ] || python3 -m venv "$REPO_PATH/.venv"
173
+ "$REPO_PATH/.venv/bin/pip" install -q --upgrade pip
174
+ "$REPO_PATH/.venv/bin/pip" install -q -r "$REPO_PATH/packages/scorer/requirements.txt"
175
+ ok "python venv ready ($("$REPO_PATH/.venv/bin/python" --version 2>&1))"
176
+
177
+ # The trained heads ARE tracked (796 KB); the MiniLM encoder is 87 MB and deliberately is not. It is a
178
+ # public model pinned by BOTH revision and sha256 in fetch-encoder.py, so fetching rather than
179
+ # vendoring it is still reproducible -- which is why .gitignore excludes `models/encoders/`.
180
+ if [ -f "$REPO_PATH/packages/scorer/models/encoders/all-MiniLM-L6-v2/model.safetensors" ]; then
181
+ ok 'encoder already present'
182
+ else
183
+ (cd "$REPO_PATH" && .venv/bin/python packages/scorer/python/fetch-encoder.py >/dev/null 2>&1) \
184
+ && ok 'encoder fetched (pinned revision, sha256-verified)' \
185
+ || warn 'could not fetch the encoder -- the local judge cannot score until it is present'
186
+ fi
187
+ else
188
+ # The control container deliberately has NO node_modules. deploy.yml computes codeVersion by importing
189
+ # code-version.mjs BY PATH -- it needs nothing but node stdlib, and verified identical to the workspace
190
+ # import. 100 MB of transitive dependencies next to the fleet's SSH key is the coupling ADR 0012 removes.
191
+ ok 'skipped (control role) -- no node_modules beside the fleet key'
192
+ fi
193
+
194
+ step 4 'Ansible, and the fleet key'
195
+ if ! is_control; then
196
+ ok 'skipped (lab role) -- the lab holds NO fleet key and runs no Ansible, by design (ADR 0012)'
197
+ else
198
+ # The control plane MANAGES the workers as well as capturing with them, and this script predates that
199
+ # half entirely -- it installed node and a checkout and left you without the thing that provisions,
200
+ # deploys, wakes and sleeps a box.
201
+ #
202
+ # pipx rather than apt: Windows-over-SSH support is ansible-core 2.18+, and Debian ships older. That
203
+ # version gap is not cosmetic -- on an older core every Windows task fails at connection time.
204
+ #
205
+ # The REQUIREMENT is enforced here rather than documented here. It used to be neither: this block installed
206
+ # `ansible-core` with no version constraint, and the "already present" branch accepted whatever was on the
207
+ # box -- so the one case the comment above warns about, an older core, passed the check silently and failed
208
+ # later at connection time, which reads like a broken worker rather than a stale controller.
209
+ #
210
+ # `collections/ansible_collections/a11y/worker/meta/runtime.yml` states the same floor as
211
+ # `requires_ansible`, so Ansible itself refuses the collection on an older core. This makes the bootstrap
212
+ # agree with it instead of leaving the two to drift.
213
+ ANSIBLE_MIN="2.18"
214
+
215
+ ansible_core_version() { ansible --version 2>/dev/null | head -1 | grep -oE '[0-9]+\.[0-9]+(\.[0-9]+)?' | head -1; }
216
+ # Sort-based compare, so 2.9 does not read as newer than 2.18 the way a string compare would. That is the
217
+ # specific wrong answer this guard has to avoid, because 2.9 is exactly what Debian ships.
218
+ version_at_least() { [ "$(printf '%s\n%s\n' "$2" "$1" | sort -V | head -1)" = "$2" ]; }
219
+
220
+ if command -v ansible-playbook >/dev/null; then
221
+ found="$(ansible_core_version)"
222
+ if [ -n "$found" ] && version_at_least "$found" "$ANSIBLE_MIN"; then
223
+ ok "ansible already present (core $found, >= $ANSIBLE_MIN)"
224
+ else
225
+ warn "ansible core ${found:-unknown} is older than $ANSIBLE_MIN — every Windows task will fail at"
226
+ warn "connection time, which looks like a broken worker rather than a stale controller. Upgrade with:"
227
+ warn " pipx upgrade ansible-core || pipx install --force 'ansible-core>=$ANSIBLE_MIN'"
228
+ fi
229
+ else
230
+ $SUDO apt-get install -y -qq pipx >/dev/null 2>&1 || $SUDO apt-get install -y -qq python3-pip >/dev/null
231
+ if command -v pipx >/dev/null; then
232
+ pipx install "ansible-core>=$ANSIBLE_MIN" >/dev/null
233
+ pipx ensurepath >/dev/null 2>&1 || true
234
+ else
235
+ $SUDO pip3 install --break-system-packages -q "ansible-core>=$ANSIBLE_MIN"
236
+ fi
237
+ export PATH="$HOME/.local/bin:$PATH"
238
+ ok "ansible installed ($(ansible --version 2>/dev/null | head -1))"
239
+ fi
240
+
241
+ # -p is load-bearing: ansible.cfg puts the repo's own collections path FIRST, so a bare install vendors
242
+ # third-party collections into the git tree. Only a11y.worker belongs there.
243
+ export PATH="$HOME/.local/bin:$PATH"
244
+ if [ -f "$REPO_PATH/packages/control/ansible/requirements.yml" ]; then
245
+ ansible-galaxy collection install -r "$REPO_PATH/packages/control/ansible/requirements.yml" \
246
+ -p "$HOME/.ansible/collections" >/dev/null
247
+ ok 'collections installed (ansible.windows, community.windows, community.general)'
248
+ else
249
+ warn 'requirements.yml not found -- is the checkout complete?'
250
+ fi
251
+
252
+ # #1870: `fleet-playbook.mjs`'s fleet-hold check (#1839/#1841) shells out to `gh` to ask whether an open
253
+ # `fleet-gated` row carries an unexpired `Fleet-hold-until:` before `fleet:deploy`/`fleet:provision`
254
+ # proceed -- so `gh` is now a CONTROL-role dependency, not just something an interactive session happens
255
+ # to have. Debian ships no `gh` package at all (unlike Node, this is not a version-skew problem, it is a
256
+ # missing package), so this follows GitHub's own published apt repository rather than guessing at a distro
257
+ # package name that does not exist.
258
+ if command -v gh >/dev/null; then
259
+ ok "gh already present ($(gh --version | head -1))"
260
+ else
261
+ $SUDO mkdir -p -m 755 /etc/apt/keyrings
262
+ curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg \
263
+ | $SUDO tee /etc/apt/keyrings/githubcli-archive-keyring.gpg >/dev/null
264
+ $SUDO chmod go+r /etc/apt/keyrings/githubcli-archive-keyring.gpg
265
+ echo "deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages/. stable main" \
266
+ | $SUDO tee /etc/apt/sources.list.d/github-cli.list >/dev/null
267
+ $SUDO apt-get update -qq
268
+ $SUDO apt-get install -y -qq gh >/dev/null
269
+ ok "gh installed ($(gh --version | head -1))"
270
+ fi
271
+
272
+ # #1875: gh on its own is not enough -- it refuses to run unauthenticated, so the fleet-hold check still
273
+ # fails closed. `ceo`'s ruling on that row: a fine-grained PAT minted by `a11ign-ai-workers`, public
274
+ # repositories read-only, NO permissions, 90-day expiry, in this file as root:root 0600 and never in git.
275
+ # `fleet-playbook.mjs` reads it into GH_TOKEN. Minting it needs a human logged in to that account on the
276
+ # web, so this step REPORTS rather than creates: a missing or loose file is a warning, never a guess.
277
+ GH_TOKEN_FILE="$HOME/.config/a11y-witness/gh-token"
278
+ if [ ! -s "$GH_TOKEN_FILE" ]; then
279
+ warn "no GitHub token at $GH_TOKEN_FILE -- fleet:deploy/fleet:provision will refuse until one is written (#1875)"
280
+ elif [ "$(stat -c '%a' "$GH_TOKEN_FILE")" != "600" ]; then
281
+ warn "$GH_TOKEN_FILE is mode $(stat -c '%a' "$GH_TOKEN_FILE"), not 600 -- chmod 600 it"
282
+ else
283
+ ok "GitHub token present at $GH_TOKEN_FILE (mode 600)"
284
+ fi
285
+
286
+ # The fleet's SSH key lives HERE, not on somebody's laptop. That is the whole point of moving the
287
+ # control plane: a key on a Mac makes that Mac load-bearing again by a different route.
288
+ #
289
+ # Generated rather than copied, so this box is self-contained. The PUBLIC half is printed, because it
290
+ # has to reach the workers -- serve-bootstrap.sh hands it to a PXE install, and ssh-key.yml installs it
291
+ # on a box that is already up.
292
+ FLEET_KEY="$HOME/.ssh/a11y-witness_ed25519"
293
+ if [ -f "$FLEET_KEY" ]; then
294
+ ok "fleet key present ($(ssh-keygen -lf "$FLEET_KEY.pub" 2>/dev/null | awk '{print $2}'))"
295
+ else
296
+ mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
297
+ ssh-keygen -t ed25519 -N '' -C 'a11ign-capture-worker' -f "$FLEET_KEY" >/dev/null
298
+ ok "fleet key generated at $FLEET_KEY"
299
+ fi
300
+ echo
301
+ echo " The workers must trust this public key. Stage it into a PXE install with"
302
+ echo " serve-bootstrap.sh, or install it on a running box with ssh-key.yml:"
303
+ echo
304
+ echo " $(cat "$FLEET_KEY.pub")"
305
+ echo
306
+ fi
307
+
308
+ step 5 'Baseline corpus'
309
+ if ! is_lab; then
310
+ ok 'skipped (control role) -- the corpus belongs to the lab'
311
+ else
312
+ # The corpus is a GIT REPO of its own -- private, because these are our internal test pages and the
313
+ # main repo is public, so committing them there would publish the benchmark the tool is validated
314
+ # against. It is versioned rather than regenerated because a capture is NOT reproducible: browserVersion
315
+ # is in the capture cache key precisely so that evidence taken under one Edge release is not confused
316
+ # with another's, and recapturing after an update gives a DIFFERENT corpus rather than the same one.
317
+ #
318
+ # Without it the lab can capture but cannot COMPARE, and evidence:check refuses rather than silently
319
+ # reporting that nothing changed.
320
+ #
321
+ # A11Y_CORPUS_URL accepts either a git remote (preferred -- versioned, and every clone is a verified
322
+ # copy) or a tar.gz URL, because a box without access to the private repo should still be able to be
323
+ # handed a bundle.
324
+ # A DEPLOY KEY for the corpus, separate from the fleet key on purpose. Two different things are being
325
+ # authorised -- reading one private repo, and reconfiguring twelve Windows machines -- and one key for
326
+ # both means revoking either revokes the other. GitHub also refuses the same deploy key on two repos.
327
+ #
328
+ # A Host alias rather than a bare IdentityFile, so this key is used for THIS clone and nothing else. On a
329
+ # box with any other GitHub credential, ssh would otherwise offer keys in whatever order it likes and the
330
+ # failure ("Permission denied (publickey)") says nothing about which one it tried.
331
+ CORPUS_KEY="$HOME/.ssh/a11y-corpus_ed25519"
332
+ if is_lab && [ ! -f "$CORPUS_KEY" ]; then
333
+ mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
334
+ ssh-keygen -t ed25519 -N '' -C 'a11y-corpus deploy key (read-only)' -f "$CORPUS_KEY" >/dev/null
335
+ if ! grep -q 'Host a11y-corpus.github.com' "$HOME/.ssh/config" 2>/dev/null; then
336
+ printf 'Host a11y-corpus.github.com\n HostName github.com\n User git\n IdentityFile %s\n IdentitiesOnly yes\n' \
337
+ "$CORPUS_KEY" >> "$HOME/.ssh/config"
338
+ chmod 600 "$HOME/.ssh/config"
339
+ fi
340
+ echo
341
+ echo " The corpus repo is PRIVATE. Add this as a READ-ONLY deploy key at"
342
+ echo " https://github.com/DanBeckDev/a11y-corpus/settings/keys (do NOT tick write access):"
343
+ echo
344
+ echo " $(cat "$CORPUS_KEY.pub")"
345
+ echo
346
+ echo " Then re-run this script; the clone below will pick it up."
347
+ echo
348
+ fi
349
+
350
+ # Seed GitHub's host key, VERIFIED rather than trusted on first use. A fresh container has no
351
+ # known_hosts at all, so the clone below fails with "Host key verification failed" -- and the warning it
352
+ # prints blames the deploy key, which is the wrong diagnosis and cost real time chasing a key that was
353
+ # already correctly installed. `ssh-keyscan >> known_hosts` on its own is trust-on-first-use over the
354
+ # network, so the scanned key is compared against GitHub's published ed25519 fingerprint and REFUSED on
355
+ # a mismatch. Verified against two independent channels (a keyscan from a known-good host and
356
+ # api.github.com/meta over HTTPS), which agree.
357
+ GITHUB_ED25519_FP='SHA256:+DiY3wvvV6TuJJhbpZisF/zLDA0zPMSvHdkr4UvCOqU'
358
+ if is_lab && ! ssh-keygen -F github.com >/dev/null 2>&1; then
359
+ mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
360
+ SCANNED="$(ssh-keyscan -t ed25519 github.com 2>/dev/null)"
361
+ SCANNED_FP="$(printf '%s\n' "$SCANNED" | ssh-keygen -lf - 2>/dev/null | awk '{print $2}')"
362
+ if [ -n "$SCANNED" ] && [ "$SCANNED_FP" = "$GITHUB_ED25519_FP" ]; then
363
+ printf '%s\n' "$SCANNED" >> "$HOME/.ssh/known_hosts"
364
+ ok 'github.com host key seeded (fingerprint verified against the published key)'
365
+ else
366
+ warn "github.com host key did not match the published fingerprint (got ${SCANNED_FP:-nothing}) --"
367
+ warn 'NOT recorded. The corpus clone will fail; investigate before working around this.'
368
+ fi
369
+ fi
370
+
371
+ CORPUS_DIR="$REPO_PATH/runs/screenreader-dataset"
372
+ CORPUS_URL="${A11Y_CORPUS_URL:-git@a11y-corpus.github.com:DanBeckDev/a11y-corpus.git}"
373
+ if [ -d "$CORPUS_DIR/captures" ]; then
374
+ ok "corpus present ($(find "$CORPUS_DIR/captures" -name '*.json' | wc -l | tr -d ' ') captures)"
375
+ # A checkout can be updated; an unpacked tarball cannot, and saying which is which beats guessing.
376
+ if [ -d "$CORPUS_DIR/.git" ]; then
377
+ git -C "$CORPUS_DIR" pull --ff-only --quiet 2>/dev/null && ok 'corpus updated from its remote' \
378
+ || warn 'corpus is a checkout but could not be updated -- check the remote and your key'
379
+ fi
380
+ elif printf '%s' "$CORPUS_URL" | grep -qE '\.git$|^git@|^ssh://'; then
381
+ mkdir -p "$REPO_PATH/runs"
382
+ # Keep git's OWN error. `2>/dev/null` here replaced "Host key verification failed" with a guess about
383
+ # the deploy key, and the guess was wrong -- the key was installed and correct, and the container
384
+ # simply had no known_hosts. A diagnostic that states a cause it did not observe is worse than none,
385
+ # because it is believed.
386
+ if CLONE_ERR="$(git clone --quiet "$CORPUS_URL" "$CORPUS_DIR" 2>&1)"; then
387
+ ok "corpus cloned ($(find "$CORPUS_DIR/captures" -name '*.json' | wc -l | tr -d ' ') captures)"
388
+ else
389
+ warn "could not clone $CORPUS_URL"
390
+ warn "git said: ${CLONE_ERR:-(no output)}"
391
+ warn 'The repo is PRIVATE, so this box needs a deploy key with access -- but read the line above'
392
+ warn 'before assuming that is the cause.'
393
+ warn 'Capture will work; evidence:check has nothing to diff against until it is present.'
394
+ fi
395
+ else
396
+ mkdir -p "$REPO_PATH/runs"
397
+ curl -fsSL "$CORPUS_URL" | tar -xz -C "$REPO_PATH/runs" \
398
+ && ok 'corpus fetched from a tarball' \
399
+ || warn "could not fetch $CORPUS_URL"
400
+ fi
401
+ fi
402
+
403
+ step 6 'Workers'
404
+ if [ -z "${A11Y_WORKERS:-}" ]; then
405
+ warn 'A11Y_WORKERS is not set. Without it the orchestrator looks for LOCAL VMs, which do not'
406
+ warn 'exist here — that path is macOS/UTM only. Set it to your bare-metal workers:'
407
+ warn ' export A11Y_WORKERS=http://192.0.2.10:8765'
408
+ else
409
+ # Reachability is checked, not assumed: this whole exercise began with an orchestrator that
410
+ # could not reach its workers and reported it as something else entirely.
411
+ reachable=0
412
+ IFS=',' read -ra WS <<< "$A11Y_WORKERS"
413
+ for w in "${WS[@]}"; do
414
+ if curl -fsS --max-time 10 "${w%/}/health" >/dev/null 2>&1; then
415
+ ok "worker reachable: $w"; reachable=$((reachable + 1))
416
+ else
417
+ warn "worker NOT reachable: $w"
418
+ fi
419
+ done
420
+ [ "$reachable" -gt 0 ] || warn 'no worker answered /health — a run would fail immediately'
421
+ fi
422
+
423
+ step 7 'This host as the workers see it'
424
+ # The page server must be addressed by LAN IP, never localhost: a worker cannot reach our
425
+ # loopback, and a capture that fetches the wrong URL reads an error page and records it as
426
+ # evidence rather than failing.
427
+ LAN_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{print $7; exit}')"
428
+ if [ -n "$LAN_IP" ]; then
429
+ ok "workers should reach this host at $LAN_IP"
430
+ else
431
+ warn 'could not determine this host LAN address; set DATASET_BASE_URL explicitly'
432
+ fi
433
+
434
+ cat <<EOF
435
+
436
+ --- Control plane ready ---
437
+
438
+ Find and adopt workers:
439
+ pnpm run fleet:discover # scan, and reconcile against inventory.yml
440
+ \$EDITOR packages/control/ansible/inventory.yml # ansible_host + mac per box
441
+ eval "\$(pnpm run --silent fleet:env)" # A11Y_WORKERS, derived from that inventory
442
+
443
+ Build one:
444
+ packages/worker-fleet/src/provisioning/bare-metal/serve-bootstrap.sh ~/.ssh/a11y-witness_ed25519.pub
445
+
446
+ Manage them (from packages/control/ansible):
447
+ ansible-playbook provision-role.yml -l <host> --check --diff
448
+ ansible-playbook deploy.yml
449
+ ansible-playbook wake.yml / sleep.yml
450
+
451
+ Capture:
452
+ pnpm run doctor
453
+ pnpm run fleet:status
454
+ pnpm run training:capture
455
+
456
+ Nothing here depends on a Mac. Re-run this script any time; every step skips itself
457
+ when it is already done, and the corpus and repo are updated in place.
458
+
459
+ NOTE: this box must be on the SAME LAYER-2 SEGMENT as the workers. Wake-on-LAN magic
460
+ packets are broadcast and do not route, so a NAT'd or separately-VLAN'd control plane
461
+ can provision a worker over SSH and then be unable to wake it. On Proxmox that means a
462
+ bridged veth on vmbr0, not the default NAT.
463
+ EOF