@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +661 -0
- package/README.md +94 -2
- package/dist/capture-client.d.mts +49 -0
- package/dist/capture-client.d.mts.map +1 -0
- package/dist/capture-client.mjs +352 -0
- package/dist/capture-client.mjs.map +1 -0
- package/dist/check-worker-code.d.mts +34 -0
- package/dist/check-worker-code.d.mts.map +1 -0
- package/dist/check-worker-code.mjs +142 -0
- package/dist/check-worker-code.mjs.map +1 -0
- package/dist/cli-flags.d.mts +71 -0
- package/dist/cli-flags.d.mts.map +1 -0
- package/dist/cli-flags.mjs +207 -0
- package/dist/cli-flags.mjs.map +1 -0
- package/dist/code-drift.d.mts +140 -0
- package/dist/code-drift.d.mts.map +1 -0
- package/dist/code-drift.mjs +284 -0
- package/dist/code-drift.mjs.map +1 -0
- package/dist/command-line-census.d.mts +33 -0
- package/dist/command-line-census.d.mts.map +1 -0
- package/dist/command-line-census.mjs +96 -0
- package/dist/command-line-census.mjs.map +1 -0
- package/dist/compare-workers.d.mts +3 -0
- package/dist/compare-workers.d.mts.map +1 -0
- package/dist/compare-workers.mjs +332 -0
- package/dist/compare-workers.mjs.map +1 -0
- package/dist/control-plane-isolation.d.mts +45 -0
- package/dist/control-plane-isolation.d.mts.map +1 -0
- package/dist/control-plane-isolation.mjs +67 -0
- package/dist/control-plane-isolation.mjs.map +1 -0
- package/dist/deploy-worker.d.mts +3 -0
- package/dist/deploy-worker.d.mts.map +1 -0
- package/dist/deploy-worker.mjs +333 -0
- package/dist/deploy-worker.mjs.map +1 -0
- package/dist/doctor.d.mts +216 -0
- package/dist/doctor.d.mts.map +1 -0
- package/dist/doctor.mjs +962 -0
- package/dist/doctor.mjs.map +1 -0
- package/dist/fleet-consistency.d.mts +235 -0
- package/dist/fleet-consistency.d.mts.map +1 -0
- package/dist/fleet-consistency.mjs +436 -0
- package/dist/fleet-consistency.mjs.map +1 -0
- package/dist/fleet-env.d.mts +228 -0
- package/dist/fleet-env.d.mts.map +1 -0
- package/dist/fleet-env.mjs +509 -0
- package/dist/fleet-env.mjs.map +1 -0
- package/dist/fleet-scripts.d.mts +11 -0
- package/dist/fleet-scripts.d.mts.map +1 -0
- package/dist/fleet-scripts.mjs +41 -0
- package/dist/fleet-scripts.mjs.map +1 -0
- package/dist/git-safe-env.d.mts +10 -0
- package/dist/git-safe-env.d.mts.map +1 -0
- package/dist/git-safe-env.mjs +44 -0
- package/dist/git-safe-env.mjs.map +1 -0
- package/dist/guest-run.d.mts +26 -0
- package/dist/guest-run.d.mts.map +1 -0
- package/dist/guest-run.mjs +164 -0
- package/dist/guest-run.mjs.map +1 -0
- package/dist/host-address.d.mts +33 -0
- package/dist/host-address.d.mts.map +1 -0
- package/dist/host-address.mjs +105 -0
- package/dist/host-address.mjs.map +1 -0
- package/dist/host-capacity.d.mts +64 -0
- package/dist/host-capacity.d.mts.map +1 -0
- package/dist/host-capacity.mjs +152 -0
- package/dist/host-capacity.mjs.map +1 -0
- package/dist/host-metrics.d.mts +116 -0
- package/dist/host-metrics.d.mts.map +1 -0
- package/dist/host-metrics.mjs +201 -0
- package/dist/host-metrics.mjs.map +1 -0
- package/dist/index.d.ts +23 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/local-vm.d.ts +125 -0
- package/dist/local-vm.d.ts.map +1 -0
- package/dist/local-vm.js +360 -0
- package/dist/local-vm.js.map +1 -0
- package/dist/measure-guard.d.mts +34 -0
- package/dist/measure-guard.d.mts.map +1 -0
- package/dist/measure-guard.mjs +73 -0
- package/dist/measure-guard.mjs.map +1 -0
- package/dist/normalise-fleet.d.mts +2 -0
- package/dist/normalise-fleet.d.mts.map +1 -0
- package/dist/normalise-fleet.mjs +76 -0
- package/dist/normalise-fleet.mjs.map +1 -0
- package/dist/npm-cli-executable.d.mts +42 -0
- package/dist/npm-cli-executable.d.mts.map +1 -0
- package/dist/npm-cli-executable.mjs +159 -0
- package/dist/npm-cli-executable.mjs.map +1 -0
- package/dist/probe-outcome.d.mts +89 -0
- package/dist/probe-outcome.d.mts.map +1 -0
- package/dist/probe-outcome.mjs +104 -0
- package/dist/probe-outcome.mjs.map +1 -0
- package/dist/protocol-guard.d.mts +34 -0
- package/dist/protocol-guard.d.mts.map +1 -0
- package/dist/protocol-guard.mjs +121 -0
- package/dist/protocol-guard.mjs.map +1 -0
- package/dist/source-walk.d.mts +12 -0
- package/dist/source-walk.d.mts.map +1 -0
- package/dist/source-walk.mjs +56 -0
- package/dist/source-walk.mjs.map +1 -0
- package/dist/transient-fault.d.mts +6 -0
- package/dist/transient-fault.d.mts.map +1 -0
- package/dist/transient-fault.mjs +86 -0
- package/dist/transient-fault.mjs.map +1 -0
- package/dist/utm-deprecated.d.mts +6 -0
- package/dist/utm-deprecated.d.mts.map +1 -0
- package/dist/utm-deprecated.mjs +23 -0
- package/dist/utm-deprecated.mjs.map +1 -0
- package/dist/worker-code-check.d.mts +29 -0
- package/dist/worker-code-check.d.mts.map +1 -0
- package/dist/worker-code-check.mjs +85 -0
- package/dist/worker-code-check.mjs.map +1 -0
- package/dist/worker-health.d.mts +56 -0
- package/dist/worker-health.d.mts.map +1 -0
- package/dist/worker-health.mjs +73 -0
- package/dist/worker-health.mjs.map +1 -0
- package/dist/worker-http.d.mts +103 -0
- package/dist/worker-http.d.mts.map +1 -0
- package/dist/worker-http.mjs +277 -0
- package/dist/worker-http.mjs.map +1 -0
- package/dist/worker-stats.d.mts +66 -0
- package/dist/worker-stats.d.mts.map +1 -0
- package/dist/worker-stats.mjs +143 -0
- package/dist/worker-stats.mjs.map +1 -0
- package/package.json +96 -4
- package/src/local-worker/autounattend.xml +280 -0
- package/src/local-worker/build-vm.sh +218 -0
- package/src/local-worker/clone-worker.sh +141 -0
- package/src/local-worker/create-utm-vm.sh +202 -0
- package/src/local-worker/fetch-windows-iso.sh +238 -0
- package/src/local-worker/first-boot.cmd +58 -0
- package/src/local-worker/worker-ctl.sh +442 -0
- package/src/provisioning/README.md +28 -0
- package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
- package/src/provisioning/bare-metal/README.md +213 -0
- package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
- package/src/provisioning/bare-metal/autounattend.xml +428 -0
- package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
- package/src/provisioning/bootstrap-control-plane.sh +463 -0
- package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
- package/src/provisioning/build-lean-worker-image.ps1 +275 -0
- package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
- package/src/provisioning/provision-nvda-worker.ps1 +827 -0
- package/src/provisioning/set-display-mode.ps1 +411 -0
- package/src/provisioning/stamp-provision-revision.ps1 +184 -0
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Stand up the CONTROL PLANE on Linux, so no run depends on a particular laptop.
|
|
3
|
+
#
|
|
4
|
+
# curl -fsSL <raw-url>/bootstrap-control-plane.sh | bash
|
|
5
|
+
#
|
|
6
|
+
# The pair to bootstrap-windows-worker.ps1: that one turns a Windows box into a capture worker,
|
|
7
|
+
# this one turns a Debian/Ubuntu box (an LXC on Proxmox, say) into the thing that drives them.
|
|
8
|
+
#
|
|
9
|
+
# ## Why this exists
|
|
10
|
+
#
|
|
11
|
+
# ADR 0001 says it already: "The control plane is portable; only capture workers are OS-bound."
|
|
12
|
+
# Portable meant *could*, not *does* — it ran on one Mac, and that Mac was in the path of every
|
|
13
|
+
# corpus run. Three separate ways that bit, all in one day:
|
|
14
|
+
#
|
|
15
|
+
# - macOS 26 blocks node from the local network by default, so every worker call failed with
|
|
16
|
+
# EHOSTUNREACH while curl and python worked fine. A privacy toggle stopped the fleet.
|
|
17
|
+
# - the dataset page server ran there, so the pages a capture reads lived on a laptop.
|
|
18
|
+
# - a four-hour corpus run needed that laptop awake, on the same network, unslept.
|
|
19
|
+
#
|
|
20
|
+
# ## What it does NOT need
|
|
21
|
+
#
|
|
22
|
+
# No utmctl, no VM management, none of the macOS-bound half of worker-fleet. Verified rather than
|
|
23
|
+
# hoped: `leaseWorker` returns at its first line when a worker is named explicitly, and
|
|
24
|
+
# `capture-screenreader-dataset.mjs` returns the explicit pool before `leaseWorkerPool` is called.
|
|
25
|
+
# So with A11Y_WORKERS set, the managed-VM path is never entered and nothing macOS-only runs.
|
|
26
|
+
#
|
|
27
|
+
# That is why this is a deployment rather than a port: `packages/lab` — the orchestrator — contains
|
|
28
|
+
# no macOS-only command at all.
|
|
29
|
+
#
|
|
30
|
+
# A11Y_REPO_URL default the public GitHub repo
|
|
31
|
+
# A11Y_REPO_PATH default ~/a11y-witness
|
|
32
|
+
# A11Y_WORKERS comma-separated worker URLs, e.g. http://192.0.2.10:8765
|
|
33
|
+
# A11Y_CORPUS_URL optional tar.gz of runs/ to seed the baseline corpus (69 MB at time of writing)
|
|
34
|
+
set -euo pipefail
|
|
35
|
+
|
|
36
|
+
REPO_URL="${A11Y_REPO_URL:-https://github.com/a11ign/a11ign.git}"
|
|
37
|
+
REPO_PATH="${A11Y_REPO_PATH:-$HOME/a11y-witness}"
|
|
38
|
+
|
|
39
|
+
# WHICH HALF OF THE CONTROL PLANE IS THIS? (A11Y_ROLE=control|lab, default both)
|
|
40
|
+
#
|
|
41
|
+
# ADR 0012 splits them, and the reason is credentials rather than tidiness: the SSH key that can
|
|
42
|
+
# reconfigure twelve Windows machines should not sit next to 100 MB of transitive dependencies and a
|
|
43
|
+
# Python venv, which are the largest supply-chain surface in the system.
|
|
44
|
+
#
|
|
45
|
+
# control ansible + the fleet key. No node_modules, no venv, no corpus. Rebuildable in a minute.
|
|
46
|
+
# lab pnpm install + venv + the corpus. Talks to workers over HTTP only. Holds NO key.
|
|
47
|
+
#
|
|
48
|
+
# `both` remains the default so a single-box setup still works and nobody is forced into two containers
|
|
49
|
+
# on day one -- but it is the thing to grow out of, not the target.
|
|
50
|
+
ROLE="${A11Y_ROLE:-both}"
|
|
51
|
+
case "$ROLE" in
|
|
52
|
+
control|lab|both) ;;
|
|
53
|
+
*) echo "A11Y_ROLE must be control, lab or both (got '$ROLE')" >&2; exit 1 ;;
|
|
54
|
+
esac
|
|
55
|
+
is_control() { [ "$ROLE" = control ] || [ "$ROLE" = both ]; }
|
|
56
|
+
is_lab() { [ "$ROLE" = lab ] || [ "$ROLE" = both ]; }
|
|
57
|
+
|
|
58
|
+
step() { printf '\n\033[36m[%s] %s\033[0m\n' "$1" "$2"; }
|
|
59
|
+
ok() { printf ' \033[32mOK %s\033[0m\n' "$1"; }
|
|
60
|
+
warn() { printf ' \033[33mWARN %s\033[0m\n' "$1"; }
|
|
61
|
+
|
|
62
|
+
step 1 'Preconditions'
|
|
63
|
+
[ "$(uname -s)" = "Linux" ] || { echo "This is the Linux control plane; run it on the host that will drive the workers." >&2; exit 1; }
|
|
64
|
+
# shellcheck disable=SC1091 # /etc/os-release is a runtime file, not an input to lint
|
|
65
|
+
if . /etc/os-release 2>/dev/null && [ -n "${PRETTY_NAME:-}" ]; then ok "$PRETTY_NAME"; else ok "$(uname -sr)"; fi
|
|
66
|
+
|
|
67
|
+
# Root or sudo, but do not assume either: an LXC console is usually already root, and demanding
|
|
68
|
+
# sudo there fails on a box that does not have it installed.
|
|
69
|
+
SUDO=""
|
|
70
|
+
# A SEPARATE variable for the environment-preserving form, not `$SUDO -E`. When SUDO is empty --
|
|
71
|
+
# which is the normal case, because an LXC console is root -- `$SUDO -E cmd` leaves `-E` as the
|
|
72
|
+
# command word, and the shell reports `-E: command not found`. That killed this script at the
|
|
73
|
+
# NodeSource step on both containers: `set -euo pipefail` aborted correctly, so it failed loudly
|
|
74
|
+
# rather than half-installing, but the cause reads as a missing binary rather than a quoting bug.
|
|
75
|
+
SUDO_E=""
|
|
76
|
+
if [ "$(id -u)" -ne 0 ]; then
|
|
77
|
+
command -v sudo >/dev/null || { echo "Not root and no sudo. Run as root." >&2; exit 1; }
|
|
78
|
+
SUDO="sudo"
|
|
79
|
+
SUDO_E="sudo -E"
|
|
80
|
+
fi
|
|
81
|
+
if [ -n "$SUDO" ]; then ok 'using sudo'; else ok 'running as root'; fi
|
|
82
|
+
ok "role: $ROLE"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
step 2 'Node.js and git'
|
|
86
|
+
if command -v node >/dev/null && node -e 'process.exit(process.versions.node.split(".")[0] >= 20 ? 0 : 1)'; then
|
|
87
|
+
ok "node already present ($(node --version))"
|
|
88
|
+
else
|
|
89
|
+
# NodeSource rather than the distro package: Debian ships a node far older than this repo needs,
|
|
90
|
+
# and a version skew here surfaces as syntax errors in the orchestrator rather than as a version
|
|
91
|
+
# complaint.
|
|
92
|
+
$SUDO apt-get update -qq
|
|
93
|
+
$SUDO apt-get install -y -qq curl ca-certificates gnupg >/dev/null
|
|
94
|
+
curl -fsSL https://deb.nodesource.com/setup_lts.x | $SUDO_E bash - >/dev/null
|
|
95
|
+
$SUDO apt-get install -y -qq nodejs >/dev/null
|
|
96
|
+
ok "node installed ($(node --version))"
|
|
97
|
+
fi
|
|
98
|
+
command -v git >/dev/null || { $SUDO apt-get install -y -qq git >/dev/null; }
|
|
99
|
+
ok "git $(git --version | awk '{print $3}')"
|
|
100
|
+
|
|
101
|
+
step 3 'Repository'
|
|
102
|
+
if [ -d "$REPO_PATH/.git" ]; then
|
|
103
|
+
git -C "$REPO_PATH" pull --ff-only
|
|
104
|
+
ok "pulled $REPO_PATH ($(git -C "$REPO_PATH" rev-parse --short HEAD))"
|
|
105
|
+
else
|
|
106
|
+
git clone --quiet "$REPO_URL" "$REPO_PATH"
|
|
107
|
+
ok "cloned to $REPO_PATH ($(git -C "$REPO_PATH" rev-parse --short HEAD))"
|
|
108
|
+
fi
|
|
109
|
+
# A LAYER IN ITS OWN REPOSITORY IS A SECOND CHECKOUT BESIDE THE CORE'S (ADR 0039 item 6, row 6c, #3396). The core
|
|
110
|
+
# checkout above is one repository; a layer that `packages/control/layers.json` gives a `remote` lives in another,
|
|
111
|
+
# at its declared path inside this one, and the control plane and the lab both read it there. This clones it when
|
|
112
|
+
# it is absent and FETCHES it when it is there, and nothing more: where it STANDS is the pair's second half, which
|
|
113
|
+
# `fleet:deploy --layer-ref=` and a lab job's `layer_refs` set and read back, so a pull here would be a third
|
|
114
|
+
# thing moving it. With no layer that declares a `remote` the loop below has nothing to do, which is today.
|
|
115
|
+
#
|
|
116
|
+
# It REFUSES to clone over a directory that is not a git checkout: the monorepo's own copy of the layer is one,
|
|
117
|
+
# and a clone on top of it would be the core's tree answering for the layer. The path is excluded from the core's
|
|
118
|
+
# `git status` (`.git/info/exclude`, which is local and never committed), because a nested clone is otherwise
|
|
119
|
+
# `??` there, which reads as "somebody is working in the lab checkout" and stops every job from pulling.
|
|
120
|
+
LAYER_ROWS="$(node -e '
|
|
121
|
+
const layers = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8")).layers;
|
|
122
|
+
for (const [name, l] of Object.entries(layers)) {
|
|
123
|
+
if (!l.remote) continue;
|
|
124
|
+
if (!/^[A-Za-z0-9._\/-]+$/.test(l.path) || l.path.includes("..")) { console.error(`layer ${name}: path ${l.path} is not a plain relative path`); process.exit(1); }
|
|
125
|
+
if (!/^https:\/\/[A-Za-z0-9._\/-]+\.git$/.test(l.remote)) { console.error(`layer ${name}: remote ${l.remote} is not an https .git URL`); process.exit(1); }
|
|
126
|
+
console.log([name, l.path, l.remote, l.branch || ""].join("\t"));
|
|
127
|
+
}' "$REPO_PATH/packages/control/layers.json")"
|
|
128
|
+
while IFS=$'\t' read -r LAYER_NAME LAYER_DIR LAYER_REMOTE LAYER_BRANCH; do
|
|
129
|
+
[ -n "$LAYER_NAME" ] || continue
|
|
130
|
+
LAYER_PATH="$REPO_PATH/$LAYER_DIR"
|
|
131
|
+
if [ -d "$LAYER_PATH/.git" ]; then
|
|
132
|
+
git -C "$LAYER_PATH" fetch --quiet origin
|
|
133
|
+
ok "layer $LAYER_NAME fetched at $LAYER_DIR (on $(git -C "$LAYER_PATH" rev-parse --short HEAD); the pin moves it, not this)"
|
|
134
|
+
elif [ -e "$LAYER_PATH" ] && [ -n "$(ls -A "$LAYER_PATH" 2>/dev/null)" ]; then
|
|
135
|
+
echo "layer $LAYER_NAME is declared at $LAYER_DIR and $LAYER_PATH exists and is not a git checkout; refusing to clone over it." >&2
|
|
136
|
+
exit 1
|
|
137
|
+
else
|
|
138
|
+
git clone --quiet ${LAYER_BRANCH:+--branch "$LAYER_BRANCH"} "$LAYER_REMOTE" "$LAYER_PATH"
|
|
139
|
+
ok "layer $LAYER_NAME cloned to $LAYER_DIR ($(git -C "$LAYER_PATH" rev-parse --short HEAD))"
|
|
140
|
+
fi
|
|
141
|
+
mkdir -p "$REPO_PATH/.git/info"
|
|
142
|
+
grep -qxF "/$LAYER_DIR/" "$REPO_PATH/.git/info/exclude" 2>/dev/null || printf '/%s/\n' "$LAYER_DIR" >> "$REPO_PATH/.git/info/exclude"
|
|
143
|
+
done <<< "$LAYER_ROWS"
|
|
144
|
+
cd "$REPO_PATH"
|
|
145
|
+
if is_lab; then
|
|
146
|
+
# #2890: `corepack pnpm install --frozen-lockfile`, the SAME spelling `roles/worker/tasks/nvda.yml` uses
|
|
147
|
+
# (pinned by `provisioning-installs-with-pnpm.test.ts`). `packageManager` in package.json names the pnpm
|
|
148
|
+
# version and `pnpm-lock.yaml` is the specification: a plain install here resolved versions the lockfile
|
|
149
|
+
# never named, and a lab built that way captured evidence no other machine could reproduce.
|
|
150
|
+
command -v corepack >/dev/null || {
|
|
151
|
+
echo "corepack is not on PATH (Node 25+ no longer ships it). Use a Node that does (24 LTS), or install corepack from its own distribution" >&2
|
|
152
|
+
exit 1
|
|
153
|
+
}
|
|
154
|
+
export COREPACK_ENABLE_DOWNLOAD_PROMPT=0
|
|
155
|
+
# ONE-TIME MIGRATION: a tree npm made carries `node_modules/.package-lock.json`, and pnpm installed over it
|
|
156
|
+
# leaves every npm-hoisted package in place -- the hoisting that lets an undeclared import resolve.
|
|
157
|
+
if [ -f node_modules/.package-lock.json ]; then
|
|
158
|
+
ok 'node_modules was made by npm: removing it once, so nothing npm hoisted survives beside pnpm'
|
|
159
|
+
rm -rf "${REPO_PATH:?}/node_modules"
|
|
160
|
+
fi
|
|
161
|
+
corepack pnpm install --frozen-lockfile --silent
|
|
162
|
+
# The remedies this script prints below say `pnpm run ...`, so make `pnpm` a command an operator can type.
|
|
163
|
+
# `corepack enable` writes beside node, which needs root where node came from a package.
|
|
164
|
+
$SUDO corepack enable pnpm
|
|
165
|
+
ok 'dependencies installed'
|
|
166
|
+
|
|
167
|
+
# The LOCAL scorer is the default judge and the only one that ships (JUDGE_BACKEND defaults to
|
|
168
|
+
# `local`), so a lab without Python is a lab that cannot score. This step was missing entirely, and
|
|
169
|
+
# nothing here failed: `witness`, `eval` and `training:score` all name `.venv/bin/python` explicitly,
|
|
170
|
+
# so the absence surfaces as a missing interpreter partway through a run rather than at setup.
|
|
171
|
+
$SUDO apt-get install -y -qq python3-venv >/dev/null
|
|
172
|
+
[ -d "$REPO_PATH/.venv" ] || python3 -m venv "$REPO_PATH/.venv"
|
|
173
|
+
"$REPO_PATH/.venv/bin/pip" install -q --upgrade pip
|
|
174
|
+
"$REPO_PATH/.venv/bin/pip" install -q -r "$REPO_PATH/packages/scorer/requirements.txt"
|
|
175
|
+
ok "python venv ready ($("$REPO_PATH/.venv/bin/python" --version 2>&1))"
|
|
176
|
+
|
|
177
|
+
# The trained heads ARE tracked (796 KB); the MiniLM encoder is 87 MB and deliberately is not. It is a
|
|
178
|
+
# public model pinned by BOTH revision and sha256 in fetch-encoder.py, so fetching rather than
|
|
179
|
+
# vendoring it is still reproducible -- which is why .gitignore excludes `models/encoders/`.
|
|
180
|
+
if [ -f "$REPO_PATH/packages/scorer/models/encoders/all-MiniLM-L6-v2/model.safetensors" ]; then
|
|
181
|
+
ok 'encoder already present'
|
|
182
|
+
else
|
|
183
|
+
(cd "$REPO_PATH" && .venv/bin/python packages/scorer/python/fetch-encoder.py >/dev/null 2>&1) \
|
|
184
|
+
&& ok 'encoder fetched (pinned revision, sha256-verified)' \
|
|
185
|
+
|| warn 'could not fetch the encoder -- the local judge cannot score until it is present'
|
|
186
|
+
fi
|
|
187
|
+
else
|
|
188
|
+
# The control container deliberately has NO node_modules. deploy.yml computes codeVersion by importing
|
|
189
|
+
# code-version.mjs BY PATH -- it needs nothing but node stdlib, and verified identical to the workspace
|
|
190
|
+
# import. 100 MB of transitive dependencies next to the fleet's SSH key is the coupling ADR 0012 removes.
|
|
191
|
+
ok 'skipped (control role) -- no node_modules beside the fleet key'
|
|
192
|
+
fi
|
|
193
|
+
|
|
194
|
+
step 4 'Ansible, and the fleet key'
|
|
195
|
+
if ! is_control; then
|
|
196
|
+
ok 'skipped (lab role) -- the lab holds NO fleet key and runs no Ansible, by design (ADR 0012)'
|
|
197
|
+
else
|
|
198
|
+
# The control plane MANAGES the workers as well as capturing with them, and this script predates that
|
|
199
|
+
# half entirely -- it installed node and a checkout and left you without the thing that provisions,
|
|
200
|
+
# deploys, wakes and sleeps a box.
|
|
201
|
+
#
|
|
202
|
+
# pipx rather than apt: Windows-over-SSH support is ansible-core 2.18+, and Debian ships older. That
|
|
203
|
+
# version gap is not cosmetic -- on an older core every Windows task fails at connection time.
|
|
204
|
+
#
|
|
205
|
+
# The REQUIREMENT is enforced here rather than documented here. It used to be neither: this block installed
|
|
206
|
+
# `ansible-core` with no version constraint, and the "already present" branch accepted whatever was on the
|
|
207
|
+
# box -- so the one case the comment above warns about, an older core, passed the check silently and failed
|
|
208
|
+
# later at connection time, which reads like a broken worker rather than a stale controller.
|
|
209
|
+
#
|
|
210
|
+
# `collections/ansible_collections/a11y/worker/meta/runtime.yml` states the same floor as
|
|
211
|
+
# `requires_ansible`, so Ansible itself refuses the collection on an older core. This makes the bootstrap
|
|
212
|
+
# agree with it instead of leaving the two to drift.
|
|
213
|
+
ANSIBLE_MIN="2.18"
|
|
214
|
+
|
|
215
|
+
ansible_core_version() { ansible --version 2>/dev/null | head -1 | grep -oE '[0-9]+\.[0-9]+(\.[0-9]+)?' | head -1; }
|
|
216
|
+
# Sort-based compare, so 2.9 does not read as newer than 2.18 the way a string compare would. That is the
|
|
217
|
+
# specific wrong answer this guard has to avoid, because 2.9 is exactly what Debian ships.
|
|
218
|
+
version_at_least() { [ "$(printf '%s\n%s\n' "$2" "$1" | sort -V | head -1)" = "$2" ]; }
|
|
219
|
+
|
|
220
|
+
if command -v ansible-playbook >/dev/null; then
|
|
221
|
+
found="$(ansible_core_version)"
|
|
222
|
+
if [ -n "$found" ] && version_at_least "$found" "$ANSIBLE_MIN"; then
|
|
223
|
+
ok "ansible already present (core $found, >= $ANSIBLE_MIN)"
|
|
224
|
+
else
|
|
225
|
+
warn "ansible core ${found:-unknown} is older than $ANSIBLE_MIN — every Windows task will fail at"
|
|
226
|
+
warn "connection time, which looks like a broken worker rather than a stale controller. Upgrade with:"
|
|
227
|
+
warn " pipx upgrade ansible-core || pipx install --force 'ansible-core>=$ANSIBLE_MIN'"
|
|
228
|
+
fi
|
|
229
|
+
else
|
|
230
|
+
$SUDO apt-get install -y -qq pipx >/dev/null 2>&1 || $SUDO apt-get install -y -qq python3-pip >/dev/null
|
|
231
|
+
if command -v pipx >/dev/null; then
|
|
232
|
+
pipx install "ansible-core>=$ANSIBLE_MIN" >/dev/null
|
|
233
|
+
pipx ensurepath >/dev/null 2>&1 || true
|
|
234
|
+
else
|
|
235
|
+
$SUDO pip3 install --break-system-packages -q "ansible-core>=$ANSIBLE_MIN"
|
|
236
|
+
fi
|
|
237
|
+
export PATH="$HOME/.local/bin:$PATH"
|
|
238
|
+
ok "ansible installed ($(ansible --version 2>/dev/null | head -1))"
|
|
239
|
+
fi
|
|
240
|
+
|
|
241
|
+
# -p is load-bearing: ansible.cfg puts the repo's own collections path FIRST, so a bare install vendors
|
|
242
|
+
# third-party collections into the git tree. Only a11y.worker belongs there.
|
|
243
|
+
export PATH="$HOME/.local/bin:$PATH"
|
|
244
|
+
if [ -f "$REPO_PATH/packages/control/ansible/requirements.yml" ]; then
|
|
245
|
+
ansible-galaxy collection install -r "$REPO_PATH/packages/control/ansible/requirements.yml" \
|
|
246
|
+
-p "$HOME/.ansible/collections" >/dev/null
|
|
247
|
+
ok 'collections installed (ansible.windows, community.windows, community.general)'
|
|
248
|
+
else
|
|
249
|
+
warn 'requirements.yml not found -- is the checkout complete?'
|
|
250
|
+
fi
|
|
251
|
+
|
|
252
|
+
# #1870: `fleet-playbook.mjs`'s fleet-hold check (#1839/#1841) shells out to `gh` to ask whether an open
|
|
253
|
+
# `fleet-gated` row carries an unexpired `Fleet-hold-until:` before `fleet:deploy`/`fleet:provision`
|
|
254
|
+
# proceed -- so `gh` is now a CONTROL-role dependency, not just something an interactive session happens
|
|
255
|
+
# to have. Debian ships no `gh` package at all (unlike Node, this is not a version-skew problem, it is a
|
|
256
|
+
# missing package), so this follows GitHub's own published apt repository rather than guessing at a distro
|
|
257
|
+
# package name that does not exist.
|
|
258
|
+
if command -v gh >/dev/null; then
|
|
259
|
+
ok "gh already present ($(gh --version | head -1))"
|
|
260
|
+
else
|
|
261
|
+
$SUDO mkdir -p -m 755 /etc/apt/keyrings
|
|
262
|
+
curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg \
|
|
263
|
+
| $SUDO tee /etc/apt/keyrings/githubcli-archive-keyring.gpg >/dev/null
|
|
264
|
+
$SUDO chmod go+r /etc/apt/keyrings/githubcli-archive-keyring.gpg
|
|
265
|
+
echo "deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages/. stable main" \
|
|
266
|
+
| $SUDO tee /etc/apt/sources.list.d/github-cli.list >/dev/null
|
|
267
|
+
$SUDO apt-get update -qq
|
|
268
|
+
$SUDO apt-get install -y -qq gh >/dev/null
|
|
269
|
+
ok "gh installed ($(gh --version | head -1))"
|
|
270
|
+
fi
|
|
271
|
+
|
|
272
|
+
# #1875: gh on its own is not enough -- it refuses to run unauthenticated, so the fleet-hold check still
|
|
273
|
+
# fails closed. `ceo`'s ruling on that row: a fine-grained PAT minted by `a11ign-ai-workers`, public
|
|
274
|
+
# repositories read-only, NO permissions, 90-day expiry, in this file as root:root 0600 and never in git.
|
|
275
|
+
# `fleet-playbook.mjs` reads it into GH_TOKEN. Minting it needs a human logged in to that account on the
|
|
276
|
+
# web, so this step REPORTS rather than creates: a missing or loose file is a warning, never a guess.
|
|
277
|
+
GH_TOKEN_FILE="$HOME/.config/a11y-witness/gh-token"
|
|
278
|
+
if [ ! -s "$GH_TOKEN_FILE" ]; then
|
|
279
|
+
warn "no GitHub token at $GH_TOKEN_FILE -- fleet:deploy/fleet:provision will refuse until one is written (#1875)"
|
|
280
|
+
elif [ "$(stat -c '%a' "$GH_TOKEN_FILE")" != "600" ]; then
|
|
281
|
+
warn "$GH_TOKEN_FILE is mode $(stat -c '%a' "$GH_TOKEN_FILE"), not 600 -- chmod 600 it"
|
|
282
|
+
else
|
|
283
|
+
ok "GitHub token present at $GH_TOKEN_FILE (mode 600)"
|
|
284
|
+
fi
|
|
285
|
+
|
|
286
|
+
# The fleet's SSH key lives HERE, not on somebody's laptop. That is the whole point of moving the
|
|
287
|
+
# control plane: a key on a Mac makes that Mac load-bearing again by a different route.
|
|
288
|
+
#
|
|
289
|
+
# Generated rather than copied, so this box is self-contained. The PUBLIC half is printed, because it
|
|
290
|
+
# has to reach the workers -- serve-bootstrap.sh hands it to a PXE install, and ssh-key.yml installs it
|
|
291
|
+
# on a box that is already up.
|
|
292
|
+
FLEET_KEY="$HOME/.ssh/a11y-witness_ed25519"
|
|
293
|
+
if [ -f "$FLEET_KEY" ]; then
|
|
294
|
+
ok "fleet key present ($(ssh-keygen -lf "$FLEET_KEY.pub" 2>/dev/null | awk '{print $2}'))"
|
|
295
|
+
else
|
|
296
|
+
mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
|
|
297
|
+
ssh-keygen -t ed25519 -N '' -C 'a11ign-capture-worker' -f "$FLEET_KEY" >/dev/null
|
|
298
|
+
ok "fleet key generated at $FLEET_KEY"
|
|
299
|
+
fi
|
|
300
|
+
echo
|
|
301
|
+
echo " The workers must trust this public key. Stage it into a PXE install with"
|
|
302
|
+
echo " serve-bootstrap.sh, or install it on a running box with ssh-key.yml:"
|
|
303
|
+
echo
|
|
304
|
+
echo " $(cat "$FLEET_KEY.pub")"
|
|
305
|
+
echo
|
|
306
|
+
fi
|
|
307
|
+
|
|
308
|
+
step 5 'Baseline corpus'
|
|
309
|
+
if ! is_lab; then
|
|
310
|
+
ok 'skipped (control role) -- the corpus belongs to the lab'
|
|
311
|
+
else
|
|
312
|
+
# The corpus is a GIT REPO of its own -- private, because these are our internal test pages and the
|
|
313
|
+
# main repo is public, so committing them there would publish the benchmark the tool is validated
|
|
314
|
+
# against. It is versioned rather than regenerated because a capture is NOT reproducible: browserVersion
|
|
315
|
+
# is in the capture cache key precisely so that evidence taken under one Edge release is not confused
|
|
316
|
+
# with another's, and recapturing after an update gives a DIFFERENT corpus rather than the same one.
|
|
317
|
+
#
|
|
318
|
+
# Without it the lab can capture but cannot COMPARE, and evidence:check refuses rather than silently
|
|
319
|
+
# reporting that nothing changed.
|
|
320
|
+
#
|
|
321
|
+
# A11Y_CORPUS_URL accepts either a git remote (preferred -- versioned, and every clone is a verified
|
|
322
|
+
# copy) or a tar.gz URL, because a box without access to the private repo should still be able to be
|
|
323
|
+
# handed a bundle.
|
|
324
|
+
# A DEPLOY KEY for the corpus, separate from the fleet key on purpose. Two different things are being
|
|
325
|
+
# authorised -- reading one private repo, and reconfiguring twelve Windows machines -- and one key for
|
|
326
|
+
# both means revoking either revokes the other. GitHub also refuses the same deploy key on two repos.
|
|
327
|
+
#
|
|
328
|
+
# A Host alias rather than a bare IdentityFile, so this key is used for THIS clone and nothing else. On a
|
|
329
|
+
# box with any other GitHub credential, ssh would otherwise offer keys in whatever order it likes and the
|
|
330
|
+
# failure ("Permission denied (publickey)") says nothing about which one it tried.
|
|
331
|
+
CORPUS_KEY="$HOME/.ssh/a11y-corpus_ed25519"
|
|
332
|
+
if is_lab && [ ! -f "$CORPUS_KEY" ]; then
|
|
333
|
+
mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
|
|
334
|
+
ssh-keygen -t ed25519 -N '' -C 'a11y-corpus deploy key (read-only)' -f "$CORPUS_KEY" >/dev/null
|
|
335
|
+
if ! grep -q 'Host a11y-corpus.github.com' "$HOME/.ssh/config" 2>/dev/null; then
|
|
336
|
+
printf 'Host a11y-corpus.github.com\n HostName github.com\n User git\n IdentityFile %s\n IdentitiesOnly yes\n' \
|
|
337
|
+
"$CORPUS_KEY" >> "$HOME/.ssh/config"
|
|
338
|
+
chmod 600 "$HOME/.ssh/config"
|
|
339
|
+
fi
|
|
340
|
+
echo
|
|
341
|
+
echo " The corpus repo is PRIVATE. Add this as a READ-ONLY deploy key at"
|
|
342
|
+
echo " https://github.com/DanBeckDev/a11y-corpus/settings/keys (do NOT tick write access):"
|
|
343
|
+
echo
|
|
344
|
+
echo " $(cat "$CORPUS_KEY.pub")"
|
|
345
|
+
echo
|
|
346
|
+
echo " Then re-run this script; the clone below will pick it up."
|
|
347
|
+
echo
|
|
348
|
+
fi
|
|
349
|
+
|
|
350
|
+
# Seed GitHub's host key, VERIFIED rather than trusted on first use. A fresh container has no
|
|
351
|
+
# known_hosts at all, so the clone below fails with "Host key verification failed" -- and the warning it
|
|
352
|
+
# prints blames the deploy key, which is the wrong diagnosis and cost real time chasing a key that was
|
|
353
|
+
# already correctly installed. `ssh-keyscan >> known_hosts` on its own is trust-on-first-use over the
|
|
354
|
+
# network, so the scanned key is compared against GitHub's published ed25519 fingerprint and REFUSED on
|
|
355
|
+
# a mismatch. Verified against two independent channels (a keyscan from a known-good host and
|
|
356
|
+
# api.github.com/meta over HTTPS), which agree.
|
|
357
|
+
GITHUB_ED25519_FP='SHA256:+DiY3wvvV6TuJJhbpZisF/zLDA0zPMSvHdkr4UvCOqU'
|
|
358
|
+
if is_lab && ! ssh-keygen -F github.com >/dev/null 2>&1; then
|
|
359
|
+
mkdir -p "$HOME/.ssh" && chmod 700 "$HOME/.ssh"
|
|
360
|
+
SCANNED="$(ssh-keyscan -t ed25519 github.com 2>/dev/null)"
|
|
361
|
+
SCANNED_FP="$(printf '%s\n' "$SCANNED" | ssh-keygen -lf - 2>/dev/null | awk '{print $2}')"
|
|
362
|
+
if [ -n "$SCANNED" ] && [ "$SCANNED_FP" = "$GITHUB_ED25519_FP" ]; then
|
|
363
|
+
printf '%s\n' "$SCANNED" >> "$HOME/.ssh/known_hosts"
|
|
364
|
+
ok 'github.com host key seeded (fingerprint verified against the published key)'
|
|
365
|
+
else
|
|
366
|
+
warn "github.com host key did not match the published fingerprint (got ${SCANNED_FP:-nothing}) --"
|
|
367
|
+
warn 'NOT recorded. The corpus clone will fail; investigate before working around this.'
|
|
368
|
+
fi
|
|
369
|
+
fi
|
|
370
|
+
|
|
371
|
+
CORPUS_DIR="$REPO_PATH/runs/screenreader-dataset"
|
|
372
|
+
CORPUS_URL="${A11Y_CORPUS_URL:-git@a11y-corpus.github.com:DanBeckDev/a11y-corpus.git}"
|
|
373
|
+
if [ -d "$CORPUS_DIR/captures" ]; then
|
|
374
|
+
ok "corpus present ($(find "$CORPUS_DIR/captures" -name '*.json' | wc -l | tr -d ' ') captures)"
|
|
375
|
+
# A checkout can be updated; an unpacked tarball cannot, and saying which is which beats guessing.
|
|
376
|
+
if [ -d "$CORPUS_DIR/.git" ]; then
|
|
377
|
+
git -C "$CORPUS_DIR" pull --ff-only --quiet 2>/dev/null && ok 'corpus updated from its remote' \
|
|
378
|
+
|| warn 'corpus is a checkout but could not be updated -- check the remote and your key'
|
|
379
|
+
fi
|
|
380
|
+
elif printf '%s' "$CORPUS_URL" | grep -qE '\.git$|^git@|^ssh://'; then
|
|
381
|
+
mkdir -p "$REPO_PATH/runs"
|
|
382
|
+
# Keep git's OWN error. `2>/dev/null` here replaced "Host key verification failed" with a guess about
|
|
383
|
+
# the deploy key, and the guess was wrong -- the key was installed and correct, and the container
|
|
384
|
+
# simply had no known_hosts. A diagnostic that states a cause it did not observe is worse than none,
|
|
385
|
+
# because it is believed.
|
|
386
|
+
if CLONE_ERR="$(git clone --quiet "$CORPUS_URL" "$CORPUS_DIR" 2>&1)"; then
|
|
387
|
+
ok "corpus cloned ($(find "$CORPUS_DIR/captures" -name '*.json' | wc -l | tr -d ' ') captures)"
|
|
388
|
+
else
|
|
389
|
+
warn "could not clone $CORPUS_URL"
|
|
390
|
+
warn "git said: ${CLONE_ERR:-(no output)}"
|
|
391
|
+
warn 'The repo is PRIVATE, so this box needs a deploy key with access -- but read the line above'
|
|
392
|
+
warn 'before assuming that is the cause.'
|
|
393
|
+
warn 'Capture will work; evidence:check has nothing to diff against until it is present.'
|
|
394
|
+
fi
|
|
395
|
+
else
|
|
396
|
+
mkdir -p "$REPO_PATH/runs"
|
|
397
|
+
curl -fsSL "$CORPUS_URL" | tar -xz -C "$REPO_PATH/runs" \
|
|
398
|
+
&& ok 'corpus fetched from a tarball' \
|
|
399
|
+
|| warn "could not fetch $CORPUS_URL"
|
|
400
|
+
fi
|
|
401
|
+
fi
|
|
402
|
+
|
|
403
|
+
step 6 'Workers'
|
|
404
|
+
if [ -z "${A11Y_WORKERS:-}" ]; then
|
|
405
|
+
warn 'A11Y_WORKERS is not set. Without it the orchestrator looks for LOCAL VMs, which do not'
|
|
406
|
+
warn 'exist here — that path is macOS/UTM only. Set it to your bare-metal workers:'
|
|
407
|
+
warn ' export A11Y_WORKERS=http://192.0.2.10:8765'
|
|
408
|
+
else
|
|
409
|
+
# Reachability is checked, not assumed: this whole exercise began with an orchestrator that
|
|
410
|
+
# could not reach its workers and reported it as something else entirely.
|
|
411
|
+
reachable=0
|
|
412
|
+
IFS=',' read -ra WS <<< "$A11Y_WORKERS"
|
|
413
|
+
for w in "${WS[@]}"; do
|
|
414
|
+
if curl -fsS --max-time 10 "${w%/}/health" >/dev/null 2>&1; then
|
|
415
|
+
ok "worker reachable: $w"; reachable=$((reachable + 1))
|
|
416
|
+
else
|
|
417
|
+
warn "worker NOT reachable: $w"
|
|
418
|
+
fi
|
|
419
|
+
done
|
|
420
|
+
[ "$reachable" -gt 0 ] || warn 'no worker answered /health — a run would fail immediately'
|
|
421
|
+
fi
|
|
422
|
+
|
|
423
|
+
step 7 'This host as the workers see it'
|
|
424
|
+
# The page server must be addressed by LAN IP, never localhost: a worker cannot reach our
|
|
425
|
+
# loopback, and a capture that fetches the wrong URL reads an error page and records it as
|
|
426
|
+
# evidence rather than failing.
|
|
427
|
+
LAN_IP="$(ip -4 route get 1.1.1.1 2>/dev/null | awk '{print $7; exit}')"
|
|
428
|
+
if [ -n "$LAN_IP" ]; then
|
|
429
|
+
ok "workers should reach this host at $LAN_IP"
|
|
430
|
+
else
|
|
431
|
+
warn 'could not determine this host LAN address; set DATASET_BASE_URL explicitly'
|
|
432
|
+
fi
|
|
433
|
+
|
|
434
|
+
cat <<EOF
|
|
435
|
+
|
|
436
|
+
--- Control plane ready ---
|
|
437
|
+
|
|
438
|
+
Find and adopt workers:
|
|
439
|
+
pnpm run fleet:discover # scan, and reconcile against inventory.yml
|
|
440
|
+
\$EDITOR packages/control/ansible/inventory.yml # ansible_host + mac per box
|
|
441
|
+
eval "\$(pnpm run --silent fleet:env)" # A11Y_WORKERS, derived from that inventory
|
|
442
|
+
|
|
443
|
+
Build one:
|
|
444
|
+
packages/worker-fleet/src/provisioning/bare-metal/serve-bootstrap.sh ~/.ssh/a11y-witness_ed25519.pub
|
|
445
|
+
|
|
446
|
+
Manage them (from packages/control/ansible):
|
|
447
|
+
ansible-playbook provision-role.yml -l <host> --check --diff
|
|
448
|
+
ansible-playbook deploy.yml
|
|
449
|
+
ansible-playbook wake.yml / sleep.yml
|
|
450
|
+
|
|
451
|
+
Capture:
|
|
452
|
+
pnpm run doctor
|
|
453
|
+
pnpm run fleet:status
|
|
454
|
+
pnpm run training:capture
|
|
455
|
+
|
|
456
|
+
Nothing here depends on a Mac. Re-run this script any time; every step skips itself
|
|
457
|
+
when it is already done, and the corpus and repo are updated in place.
|
|
458
|
+
|
|
459
|
+
NOTE: this box must be on the SAME LAYER-2 SEGMENT as the workers. Wake-on-LAN magic
|
|
460
|
+
packets are broadcast and do not route, so a NAT'd or separately-VLAN'd control plane
|
|
461
|
+
can provision a worker over SSH and then be unable to wake it. On Proxmox that means a
|
|
462
|
+
bridged veth on vmbr0, not the default NAT.
|
|
463
|
+
EOF
|