@a11ign/screenreader-fleet 0.0.0-reserved.0 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/LICENSE +661 -0
  2. package/README.md +94 -2
  3. package/dist/capture-client.d.mts +49 -0
  4. package/dist/capture-client.d.mts.map +1 -0
  5. package/dist/capture-client.mjs +352 -0
  6. package/dist/capture-client.mjs.map +1 -0
  7. package/dist/check-worker-code.d.mts +34 -0
  8. package/dist/check-worker-code.d.mts.map +1 -0
  9. package/dist/check-worker-code.mjs +173 -0
  10. package/dist/check-worker-code.mjs.map +1 -0
  11. package/dist/cli-flags.d.mts +71 -0
  12. package/dist/cli-flags.d.mts.map +1 -0
  13. package/dist/cli-flags.mjs +207 -0
  14. package/dist/cli-flags.mjs.map +1 -0
  15. package/dist/code-drift.d.mts +140 -0
  16. package/dist/code-drift.d.mts.map +1 -0
  17. package/dist/code-drift.mjs +284 -0
  18. package/dist/code-drift.mjs.map +1 -0
  19. package/dist/command-line-census.d.mts +33 -0
  20. package/dist/command-line-census.d.mts.map +1 -0
  21. package/dist/command-line-census.mjs +96 -0
  22. package/dist/command-line-census.mjs.map +1 -0
  23. package/dist/compare-workers.d.mts +3 -0
  24. package/dist/compare-workers.d.mts.map +1 -0
  25. package/dist/compare-workers.mjs +332 -0
  26. package/dist/compare-workers.mjs.map +1 -0
  27. package/dist/control-plane-isolation.d.mts +45 -0
  28. package/dist/control-plane-isolation.d.mts.map +1 -0
  29. package/dist/control-plane-isolation.mjs +67 -0
  30. package/dist/control-plane-isolation.mjs.map +1 -0
  31. package/dist/deploy-worker.d.mts +3 -0
  32. package/dist/deploy-worker.d.mts.map +1 -0
  33. package/dist/deploy-worker.mjs +333 -0
  34. package/dist/deploy-worker.mjs.map +1 -0
  35. package/dist/doctor.d.mts +216 -0
  36. package/dist/doctor.d.mts.map +1 -0
  37. package/dist/doctor.mjs +962 -0
  38. package/dist/doctor.mjs.map +1 -0
  39. package/dist/fleet-consistency.d.mts +235 -0
  40. package/dist/fleet-consistency.d.mts.map +1 -0
  41. package/dist/fleet-consistency.mjs +436 -0
  42. package/dist/fleet-consistency.mjs.map +1 -0
  43. package/dist/fleet-env.d.mts +228 -0
  44. package/dist/fleet-env.d.mts.map +1 -0
  45. package/dist/fleet-env.mjs +509 -0
  46. package/dist/fleet-env.mjs.map +1 -0
  47. package/dist/fleet-scripts.d.mts +11 -0
  48. package/dist/fleet-scripts.d.mts.map +1 -0
  49. package/dist/fleet-scripts.mjs +41 -0
  50. package/dist/fleet-scripts.mjs.map +1 -0
  51. package/dist/git-safe-env.d.mts +10 -0
  52. package/dist/git-safe-env.d.mts.map +1 -0
  53. package/dist/git-safe-env.mjs +44 -0
  54. package/dist/git-safe-env.mjs.map +1 -0
  55. package/dist/guest-run.d.mts +26 -0
  56. package/dist/guest-run.d.mts.map +1 -0
  57. package/dist/guest-run.mjs +164 -0
  58. package/dist/guest-run.mjs.map +1 -0
  59. package/dist/host-address.d.mts +33 -0
  60. package/dist/host-address.d.mts.map +1 -0
  61. package/dist/host-address.mjs +105 -0
  62. package/dist/host-address.mjs.map +1 -0
  63. package/dist/host-capacity.d.mts +64 -0
  64. package/dist/host-capacity.d.mts.map +1 -0
  65. package/dist/host-capacity.mjs +152 -0
  66. package/dist/host-capacity.mjs.map +1 -0
  67. package/dist/host-metrics.d.mts +116 -0
  68. package/dist/host-metrics.d.mts.map +1 -0
  69. package/dist/host-metrics.mjs +201 -0
  70. package/dist/host-metrics.mjs.map +1 -0
  71. package/dist/index.d.ts +23 -0
  72. package/dist/index.d.ts.map +1 -0
  73. package/dist/index.js +25 -0
  74. package/dist/index.js.map +1 -0
  75. package/dist/local-vm.d.ts +125 -0
  76. package/dist/local-vm.d.ts.map +1 -0
  77. package/dist/local-vm.js +360 -0
  78. package/dist/local-vm.js.map +1 -0
  79. package/dist/measure-guard.d.mts +34 -0
  80. package/dist/measure-guard.d.mts.map +1 -0
  81. package/dist/measure-guard.mjs +73 -0
  82. package/dist/measure-guard.mjs.map +1 -0
  83. package/dist/normalise-fleet.d.mts +2 -0
  84. package/dist/normalise-fleet.d.mts.map +1 -0
  85. package/dist/normalise-fleet.mjs +76 -0
  86. package/dist/normalise-fleet.mjs.map +1 -0
  87. package/dist/npm-cli-executable.d.mts +42 -0
  88. package/dist/npm-cli-executable.d.mts.map +1 -0
  89. package/dist/npm-cli-executable.mjs +159 -0
  90. package/dist/npm-cli-executable.mjs.map +1 -0
  91. package/dist/probe-outcome.d.mts +89 -0
  92. package/dist/probe-outcome.d.mts.map +1 -0
  93. package/dist/probe-outcome.mjs +104 -0
  94. package/dist/probe-outcome.mjs.map +1 -0
  95. package/dist/protocol-guard.d.mts +34 -0
  96. package/dist/protocol-guard.d.mts.map +1 -0
  97. package/dist/protocol-guard.mjs +121 -0
  98. package/dist/protocol-guard.mjs.map +1 -0
  99. package/dist/source-walk.d.mts +12 -0
  100. package/dist/source-walk.d.mts.map +1 -0
  101. package/dist/source-walk.mjs +56 -0
  102. package/dist/source-walk.mjs.map +1 -0
  103. package/dist/transient-fault.d.mts +6 -0
  104. package/dist/transient-fault.d.mts.map +1 -0
  105. package/dist/transient-fault.mjs +86 -0
  106. package/dist/transient-fault.mjs.map +1 -0
  107. package/dist/utm-deprecated.d.mts +6 -0
  108. package/dist/utm-deprecated.d.mts.map +1 -0
  109. package/dist/utm-deprecated.mjs +23 -0
  110. package/dist/utm-deprecated.mjs.map +1 -0
  111. package/dist/worker-code-check.d.mts +29 -0
  112. package/dist/worker-code-check.d.mts.map +1 -0
  113. package/dist/worker-code-check.mjs +78 -0
  114. package/dist/worker-code-check.mjs.map +1 -0
  115. package/dist/worker-health.d.mts +56 -0
  116. package/dist/worker-health.d.mts.map +1 -0
  117. package/dist/worker-health.mjs +73 -0
  118. package/dist/worker-health.mjs.map +1 -0
  119. package/dist/worker-http.d.mts +103 -0
  120. package/dist/worker-http.d.mts.map +1 -0
  121. package/dist/worker-http.mjs +277 -0
  122. package/dist/worker-http.mjs.map +1 -0
  123. package/dist/worker-stats.d.mts +66 -0
  124. package/dist/worker-stats.d.mts.map +1 -0
  125. package/dist/worker-stats.mjs +143 -0
  126. package/dist/worker-stats.mjs.map +1 -0
  127. package/package.json +96 -4
  128. package/src/local-worker/autounattend.xml +280 -0
  129. package/src/local-worker/build-vm.sh +218 -0
  130. package/src/local-worker/clone-worker.sh +141 -0
  131. package/src/local-worker/create-utm-vm.sh +202 -0
  132. package/src/local-worker/fetch-windows-iso.sh +238 -0
  133. package/src/local-worker/first-boot.cmd +58 -0
  134. package/src/local-worker/worker-ctl.sh +442 -0
  135. package/src/provisioning/README.md +28 -0
  136. package/src/provisioning/apply-foreground-lock-timeout.ps1 +71 -0
  137. package/src/provisioning/bare-metal/README.md +213 -0
  138. package/src/provisioning/bare-metal/a11y-bootstrap.service +58 -0
  139. package/src/provisioning/bare-metal/autounattend.xml +428 -0
  140. package/src/provisioning/bare-metal/serve-bootstrap.sh +86 -0
  141. package/src/provisioning/bootstrap-control-plane.sh +463 -0
  142. package/src/provisioning/bootstrap-windows-worker.ps1 +649 -0
  143. package/src/provisioning/build-lean-worker-image.ps1 +275 -0
  144. package/src/provisioning/diagnose-nvda-worker.ps1 +174 -0
  145. package/src/provisioning/provision-nvda-worker.ps1 +827 -0
  146. package/src/provisioning/set-display-mode.ps1 +411 -0
  147. package/src/provisioning/stamp-provision-revision.ps1 +184 -0
@@ -0,0 +1,649 @@
1
+ # Bootstrap a freshly-installed Windows box into an NVDA capture worker.
2
+ #
3
+ # Run this ONCE, in the VM, in an elevated PowerShell, right after Windows setup:
4
+ #
5
+ # Set-ExecutionPolicy -Scope Process Bypass -Force
6
+ # irm https://raw.githubusercontent.com/a11ign/a11ign/main/packages/worker-fleet/src/provisioning/bootstrap-windows-worker.ps1 | iex
7
+ #
8
+ # ...or, if you already have the repo, just run this file. It installs the
9
+ # prerequisites, makes the box reachable over SSH, clones the repo, and then hands
10
+ # off to provision-nvda-worker.ps1 (which does the NVDA/OS configuration and is the
11
+ # tested part of this pair).
12
+ #
13
+ # STATUS: the individual steps are the ones we ran by hand to build the existing
14
+ # worker, but this script has NOT yet been run end-to-end on a fresh Windows install.
15
+ # Expect to babysit it the first time; fix and commit what it gets wrong.
16
+ #
17
+ # Style constraints match the other scripts: `#` line comments and env-var config, no
18
+ # `<# #>` block comment and no param() block -- see diagnose-nvda-worker.ps1 beside this file.
19
+ # A11Y_REPO_URL (default the public GitHub repo)
20
+ # A11Y_REPO_PATH (default %USERPROFILE%\a11y-witness)
21
+ #
22
+ # Auto-logon needs NO configuration and NO password: provisioning gives the console account a
23
+ # blank password and points Winlogon at it. Nothing is stored, so there is nothing to distribute
24
+ # across a fleet. It requires a LOCAL account -- a Microsoft account cannot hold a blank password.
25
+
26
+ $ErrorActionPreference = 'Stop'
27
+
28
+ $RepoUrl = if ($env:A11Y_REPO_URL) { $env:A11Y_REPO_URL } else { 'https://github.com/a11ign/a11ign.git' }
29
+ $RepoPath = if ($env:A11Y_REPO_PATH) { $env:A11Y_REPO_PATH } else { Join-Path $env:USERPROFILE 'a11y-witness' }
30
+
31
+ # What each step did, so the end of a re-run says which steps were already good and which
32
+ # were retried. Without this the second run looks identical to the first and you cannot tell
33
+ # whether it fixed anything.
34
+ $script:outcomes = [ordered]@{}
35
+ function Record($name, $state) { $script:outcomes[$name] = $state }
36
+
37
+ function Step($n, $msg) { Write-Host "`n[$n] $msg" -ForegroundColor Cyan }
38
+ function OK($msg) { Write-Host " OK $msg" -ForegroundColor Green }
39
+ function Warn($msg) { Write-Host " WARN $msg" -ForegroundColor Yellow }
40
+
41
+ # Same rationale as provision-nvda-worker.ps1: native tools write to stderr as a
42
+ # matter of course, and with $ErrorActionPreference='Stop' PowerShell turns any
43
+ # stderr line into a terminating error. Gate on the exit code instead.
44
+ function Invoke-Native($exe, [string[]] $cmdArgs, [string] $what, [int] $tail = 4) {
45
+ $prev = $ErrorActionPreference
46
+ $ErrorActionPreference = 'Continue'
47
+ try {
48
+ $out = & $exe @cmdArgs 2>&1
49
+ $code = $LASTEXITCODE
50
+ $out | Select-Object -Last $tail | ForEach-Object { Write-Host " $_" }
51
+ if ($code -ne 0) { throw "$what failed (exit $code)." }
52
+ } finally { $ErrorActionPreference = $prev }
53
+ }
54
+
55
+ Step 1 'Check elevation'
56
+ $elevated = (New-Object Security.Principal.WindowsPrincipal(
57
+ [Security.Principal.WindowsIdentity]::GetCurrent())).IsInRole(
58
+ [Security.Principal.WindowsBuiltinRole]::Administrator)
59
+ if (-not $elevated) { throw 'Run this in an ELEVATED PowerShell (needed for SSH + firewall + policy steps).' }
60
+ OK 'elevated'
61
+
62
+ Step 2 'Install Node.js and Git (direct download, no winget)'
63
+ # Do NOT use winget here. On a freshly installed Windows it does not exist: winget ships
64
+ # as the "App Installer" Store package, which is not registered on a brand-new image, so
65
+ # the call fails with "'winget' is not recognized" and the whole bootstrap dies. Fetching
66
+ # the official archives directly has no Store dependency and no installer UI to hang on.
67
+ #
68
+ # Archives rather than MSIs on purpose: Node's current LTS line publishes
69
+ # win-arm64-zip but NO arm64 .msi, so the zip is the only version-agnostic choice.
70
+ $ProgressPreference = 'SilentlyContinue' # or Invoke-WebRequest crawls
71
+
72
+ # Which architecture to fetch binaries for. Every download below was hardcoded to arm64,
73
+ # because this script was written from the UTM guests on an Apple-silicon Mac -- so on an
74
+ # x64 box it downloaded ARM binaries that cannot execute, and the failure surfaces as
75
+ # "node is not recognized" AFTER a successful-looking install.
76
+ #
77
+ # `OSArchitecture`, not $env:PROCESSOR_ARCHITECTURE: the env var reports the *process*
78
+ # architecture, so an x64 PowerShell emulated on ARM64 Windows reports AMD64 and would
79
+ # install the wrong Node. OSArchitecture asks the OS.
80
+ #
81
+ # The three projects spell the same architecture three different ways, which is exactly
82
+ # the kind of thing that looks right in review and 404s at runtime -- so each mapping is
83
+ # named rather than derived.
84
+ function Get-OsArchitecture {
85
+ # RuntimeInformation is the RIGHT answer and is NOT reliably available here. This script is
86
+ # run via `irm | iex`, which means Windows PowerShell 5.1 on .NET Framework, where the
87
+ # System.Runtime.InteropServices.RuntimeInformation assembly is not loaded by default -- so
88
+ # the static property yields $null and `.ToString()` on it dies with "You cannot call a
89
+ # method on a null-valued expression", an error that names neither the variable nor the
90
+ # line. Try it, never depend on it.
91
+ try {
92
+ $ri = [System.Runtime.InteropServices.RuntimeInformation]::OSArchitecture
93
+ if ($ri) { return $ri.ToString().ToLower() }
94
+ } catch {
95
+ # Type not resolvable on this runtime. Fall through to the environment.
96
+ }
97
+ # PROCESSOR_ARCHITEW6432 is set ONLY inside a 32-bit process on a 64-bit OS, and it carries
98
+ # the OS architecture -- so reading it first is what makes this report the machine rather
99
+ # than the process. Preserving that distinction is the entire reason for preferring
100
+ # RuntimeInformation in the first place; the fallback must not quietly lose it.
101
+ $pa = if ($env:PROCESSOR_ARCHITEW6432) { $env:PROCESSOR_ARCHITEW6432 } else { $env:PROCESSOR_ARCHITECTURE }
102
+ switch ($pa) {
103
+ 'AMD64' { return 'x64' }
104
+ 'ARM64' { return 'arm64' }
105
+ 'x86' { return 'x86' }
106
+ default { return "$pa".ToLower() } # quoted: .ToLower() on a $null would repeat the bug
107
+ }
108
+ }
109
+ $osArch = Get-OsArchitecture
110
+ if ($osArch -notin @('x64', 'arm64')) { throw "unsupported architecture '$osArch' (expected x64 or arm64)" }
111
+ $nodeArch = $osArch # node-vX-win-x64.zip / -win-arm64.zip
112
+ $minGitArch = if ($osArch -eq 'x64') { '64-bit' } else { 'arm64' } # MinGit-X-64-bit.zip / -arm64.zip
113
+ $openSshArch = if ($osArch -eq 'x64') { 'Win64' } else { 'ARM64' } # OpenSSH-Win64.msi / OpenSSH-ARM64.msi
114
+ OK "architecture $osArch (node=$nodeArch, mingit=$minGitArch, openssh=$openSshArch)"
115
+
116
+ function Get-Archive($url, $outFile) {
117
+ Invoke-WebRequest -Uri $url -OutFile $outFile -UseBasicParsing
118
+ if (-not (Test-Path $outFile)) { throw "download failed: $url" }
119
+ }
120
+
121
+ $tmp = Join-Path $env:TEMP 'a11y-bootstrap'
122
+ New-Item -ItemType Directory -Force -Path $tmp | Out-Null
123
+
124
+ # THE NODE BUILD THIS FLEET RUNS -- pinned, and PINNED EQUAL to `roles/worker/defaults/main.yml`'s
125
+ # `worker_node_version`, which is the tested copy. `node-pin-parity.test.ts` refuses a change to one
126
+ # without the other, for `edge-pin-parity.test.ts`'s reasons: the bootstrap runs on a box with no
127
+ # Ansible, before the fleet can reach it, so the copy cannot be deleted or derived -- only pinned.
128
+ #
129
+ # IT USED TO RESOLVE THE CURRENT LTS FROM nodejs.org, exactly as `packages.yml` did, and that is how a
130
+ # fleet ends up on two runtimes: a resolve pins within a run and never across runs, and a box keeps
131
+ # whatever it first received. Measured 2026-09-23T06:55Z: workers 2-6 on v24.19.0, workers 7-11 on
132
+ # v24.20.0, with every other reported field identical (#2063).
133
+ $NodeVersion = if ($env:A11Y_NODE_VERSION) { $env:A11Y_NODE_VERSION } else { 'v24.20.0' }
134
+ # Per ARCHITECTURE, because the zip differs per architecture and one hash could only verify one of them.
135
+ $NodeSha = @{
136
+ 'x64' = '6cac9ffbca8f6a47091e4b5c772e0606049c3871cb67d900c0cedde630e545ba'
137
+ 'arm64' = '31c6799744de8a54601643098040c68c3697e56c94e407d61d0e5fa5f34191d7'
138
+ }
139
+
140
+ # Check the INSTALL PATH, not just the command. `Get-Command node` consults this session's PATH,
141
+ # and a session that started before the machine PATH was updated -- or a fresh account's first
142
+ # logon -- does not have it, so a perfectly good install looks absent and gets reinstalled.
143
+ $nodeHome = Join-Path $env:ProgramFiles 'nodejs'
144
+ $nodeExe = Join-Path $nodeHome 'node.exe'
145
+ # THE VERSION, NOT THE PRESENCE. "is there a node" can only ever answer once, so a box that came up on
146
+ # the wrong build keeps it for ever -- which is the drift the pin above exists to end. Same gate the role
147
+ # now uses, same reason.
148
+ $haveNode = if (Test-Path $nodeExe) { (& $nodeExe --version).Trim() }
149
+ elseif (Get-Command node -ErrorAction SilentlyContinue) { (& node --version).Trim() }
150
+ else { '(absent)' }
151
+ if ($haveNode -eq $NodeVersion) {
152
+ OK "node already present ($haveNode, the pinned build)"
153
+ # REPAIR the ACL even when we did not install it. An existing install may carry the permissions
154
+ # of whichever account put it there -- see the Move-Item note below -- and a re-run that skips
155
+ # the install would otherwise never fix a box that is already broken. Idempotent, so it costs
156
+ # nothing to do every time, and it makes "run it again" a real remedy rather than a hope.
157
+ if (Test-Path $nodeHome) {
158
+ icacls $nodeHome /reset /T /C /Q | Out-Null
159
+ Record 'node' 'already present (permissions reset)'
160
+ } else {
161
+ Record 'node' 'already present'
162
+ }
163
+ }
164
+ else {
165
+ Record 'node' $(if ($haveNode -eq '(absent)') { 'installed' } else { "upgraded from $haveNode" })
166
+ if (-not $NodeSha.ContainsKey($nodeArch)) { throw "no pinned Node checksum for architecture $nodeArch" }
167
+ $zip = Join-Path $tmp 'node.zip'
168
+ Get-Archive "https://nodejs.org/dist/$NodeVersion/node-$NodeVersion-win-$nodeArch.zip" $zip
169
+ # VERIFIED BEFORE IT IS UNPACKED. The pin is a claim about BYTES; without this it is a claim about a
170
+ # hostname, which is the reason `edge-version.yml` pins its MSI by checksum too.
171
+ $got = (Get-FileHash -Path $zip -Algorithm SHA256).Hash.ToLower()
172
+ if ($got -ne $NodeSha[$nodeArch]) {
173
+ throw "node-$NodeVersion-win-$nodeArch.zip hashed $got, expected $($NodeSha[$nodeArch]) -- refusing to install unverified bytes"
174
+ }
175
+ Expand-Archive -Path $zip -DestinationPath $tmp -Force
176
+ $src = Get-ChildItem $tmp -Directory -Filter "node-*-win-$nodeArch" | Select-Object -First 1
177
+ # Named, because `$src.FullName` on a $null gives the same useless "null-valued expression"
178
+ # as the bug above. If Node ever changes its archive's top-level directory name, this must
179
+ # say WHICH expectation broke.
180
+ if (-not $src) { throw "expanded Node archive has no 'node-*-win-$nodeArch' directory under $tmp" }
181
+ $dest = $nodeHome
182
+ # run-server.cmd looks for "%ProgramFiles%\nodejs\node.exe" first, so install there.
183
+ #
184
+ # DELETE ONLY AFTER the replacement is in hand. This used to remove the existing install and
185
+ # THEN move the new one in, so any failure between the two -- a bad download, a failed expand,
186
+ # a half-written archive -- left the box with NO node at all. That is worse than the state it
187
+ # started in, and it is silent: the next thing to notice is run-server.cmd's window opening and
188
+ # closing in two seconds because `node` is not a recognised command.
189
+ #
190
+ # Verify the new tree actually contains node.exe before touching the old one, then swap.
191
+ if (-not (Test-Path (Join-Path $src.FullName 'node.exe'))) {
192
+ throw "expanded Node archive at $($src.FullName) has no node.exe -- refusing to replace a working install"
193
+ }
194
+ if (Test-Path $dest) {
195
+ $old = "$dest.old-$PID"
196
+ Rename-Item $dest $old -Force
197
+ try {
198
+ Move-Item $src.FullName $dest
199
+ Remove-Item $old -Recurse -Force -ErrorAction SilentlyContinue
200
+ } catch {
201
+ # Put it back rather than leaving the box with nothing.
202
+ if (Test-Path $old) { Rename-Item $old $dest -Force }
203
+ throw
204
+ }
205
+ } else {
206
+ Move-Item $src.FullName $dest
207
+ }
208
+
209
+ # RESTORE INHERITED PERMISSIONS. Move-Item on the same volume PRESERVES the source ACL rather
210
+ # than inheriting the destination's -- so a tree moved out of one account's %TEMP% into
211
+ # C:\Program Files carries that account's permissions, and nobody else can execute it.
212
+ #
213
+ # Measured: node installed by the first (Microsoft) account, then run by the worker account's
214
+ # UNELEVATED scheduled task -> "Access is denied", exit 5. Elevated shells worked throughout,
215
+ # because an administrator bypasses the ACL -- which is exactly why every manual test passed and
216
+ # only the task failed.
217
+ icacls $dest /reset /T /C /Q | Out-Null
218
+ OK "permissions on $dest reset to inherit (a Move-Item keeps the SOURCE acl)"
219
+
220
+ OK "node installed to $dest ($NodeVersion)"
221
+ }
222
+
223
+ if (Get-Command git -ErrorAction SilentlyContinue) { OK "git already present"; Record 'git' 'already present' }
224
+ else {
225
+ Record 'git' 'installed'
226
+ # MinGit is the portable Git build: a zip, no installer, which is all we need to clone.
227
+ $rel = Invoke-RestMethod -Uri 'https://api.github.com/repos/git-for-windows/git/releases/latest' -UseBasicParsing
228
+ # `[0-9.]+` rather than `.*` between the name and the architecture. Git for Windows also
229
+ # ships MinGit-<ver>-busybox-64-bit.zip, which a `.*` matches just as happily -- so the
230
+ # build you got would depend on GitHub's asset ordering rather than on this code. That is
231
+ # the "check that cannot discriminate" shape, and it would have installed a busybox Git
232
+ # that looks identical until something needs a real coreutil.
233
+ $asset = $rel.assets | Where-Object { $_.name -match "^MinGit-[0-9.]+-$([regex]::Escape($minGitArch))\.zip$" } | Select-Object -First 1
234
+ if (-not $asset) { throw "no MinGit $minGitArch asset found" }
235
+ $zip = Join-Path $tmp 'mingit.zip'
236
+ Get-Archive $asset.browser_download_url $zip
237
+ $dest = Join-Path $env:ProgramFiles 'MinGit'
238
+ if (Test-Path $dest) { Remove-Item $dest -Recurse -Force }
239
+ Expand-Archive -Path $zip -DestinationPath $dest -Force
240
+ icacls $dest /reset /T /C /Q | Out-Null
241
+ OK "git installed to $dest ($($asset.name))"
242
+ }
243
+
244
+ # Put both on the MACHINE path so the scheduled task's session sees them too, then
245
+ # refresh this process so the rest of the bootstrap can use them immediately.
246
+ $machinePath = [Environment]::GetEnvironmentVariable('Path', 'Machine')
247
+ foreach ($p in @((Join-Path $env:ProgramFiles 'nodejs'), (Join-Path $env:ProgramFiles 'MinGit\cmd'))) {
248
+ if ($machinePath -notlike "*$p*") { $machinePath = "$machinePath;$p" }
249
+ }
250
+ [Environment]::SetEnvironmentVariable('Path', $machinePath, 'Machine')
251
+ $env:Path = $machinePath + ';' + [Environment]::GetEnvironmentVariable('Path', 'User')
252
+ OK "node $(& node --version 2>&1), git $(& git --version 2>&1)"
253
+
254
+ Step 3 'Install the OpenSSH server (best effort)'
255
+ # Best effort on purpose: the worker is driven over HTTP on 8765, and on a local VM the
256
+ # UTM guest agent (`utmctl exec` / `utmctl file pull`) is a perfectly good control
257
+ # channel. SSH is a convenience, so a failure here must not abort provisioning.
258
+ #
259
+ # Deliberately NOT Add-WindowsCapability: that route can stall indefinitely on Windows
260
+ # Update (documented in packages/nvda-worker/README.md). Use the Win32-OpenSSH release.
261
+ try {
262
+ if (Get-Service sshd -ErrorAction SilentlyContinue) { OK 'sshd already present'; Record 'sshd' 'already present' }
263
+ else {
264
+ Record 'sshd' 'installed'
265
+ $rel = Invoke-RestMethod -Uri 'https://api.github.com/repos/PowerShell/Win32-OpenSSH/releases/latest' -UseBasicParsing
266
+ $asset = $rel.assets | Where-Object { $_.name -match "^OpenSSH-$([regex]::Escape($openSshArch)).*\.msi$" } | Select-Object -First 1
267
+ if (-not $asset) { throw "no OpenSSH $openSshArch msi asset found" }
268
+ $msi = Join-Path $tmp 'openssh.msi'
269
+ Get-Archive $asset.browser_download_url $msi
270
+ # Start-Process -Wait, NOT `& msiexec`. msiexec.exe is a GUI-subsystem binary, so invoking
271
+ # it through the call operator returns IMMEDIATELY and $LASTEXITCODE is meaningless -- the
272
+ # install had not happened yet when the next line ran, and `Set-Service -Name sshd` failed
273
+ # with "not found on computer '.'", which reads like a broken MSI rather than a race.
274
+ # 3010 is "success, reboot required" and is not a failure.
275
+ $mi = Start-Process 'msiexec.exe' -ArgumentList @('/i', "`"$msi`"", '/qn', '/norestart') -Wait -PassThru
276
+ if ($mi.ExitCode -notin @(0, 3010)) { throw "OpenSSH msi failed (exit $($mi.ExitCode))" }
277
+ # Service registration can lag the installer's exit. Poll for the CONDITION rather than
278
+ # guessing a sleep -- the same rule the capture path already follows.
279
+ $deadline = (Get-Date).AddSeconds(30)
280
+ while (-not (Get-Service sshd -ErrorAction SilentlyContinue) -and (Get-Date) -lt $deadline) {
281
+ Start-Sleep -Milliseconds 500
282
+ }
283
+ if (-not (Get-Service sshd -ErrorAction SilentlyContinue)) {
284
+ throw "OpenSSH msi reported success (exit $($mi.ExitCode)) but no sshd service appeared within 30s"
285
+ }
286
+ }
287
+ Set-Service -Name sshd -StartupType Automatic
288
+ Start-Service sshd
289
+ if (-not (Get-NetFirewallRule -Name 'sshd-a11y' -ErrorAction SilentlyContinue)) {
290
+ New-NetFirewallRule -Name 'sshd-a11y' -DisplayName 'OpenSSH Server (a11ign)' `
291
+ -Enabled True -Direction Inbound -Protocol TCP -Action Allow -LocalPort 22 | Out-Null
292
+ }
293
+ # PowerShell as the SSH shell, not cmd.
294
+ #
295
+ # Windows OpenSSH ships with cmd.exe as DefaultShell. Ansible's Windows modules ARE PowerShell, so with
296
+ # cmd every task pays an extra quoting layer for nothing, and `ansible_shell_type: powershell` -- which
297
+ # the fleet playbooks set -- answers every task with a parse error that points at the YAML rather than at
298
+ # the shell. Setting it here means a box is manageable the moment it finishes bootstrapping, rather than
299
+ # after somebody remembers this.
300
+ #
301
+ # In its OWN try, because the enclosing catch records the failure as 'sshd'. A key that could not be
302
+ # written reported as "OpenSSH setup failed" would send the next person to look at a service that is
303
+ # running perfectly -- the two-states-reported-as-one shape this project keeps paying for.
304
+ try {
305
+ $sshReg = 'HKLM:\SOFTWARE\OpenSSH'
306
+ if (-not (Test-Path $sshReg)) { New-Item -Path $sshReg -Force | Out-Null }
307
+ $pwsh = Join-Path $env:SystemRoot 'System32\WindowsPowerShell\v1.0\powershell.exe'
308
+ New-ItemProperty -Path $sshReg -Name DefaultShell -Value $pwsh -PropertyType String -Force | Out-Null
309
+ OK 'ssh DefaultShell set to PowerShell (the fleet playbooks require it)'
310
+ Record 'ssh shell' 'PowerShell'
311
+ } catch {
312
+ Record 'ssh shell' "FAILED - $($_.Exception.Message)"
313
+ Warn "Could not set ssh DefaultShell ($($_.Exception.Message)). Ansible plays will fail against this"
314
+ Warn 'box until it is set; see packages/control/ansible/README.md for the one-liner.'
315
+ }
316
+
317
+ # The operator's public key, if one was supplied. This is what turns "6-12 console visits" into one.
318
+ #
319
+ # `witness` is an Administrator, and Windows OpenSSH IGNORES a per-user authorized_keys for admin
320
+ # accounts in favour of one shared file with strict ACLs. Getting that wrong is the most common reason
321
+ # key auth "silently fails" here: sshd reads nothing, falls back to offering password auth, and the
322
+ # blank-password account cannot use it -- so the symptom is a permission denial that looks like a bad key.
323
+ # Its own try, for the same reason as above: a key problem must not be reported as an sshd problem.
324
+ try {
325
+ # Two ways in, one code path. A11Y_OPERATOR_KEY is what a human sets before running this by hand;
326
+ # the file is what first-boot.cmd stages off the install media, because the unattended path may run
327
+ # this through a scheduled task -- a fresh session that inherits no environment at all.
328
+ $stagedKey = Join-Path $env:ProgramData 'a11y-witness\operator-key.pub'
329
+ $operatorKey = if ($env:A11Y_OPERATOR_KEY) { $env:A11Y_OPERATOR_KEY }
330
+ elseif (Test-Path $stagedKey) { Get-Content -Path $stagedKey -Raw }
331
+ else { $null }
332
+ if ($operatorKey) {
333
+ $adminKeys = Join-Path $env:ProgramData 'ssh\administrators_authorized_keys'
334
+ $key = $operatorKey.Trim()
335
+ if ($key -notmatch '^(ssh-|ecdsa-)') {
336
+ Warn "The operator key does not look like a public key (expected ssh-... or ecdsa-...); ignoring it."
337
+ Warn " read from: $(if ($env:A11Y_OPERATOR_KEY) { 'A11Y_OPERATOR_KEY' } else { $stagedKey })"
338
+ Record 'operator key' 'REFUSED - not a public key'
339
+ } else {
340
+ if (-not (Test-Path $adminKeys)) { New-Item -ItemType File -Path $adminKeys -Force | Out-Null }
341
+ if (@(Get-Content -Path $adminKeys -ErrorAction SilentlyContinue) -contains $key) {
342
+ OK 'operator key already installed'
343
+ Record 'operator key' 'already present'
344
+ } else {
345
+ Add-Content -Path $adminKeys -Value $key
346
+ OK "operator key added to $adminKeys"
347
+ Record 'operator key' 'installed'
348
+ }
349
+ # sshd refuses a key file that any non-admin can write, and refuses it without telling the client.
350
+ icacls.exe $adminKeys /inheritance:r /grant 'Administrators:F' /grant 'SYSTEM:F' | Out-Null
351
+ }
352
+ } else {
353
+ Record 'operator key' "not supplied (set A11Y_OPERATOR_KEY, or stage $stagedKey, to manage this box with Ansible)"
354
+ }
355
+ } catch {
356
+ Record 'operator key' "FAILED - $($_.Exception.Message)"
357
+ Warn "Could not install the operator key ($($_.Exception.Message)). sshd is unaffected; install it"
358
+ Warn 'from the control plane with: ansible-playbook ssh-key.yml -l <host> -e a11y_operator_key=...'
359
+ }
360
+
361
+ OK "sshd $((Get-Service sshd).Status), port 22 allowed"
362
+ Record 'sshd' "running on port 22"
363
+ } catch {
364
+ # Non-fatal by design, and now explicitly RETRYABLE: the summary names it, and a re-run
365
+ # re-enters this step because its guard is "is there an sshd service", which there is not.
366
+ Record 'sshd' "FAILED - $($_.Exception.Message)"
367
+ Warn "OpenSSH setup failed ($($_.Exception.Message)). Continuing -- the worker does not need it."
368
+ Warn "Re-run this script to retry it; every other step skips itself when already done."
369
+ }
370
+
371
+ Step 4 'Clone the repo'
372
+ if (Test-Path (Join-Path $RepoPath '.git')) {
373
+ # PULL on a re-run. Without this, a second attempt keeps running the same buggy checkout and
374
+ # fails identically on something already fixed upstream -- which is exactly what the stale
375
+ # provisioning paths would have done.
376
+ Invoke-Native 'git' @('-C', $RepoPath, 'pull', '--ff-only') 'git pull' 2
377
+ OK "already cloned at $RepoPath (pulled)"
378
+ Record 'repo' 'pulled'
379
+ } else {
380
+ Invoke-Native 'git' @('clone', $RepoUrl, $RepoPath) 'git clone' 2
381
+ OK "cloned to $RepoPath"
382
+ Record 'repo' 'cloned'
383
+ }
384
+
385
+ # A LAYER THAT LIVES IN ITS OWN REPOSITORY IS A SECOND CLONE, at the path the monorepo used (ADR 0039 item 6,
386
+ # #3395). Which layers is read from the manifest in the checkout just made -- `packages/control/layers.json`, the
387
+ # one place that says -- and never restated here, because this script is run as `irm <url> | iex` and a list
388
+ # written into it would outlive the layer it names. A layer with no `remote` is inside the core checkout and is
389
+ # skipped; so is a core that predates the manifest, which can have no layer of its own. Like the core, it is
390
+ # cloned at its default branch: the PIN is the deploy's (`tasks/layer-checkouts.yml`), which has the layer's commit.
391
+ $layersManifest = Join-Path $RepoPath 'packages\control\layers.json'
392
+ if (Test-Path $layersManifest) {
393
+ $layers = (Get-Content -Raw $layersManifest | ConvertFrom-Json).layers
394
+ foreach ($layer in $layers.PSObject.Properties) {
395
+ if (-not $layer.Value.remote) { continue }
396
+ $layerPath = Join-Path $RepoPath ($layer.Value.path -replace '/', '\')
397
+ if (Test-Path (Join-Path $layerPath '.git')) {
398
+ Invoke-Native 'git' @('-C', $layerPath, 'pull', '--ff-only') "pull of layer $($layer.Name)" 2
399
+ OK "layer $($layer.Name) already cloned at $layerPath (pulled)"
400
+ Record "layer $($layer.Name)" 'pulled'
401
+ } else {
402
+ Invoke-Native 'git' @('clone', $layer.Value.remote, $layerPath) "clone of layer $($layer.Name)" 2
403
+ OK "layer $($layer.Name) cloned to $layerPath"
404
+ Record "layer $($layer.Name)" 'cloned'
405
+ }
406
+ }
407
+ } else {
408
+ OK "no layers manifest at $layersManifest (a checkout older than #3394): no separate layer to clone"
409
+ }
410
+
411
+ Step 5 'Pin Edge and stop its updater — BEFORE the box has time to update itself'
412
+ # THE STEP THAT WAS MISSING, and a11y-worker-7 is what it cost.
413
+ #
414
+ # `browserVersion` is part of the capture cache key, so two Edge builds must never write into one corpus,
415
+ # and `fleet-consistency` lists it FIRST in MUST_MATCH — a split there does not degrade one box, it makes
416
+ # every capture run refuse to start.
417
+ #
418
+ # Until 2026-09-04 nothing in first boot mentioned Edge. The box installed Windows, came up with a
419
+ # network, and Edge updated itself while `npm install` was still running. By the time
420
+ # `roles/worker/tasks/edge-version.yml` ran it found a NEWER build than the pin, and one box was
421
+ # reimaged over it -- because a pin that arrives after the update is not a pin.
422
+ #
423
+ # The reimage was NOT necessary, and this comment used to say it was. Chromium declines to install over
424
+ # a newer build BY DEFAULT; `ALLOWDOWNGRADE=1` is the supported way to say otherwise. See the version
425
+ # check below, which now rolls back rather than refusing.
426
+ #
427
+ # So the same two levers the role uses, moved to the earliest moment they work. This is a PORT: keep it in
428
+ # step with `edge-version.yml`, which is the tested copy and the one `provisionRevision` hashes.
429
+ #
430
+ # ORDER IS THE WHOLE OF IT. Stop the updater first, then install. Doing it the other way leaves a live
431
+ # updater to finish the update it had already started, in the window between installing and believing it.
432
+ $EdgeVersion = if ($env:A11Y_EDGE_VERSION) { $env:A11Y_EDGE_VERSION } else { '152.0.4191.66' }
433
+ $EdgeExe = 'C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe'
434
+ $EdgeMsiUrl = 'https://msedge.sf.dl.delivery.mp.microsoft.com/filestreamingservice/files/2757513f-2ec6-4bb1-b23a-5738a92d7a56/MicrosoftEdgeEnterpriseX64.msi'
435
+ $EdgeMsiSha = '5bcf8cd57351ddaa94419f800f8d90b8cbaf2b485b49d9ddb07d055b76a518df'
436
+
437
+ # BY PREFIX, NOT BY NAME. The role learned this: a freshly installed guest carried a third task,
438
+ # `MicrosoftEdgeUpdateBrowserReplacementTask`, sitting Ready while the two known ones were disabled. A task
439
+ # called "browser replacement" is precisely what this exists to stop, and an enumerated list cannot see it.
440
+ $tasks = Get-ScheduledTask -TaskName 'MicrosoftEdgeUpdate*' -ErrorAction SilentlyContinue
441
+ foreach ($t in $tasks) {
442
+ # STOP BEFORE DISABLE. `Disable-ScheduledTask` stops a task STARTING again; it does not terminate an
443
+ # instance already RUNNING. Measured on a11y-worker-5: the verify refused with "MicrosoftEdgeUpdateTask
444
+ # MachineCore is Running, expected Disabled" — on the one guest that had updated past the pin.
445
+ if ($t.State -eq 'Running') { Stop-ScheduledTask -TaskName $t.TaskName -TaskPath $t.TaskPath | Out-Null }
446
+ if ($t.State -ne 'Disabled') { Disable-ScheduledTask -TaskName $t.TaskName -TaskPath $t.TaskPath | Out-Null }
447
+ }
448
+ OK "$(@($tasks).Count) Edge updater task(s) stopped and disabled"
449
+ foreach ($svc in 'edgeupdate','edgeupdatem','MicrosoftEdgeElevationService') {
450
+ try {
451
+ $s = Get-Service -Name $svc -ErrorAction Stop
452
+ if ($s.Status -ne 'Stopped') { Stop-Service -Name $svc -Force -ErrorAction SilentlyContinue }
453
+ Set-Service -Name $svc -StartupType Disabled -ErrorAction SilentlyContinue
454
+ } catch { } # absent on a box where Edge has not registered them yet; nothing to disable
455
+ }
456
+ OK 'Edge updater services stopped and disabled'
457
+
458
+ # Read the build from the BINARY, never the registry. Edge's own ClientState `pv` read .101 while the file
459
+ # read .93 on one box — the vendor's bookkeeping disagreeing with the file. Trust the binary.
460
+ $have = if (Test-Path -LiteralPath $EdgeExe) { (Get-Item -LiteralPath $EdgeExe).VersionInfo.ProductVersion } else { '(absent)' }
461
+ if ($have -eq $EdgeVersion) {
462
+ OK "Edge is already $EdgeVersion"
463
+ Record 'edge' 'already pinned'
464
+ } else {
465
+ # A NEWER BUILD IS ROLLED BACK HERE, NOT REFUSED -- AND THIS BRANCH USED TO SAY THE OPPOSITE.
466
+ #
467
+ # It said a newer Edge could not be brought back and the box had to be reimaged. That is false, and the
468
+ # phrase itself is banned by `edge-pin-parity.test.ts` -- a substring guard cannot tell a quotation from
469
+ # a claim, so the wrong sentence may not appear here even to be disowned.
470
+ # `ALLOWDOWNGRADE=1` is Microsoft's own supported enterprise rollback, and the role's
471
+ # refusal message has listed it as remedy (2) since it was corrected. This copy was ported from that
472
+ # message BEFORE the correction and kept the wrong half -- the fact-stated-twice shape, introduced by
473
+ # the very act of porting.
474
+ #
475
+ # First boot is also the one moment where a rollback is unambiguously free. The box has taken zero
476
+ # captures, so there is no corpus to invalidate and nothing to weigh against the documented risk
477
+ # ("exposure to known security issues"), which is a reason to prefer following a pin FORWARD on a box
478
+ # that is already working -- not a reason to leave a new box stranded.
479
+ #
480
+ # And stranded is what it was. `browserVersion` is first in `fleet-consistency`'s MUST_MATCH, so a box
481
+ # that loses the race -- fresh Windows ships consumer Edge and it self-updates while `npm install` is
482
+ # still running -- could never join the fleet. Measured 2026-09-04: three boxes won that race and came
483
+ # up at the pin, the fourth came up at 152.0.4191.66 -- a build the enterprise channel had not published
484
+ # YET. It appeared hours later, so following forward was the right answer all along, and the rollback was
485
+ # started on a measurement that expired between taking it and acting on it. A box can be AHEAD of the
486
+ # channel; re-query before concluding otherwise. What stays true is that first boot must not STRAND the
487
+ # box while that is the case, which is what this branch is for.
488
+ $rollingBack = $have -ne '(absent)' -and [version]$have -gt [version]$EdgeVersion
489
+ if ($rollingBack) { Warn "Edge is $have, newer than the pin $EdgeVersion -- rolling back" }
490
+ $msi = Join-Path $env:TEMP "MicrosoftEdgeEnterpriseX64-$EdgeVersion.msi"
491
+ Invoke-WebRequest -UseBasicParsing -Uri $EdgeMsiUrl -OutFile $msi
492
+ $sha = (Get-FileHash -LiteralPath $msi -Algorithm SHA256).Hash.ToLower()
493
+ # BY HASH, because the URL is a delivery endpoint and what it serves can change under a stable address.
494
+ if ($sha -ne $EdgeMsiSha) { throw "Edge MSI hash $sha does not match the pinned $EdgeMsiSha" }
495
+ Get-Process msedge -ErrorAction SilentlyContinue | Stop-Process -Force -ErrorAction SilentlyContinue
496
+ # NOT `$args` -- that is a PowerShell automatic variable, and shadowing it inside a script is the kind
497
+ # of thing that works until something in this block calls a function.
498
+ $msiArgs = @('/i', $msi, '/quiet', '/norestart')
499
+ if ($rollingBack) { $msiArgs += 'ALLOWDOWNGRADE=1' }
500
+ Invoke-Native 'msiexec.exe' $msiArgs $(if ($rollingBack) { 'roll Edge back to the pin' } else { 'install pinned Edge' })
501
+ # THE MIDDLE STEP IS NOT OPTIONAL. A Chromium install stages the new launcher as `new_msedge.exe` and
502
+ # leaves `msedge.exe` alone, because the running browser holds it; the rename is a separate operation the
503
+ # updater performs later — and we have just disabled the updater. Measured on a11y-worker-3: win_package
504
+ # reported success, left the old build in place, and a FULL REBOOT did not complete it.
505
+ $setup = Get-ChildItem 'C:\Program Files (x86)\Microsoft\Edge\Application' -Filter setup.exe -Recurse `
506
+ -ErrorAction SilentlyContinue | Select-Object -First 1 -ExpandProperty FullName
507
+ if ($setup) {
508
+ Start-Process -FilePath $setup -ArgumentList '--rename-chrome-exe','--system-level' -Wait -NoNewWindow
509
+ } else { Warn 'no setup.exe found to complete the rename' }
510
+ $now = if (Test-Path -LiteralPath $EdgeExe) { (Get-Item -LiteralPath $EdgeExe).VersionInfo.ProductVersion } else { '(absent)' }
511
+ if ($now -eq $EdgeVersion) { OK "Edge pinned at $EdgeVersion"; Record 'edge' 'pinned' }
512
+ else {
513
+ # A ROLLBACK THAT DID NOT LAND MUST SAY SO IN THOSE WORDS. "wrong (152.0.4191.66)" after a rollback
514
+ # attempt reads as an install that half-worked; it means msiexec declined the downgrade, which is a
515
+ # different fault with a different remedy.
516
+ Warn "Edge reads $now after $(if ($rollingBack) { 'rollback' } else { 'install' }), wanted $EdgeVersion"
517
+ Record 'edge' "$(if ($rollingBack) { 'ROLLBACK REFUSED' } else { 'wrong' }) ($now)"
518
+ }
519
+ }
520
+
521
+ Step 6 'Hand off to the provisioning script'
522
+ # Located by SEARCH, not by a hardcoded path. This read 'scripts\provision-nvda-worker.ps1'
523
+ # and the repo has since moved everything under packages/ -- so the bootstrap cloned
524
+ # successfully and then died here with "Not found", after doing all the expensive work.
525
+ # A path spelled out in one script and owned by another is exactly the coupling that rots,
526
+ # and the same restructure silently broke the provision-revision hash list below.
527
+ $provision = Get-ChildItem -Path $RepoPath -Filter 'provision-nvda-worker.ps1' -Recurse -File `
528
+ -ErrorAction SilentlyContinue | Select-Object -First 1 -ExpandProperty FullName
529
+ if (-not $provision) { throw "provision-nvda-worker.ps1 not found anywhere under $RepoPath" }
530
+ OK "provisioning script at $provision"
531
+ $env:A11Y_REPO_PATH = $RepoPath
532
+ & powershell -NoProfile -ExecutionPolicy Bypass -File $provision
533
+ $provisionExit = $LASTEXITCODE
534
+ # 75 means provisioning created the local worker account, armed the a11ybootstrap task, and is
535
+ # rebooting to continue as that user. It is NOT a failure, and we must not fall through to the
536
+ # final step -- that unregisters a11ybootstrap, disarming the handoff we are relying on three
537
+ # lines after arming it.
538
+ # A FLAG, not `exit`. This script is designed to be run as `irm <url> | iex`, and Invoke-Expression
539
+ # evaluates in the CALLER's session -- so `exit` here does not end the script, it terminates the
540
+ # operator's entire PowerShell window. Which it did.
541
+ #
542
+ # `return` would work in both invocation modes, but its behaviour under iex is exactly the kind of
543
+ # thing that is obvious right up until it is wrong, and there is no way to test PowerShell from the
544
+ # machine this is written on. A guarded block cannot be ambiguous.
545
+ $handedOff = $false
546
+ if ($provisionExit -eq 75) {
547
+ OK 'provisioning handed off to the worker account; the machine is rebooting to continue'
548
+ $handedOff = $true
549
+ } elseif ($provisionExit -ne 0) {
550
+ throw "Provisioning failed (exit $provisionExit)."
551
+ }
552
+
553
+ # A fresh install needs ONE reboot before it can capture. Observed on two independent
554
+ # clean builds: provisioning completes, /health answers, NVDA connects -- and every read
555
+ # comes back empty (0 phrases, no error). After a single reboot, capture works.
556
+ #
557
+ # It is NOT ForegroundLockTimeout: that is applied live via SystemParametersInfo by both
558
+ # provisioning and run-server.cmd, and the logs confirm 0 in the session that still failed.
559
+ # The remaining cause is something about the first-logon session (guest-tools drivers
560
+ # settling, or first-logon shell state holding the foreground) and is not yet pinned down.
561
+ # The remedy is reliable and reproducible, so take it: auto-logon plus the at-logon trigger
562
+ # bring the worker back on their own, ~65s later.
563
+ # Skipped entirely on the handoff path: provisioning has already armed a11ybootstrap and
564
+ # scheduled the reboot, and the block below would UNREGISTER that task -- dismantling the
565
+ # handoff moments after it was set up.
566
+ if (-not $handedOff) {
567
+ Step 6 'Reboot to finish (a fresh install cannot capture until it has restarted once)'
568
+
569
+ # What this run actually did. On a re-run most lines should read "already present", and the
570
+ # ones that do not are what changed -- without this, a second run looks identical to the first
571
+ # and you cannot tell whether it fixed anything.
572
+ Write-Host "`n --- what this run did ---" -ForegroundColor Cyan
573
+ foreach ($k in $script:outcomes.Keys) {
574
+ $v = $script:outcomes[$k]
575
+ $colour = if ("$v" -like 'FAILED*') { 'Red' } elseif ("$v" -like '*already*') { 'DarkGray' } else { 'Green' }
576
+ Write-Host (" {0,-8} {1}" -f $k, $v) -ForegroundColor $colour
577
+ }
578
+ $failed = $script:outcomes.Keys | Where-Object { "$($script:outcomes[$_])" -like 'FAILED*' }
579
+ if ($failed) { Warn "re-run this script to retry: $($failed -join ', ')" }
580
+
581
+ # Remove the first-run continuation task, if provisioning left one. Getting this far means setup
582
+ # succeeded, and leaving it armed would re-run the whole bootstrap at EVERY logon.
583
+ #
584
+ # That is not merely untidy, it is a boot loop: this script reboots when /health is not answering,
585
+ # and /health is never answering at the moment a logon task fires. Run once, then disarm.
586
+ #
587
+ # Deliberately NOT turned into a general update-on-boot mechanism, tempting as that is. Three
588
+ # reasons, all of which matter more than the convenience:
589
+ # - `workerCode` is recorded on every capture so you know what produced it. A worker that
590
+ # silently updates itself on reboot can span two code versions inside one corpus run, and the
591
+ # provenance stops meaning anything.
592
+ # - `worker:deploy` exists and VERIFIES over /health.code, which shares no failure mode with the
593
+ # push. An unattended self-update has no such check.
594
+ # - one bad push would then brick every box in the fleet at its next restart, simultaneously.
595
+ if (Get-ScheduledTask -TaskName 'a11ybootstrap' -ErrorAction SilentlyContinue) {
596
+ Unregister-ScheduledTask -TaskName 'a11ybootstrap' -Confirm:$false -ErrorAction SilentlyContinue
597
+ OK "removed the 'a11ybootstrap' first-run task -- setup is complete, it will not run again"
598
+ Record 'first-run task' 'removed'
599
+ }
600
+
601
+ # Only reboot if the worker is not ALREADY answering. The reboot exists because a freshly
602
+ # installed Windows cannot capture until it has restarted once -- it is not a general remedy,
603
+ # and rebooting a healthy box on every re-run turns "run it again to fix one step" into an
604
+ # outage. Checked over HTTP because that is the channel the worker actually serves on.
605
+ $alreadyServing = $false
606
+ # `if/else`, not the `? :` ternary: this runs on Windows PowerShell 5.1, where the ternary is a
607
+ # PARSE error -- so it would not fail at this line, it would refuse to load the whole script.
608
+ $workerPort = if ($env:A11Y_PORT) { $env:A11Y_PORT } else { 8765 }
609
+ try {
610
+ $probe = Invoke-WebRequest -Uri "http://127.0.0.1:$workerPort/health" -UseBasicParsing -TimeoutSec 3
611
+ $alreadyServing = $probe.StatusCode -eq 200
612
+ } catch {
613
+ # Not answering is the normal case on a first run; it is why we are about to reboot.
614
+ }
615
+ if ($alreadyServing) {
616
+ OK 'worker is already serving /health -- skipping the reboot'
617
+ } else {
618
+ OK 'rebooting now; the worker restarts itself via auto-logon + the at-logon task'
619
+ Start-Process -FilePath 'shutdown.exe' -ArgumentList '/r','/t','5' -NoNewWindow
620
+ }
621
+
622
+ Write-Host @"
623
+
624
+ --- Bootstrap complete ---
625
+
626
+ Reach it from your Mac:
627
+
628
+ A11Y_WORKER=http://<vm-ip>:8765 npm run witness -- https://example.com --task "..."
629
+
630
+ Manage it with the fleet playbooks:
631
+
632
+ Set A11Y_OPERATOR_KEY before running this script and the key is installed for you, which
633
+ is what makes every later operation unattended:
634
+
635
+ `$env:A11Y_OPERATOR_KEY = 'ssh-ed25519 AAAA... you@host'
636
+
637
+ If it was not set, install it once from the control plane:
638
+
639
+ ansible-playbook ssh-key.yml -l <host> -e a11y_operator_key="`$(cat ~/.ssh/id_ed25519.pub)"
640
+
641
+ Then add this box to packages/control/ansible/inventory.yml -- the ONE place the fleet
642
+ is defined. "npm run fleet:env" derives A11Y_WORKERS from it.
643
+
644
+ Auto-logon, so the interactive session survives a reboot:
645
+ Auto-logon is configured automatically and needs NO password: provisioning gives the
646
+ console account a blank password, which LimitBlankPasswordUse confines to console
647
+ logon only.
648
+ "@
649
+ }